Add sentiment_engine with CryptoSentimentCalibrator fixes - improved keyword lists, lowered FinBERT threshold, added neutral handling

This commit is contained in:
Codex
2026-09-14 13:30:05 +02:00
parent 19a7812094
commit a276aeaded
149 changed files with 35226 additions and 0 deletions

View File

@@ -0,0 +1,501 @@
"""Mock models for testing without external dependencies"""
import asyncio
import logging
import torch
from typing import Dict, List, Optional, Tuple, Any
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from sentiment_engine.schemas.processed import EventClassification, EventType
logger = logging.getLogger(__name__)
class MockSentimentModel:
"""Mock sentiment model for testing without external dependencies"""
def __init__(self, device: str = "cpu"):
self.device = device
def __call__(self, **inputs):
"""Mock forward pass"""
batch_size = inputs["input_ids"].shape[0]
# Return mock logits: [batch_size, 3] for negative, neutral, positive
logits = torch.randn(batch_size, 3, device=self.device)
return type('Outputs', (), {'logits': logits})()
class MockEmotionModel:
"""Mock emotion model for testing"""
def __init__(self, device: str = "cpu"):
self.device = device
def __call__(self, **inputs):
"""Mock forward pass"""
batch_size = inputs["input_ids"].shape[0]
# Return mock logits: [batch_size, 6] for 6 emotions
logits = torch.randn(batch_size, 6, device=self.device)
return type('Outputs', (), {'logits': logits})()
class MockTokenizer:
"""Mock tokenizer for testing"""
def __init__(self):
self.vocab_size = 30522
def __call__(self, text, return_tensors="pt", truncation=True, max_length=512, padding=True):
"""Mock tokenization"""
if isinstance(text, list):
batch_size = len(text)
else:
batch_size = 1
text = [text]
# Create mock input_ids and attention_mask
seq_len = min(max(len(t.split()) for t in text) + 2, 512)
input_ids = torch.randint(1, 1000, (batch_size, 512))
attention_mask = torch.ones_like(input_ids)
return {
"input_ids": input_ids,
"attention_mask": attention_mask
}
@classmethod
def from_pretrained(cls, model_name: str):
return MockTokenizer()
def save_pretrained(self, path: str):
pass
class MockModel:
def __init__(self, device="cpu"):
self.device = device
def to(self, device):
self.device = device
return self
def eval(self):
return self
def __call__(self, **inputs):
batch_size = inputs["input_ids"].shape[0]
logits = torch.randn(batch_size, 3) # 3 classes: neg, neu, pos
return type('Outputs', (), {'logits': logits})()
class MockSentimentEmotionAnalyzer:
"""Mock sentiment/emotion analyzer for testing"""
def __init__(self, device: str = "cpu"):
self.device = device
self._tokenizer = None
self._model = None
self._emotion_model = None
self._emotion_tokenizer = None
self._labels = ["negative", "neutral", "positive"]
self._emotion_labels = ["joy", "fear", "anger", "greed", "sadness", "neutral"]
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
# Create mock sentiment scores
sentiment_scores = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_scores = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
sentiment_results[asset_id] = sentiment_scores
emotion_results[asset_id] = emotion_scores
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
# Create mock sentiment scores
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
# Create mock sentiment scores
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
# Create mock sentiment scores
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
async def analyze(
self,
text: str,
asset_mentions: List[Dict]
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
"""Mock sentiment/emotion analysis"""
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
# Simple heuristic based on text content
text_lower = text.lower() if isinstance(text, str) else ""
# Simple keyword-based sentiment
positive_words = ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"]
negative_words = ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"]
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text_lower)
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text_lower)
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
# Create mock sentiment scores
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
# Simple emotions
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self) -> None:
"""Mock initialization"""
pass
def create_mock_event_classifier():
"""Create mock event classifier"""
classifier = type('MockEventClassifier', (), {
'EVENT_KEYWORDS': {
'listing': ["listing", "listed", "debut", "launch", "goes live", "trading starts"],
'hack': ["hack", "hacked", "exploit", "exploited", "breach", "stolen", "theft"],
'regulatory': ["sec", "cftc", "regulation", "regulatory", "compliance"],
},
'EVENT_TYPES': ["listing", "hack", "regulatory", "delisting", "governance",
"upgrade", "partnership", "earnings", "macro", "liquidation", "whale", "manipulation"]
})()
return classifier
def create_mock_asset_mapper():
"""Create mock asset mapper"""
mapper = type('MockAssetMapper', (), {
'aliases': {"VITALIK": "ETH", "CZ": "BNB", "ELON": "DOGE", "SAYLOR": "BTC"},
'known_entities': {
"BTC": {"name": "Bitcoin", "type": "crypto", "contracts": []},
"ETH": {"name": "Ethereum", "type": "crypto", "contracts": ["0xC02aaA39b223FE8D0A0e5C4F27eAD9083C756Cc2"]},
"SOL": {"name": "Solana", "type": "crypto", "contracts": ["So11111111111111111111111111111111111111112"]},
}
})()
return mapper
def create_mock_entity_extractor():
"""Create mock entity extractor"""
from sentiment_engine.nlp.entity_extraction import EntityExtractor, AssetMapper
asset_mapper = type('MockAssetMapper', (), {
'aliases': {"VITALIK": "ETH", "CZ": "BNB", "ELON": "DOGE", "SAYLOR": "BTC"},
'known_entities': {
"BTC": {"name": "Bitcoin", "type": "crypto", "contracts": []},
"ETH": {"name": "Ethereum", "type": "crypto", "contracts": ["0xC02aaA39b223FE8D0A0e5C4F27eAD9083C756Cc2"]},
"SOL": {"name": "Solana", "type": "crypto", "contracts": ["So11111111111111111111111111111111111111112"]},
}
})()
from sentiment_engine.nlp.entity_extraction import EntityExtractor, AssetMapper
extractor = EntityExtractor(asset_mapper)
# Override initialize to not load spaCy
extractor.initialize = lambda: None
return extractor
# Export all mocks
__all__ = [
"MockSentimentModel",
"MockEmotionModel",
"MockTokenizer",
"MockModel",
"MockTokenizer",
"MockSentimentEmotionAnalyzer",
"MockModel",
"MockAssetMapper",
"create_mock_sentiment_analyzer",
"create_mock_event_classifier",
"create_mock_asset_mapper",
"create_mock_entity_extractor",
]