Add sentiment_engine with CryptoSentimentCalibrator fixes - improved keyword lists, lowered FinBERT threshold, added neutral handling

This commit is contained in:
Codex
2026-09-14 13:30:05 +02:00
parent 19a7812094
commit a276aeaded
149 changed files with 35226 additions and 0 deletions

View File

@@ -0,0 +1,652 @@
"""Tests for mock models"""
import pytest
import torch
import sys
import os
sys.path.insert(0, os.path.join(os.path.dirname(__file__), '../../src'))
# Mock classes defined in this file
# (moved here to avoid import issues)
class MockSentimentModel:
"""Mock sentiment model for testing without external dependencies"""
def __init__(self, device: str = "cpu"):
self.device = device
def __call__(self, **inputs):
batch_size = inputs["input_ids"].shape[0]
logits = torch.randn(batch_size, 3, device=self.device)
return type('Outputs', (), {'logits': logits})()
class MockEmotionModel:
def __init__(self, device: str = "cpu"):
self.device = device
def __call__(self, **inputs):
batch_size = inputs["input_ids"].shape[0]
logits = torch.randn(batch_size, 6, device=self.device)
return type('Outputs', (), {'logits': logits})()
class MockTokenizer:
def __init__(self):
self.vocab_size = 30522
def __call__(self, text, return_tensors="pt", truncation=True, max_length=512, padding=True):
if isinstance(text, list):
batch_size = len(text)
else:
batch_size = 1
text = [text]
seq_len = min(max(len(t.split()) for t in text) + 2, 512)
input_ids = torch.randint(1, 1000, (batch_size, 512))
attention_mask = torch.ones_like(input_ids)
return {
"input_ids": input_ids,
"attention_mask": attention_mask
}
@classmethod
def from_pretrained(cls, model_name: str):
return MockTokenizer()
def save_pretrained(self, path: str):
pass
class MockModel:
def __init__(self, device="cpu"):
self.device = device
def to(self, device):
self.device = device
return self
def eval(self):
return self
def __call__(self, **inputs):
batch_size = inputs["input_ids"].shape[0]
logits = torch.randn(batch_size, 3)
return type('Outputs', (), {'logits': logits})()
def create_mock_sentiment_analyzer(device: str = "cpu"):
class MockSentimentEmotionAnalyzer:
def __init__(self, device: str = "cpu"):
self.device = device
self._tokenizer = None
self._model = None
self._emotion_model = None
self._emotion_tokenizer = None
self._labels = ["negative", "neutral", "positive"]
self._emotion_labels = ["joy", "fear", "anger", "greed", "sadness", "neutral"]
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
async def analyze(
self,
text: str,
asset_mentions: list
):
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores
from typing import Dict, Any, List, Tuple
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
span = mention.get("span", (0, 0))
text_lower = text.lower() if isinstance(text, str) else ""
pos_count = sum(1 for kw in ["rally", "surge", "pump", "moon", "bullish", "profit", "gain", "win", "success", "breakthrough"] if kw in text.lower())
neg_count = sum(1 for kw in ["crash", "dump", "panic", "fear", "scared", "worried", "risk", "danger", "collapse", "liquidation"] if kw in text.lower())
polarity = (pos_count - neg_count) * 0.3
polarity = max(-1.0, min(1.0, polarity))
confidence = min(0.9, 0.3 + abs(polarity) * 0.5)
sentiment_results = {}
emotion_results = {}
for mention in asset_mentions:
asset_id = mention.get("asset_id")
sentiment_results[asset_id] = type('SentimentScores', (), {
'polarity': polarity,
'confidence': confidence,
'positive_prob': max(0, polarity),
'negative_prob': max(0, -polarity),
'neutral_prob': 1 - abs(polarity)
})()
emotion_results[asset_id] = type('EmotionScores', (), {
'joy': 0.5 if polarity > 0 else 0.1,
'fear': 0.5 if polarity < 0 else 0.1,
'anger': 0.1,
'greed': 0.5 if polarity > 0.2 else 0.1,
'sadness': 0.5 if polarity < -0.2 else 0.1,
'intensity': 0.5
})()
return sentiment_results, emotion_results
async def initialize(self):
pass
analyzer = type('MockSentimentEmotionAnalyzer', (), {
'device': 'cpu',
'_tokenizer': None,
'_model': None,
'_emotion_model': None,
'_emotion_tokenizer': None,
'_labels': ["negative", "neutral", "positive"],
'_emotion_labels': ["joy", "fear", "anger", "greed", "sadness", "neutral"],
'initialize': lambda self: None,
'analyze': lambda self, text, asset_mentions: None,
})()
return analyzer
def create_mock_event_classifier():
classifier = type('MockEventClassifier', (), {
'EVENT_KEYWORDS': {
'listing': ["listing", "listed", "debut", "launch", "goes live", "trading starts"],
'hack': ["hack", "hacked", "exploit", "exploited", "breach", "stolen", "theft"],
'regulatory': ["sec", "cftc", "regulation", "regulatory", "compliance"],
},
'EVENT_TYPES': ["listing", "hack", "regulatory", "delisting", "governance",
"upgrade", "partnership", "earnings", "macro", "liquidation", "whale", "manipulation"],
'_classify_sync': lambda self, text, asset_mentions: [
type('EventClassification', (), {
'event_type': type('EventType', (), {'value': 'listing'})(),
'confidence': 0.8,
'assets_involved': ['BTC'],
'key_details': {},
'severity': 0.5
})()
]
})()
return classifier
def create_mock_asset_mapper():
mapper = type('MockAssetMapper', (), {
'aliases': {"VITALIK": "ETH", "CZ": "BNB", "ELON": "DOGE", "SAYLOR": "BTC"},
'known_entities': {
"BTC": {"name": "Bitcoin", "type": "crypto", "contracts": []},
"ETH": {"name": "Ethereum", "type": "crypto", "contracts": ["0xC02aaA39b223FE8D0A0e5C4F27eAD9083C756Cc2"]},
"SOL": {"name": "Solana", "type": "crypto", "contracts": ["So11111111111111111111111111111111111111112"]},
},
'map_ticker': lambda self, ticker: ("BTC", 0.9) if ticker == "BTC" else ("ETH", 0.7) if ticker == "VITALIK" else ("UNKNOWNTICKER", 0.5),
'map_contract': lambda self, address: ("ETH", 0.99, "ethereum") if address == "0xC02aaA39b223FE8D0A0e5C4F27eAD9083C756Cc2" else ("UNKNOWN", 0.3, None),
})()
return mapper
def create_mock_entity_extractor():
from sentiment_engine.nlp.entity_extraction import EntityExtractor, AssetMapper
asset_mapper = type('MockAssetMapper', (), {
'aliases': {"VITALIK": "ETH", "CZ": "BNB", "ELON": "DOGE", "SAYLOR": "BTC"},
'known_entities': {
"BTC": {"name": "Bitcoin", "type": "crypto", "contracts": []},
"ETH": {"name": "Ethereum", "type": "crypto", "contracts": ["0xC02aaA39b223FE8D0A0e5C4F27eAD9083C756Cc2"]},
"SOL": {"name": "Solana", "type": "crypto", "contracts": ["So11111111111111111111111111111111111111112"]},
}
})()
from sentiment_engine.nlp.entity_extraction import EntityExtractor, AssetMapper
extractor = EntityExtractor(asset_mapper)
ext.initialize = lambda: None
return ext
# Export all mocks
__all__ = [
"MockSentimentModel",
"MockEmotionModel",
"MockTokenizer",
"MockModel",
"MockTokenizer",
"MockSentimentEmotionAnalyzer",
"MockModel",
"MockAssetMapper",
"create_mock_sentiment_analyzer",
"create_mock_event_classifier",
"create_mock_asset_mapper",
"create_mock_entity_extractor",
]