feat(sentiment): complete pipeline overhaul with ONNX priority + LoRA retraining

- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets
- Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock
- ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text
- LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral)
- Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x)
- 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP
- Early stopping (patience=3) on both LoRA trainings
- Human-in-the-loop verification CLI tool created
- Disk-conscious: save_total_limit=1, adapters 6-8MB each

Pipeline now correctly classifies:
- BTC breaks 100k → +0.54 Bullish ✅
- Major hack → -0.23 Bearish ✅
- HODL → +0.91 Bullish ✅
- Rug pull → -0.30 Bearish ✅
- SEC sues → -0.30 Bearish ✅
- ETF approval → +0.32 Bullish ✅
- Whale accumulation → +0.31 Bullish ✅

Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each)
Training data: 518 carefully labeled samples (190 real + 328 synthetic)
Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA
Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
This commit is contained in:
Codex
2026-09-27 04:34:49 +02:00
parent 2ea14bd465
commit c32db97d57
178 changed files with 64849 additions and 42 deletions

View File

@@ -0,0 +1,285 @@
"""
Comprehensive tests for mock models and test utilities.
"""
import pytest
import asyncio
import numpy as np
from unittest.mock import AsyncMock, MagicMock, patch
from sentiment_engine.utils.mock_models import (
MockTokenizer, MockSentimentModel, MockEmotionModel,
MockEventModel, MockNERModel, create_mock_pipeline
)
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores, EventClassification, EventType
from sentiment_engine.schemas.payload import NormalizedPayload, SourceType, AssetMention
class TestMockTokenizer:
"""Tests for MockTokenizer"""
def test_single_text(self):
"""Should tokenize single text"""
tokenizer = MockTokenizer()
result = tokenizer("test text")
assert "input_ids" in result
assert "attention_mask" in result
assert "token_type_ids" in result
def test_batch_text(self):
"""Should tokenize batch of texts"""
tokenizer = MockTokenizer()
result = tokenizer(["text1", "text2", "text3"])
assert result["input_ids"].shape[0] == 3
def test_truncation(self):
"""Should respect truncation"""
tokenizer = MockTokenizer()
long_text = "word " * 1000
result = tokenizer(long_text, max_length=128, truncation=True)
assert result["input_ids"].shape[1] <= 128
def test_padding(self):
"""Should pad to max_length"""
tokenizer = MockTokenizer()
result = tokenizer("short", max_length=128, padding=True)
assert result["input_ids"].shape[1] == 128
def test_return_tensors_pt(self):
"""Should return PyTorch tensors when requested"""
import torch
tokenizer = MockTokenizer()
result = tokenizer("test", return_tensors="pt")
assert isinstance(result["input_ids"], torch.Tensor)
def test_return_tensors_np(self):
"""Should return numpy arrays when requested"""
tokenizer = MockTokenizer()
result = tokenizer("test", return_tensors="np")
assert isinstance(result["input_ids"], np.ndarray)
def test_from_pretrained(self):
"""from_pretrained should return new instance"""
tokenizer = MockTokenizer.from_pretrained("test-model")
assert isinstance(tokenizer, MockTokenizer)
class TestMockSentimentModel:
"""Tests for MockSentimentModel"""
def test_returns_logits(self):
"""Should return logits"""
model = MockSentimentModel()
result = model(input_ids=np.ones((2, 10)), attention_mask=np.ones((2, 10)))
assert hasattr(result, 'logits')
assert result.logits.shape == (2, 3)
def test_to_device(self):
"""to() should return self"""
model = MockSentimentModel()
result = model.to("cuda")
assert result is model
def test_eval_mode(self):
"""eval() should return self"""
model = MockSentimentModel()
result = model.eval()
assert result is model
class TestMockEmotionModel:
"""Tests for MockEmotionModel"""
def test_returns_logits(self):
"""Should return logits for 6 emotions"""
model = MockEmotionModel()
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
assert hasattr(result, 'logits')
assert result.logits.shape == (1, 6)
class TestMockEventModel:
"""Tests for MockEventModel"""
def test_returns_logits(self):
"""Should return logits for 12 events"""
model = MockEventModel()
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
assert hasattr(result, 'logits')
assert result.logits.shape == (1, 12)
class TestMockNERModel:
"""Tests for MockNERModel"""
def test_extract_entities(self):
"""Should extract entities"""
model = MockNERModel()
entities = model.extract_entities("Bitcoin and Ethereum surge")
assert isinstance(entities, list)
assert len(entities) >= 0
class TestMockPipeline:
"""Tests for create_mock_pipeline"""
def test_creates_full_pipeline(self):
"""Should create complete mock pipeline"""
pipeline = create_mock_pipeline()
assert hasattr(pipeline, 'entity_extractor')
assert hasattr(pipeline, 'sentiment_analyzer')
assert hasattr(pipeline, 'event_classifier')
assert hasattr(pipeline, 'temporal_anchorer')
assert hasattr(pipeline, 'credibility_scorer')
@pytest.mark.asyncio
async def test_mock_pipeline_process(self):
"""Mock pipeline should process payloads"""
pipeline = create_mock_pipeline()
payload = NormalizedPayload(
source_id="test",
source_type=SourceType.NEWS,
source_credibility_base=0.8,
ingest_ts=1700000000.0,
publish_ts=1700000000.0,
content_length=100,
raw_text="Bitcoin surges!",
metadata={}
)
result = await pipeline.process(payload)
assert hasattr(result, 'entities')
assert hasattr(result, 'sentiment_per_asset')
assert hasattr(result, 'events')
class TestMockModelIntegration:
"""Integration tests for mock models"""
@pytest.mark.asyncio
async def test_mock_tokenizer_with_sentiment_model(self):
"""Mock tokenizer should work with sentiment model"""
tokenizer = MockTokenizer()
model = MockSentimentModel()
text = "Bitcoin surges!"
inputs = tokenizer(text, return_tensors="np")
result = model(**inputs)
assert result.logits.shape == (1, 3)
@pytest.mark.asyncio
async def test_mock_pipeline_end_to_end(self):
"""Full mock pipeline should work end-to-end"""
pipeline = create_mock_pipeline()
payload = NormalizedPayload(
source_id="test",
source_type=SourceType.NEWS,
source_credibility_base=0.8,
ingest_ts=1700000000.0,
publish_ts=1700000000.0,
content_length=100,
raw_text="Bitcoin surges to new high!",
metadata={}
)
result = await pipeline.process(payload)
assert hasattr(result, 'entities')
assert hasattr(result, 'sentiment_per_asset')
assert hasattr(result, 'emotions_per_asset')
assert hasattr(result, 'events')
assert hasattr(result, 'temporal')
assert hasattr(result, 'credibility')
assert isinstance(result.processing_latency_ms, float)
class TestMockModelEdgeCases:
"""Edge case tests for mock models"""
def test_mock_tokenizer_empty_text(self):
"""Should handle empty text"""
tokenizer = MockTokenizer()
result = tokenizer("")
assert "input_ids" in result
def test_mock_tokenizer_very_long(self):
"""Should handle very long text"""
tokenizer = MockTokenizer()
long_text = "word " * 10000
result = tokenizer(long_text, truncation=True, max_length=512)
assert result["input_ids"].shape[1] == 512
def test_mock_model_batch_size(self):
"""Should handle various batch sizes"""
model = MockSentimentModel()
for batch_size in [1, 2, 4, 8, 16, 32]:
inputs = {
"input_ids": np.ones((batch_size, 128)),
"attention_mask": np.ones((batch_size, 128))
}
result = model(**inputs)
assert result.logits.shape == (batch_size, 3)
def test_mock_model_different_devices(self):
"""Should work on different devices"""
model = MockSentimentModel()
for device in ["cpu", "cuda"]:
model.to(device)
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
assert result.logits.shape == (1, 3)
class TestMockModelCompatibility:
"""Tests for compatibility with real model interfaces"""
def test_tokenizer_interface(self):
"""MockTokenizer should match HF tokenizer interface"""
tokenizer = MockTokenizer()
# Should have required methods
assert hasattr(tokenizer, '__call__')
assert hasattr(tokenizer, 'from_pretrained')
assert hasattr(tokenizer, 'save_pretrained')
def test_model_interface(self):
"""MockSentimentModel should match HF model interface"""
model = MockSentimentModel()
assert hasattr(model, 'to')
assert hasattr(model, 'eval')
assert hasattr(model, '__call__')
def test_output_structure(self):
"""Output should match HF model output structure"""
model = MockSentimentModel()
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
# Should have logits attribute
assert hasattr(result, 'logits')
# Logits should be 2D: (batch, num_labels)
assert len(result.logits.shape) == 2
if __name__ == "__main__":
pytest.main([__file__, "-v"])