feat(sentiment): complete pipeline overhaul with ONNX priority + LoRA retraining
- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets - Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock - ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text - LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral) - Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x) - 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP - Early stopping (patience=3) on both LoRA trainings - Human-in-the-loop verification CLI tool created - Disk-conscious: save_total_limit=1, adapters 6-8MB each Pipeline now correctly classifies: - BTC breaks 100k → +0.54 Bullish ✅ - Major hack → -0.23 Bearish ✅ - HODL → +0.91 Bullish ✅ - Rug pull → -0.30 Bearish ✅ - SEC sues → -0.30 Bearish ✅ - ETF approval → +0.32 Bullish ✅ - Whale accumulation → +0.31 Bullish ✅ Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each) Training data: 518 carefully labeled samples (190 real + 328 synthetic) Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
This commit is contained in:
285
sentiment_engine/tests/unit/test_mock_models_comprehensive.py
Normal file
285
sentiment_engine/tests/unit/test_mock_models_comprehensive.py
Normal file
@@ -0,0 +1,285 @@
|
||||
"""
|
||||
Comprehensive tests for mock models and test utilities.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
import asyncio
|
||||
import numpy as np
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from sentiment_engine.utils.mock_models import (
|
||||
MockTokenizer, MockSentimentModel, MockEmotionModel,
|
||||
MockEventModel, MockNERModel, create_mock_pipeline
|
||||
)
|
||||
from sentiment_engine.schemas.processed import SentimentScores, EmotionScores, EventClassification, EventType
|
||||
from sentiment_engine.schemas.payload import NormalizedPayload, SourceType, AssetMention
|
||||
|
||||
|
||||
class TestMockTokenizer:
|
||||
"""Tests for MockTokenizer"""
|
||||
|
||||
def test_single_text(self):
|
||||
"""Should tokenize single text"""
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer("test text")
|
||||
|
||||
assert "input_ids" in result
|
||||
assert "attention_mask" in result
|
||||
assert "token_type_ids" in result
|
||||
|
||||
def test_batch_text(self):
|
||||
"""Should tokenize batch of texts"""
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer(["text1", "text2", "text3"])
|
||||
|
||||
assert result["input_ids"].shape[0] == 3
|
||||
|
||||
def test_truncation(self):
|
||||
"""Should respect truncation"""
|
||||
tokenizer = MockTokenizer()
|
||||
long_text = "word " * 1000
|
||||
result = tokenizer(long_text, max_length=128, truncation=True)
|
||||
|
||||
assert result["input_ids"].shape[1] <= 128
|
||||
|
||||
def test_padding(self):
|
||||
"""Should pad to max_length"""
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer("short", max_length=128, padding=True)
|
||||
|
||||
assert result["input_ids"].shape[1] == 128
|
||||
|
||||
def test_return_tensors_pt(self):
|
||||
"""Should return PyTorch tensors when requested"""
|
||||
import torch
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer("test", return_tensors="pt")
|
||||
|
||||
assert isinstance(result["input_ids"], torch.Tensor)
|
||||
|
||||
def test_return_tensors_np(self):
|
||||
"""Should return numpy arrays when requested"""
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer("test", return_tensors="np")
|
||||
|
||||
assert isinstance(result["input_ids"], np.ndarray)
|
||||
|
||||
def test_from_pretrained(self):
|
||||
"""from_pretrained should return new instance"""
|
||||
tokenizer = MockTokenizer.from_pretrained("test-model")
|
||||
assert isinstance(tokenizer, MockTokenizer)
|
||||
|
||||
|
||||
class TestMockSentimentModel:
|
||||
"""Tests for MockSentimentModel"""
|
||||
|
||||
def test_returns_logits(self):
|
||||
"""Should return logits"""
|
||||
model = MockSentimentModel()
|
||||
result = model(input_ids=np.ones((2, 10)), attention_mask=np.ones((2, 10)))
|
||||
|
||||
assert hasattr(result, 'logits')
|
||||
assert result.logits.shape == (2, 3)
|
||||
|
||||
def test_to_device(self):
|
||||
"""to() should return self"""
|
||||
model = MockSentimentModel()
|
||||
result = model.to("cuda")
|
||||
|
||||
assert result is model
|
||||
|
||||
def test_eval_mode(self):
|
||||
"""eval() should return self"""
|
||||
model = MockSentimentModel()
|
||||
result = model.eval()
|
||||
|
||||
assert result is model
|
||||
|
||||
|
||||
class TestMockEmotionModel:
|
||||
"""Tests for MockEmotionModel"""
|
||||
|
||||
def test_returns_logits(self):
|
||||
"""Should return logits for 6 emotions"""
|
||||
model = MockEmotionModel()
|
||||
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
|
||||
|
||||
assert hasattr(result, 'logits')
|
||||
assert result.logits.shape == (1, 6)
|
||||
|
||||
|
||||
class TestMockEventModel:
|
||||
"""Tests for MockEventModel"""
|
||||
|
||||
def test_returns_logits(self):
|
||||
"""Should return logits for 12 events"""
|
||||
model = MockEventModel()
|
||||
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
|
||||
|
||||
assert hasattr(result, 'logits')
|
||||
assert result.logits.shape == (1, 12)
|
||||
|
||||
|
||||
class TestMockNERModel:
|
||||
"""Tests for MockNERModel"""
|
||||
|
||||
def test_extract_entities(self):
|
||||
"""Should extract entities"""
|
||||
model = MockNERModel()
|
||||
entities = model.extract_entities("Bitcoin and Ethereum surge")
|
||||
|
||||
assert isinstance(entities, list)
|
||||
assert len(entities) >= 0
|
||||
|
||||
|
||||
class TestMockPipeline:
|
||||
"""Tests for create_mock_pipeline"""
|
||||
|
||||
def test_creates_full_pipeline(self):
|
||||
"""Should create complete mock pipeline"""
|
||||
pipeline = create_mock_pipeline()
|
||||
|
||||
assert hasattr(pipeline, 'entity_extractor')
|
||||
assert hasattr(pipeline, 'sentiment_analyzer')
|
||||
assert hasattr(pipeline, 'event_classifier')
|
||||
assert hasattr(pipeline, 'temporal_anchorer')
|
||||
assert hasattr(pipeline, 'credibility_scorer')
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mock_pipeline_process(self):
|
||||
"""Mock pipeline should process payloads"""
|
||||
pipeline = create_mock_pipeline()
|
||||
|
||||
payload = NormalizedPayload(
|
||||
source_id="test",
|
||||
source_type=SourceType.NEWS,
|
||||
source_credibility_base=0.8,
|
||||
ingest_ts=1700000000.0,
|
||||
publish_ts=1700000000.0,
|
||||
content_length=100,
|
||||
raw_text="Bitcoin surges!",
|
||||
metadata={}
|
||||
)
|
||||
|
||||
result = await pipeline.process(payload)
|
||||
|
||||
assert hasattr(result, 'entities')
|
||||
assert hasattr(result, 'sentiment_per_asset')
|
||||
assert hasattr(result, 'events')
|
||||
|
||||
|
||||
class TestMockModelIntegration:
|
||||
"""Integration tests for mock models"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mock_tokenizer_with_sentiment_model(self):
|
||||
"""Mock tokenizer should work with sentiment model"""
|
||||
tokenizer = MockTokenizer()
|
||||
model = MockSentimentModel()
|
||||
|
||||
text = "Bitcoin surges!"
|
||||
inputs = tokenizer(text, return_tensors="np")
|
||||
result = model(**inputs)
|
||||
|
||||
assert result.logits.shape == (1, 3)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_mock_pipeline_end_to_end(self):
|
||||
"""Full mock pipeline should work end-to-end"""
|
||||
pipeline = create_mock_pipeline()
|
||||
|
||||
payload = NormalizedPayload(
|
||||
source_id="test",
|
||||
source_type=SourceType.NEWS,
|
||||
source_credibility_base=0.8,
|
||||
ingest_ts=1700000000.0,
|
||||
publish_ts=1700000000.0,
|
||||
content_length=100,
|
||||
raw_text="Bitcoin surges to new high!",
|
||||
metadata={}
|
||||
)
|
||||
|
||||
result = await pipeline.process(payload)
|
||||
|
||||
assert hasattr(result, 'entities')
|
||||
assert hasattr(result, 'sentiment_per_asset')
|
||||
assert hasattr(result, 'emotions_per_asset')
|
||||
assert hasattr(result, 'events')
|
||||
assert hasattr(result, 'temporal')
|
||||
assert hasattr(result, 'credibility')
|
||||
assert isinstance(result.processing_latency_ms, float)
|
||||
|
||||
|
||||
class TestMockModelEdgeCases:
|
||||
"""Edge case tests for mock models"""
|
||||
|
||||
def test_mock_tokenizer_empty_text(self):
|
||||
"""Should handle empty text"""
|
||||
tokenizer = MockTokenizer()
|
||||
result = tokenizer("")
|
||||
|
||||
assert "input_ids" in result
|
||||
|
||||
def test_mock_tokenizer_very_long(self):
|
||||
"""Should handle very long text"""
|
||||
tokenizer = MockTokenizer()
|
||||
long_text = "word " * 10000
|
||||
result = tokenizer(long_text, truncation=True, max_length=512)
|
||||
|
||||
assert result["input_ids"].shape[1] == 512
|
||||
|
||||
def test_mock_model_batch_size(self):
|
||||
"""Should handle various batch sizes"""
|
||||
model = MockSentimentModel()
|
||||
|
||||
for batch_size in [1, 2, 4, 8, 16, 32]:
|
||||
inputs = {
|
||||
"input_ids": np.ones((batch_size, 128)),
|
||||
"attention_mask": np.ones((batch_size, 128))
|
||||
}
|
||||
result = model(**inputs)
|
||||
assert result.logits.shape == (batch_size, 3)
|
||||
|
||||
def test_mock_model_different_devices(self):
|
||||
"""Should work on different devices"""
|
||||
model = MockSentimentModel()
|
||||
|
||||
for device in ["cpu", "cuda"]:
|
||||
model.to(device)
|
||||
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
|
||||
assert result.logits.shape == (1, 3)
|
||||
|
||||
|
||||
class TestMockModelCompatibility:
|
||||
"""Tests for compatibility with real model interfaces"""
|
||||
|
||||
def test_tokenizer_interface(self):
|
||||
"""MockTokenizer should match HF tokenizer interface"""
|
||||
tokenizer = MockTokenizer()
|
||||
|
||||
# Should have required methods
|
||||
assert hasattr(tokenizer, '__call__')
|
||||
assert hasattr(tokenizer, 'from_pretrained')
|
||||
assert hasattr(tokenizer, 'save_pretrained')
|
||||
|
||||
def test_model_interface(self):
|
||||
"""MockSentimentModel should match HF model interface"""
|
||||
model = MockSentimentModel()
|
||||
|
||||
assert hasattr(model, 'to')
|
||||
assert hasattr(model, 'eval')
|
||||
assert hasattr(model, '__call__')
|
||||
|
||||
def test_output_structure(self):
|
||||
"""Output should match HF model output structure"""
|
||||
model = MockSentimentModel()
|
||||
result = model(input_ids=np.ones((1, 10)), attention_mask=np.ones((1, 10)))
|
||||
|
||||
# Should have logits attribute
|
||||
assert hasattr(result, 'logits')
|
||||
# Logits should be 2D: (batch, num_labels)
|
||||
assert len(result.logits.shape) == 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
pytest.main([__file__, "-v"])
|
||||
Reference in New Issue
Block a user