Files
sentiment-engine/sentiment_engine/tests/unit/test_text_utils.py
Codex c32db97d57 feat(sentiment): complete pipeline overhaul with ONNX priority + LoRA retraining
- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets
- Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock
- ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text
- LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral)
- Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x)
- 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP
- Early stopping (patience=3) on both LoRA trainings
- Human-in-the-loop verification CLI tool created
- Disk-conscious: save_total_limit=1, adapters 6-8MB each

Pipeline now correctly classifies:
- BTC breaks 100k → +0.54 Bullish ✅
- Major hack → -0.23 Bearish ✅
- HODL → +0.91 Bullish ✅
- Rug pull → -0.30 Bearish ✅
- SEC sues → -0.30 Bearish ✅
- ETF approval → +0.32 Bullish ✅
- Whale accumulation → +0.31 Bullish ✅

Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each)
Training data: 518 carefully labeled samples (190 real + 328 synthetic)
Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA
Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
2026-09-27 04:34:49 +02:00

123 lines
4.0 KiB
Python

"""Tests for text processing utilities"""
import pytest
from sentiment_engine.utils.text import clean_html, extract_tickers, extract_cashtags, detect_language, normalize_text, split_into_sentences, compute_token_proximity
class TestTextCleaning:
"""Test text cleaning functions"""
def test_clean_html_removes_tags(self):
text = "<p>Hello <b>world</b></p>"
cleaned = clean_html(text)
assert "<p>" not in cleaned
assert "<b>" not in cleaned
assert "Hello world" in cleaned
def test_clean_html_unescapes_entities(self):
text = "<div> & \"quoted\""
cleaned = clean_html(text)
assert "<div>" not in cleaned # HTML tags are removed
assert "&" in cleaned # HTML entities are unescaped
assert '"quoted"' in cleaned
def test_clean_html_normalizes_whitespace(self):
text = "Hello world\n\n\n\t\tagain"
cleaned = clean_html(text)
assert "Hello world again" == cleaned
def test_normalize_text_removes_urls(self):
text = "Check out https://example.com and http://test.org"
normalized = normalize_text(text)
assert "https://example.com" not in normalized
assert "http://test.org" not in normalized
class TestTickerExtraction:
"""Test ticker and cashtag extraction"""
def test_extract_tickers_basic(self):
text = "BTC and ETH are pumping"
tickers = extract_tickers(text)
assert "BTC" in tickers
assert "ETH" in tickers
def test_extract_tickers_with_dollar(self):
text = "$BTC $ETH $SOL"
tickers = extract_tickers(text)
assert "BTC" in tickers
assert "ETH" in tickers
assert "SOL" in tickers
def test_extract_tickers_filters_false_positives(self):
text = "THE CEO OF API COMPANY SAYS BTC"
tickers = extract_tickers(text)
assert "THE" not in tickers
assert "CEO" not in tickers
assert "API" not in tickers
assert "BTC" in tickers
def test_extract_cashtags(self):
text = "Buying $BTC and $ETH today"
cashtags = extract_cashtags(text)
assert "$BTC" in cashtags
assert "$ETH" in cashtags
def test_extract_cashtags_case_insensitive(self):
text = "Buying $btc and $Eth"
cashtags = extract_cashtags(text)
assert "$BTC" in cashtags
assert "$ETH" in cashtags
class TestLanguageDetection:
"""Test language detection"""
def test_detect_english(self):
text = "Bitcoin surges to new all-time high as institutional adoption accelerates"
lang = detect_language(text)
assert lang == "en"
def test_detect_short_text_defaults_en(self):
text = "BTC up"
lang = detect_language(text)
assert lang == "en"
class TestSentenceSplitting:
"""Test sentence splitting"""
def test_split_sentences(self):
text = "First sentence. Second sentence! Third sentence?"
sentences = split_into_sentences(text)
assert len(sentences) == 3
assert "First sentence" in sentences[0]
assert "Second sentence" in sentences[1]
assert "Third sentence" in sentences[2]
class TestTokenProximity:
"""Test token proximity computation"""
def test_proximity_close(self):
sentence = "BTC surges to new highs"
keywords = ["surges", "pumps", "moon"]
proximity = compute_token_proximity(sentence, keywords, "BTC")
assert proximity > 0.5 # "surges" is close to "BTC"
def test_proximity_far(self):
sentence = "The asset BTC which we mentioned earlier surges"
keywords = ["surges"]
proximity = compute_token_proximity(sentence, keywords, "BTC")
assert proximity < 1.0 # Further away
def test_proximity_no_match(self):
sentence = "ETH pumps hard"
keywords = ["surges"]
proximity = compute_token_proximity(sentence, keywords, "BTC")
assert proximity == 0.0 # BTC not in sentence
if __name__ == "__main__":
pytest.main([__file__, "-v"])