123 lines
4.0 KiB
Python
123 lines
4.0 KiB
Python
|
|
"""Tests for text processing utilities"""
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
from sentiment_engine.utils.text import clean_html, extract_tickers, extract_cashtags, detect_language, normalize_text, split_into_sentences, compute_token_proximity
|
||
|
|
|
||
|
|
|
||
|
|
class TestTextCleaning:
|
||
|
|
"""Test text cleaning functions"""
|
||
|
|
|
||
|
|
def test_clean_html_removes_tags(self):
|
||
|
|
text = "<p>Hello <b>world</b></p>"
|
||
|
|
cleaned = clean_html(text)
|
||
|
|
assert "<p>" not in cleaned
|
||
|
|
assert "<b>" not in cleaned
|
||
|
|
assert "Hello world" in cleaned
|
||
|
|
|
||
|
|
def test_clean_html_unescapes_entities(self):
|
||
|
|
text = "<div> & \"quoted\""
|
||
|
|
cleaned = clean_html(text)
|
||
|
|
assert "<div>" not in cleaned # HTML tags are removed
|
||
|
|
assert "&" in cleaned # HTML entities are unescaped
|
||
|
|
assert '"quoted"' in cleaned
|
||
|
|
|
||
|
|
def test_clean_html_normalizes_whitespace(self):
|
||
|
|
text = "Hello world\n\n\n\t\tagain"
|
||
|
|
cleaned = clean_html(text)
|
||
|
|
assert "Hello world again" == cleaned
|
||
|
|
|
||
|
|
def test_normalize_text_removes_urls(self):
|
||
|
|
text = "Check out https://example.com and http://test.org"
|
||
|
|
normalized = normalize_text(text)
|
||
|
|
assert "https://example.com" not in normalized
|
||
|
|
assert "http://test.org" not in normalized
|
||
|
|
|
||
|
|
|
||
|
|
class TestTickerExtraction:
|
||
|
|
"""Test ticker and cashtag extraction"""
|
||
|
|
|
||
|
|
def test_extract_tickers_basic(self):
|
||
|
|
text = "BTC and ETH are pumping"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
|
||
|
|
def test_extract_tickers_with_dollar(self):
|
||
|
|
text = "$BTC $ETH $SOL"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
assert "SOL" in tickers
|
||
|
|
|
||
|
|
def test_extract_tickers_filters_false_positives(self):
|
||
|
|
text = "THE CEO OF API COMPANY SAYS BTC"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
assert "THE" not in tickers
|
||
|
|
assert "CEO" not in tickers
|
||
|
|
assert "API" not in tickers
|
||
|
|
assert "BTC" in tickers
|
||
|
|
|
||
|
|
def test_extract_cashtags(self):
|
||
|
|
text = "Buying $BTC and $ETH today"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
assert "$BTC" in cashtags
|
||
|
|
assert "$ETH" in cashtags
|
||
|
|
|
||
|
|
def test_extract_cashtags_case_insensitive(self):
|
||
|
|
text = "Buying $btc and $Eth"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
assert "$BTC" in cashtags
|
||
|
|
assert "$ETH" in cashtags
|
||
|
|
|
||
|
|
|
||
|
|
class TestLanguageDetection:
|
||
|
|
"""Test language detection"""
|
||
|
|
|
||
|
|
def test_detect_english(self):
|
||
|
|
text = "Bitcoin surges to new all-time high as institutional adoption accelerates"
|
||
|
|
lang = detect_language(text)
|
||
|
|
assert lang == "en"
|
||
|
|
|
||
|
|
def test_detect_short_text_defaults_en(self):
|
||
|
|
text = "BTC up"
|
||
|
|
lang = detect_language(text)
|
||
|
|
assert lang == "en"
|
||
|
|
|
||
|
|
|
||
|
|
class TestSentenceSplitting:
|
||
|
|
"""Test sentence splitting"""
|
||
|
|
|
||
|
|
def test_split_sentences(self):
|
||
|
|
text = "First sentence. Second sentence! Third sentence?"
|
||
|
|
sentences = split_into_sentences(text)
|
||
|
|
assert len(sentences) == 3
|
||
|
|
assert "First sentence" in sentences[0]
|
||
|
|
assert "Second sentence" in sentences[1]
|
||
|
|
assert "Third sentence" in sentences[2]
|
||
|
|
|
||
|
|
|
||
|
|
class TestTokenProximity:
|
||
|
|
"""Test token proximity computation"""
|
||
|
|
|
||
|
|
def test_proximity_close(self):
|
||
|
|
sentence = "BTC surges to new highs"
|
||
|
|
keywords = ["surges", "pumps", "moon"]
|
||
|
|
proximity = compute_token_proximity(sentence, keywords, "BTC")
|
||
|
|
assert proximity > 0.5 # "surges" is close to "BTC"
|
||
|
|
|
||
|
|
def test_proximity_far(self):
|
||
|
|
sentence = "The asset BTC which we mentioned earlier surges"
|
||
|
|
keywords = ["surges"]
|
||
|
|
proximity = compute_token_proximity(sentence, keywords, "BTC")
|
||
|
|
assert proximity < 1.0 # Further away
|
||
|
|
|
||
|
|
def test_proximity_no_match(self):
|
||
|
|
sentence = "ETH pumps hard"
|
||
|
|
keywords = ["surges"]
|
||
|
|
proximity = compute_token_proximity(sentence, keywords, "BTC")
|
||
|
|
assert proximity == 0.0 # BTC not in sentence
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
pytest.main([__file__, "-v"])
|