"""Tests for text processing utilities""" import pytest from sentiment_engine.utils.text import clean_html, extract_tickers, extract_cashtags, detect_language, normalize_text, split_into_sentences, compute_token_proximity class TestTextCleaning: """Test text cleaning functions""" def test_clean_html_removes_tags(self): text = "

Hello world

" cleaned = clean_html(text) assert "

" not in cleaned assert "" not in cleaned assert "Hello world" in cleaned def test_clean_html_unescapes_entities(self): text = "

& \"quoted\"" cleaned = clean_html(text) assert "
" not in cleaned # HTML tags are removed assert "&" in cleaned # HTML entities are unescaped assert '"quoted"' in cleaned def test_clean_html_normalizes_whitespace(self): text = "Hello world\n\n\n\t\tagain" cleaned = clean_html(text) assert "Hello world again" == cleaned def test_normalize_text_removes_urls(self): text = "Check out https://example.com and http://test.org" normalized = normalize_text(text) assert "https://example.com" not in normalized assert "http://test.org" not in normalized class TestTickerExtraction: """Test ticker and cashtag extraction""" def test_extract_tickers_basic(self): text = "BTC and ETH are pumping" tickers = extract_tickers(text) assert "BTC" in tickers assert "ETH" in tickers def test_extract_tickers_with_dollar(self): text = "$BTC $ETH $SOL" tickers = extract_tickers(text) assert "BTC" in tickers assert "ETH" in tickers assert "SOL" in tickers def test_extract_tickers_filters_false_positives(self): text = "THE CEO OF API COMPANY SAYS BTC" tickers = extract_tickers(text) assert "THE" not in tickers assert "CEO" not in tickers assert "API" not in tickers assert "BTC" in tickers def test_extract_cashtags(self): text = "Buying $BTC and $ETH today" cashtags = extract_cashtags(text) assert "$BTC" in cashtags assert "$ETH" in cashtags def test_extract_cashtags_case_insensitive(self): text = "Buying $btc and $Eth" cashtags = extract_cashtags(text) assert "$BTC" in cashtags assert "$ETH" in cashtags class TestLanguageDetection: """Test language detection""" def test_detect_english(self): text = "Bitcoin surges to new all-time high as institutional adoption accelerates" lang = detect_language(text) assert lang == "en" def test_detect_short_text_defaults_en(self): text = "BTC up" lang = detect_language(text) assert lang == "en" class TestSentenceSplitting: """Test sentence splitting""" def test_split_sentences(self): text = "First sentence. Second sentence! Third sentence?" sentences = split_into_sentences(text) assert len(sentences) == 3 assert "First sentence" in sentences[0] assert "Second sentence" in sentences[1] assert "Third sentence" in sentences[2] class TestTokenProximity: """Test token proximity computation""" def test_proximity_close(self): sentence = "BTC surges to new highs" keywords = ["surges", "pumps", "moon"] proximity = compute_token_proximity(sentence, keywords, "BTC") assert proximity > 0.5 # "surges" is close to "BTC" def test_proximity_far(self): sentence = "The asset BTC which we mentioned earlier surges" keywords = ["surges"] proximity = compute_token_proximity(sentence, keywords, "BTC") assert proximity < 1.0 # Further away def test_proximity_no_match(self): sentence = "ETH pumps hard" keywords = ["surges"] proximity = compute_token_proximity(sentence, keywords, "BTC") assert proximity == 0.0 # BTC not in sentence if __name__ == "__main__": pytest.main([__file__, "-v"])