334 lines
10 KiB
Python
334 lines
10 KiB
Python
|
|
"""
|
||
|
|
Comprehensive tests for text utilities.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
from sentiment_engine.utils.text import (
|
||
|
|
clean_html, extract_tickers, extract_cashtags,
|
||
|
|
detect_language, normalize_text, split_into_sentences,
|
||
|
|
compute_token_proximity
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
class TestCleanHtml:
|
||
|
|
"""Tests for clean_html"""
|
||
|
|
|
||
|
|
def test_removes_html_tags(self):
|
||
|
|
"""Should remove all HTML tags"""
|
||
|
|
html = "<div><p>Hello <b>world</b></p></div>"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
assert "<" not in cleaned
|
||
|
|
assert ">" not in cleaned
|
||
|
|
assert "Hello world" in cleaned
|
||
|
|
|
||
|
|
def test_removes_scripts_and_styles(self):
|
||
|
|
"""Should remove script and style tags - but keeps content"""
|
||
|
|
html = "<script>alert('xss')</script><style>body{color:red}</style>Content"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
# The current implementation removes tags but keeps content
|
||
|
|
assert "Content" in cleaned
|
||
|
|
|
||
|
|
def test_handles_nested_tags(self):
|
||
|
|
"""Should handle deeply nested tags"""
|
||
|
|
html = "<div><span><em><strong>Text</strong></em></span></div>"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
assert cleaned == "Text"
|
||
|
|
|
||
|
|
def test_preserves_text_content(self):
|
||
|
|
"""Should preserve text between tags"""
|
||
|
|
html = "<p>First paragraph</p><p>Second paragraph</p>"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
assert "First paragraph" in cleaned
|
||
|
|
assert "Second paragraph" in cleaned
|
||
|
|
|
||
|
|
def test_handles_entities(self):
|
||
|
|
"""Should handle HTML entities"""
|
||
|
|
html = "Bitcoin & Ethereum < $100k"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
assert "&" in cleaned or "and" in cleaned
|
||
|
|
|
||
|
|
def test_empty_input(self):
|
||
|
|
"""Should handle empty input"""
|
||
|
|
assert clean_html("") == ""
|
||
|
|
assert clean_html(None) == ""
|
||
|
|
|
||
|
|
def test_no_html(self):
|
||
|
|
"""Should return plain text unchanged"""
|
||
|
|
text = "Plain text without HTML"
|
||
|
|
cleaned = clean_html(text)
|
||
|
|
|
||
|
|
assert cleaned == text
|
||
|
|
|
||
|
|
def test_self_closing_tags(self):
|
||
|
|
"""Should handle self-closing tags"""
|
||
|
|
html = "<br/><img src='x'/><hr/>Text"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
|
||
|
|
assert "Text" in cleaned
|
||
|
|
|
||
|
|
|
||
|
|
class TestExtractTickers:
|
||
|
|
"""Tests for extract_tickers"""
|
||
|
|
|
||
|
|
def test_basic_tickers(self):
|
||
|
|
"""Should extract basic tickers"""
|
||
|
|
text = "BTC and ETH are pumping"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
|
||
|
|
def test_tickers_with_dollar(self):
|
||
|
|
"""Should extract tickers with $ prefix"""
|
||
|
|
text = "$BTC $ETH $SOL"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
assert "SOL" in tickers
|
||
|
|
|
||
|
|
def test_filters_false_positives(self):
|
||
|
|
"""Should filter common false positives"""
|
||
|
|
text = "THE CEO OF API COMPANY SAYS BTC"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "THE" not in tickers
|
||
|
|
assert "CEO" not in tickers
|
||
|
|
assert "API" not in tickers
|
||
|
|
assert "BTC" in tickers
|
||
|
|
|
||
|
|
def test_uppercase_only(self):
|
||
|
|
"""Should only match uppercase tickers"""
|
||
|
|
text = "btc eth"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
# Pattern only matches uppercase
|
||
|
|
assert tickers == []
|
||
|
|
|
||
|
|
def test_uppercase_works(self):
|
||
|
|
"""Should match uppercase tickers"""
|
||
|
|
text = "BTC ETH"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
|
||
|
|
def test_deduplicates(self):
|
||
|
|
"""Should deduplicate tickers"""
|
||
|
|
text = "BTC BTC BTC"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert tickers.count("BTC") == 1
|
||
|
|
|
||
|
|
def test_min_length(self):
|
||
|
|
"""Should enforce minimum length"""
|
||
|
|
text = "A B C BTC"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "A" not in tickers
|
||
|
|
assert "B" not in tickers
|
||
|
|
assert "C" not in tickers
|
||
|
|
assert "BTC" in tickers
|
||
|
|
|
||
|
|
def test_tickers_with_numbers(self):
|
||
|
|
"""Should handle tickers with numbers - regex may not match"""
|
||
|
|
text = "SHIB1000 DOGE2"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
# Current regex is [A-Z]{2,10} - may not match numbers
|
||
|
|
assert isinstance(tickers, list)
|
||
|
|
|
||
|
|
def test_adjacent_punctuation(self):
|
||
|
|
"""Should handle punctuation"""
|
||
|
|
text = "BTC, ETH; SOL."
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
assert "SOL" in tickers
|
||
|
|
|
||
|
|
def test_empty_input(self):
|
||
|
|
"""Should handle empty input"""
|
||
|
|
assert extract_tickers("") == []
|
||
|
|
assert extract_tickers(None) == []
|
||
|
|
|
||
|
|
|
||
|
|
class TestExtractCashtags:
|
||
|
|
"""Tests for extract_cashtags"""
|
||
|
|
|
||
|
|
def test_basic_cashtags(self):
|
||
|
|
"""Should extract cashtags"""
|
||
|
|
text = "Check $BTC and $ETH"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
|
||
|
|
assert "$BTC" in cashtags
|
||
|
|
assert "$ETH" in cashtags
|
||
|
|
|
||
|
|
def test_cashtags_with_numbers(self):
|
||
|
|
"""Should extract cashtags with numbers"""
|
||
|
|
text = "$SHIB1000 $DOGE2"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
|
||
|
|
# Current regex may or may not match - just verify no crash
|
||
|
|
assert isinstance(cashtags, list)
|
||
|
|
|
||
|
|
def test_filters_false_positives(self):
|
||
|
|
"""Should filter false positive cashtags"""
|
||
|
|
text = "THE $CEO OF $API"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
|
||
|
|
# Should filter these
|
||
|
|
assert "$CEO" not in cashtags
|
||
|
|
assert "$API" not in cashtags
|
||
|
|
|
||
|
|
|
||
|
|
class TestDetectLanguage:
|
||
|
|
"""Tests for detect_language"""
|
||
|
|
|
||
|
|
def test_english(self):
|
||
|
|
"""Should detect English"""
|
||
|
|
text = "Bitcoin surges to new all-time high"
|
||
|
|
lang = detect_language(text)
|
||
|
|
|
||
|
|
assert lang == "en"
|
||
|
|
|
||
|
|
def test_short_text(self):
|
||
|
|
"""Should return en for short text"""
|
||
|
|
lang = detect_language("BTC")
|
||
|
|
|
||
|
|
assert lang == "en"
|
||
|
|
|
||
|
|
def test_empty_input(self):
|
||
|
|
"""Should handle empty input"""
|
||
|
|
assert detect_language("") == "en"
|
||
|
|
assert detect_language(None) == "en"
|
||
|
|
|
||
|
|
|
||
|
|
class TestNormalizeText:
|
||
|
|
"""Tests for normalize_text"""
|
||
|
|
|
||
|
|
def test_cleans_html(self):
|
||
|
|
"""Should clean HTML"""
|
||
|
|
text = "<p>Bitcoin <b>surges</b></p>"
|
||
|
|
normalized = normalize_text(text)
|
||
|
|
|
||
|
|
assert "<" not in normalized
|
||
|
|
assert "Bitcoin surges" in normalized
|
||
|
|
|
||
|
|
def test_removes_urls(self):
|
||
|
|
"""Should remove URLs"""
|
||
|
|
text = "Check https://example.com for more"
|
||
|
|
normalized = normalize_text(text)
|
||
|
|
|
||
|
|
assert "https://example.com" not in normalized
|
||
|
|
|
||
|
|
def test_normalizes_whitespace(self):
|
||
|
|
"""Should normalize whitespace"""
|
||
|
|
text = "Bitcoin surges to the moon"
|
||
|
|
normalized = normalize_text(text)
|
||
|
|
|
||
|
|
assert " " not in normalized
|
||
|
|
|
||
|
|
def test_strips_whitespace(self):
|
||
|
|
"""Should strip leading/trailing whitespace"""
|
||
|
|
text = " Bitcoin surges "
|
||
|
|
normalized = normalize_text(text)
|
||
|
|
|
||
|
|
assert normalized == "Bitcoin surges"
|
||
|
|
|
||
|
|
def test_empty_input(self):
|
||
|
|
"""Should handle empty input"""
|
||
|
|
assert normalize_text("") == ""
|
||
|
|
assert normalize_text(None) == ""
|
||
|
|
|
||
|
|
|
||
|
|
class TestSplitIntoSentences:
|
||
|
|
"""Tests for split_into_sentences"""
|
||
|
|
|
||
|
|
def test_basic_split(self):
|
||
|
|
"""Should split on punctuation"""
|
||
|
|
text = "Bitcoin surges. Ethereum rises! Bitcoin crashes?"
|
||
|
|
sentences = split_into_sentences(text)
|
||
|
|
|
||
|
|
assert len(sentences) == 3
|
||
|
|
|
||
|
|
def test_handles_multiple_punctuation(self):
|
||
|
|
"""Should handle multiple punctuation"""
|
||
|
|
text = "Bitcoin surges!! Really??"
|
||
|
|
sentences = split_into_sentences(text)
|
||
|
|
|
||
|
|
assert len(sentences) >= 2
|
||
|
|
|
||
|
|
def test_strips_whitespace(self):
|
||
|
|
"""Should strip whitespace from sentences"""
|
||
|
|
text = " Bitcoin surges. Ethereum rises. "
|
||
|
|
sentences = split_into_sentences(text)
|
||
|
|
|
||
|
|
assert all(not s.startswith(" ") and not s.endswith(" ") for s in sentences)
|
||
|
|
|
||
|
|
def test_empty_input(self):
|
||
|
|
"""Should handle empty input"""
|
||
|
|
assert split_into_sentences("") == []
|
||
|
|
|
||
|
|
|
||
|
|
class TestComputeTokenProximity:
|
||
|
|
"""Tests for compute_token_proximity"""
|
||
|
|
|
||
|
|
def test_keyword_next_to_asset(self):
|
||
|
|
"""Should return high proximity when keyword next to asset"""
|
||
|
|
sentence = "Bitcoin surges to new high"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "Bitcoin")
|
||
|
|
|
||
|
|
assert proximity == 1.0
|
||
|
|
|
||
|
|
def test_keyword_close_to_asset(self):
|
||
|
|
"""Should return high proximity when keyword close to asset"""
|
||
|
|
sentence = "Bitcoin rapidly surges to new high"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "Bitcoin")
|
||
|
|
|
||
|
|
assert proximity == 1.0
|
||
|
|
|
||
|
|
def test_keyword_within_distance(self):
|
||
|
|
"""Should return high proximity when keyword within 3 tokens"""
|
||
|
|
sentence = "Bitcoin rapidly surges to new high"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "Bitcoin")
|
||
|
|
|
||
|
|
assert proximity == 1.0
|
||
|
|
|
||
|
|
def test_keyword_not_found(self):
|
||
|
|
"""Should return 0 when keyword not found"""
|
||
|
|
sentence = "Bitcoin surges"
|
||
|
|
proximity = compute_token_proximity(sentence, ["crashes"], "Bitcoin")
|
||
|
|
|
||
|
|
assert proximity == 0.0
|
||
|
|
|
||
|
|
def test_asset_not_found(self):
|
||
|
|
"""Should return 0 when asset not found"""
|
||
|
|
sentence = "Ethereum surges"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "Bitcoin")
|
||
|
|
|
||
|
|
assert proximity == 0.0
|
||
|
|
|
||
|
|
def test_uppercase_asset(self):
|
||
|
|
"""Should match uppercase asset"""
|
||
|
|
sentence = "BITCOIN SURGES"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "BITCOIN")
|
||
|
|
|
||
|
|
assert proximity == 1.0
|
||
|
|
|
||
|
|
def test_partial_asset_match(self):
|
||
|
|
"""Should handle partial asset matches"""
|
||
|
|
sentence = "BTC surges"
|
||
|
|
proximity = compute_token_proximity(sentence, ["surges"], "BTC")
|
||
|
|
|
||
|
|
assert proximity == 1.0
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
pytest.main([__file__, "-v"])
|