564 lines
21 KiB
Python
564 lines
21 KiB
Python
|
|
"""
|
||
|
|
Property-based tests using Hypothesis for comprehensive edge case coverage.
|
||
|
|
These tests generate thousands of test cases automatically.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
from hypothesis import given, strategies as st, settings, assume, example
|
||
|
|
import re
|
||
|
|
|
||
|
|
from sentiment_engine.utils.text import (
|
||
|
|
clean_html, extract_tickers, extract_cashtags,
|
||
|
|
detect_language, normalize_whitespace, truncate_text
|
||
|
|
)
|
||
|
|
from sentiment_engine.nlp.entity_extraction import EntityExtractor, AssetMapper
|
||
|
|
from sentiment_engine.nlp.temporal import TemporalAnchorer
|
||
|
|
from sentiment_engine.nlp.credibility import CredibilityScorer
|
||
|
|
from sentiment_engine.schemas.payload import NormalizedPayload, SourceType, AssetMention, EngagementMetrics
|
||
|
|
from sentiment_engine.schemas.processed import ProcessedItem, EntityExtraction, SentimentScores, EmotionScores
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# TEXT UTILS PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestTextUtilsProperties:
|
||
|
|
"""Property-based tests for text utilities"""
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=1000))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_clean_html_idempotent(self, text):
|
||
|
|
"""clean_html should be idempotent"""
|
||
|
|
cleaned = clean_html(text)
|
||
|
|
assert clean_html(cleaned) == cleaned
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=1000))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_clean_html_removes_tags(self, text):
|
||
|
|
"""clean_html should remove all HTML tags"""
|
||
|
|
html = f"<div><p>{text}</p></div>"
|
||
|
|
cleaned = clean_html(html)
|
||
|
|
assert "<" not in cleaned or ">" not in cleaned or "<" in cleaned
|
||
|
|
|
||
|
|
@given(st.text(alphabet=st.characters(blacklist_categories=('Cc', 'Cs')), min_size=0, max_size=500))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_normalize_whitespace_collapse(self, text):
|
||
|
|
"""normalize_whitespace should collapse multiple spaces"""
|
||
|
|
normalized = normalize_whitespace(text)
|
||
|
|
assert " " not in normalized
|
||
|
|
assert normalized == normalized.strip()
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=200))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_truncate_text_length(self, text):
|
||
|
|
"""truncate_text should not exceed max_length"""
|
||
|
|
max_len = 50
|
||
|
|
truncated = truncate_text(text, max_len)
|
||
|
|
assert len(truncated) <= max_len + 3 # +3 for "..."
|
||
|
|
|
||
|
|
@given(st.lists(st.text(min_size=1, max_size=10), min_size=0, max_size=20))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_extract_tickers_preserves_case(self, words):
|
||
|
|
"""extract_tickers should preserve ticker case"""
|
||
|
|
text = " ".join(words)
|
||
|
|
# Add some explicit tickers
|
||
|
|
text = f"BTC ETH {text} SOL"
|
||
|
|
tickers = extract_tickers(text)
|
||
|
|
assert "BTC" in tickers
|
||
|
|
assert "ETH" in tickers
|
||
|
|
assert "SOL" in tickers
|
||
|
|
|
||
|
|
@given(st.text(alphabet="ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789$", min_size=2, max_size=10))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_extract_cashtags_format(self, cashtag):
|
||
|
|
"""extract_cashtags should find $TICKER format"""
|
||
|
|
if cashtag.startswith("$") and len(cashtag) >= 3:
|
||
|
|
text = f"Check {cashtag} now"
|
||
|
|
cashtags = extract_cashtags(text)
|
||
|
|
assert cashtag in cashtags
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# ENTITY EXTRACTION PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestEntityExtractionProperties:
|
||
|
|
"""Property-based tests for entity extraction"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def extractor(self):
|
||
|
|
return EntityExtractor(AssetMapper())
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=500))
|
||
|
|
@settings(max_examples=300)
|
||
|
|
async def test_extract_all_returns_list(self, extractor, text):
|
||
|
|
"""extract_all should always return a list"""
|
||
|
|
await extractor.initialize()
|
||
|
|
entities = await extractor.extract_all(text)
|
||
|
|
assert isinstance(entities, list)
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=500))
|
||
|
|
@settings(max_examples=300)
|
||
|
|
async def test_extract_all_entities_have_required_fields(self, extractor, text):
|
||
|
|
"""All extracted entities should have required fields"""
|
||
|
|
await extractor.initialize()
|
||
|
|
entities = await extractor.extract_all(text)
|
||
|
|
for entity in entities:
|
||
|
|
assert hasattr(entity, 'asset_id')
|
||
|
|
assert hasattr(entity, 'mention_span')
|
||
|
|
assert hasattr(entity, 'confidence')
|
||
|
|
assert hasattr(entity, 'entity_type')
|
||
|
|
assert hasattr(entity, 'canonical_name')
|
||
|
|
assert 0 <= entity.confidence <= 1
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=500))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
async def test_deduplication_removes_overlaps(self, extractor, text):
|
||
|
|
"""Deduplication should remove overlapping mentions"""
|
||
|
|
await extractor.initialize()
|
||
|
|
entities = await extractor.extract_all(text)
|
||
|
|
# Check no overlapping spans
|
||
|
|
spans = [(e.mention_span[0], e.mention_span[1]) for e in entities]
|
||
|
|
for i, (s1, e1) in enumerate(spans):
|
||
|
|
for j, (s2, e2) in enumerate(spans):
|
||
|
|
if i != j:
|
||
|
|
assert not (s1 < e2 and s2 < e1), "Overlapping spans found"
|
||
|
|
|
||
|
|
@given(st.lists(
|
||
|
|
st.text(alphabet="ABCDEFGHIJKLMNOPQRSTUVWXYZ", min_size=2, max_size=5),
|
||
|
|
min_size=1, max_size=10
|
||
|
|
))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_asset_mapper_known_tickers(self, tickers):
|
||
|
|
"""AssetMapper should map known tickers with high confidence"""
|
||
|
|
mapper = AssetMapper()
|
||
|
|
for ticker in set(tickers):
|
||
|
|
asset_id, confidence = mapper.map_ticker(ticker)
|
||
|
|
if ticker in {"BTC", "ETH", "SOL", "AVAX", "MATIC", "DOT", "LINK", "UNI", "AAVE", "ARB", "OP", "SUI"}:
|
||
|
|
assert confidence >= 0.9
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# TEMPORAL ANCHORING PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestTemporalAnchoringProperties:
|
||
|
|
"""Property-based tests for temporal anchoring"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def anchorer(self):
|
||
|
|
return TemporalAnchorer()
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=500))
|
||
|
|
@settings(max_examples=300)
|
||
|
|
def test_anchor_returns_valid_object(self, anchorer, text):
|
||
|
|
"""anchor should return a valid TemporalAnchor"""
|
||
|
|
anchor = anchorer.anchor(text)
|
||
|
|
assert hasattr(anchor, 'time_horizon')
|
||
|
|
assert hasattr(anchor, 'is_breaking')
|
||
|
|
assert hasattr(anchor, 'is_scheduled')
|
||
|
|
assert anchor.time_horizon in ["immediate", "near", "medium", "long"]
|
||
|
|
assert isinstance(anchor.is_breaking, bool)
|
||
|
|
assert isinstance(anchor.is_scheduled, bool)
|
||
|
|
|
||
|
|
@given(st.text(alphabet="abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789 :,-.", min_size=0, max_size=200))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_breaking_detection_consistency(self, anchorer, text):
|
||
|
|
"""Breaking detection should be consistent"""
|
||
|
|
anchor1 = anchorer.anchor(text)
|
||
|
|
anchor2 = anchorer.anchor(text)
|
||
|
|
assert anchor1.is_breaking == anchor2.is_breaking
|
||
|
|
|
||
|
|
@given(st.text(min_size=0, max_size=200))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_horizon_ordering(self, anchorer, text):
|
||
|
|
"""Horizon should follow expected ordering"""
|
||
|
|
horizon_order = {"immediate": 0, "near": 1, "medium": 2, "long": 3}
|
||
|
|
anchor = anchorer.anchor(text)
|
||
|
|
assert anchor.time_horizon in horizon_order
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# CREDIBILITY SCORING PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestCredibilityScoringProperties:
|
||
|
|
"""Property-based tests for credibility scoring"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def scorer(self):
|
||
|
|
return CredibilityScorer()
|
||
|
|
|
||
|
|
@given(st.text(min_size=10, max_size=1000))
|
||
|
|
@settings(max_examples=300)
|
||
|
|
def test_composite_score_bounds(self, scorer, text):
|
||
|
|
"""Composite score should be in [0, 1]"""
|
||
|
|
scorer.load_registry({"test": {"base_credibility": 0.5}})
|
||
|
|
cred = scorer.compute_composite(
|
||
|
|
source_id="test",
|
||
|
|
text=text,
|
||
|
|
metadata={},
|
||
|
|
asset_id="BTC",
|
||
|
|
event_type="listing"
|
||
|
|
)
|
||
|
|
assert 0 <= cred.composite <= 1
|
||
|
|
|
||
|
|
@given(
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1)
|
||
|
|
)
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_composite_formula(self, scorer, source_base, content_quality,
|
||
|
|
engagement_auth, cross_source, historical):
|
||
|
|
"""Composite should match weighted formula"""
|
||
|
|
composite = (
|
||
|
|
0.3 * source_base +
|
||
|
|
0.25 * content_quality +
|
||
|
|
0.2 * engagement_auth +
|
||
|
|
0.15 * cross_source +
|
||
|
|
0.1 * historical
|
||
|
|
)
|
||
|
|
cred = CredibilityScore.compute(
|
||
|
|
source_base=source_base,
|
||
|
|
content_quality=content_quality,
|
||
|
|
engagement_authenticity=engagement_auth,
|
||
|
|
cross_source=cross_source,
|
||
|
|
historical=historical
|
||
|
|
)
|
||
|
|
assert abs(cred.composite - min(1.0, composite)) < 0.001
|
||
|
|
|
||
|
|
@given(st.text(min_size=100, max_size=2000))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_content_quality_increases_with_length(self, scorer, text):
|
||
|
|
"""Content quality should generally increase with length (up to a point)"""
|
||
|
|
short_text = text[:50]
|
||
|
|
long_text = text
|
||
|
|
short_score = scorer.score_content_quality(short_text, {})
|
||
|
|
long_score = scorer.score_content_quality(long_text, {})
|
||
|
|
# Longer text should not score significantly lower
|
||
|
|
assert long_score >= short_score - 0.2
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# SCHEMA VALIDATION PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestSchemaProperties:
|
||
|
|
"""Property-based tests for Pydantic schema validation"""
|
||
|
|
|
||
|
|
@given(
|
||
|
|
st.text(min_size=1, max_size=100),
|
||
|
|
st.text(min_size=1, max_size=5000),
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=1000000000, max_value=2000000000),
|
||
|
|
)
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_normalized_payload_creation(self, source_id, raw_text, credibility, ingest_ts):
|
||
|
|
"""NormalizedPayload should accept valid inputs"""
|
||
|
|
payload = NormalizedPayload(
|
||
|
|
source_id=source_id,
|
||
|
|
source_type=SourceType.NEWS,
|
||
|
|
source_credibility_base=credibility,
|
||
|
|
ingest_ts=ingest_ts,
|
||
|
|
publish_ts=ingest_ts,
|
||
|
|
content_length=len(raw_text),
|
||
|
|
raw_text=raw_text,
|
||
|
|
metadata={}
|
||
|
|
)
|
||
|
|
assert payload.source_id == source_id
|
||
|
|
assert payload.raw_text == raw_text.strip()
|
||
|
|
|
||
|
|
@given(
|
||
|
|
st.text(min_size=1, max_size=100),
|
||
|
|
st.floats(min_value=-1, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
)
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_sentiment_scores_bounds(self, asset_id, polarity, confidence):
|
||
|
|
"""SentimentScores should enforce bounds"""
|
||
|
|
scores = SentimentScores(
|
||
|
|
polarity=max(-1, min(1, polarity)),
|
||
|
|
confidence=max(0, min(1, confidence)),
|
||
|
|
positive_prob=max(0, min(1, (polarity + 1) / 2)),
|
||
|
|
negative_prob=max(0, min(1, (1 - polarity) / 2)),
|
||
|
|
neutral_prob=1 - abs(polarity)
|
||
|
|
)
|
||
|
|
assert -1 <= scores.polarity <= 1
|
||
|
|
assert 0 <= scores.confidence <= 1
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# SIGNAL PROCESSING PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestSignalProcessingProperties:
|
||
|
|
"""Property-based tests for signal processing"""
|
||
|
|
|
||
|
|
@given(st.floats(min_value=0, max_value=1))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_fear_greed_bounds(self, value):
|
||
|
|
"""Fear/greed index should be in [0, 100]"""
|
||
|
|
from sentiment_engine.signal.processor import FearGreedProcessor
|
||
|
|
# The index is derived from normalized inputs
|
||
|
|
assert 0 <= value <= 1
|
||
|
|
|
||
|
|
@given(st.floats(min_value=0, max_value=100))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_velocity_non_negative(self, value):
|
||
|
|
"""Velocity should be non-negative"""
|
||
|
|
assert value >= 0
|
||
|
|
|
||
|
|
@given(
|
||
|
|
st.floats(min_value=0, max_value=1),
|
||
|
|
st.floats(min_value=0, max_value=1)
|
||
|
|
)
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_fusion_weighted_average(self, w1, w2):
|
||
|
|
"""Fusion should produce weighted average"""
|
||
|
|
# Normalize weights
|
||
|
|
total = w1 + w2
|
||
|
|
if total > 0:
|
||
|
|
w1, w2 = w1/total, w2/total
|
||
|
|
result = w1 * 0.5 + w2 * 0.8
|
||
|
|
assert 0 <= result <= 1
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# LABELING PIPELINE PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestLabelingPipelineProperties:
|
||
|
|
"""Property-based tests for labeling pipeline"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def runner(self):
|
||
|
|
from labeling_pipeline import LabelingPipelineRunner
|
||
|
|
return LabelingPipelineRunner()
|
||
|
|
|
||
|
|
@given(st.text(min_size=20, max_size=500))
|
||
|
|
@settings(max_examples=100)
|
||
|
|
async def test_labeling_returns_valid_structure(self, runner, text):
|
||
|
|
"""Labeling should return valid structure"""
|
||
|
|
from labeling_pipeline import LabelingPipeline
|
||
|
|
pipeline = LabelingPipeline()
|
||
|
|
result = await pipeline.label_text(text)
|
||
|
|
|
||
|
|
assert "labels" in result
|
||
|
|
assert "confidence" in result
|
||
|
|
assert "verified" in result
|
||
|
|
assert "verification_details" in result
|
||
|
|
assert result["labels"]["sentiment"] in ["Bearish", "Bullish", "Neutral"]
|
||
|
|
assert result["labels"]["event_type"] in [
|
||
|
|
"listing", "delisting", "hack", "regulatory", "governance",
|
||
|
|
"upgrade", "partnership", "earnings", "macro",
|
||
|
|
"liquidation", "whale", "manipulation"
|
||
|
|
]
|
||
|
|
|
||
|
|
@given(st.text(min_size=20, max_size=500))
|
||
|
|
@settings(max_examples=100)
|
||
|
|
async def test_confidence_bounds(self, runner, text):
|
||
|
|
"""Confidence scores should be in [0, 1]"""
|
||
|
|
from labeling_pipeline import LabelingPipeline
|
||
|
|
pipeline = LabelingPipeline()
|
||
|
|
result = await pipeline.label_text(text)
|
||
|
|
|
||
|
|
assert 0 <= result["confidence"]["sentiment"] <= 1
|
||
|
|
assert 0 <= result["confidence"]["event"] <= 1
|
||
|
|
assert 0 <= result["confidence"]["verification"] <= 1
|
||
|
|
assert 0 <= result["confidence"]["overall"] <= 1
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# ONNX MODEL PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestONNXModelProperties:
|
||
|
|
"""Property-based tests for ONNX model inference"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def finbert_session(self):
|
||
|
|
import onnxruntime as ort
|
||
|
|
return ort.InferenceSession(
|
||
|
|
"models/onnx/finbert/model.onnx",
|
||
|
|
providers=['CPUExecutionProvider']
|
||
|
|
)
|
||
|
|
|
||
|
|
@given(st.integers(min_value=1, max_value=10), st.integers(min_value=16, max_value=256))
|
||
|
|
@settings(max_examples=100)
|
||
|
|
def test_finbert_input_shapes(self, finbert_session, batch_size, seq_len):
|
||
|
|
"""FinBERT should accept various input shapes"""
|
||
|
|
import numpy as np
|
||
|
|
input_ids = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
attention_mask = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
token_type_ids = np.zeros((batch_size, seq_len), dtype=np.int64)
|
||
|
|
|
||
|
|
outputs = finbert_session.run(None, {
|
||
|
|
"input_ids": input_ids,
|
||
|
|
"attention_mask": attention_mask,
|
||
|
|
"token_type_ids": token_type_ids
|
||
|
|
})
|
||
|
|
assert outputs[0].shape == (batch_size, 3)
|
||
|
|
|
||
|
|
@given(st.integers(min_value=1, max_value=10), st.integers(min_value=16, max_value=256))
|
||
|
|
@settings(max_examples=100)
|
||
|
|
def test_bert_events_input_shapes(self, batch_size, seq_len):
|
||
|
|
"""BERT Events should accept various input shapes"""
|
||
|
|
import onnxruntime as ort
|
||
|
|
import numpy as np
|
||
|
|
session = ort.InferenceSession(
|
||
|
|
"models/onnx/bert-base-event/model.onnx",
|
||
|
|
providers=['CPUExecutionProvider']
|
||
|
|
)
|
||
|
|
input_ids = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
attention_mask = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
token_type_ids = np.zeros((batch_size, seq_len), dtype=np.int64)
|
||
|
|
|
||
|
|
outputs = session.run(None, {
|
||
|
|
"input_ids": input_ids,
|
||
|
|
"attention_mask": attention_mask,
|
||
|
|
"token_type_ids": token_type_ids
|
||
|
|
})
|
||
|
|
assert outputs[0].shape == (batch_size, 12)
|
||
|
|
|
||
|
|
@given(st.integers(min_value=1, max_value=10), st.integers(min_value=16, max_value=256))
|
||
|
|
@settings(max_examples=100)
|
||
|
|
def test_distilroberta_emotion_input_shapes(self, batch_size, seq_len):
|
||
|
|
"""DistilRoBERTa Emotion should accept various input shapes (no token_type_ids)"""
|
||
|
|
import onnxruntime as ort
|
||
|
|
import numpy as np
|
||
|
|
session = ort.InferenceSession(
|
||
|
|
"models/onnx/distilroberta-emotion/model.onnx",
|
||
|
|
providers=['CPUExecutionProvider']
|
||
|
|
)
|
||
|
|
input_ids = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
attention_mask = np.ones((batch_size, seq_len), dtype=np.int64)
|
||
|
|
|
||
|
|
outputs = session.run(None, {
|
||
|
|
"input_ids": input_ids,
|
||
|
|
"attention_mask": attention_mask
|
||
|
|
})
|
||
|
|
assert outputs[0].shape == (batch_size, 6)
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# CONNECTOR PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestConnectorProperties:
|
||
|
|
"""Property-based tests for connectors"""
|
||
|
|
|
||
|
|
@given(st.text(min_size=1, max_size=200))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_rss_feed_url_validation(self, url):
|
||
|
|
"""RSS feed URLs should be valid"""
|
||
|
|
# Simple validation
|
||
|
|
if url.startswith(("http://", "https://")):
|
||
|
|
assert "." in url
|
||
|
|
|
||
|
|
@given(st.integers(min_value=1, max_value=1000))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_poll_interval_reasonable(self, interval):
|
||
|
|
"""Poll intervals should be reasonable (1 min to 24 hours)"""
|
||
|
|
assert 60 <= interval <= 86400
|
||
|
|
|
||
|
|
@given(st.floats(min_value=0.01, max_value=100))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_rate_limit_reasonable(self, rps):
|
||
|
|
"""Rate limits should be reasonable"""
|
||
|
|
assert 0.01 <= rps <= 100
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# CATALOGUE PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestCatalogueProperties:
|
||
|
|
"""Property-based tests for catalogue"""
|
||
|
|
|
||
|
|
@given(st.text(min_size=1, max_size=100))
|
||
|
|
@settings(max_examples=200)
|
||
|
|
def test_source_id_format(self, source_id):
|
||
|
|
"""Source IDs should follow format"""
|
||
|
|
# Should not contain special chars except : . -
|
||
|
|
import re
|
||
|
|
assert re.match(r'^[a-zA-Z0-9.:_-]+$', source_id)
|
||
|
|
|
||
|
|
@given(st.floats(min_value=0, max_value=1))
|
||
|
|
@settings(max_examples=500)
|
||
|
|
def test_credibility_bounds(self, credibility):
|
||
|
|
"""Credibility should be in [0, 1]"""
|
||
|
|
assert 0 <= credibility <= 1
|
||
|
|
|
||
|
|
|
||
|
|
# ============================================================
|
||
|
|
# INTEGRATION PROPERTY TESTS
|
||
|
|
# ============================================================
|
||
|
|
|
||
|
|
class TestIntegrationProperties:
|
||
|
|
"""Property-based tests for end-to-end integration"""
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def pipeline(self):
|
||
|
|
from sentiment_engine.nlp.pipeline import NLPProcessingPipeline
|
||
|
|
return NLPProcessingPipeline()
|
||
|
|
|
||
|
|
@given(
|
||
|
|
st.text(min_size=10, max_size=500),
|
||
|
|
st.sampled_from(["news", "social", "regulatory", "exchange_ann", "on_chain"]),
|
||
|
|
)
|
||
|
|
@settings(max_examples=100)
|
||
|
|
async def test_pipeline_processes_any_text(self, pipeline, text, source_type):
|
||
|
|
"""Pipeline should process any valid text without crashing"""
|
||
|
|
from sentiment_engine.schemas.payload import NormalizedPayload, SourceType, AssetMention
|
||
|
|
await pipeline.initialize()
|
||
|
|
|
||
|
|
payload = NormalizedPayload(
|
||
|
|
source_id="test",
|
||
|
|
source_type=SourceType(source_type),
|
||
|
|
source_credibility_base=0.5,
|
||
|
|
ingest_ts=1700000000.0,
|
||
|
|
publish_ts=1700000000.0,
|
||
|
|
content_length=len(text),
|
||
|
|
raw_text=text,
|
||
|
|
asset_mentions=[],
|
||
|
|
metadata={}
|
||
|
|
)
|
||
|
|
|
||
|
|
result = await pipeline.process(payload)
|
||
|
|
assert isinstance(result, ProcessedItem)
|
||
|
|
assert result.source_id == "test"
|
||
|
|
assert hasattr(result, 'entities')
|
||
|
|
assert hasattr(result, 'sentiment_per_asset')
|
||
|
|
assert hasattr(result, 'events')
|
||
|
|
assert hasattr(result, 'temporal')
|
||
|
|
assert hasattr(result, 'credibility')
|
||
|
|
|
||
|
|
@given(st.text(min_size=10, max_size=500))
|
||
|
|
@settings(max_examples=50)
|
||
|
|
async def test_pipeline_latency_reasonable(self, pipeline, text):
|
||
|
|
"""Pipeline latency should be reasonable (< 10 seconds)"""
|
||
|
|
from sentiment_engine.schemas.payload import NormalizedPayload, SourceType
|
||
|
|
await pipeline.initialize()
|
||
|
|
|
||
|
|
payload = NormalizedPayload(
|
||
|
|
source_id="test",
|
||
|
|
source_type=SourceType.NEWS,
|
||
|
|
source_credibility_base=0.5,
|
||
|
|
ingest_ts=1700000000.0,
|
||
|
|
publish_ts=1700000000.0,
|
||
|
|
content_length=len(text),
|
||
|
|
raw_text=text,
|
||
|
|
asset_mentions=[],
|
||
|
|
metadata={}
|
||
|
|
)
|
||
|
|
|
||
|
|
result = await pipeline.process(payload)
|
||
|
|
assert result.processing_latency_ms < 10000 # 10 seconds
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
pytest.main([__file__, "-v", "--tb=short"])
|