Files
sentiment-engine/sentiment_engine/pyproject.toml
Codex c32db97d57 feat(sentiment): complete pipeline overhaul with ONNX priority + LoRA retraining
- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets
- Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock
- ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text
- LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral)
- Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x)
- 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP
- Early stopping (patience=3) on both LoRA trainings
- Human-in-the-loop verification CLI tool created
- Disk-conscious: save_total_limit=1, adapters 6-8MB each

Pipeline now correctly classifies:
- BTC breaks 100k → +0.54 Bullish ✅
- Major hack → -0.23 Bearish ✅
- HODL → +0.91 Bullish ✅
- Rug pull → -0.30 Bearish ✅
- SEC sues → -0.30 Bearish ✅
- ETF approval → +0.32 Bullish ✅
- Whale accumulation → +0.31 Bullish ✅

Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each)
Training data: 518 carefully labeled samples (190 real + 328 synthetic)
Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA
Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
2026-09-27 04:34:49 +02:00

114 lines
2.5 KiB
TOML

[build-system]
requires = ["setuptools>=68.0", "wheel"]
build-backend = "setuptools.build_meta"
[project]
name = "sentiment-engine"
version = "2.0.0"
description = "Real-time sentiment analysis engine for DOLPHIN NG5"
readme = "README.md"
requires-python = ">=3.12"
dependencies = [
"numpy>=1.26",
"pandas>=2.1",
"pydantic>=2.7",
"pydantic-settings>=2.3",
"aiohttp>=3.9",
"aiokafka>=0.8",
"nats-py>=2.6",
"redis>=5.0",
"clickhouse-connect>=0.7",
"hazelcast-python-client>=5.6",
"prefect>=3.0",
"feedparser>=6.0",
"tweepy>=4.14",
"asyncpraw>=7.7",
"discord.py>=2.3",
"aiogram>=3.4",
"transformers>=4.40",
"torch>=2.3",
"sentence-transformers>=3.0",
"spacy>=3.7",
"rapidfuzz>=3.7",
"fasttext>=0.9",
"scikit-learn>=1.4",
"scipy>=1.12",
"pyyaml>=6.0",
"python-dotenv>=1.0",
"structlog>=24.1",
"opentelemetry-api>=1.24",
"opentelemetry-sdk>=1.24",
"opentelemetry-exporter-otlp>=1.24",
"prometheus-client>=0.19",
"pydantic-extra-types>=2.6",
"textual>=0.52",
"rich>=13.7",
"duckdb>=1.0", # Source catalogue
]
[project.optional-dependencies]
dev = [
"pytest>=8.0",
"pytest-asyncio>=0.23",
"pytest-cov>=5.0",
"ruff>=0.5",
"mypy>=1.10",
"pre-commit>=3.7",
]
gpu = [
"torch[cuda]>=2.3",
"sentence-transformers[cuda]>=3.0",
]
crawl = [
"scrapy>=2.11",
"chromedp>=0.0",
]
tui = [
"textual>=0.52",
"rich>=13.7",
]
[tool.setuptools.packages.find]
where = ["src"]
include = ["sentiment_engine*"]
[tool.ruff]
line-length = 100
target-version = "py312"
select = ["E", "F", "I", "UP", "W", "C90", "ANN", "T20", "PTH", "ERA", "PL", "TRY", "PD", "NPY", "PERF", "RET", "ASYNC"]
ignore = ["ANN101", "ANN102", "ANN201", "ANN202", "ANN204", "T201", "T203"]
[tool.ruff.format]
quote-style = "double"
indent-style = "space"
[tool.mypy]
python_version = "3.12"
strict = true
warn_return_any = true
warn_unused_configs = true
disallow_untyped_defs = true
disallow_incomplete_defs = true
check_untyped_defs = true
no_implicit_optional = true
ignore_missing_imports = false
[tool.pytest.ini_options]
asyncio_mode = "auto"
testpaths = ["tests"]
python_files = ["test_*.py"]
python_classes = ["Test*"]
python_functions = ["test_*"]
[tool.coverage.run]
source = ["src/sentiment_engine"]
omit = ["*/tests/*", "*/conftest.py"]
[tool.coverage.report]
exclude_lines = [
"pragma: no cover",
"def __repr__",
"raise NotImplementedError",
"if __name__ == .__main__.:",
]