- Added 30 new sources (5 RSS + 25 Telegram) for previously ZERO-coverage assets - Fixed model loading priority: ONNX > LoRA v2 > PyTorch > Mock - ONNX FinBERT (pre-trained on 1.2M financial docs) now PRIMARY - best for real-world text - LoRA v2 models trained on 518 carefully labeled samples (balanced Bearish/Bullish/Neutral) - Emotion LoRA v2 trained with weighted loss (greed/fear 2x, joy 1.5x) - 30 new sources: STX, FET, XTZ, ENJ, ETC, TRX, ONG, DASH, LTC, ZIL, NEAR, APT, SUI, ICP - Early stopping (patience=3) on both LoRA trainings - Human-in-the-loop verification CLI tool created - Disk-conscious: save_total_limit=1, adapters 6-8MB each Pipeline now correctly classifies: - BTC breaks 100k → +0.54 Bullish ✅ - Major hack → -0.23 Bearish ✅ - HODL → +0.91 Bullish ✅ - Rug pull → -0.30 Bearish ✅ - SEC sues → -0.30 Bearish ✅ - ETF approval → +0.32 Bullish ✅ - Whale accumulation → +0.31 Bullish ✅ Models: ONNX FinBERT (PRIORITY 1) + LoRA v2 adapters (6-8MB each) Training data: 518 carefully labeled samples (190 real + 328 synthetic) Early stopping (patience=3) on both FinBERT and DistilRoBERTa LoRA Emotion LoRA v2: weighted loss (greed/fear 2x, joy 1.5x) + early stopping
114 lines
2.5 KiB
TOML
114 lines
2.5 KiB
TOML
[build-system]
|
|
requires = ["setuptools>=68.0", "wheel"]
|
|
build-backend = "setuptools.build_meta"
|
|
|
|
[project]
|
|
name = "sentiment-engine"
|
|
version = "2.0.0"
|
|
description = "Real-time sentiment analysis engine for DOLPHIN NG5"
|
|
readme = "README.md"
|
|
requires-python = ">=3.12"
|
|
dependencies = [
|
|
"numpy>=1.26",
|
|
"pandas>=2.1",
|
|
"pydantic>=2.7",
|
|
"pydantic-settings>=2.3",
|
|
"aiohttp>=3.9",
|
|
"aiokafka>=0.8",
|
|
"nats-py>=2.6",
|
|
"redis>=5.0",
|
|
"clickhouse-connect>=0.7",
|
|
"hazelcast-python-client>=5.6",
|
|
"prefect>=3.0",
|
|
"feedparser>=6.0",
|
|
"tweepy>=4.14",
|
|
"asyncpraw>=7.7",
|
|
"discord.py>=2.3",
|
|
"aiogram>=3.4",
|
|
"transformers>=4.40",
|
|
"torch>=2.3",
|
|
"sentence-transformers>=3.0",
|
|
"spacy>=3.7",
|
|
"rapidfuzz>=3.7",
|
|
"fasttext>=0.9",
|
|
"scikit-learn>=1.4",
|
|
"scipy>=1.12",
|
|
"pyyaml>=6.0",
|
|
"python-dotenv>=1.0",
|
|
"structlog>=24.1",
|
|
"opentelemetry-api>=1.24",
|
|
"opentelemetry-sdk>=1.24",
|
|
"opentelemetry-exporter-otlp>=1.24",
|
|
"prometheus-client>=0.19",
|
|
"pydantic-extra-types>=2.6",
|
|
"textual>=0.52",
|
|
"rich>=13.7",
|
|
"duckdb>=1.0", # Source catalogue
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
dev = [
|
|
"pytest>=8.0",
|
|
"pytest-asyncio>=0.23",
|
|
"pytest-cov>=5.0",
|
|
"ruff>=0.5",
|
|
"mypy>=1.10",
|
|
"pre-commit>=3.7",
|
|
]
|
|
gpu = [
|
|
"torch[cuda]>=2.3",
|
|
"sentence-transformers[cuda]>=3.0",
|
|
]
|
|
crawl = [
|
|
"scrapy>=2.11",
|
|
"chromedp>=0.0",
|
|
]
|
|
tui = [
|
|
"textual>=0.52",
|
|
"rich>=13.7",
|
|
]
|
|
|
|
[tool.setuptools.packages.find]
|
|
where = ["src"]
|
|
include = ["sentiment_engine*"]
|
|
|
|
[tool.ruff]
|
|
line-length = 100
|
|
target-version = "py312"
|
|
select = ["E", "F", "I", "UP", "W", "C90", "ANN", "T20", "PTH", "ERA", "PL", "TRY", "PD", "NPY", "PERF", "RET", "ASYNC"]
|
|
ignore = ["ANN101", "ANN102", "ANN201", "ANN202", "ANN204", "T201", "T203"]
|
|
|
|
[tool.ruff.format]
|
|
quote-style = "double"
|
|
indent-style = "space"
|
|
|
|
[tool.mypy]
|
|
python_version = "3.12"
|
|
strict = true
|
|
warn_return_any = true
|
|
warn_unused_configs = true
|
|
disallow_untyped_defs = true
|
|
disallow_incomplete_defs = true
|
|
check_untyped_defs = true
|
|
no_implicit_optional = true
|
|
ignore_missing_imports = false
|
|
|
|
[tool.pytest.ini_options]
|
|
asyncio_mode = "auto"
|
|
testpaths = ["tests"]
|
|
python_files = ["test_*.py"]
|
|
python_classes = ["Test*"]
|
|
python_functions = ["test_*"]
|
|
|
|
[tool.coverage.run]
|
|
source = ["src/sentiment_engine"]
|
|
omit = ["*/tests/*", "*/conftest.py"]
|
|
|
|
[tool.coverage.report]
|
|
exclude_lines = [
|
|
"pragma: no cover",
|
|
"def __repr__",
|
|
"raise NotImplementedError",
|
|
"if __name__ == .__main__.:",
|
|
]
|