Add sentiment_engine with CryptoSentimentCalibrator fixes - improved keyword lists, lowered FinBERT threshold, added neutral handling

This commit is contained in:
Codex
2026-09-14 13:30:05 +02:00
parent 19a7812094
commit a276aeaded
149 changed files with 35226 additions and 0 deletions

View File

@@ -0,0 +1,128 @@
"""Web crawl Prefect flow"""
import asyncio
import logging
import subprocess
import tempfile
from pathlib import Path
from typing import List
from prefect import flow, task
from sentiment_engine.utils.config import get_settings
logger = logging.getLogger(__name__)
@task(retries=1, retry_delay_seconds=300)
async def run_hister_crawl(
seed_urls: List[str],
allowed_domains: List[str],
max_depth: int = 2,
job_timeout: int = 3600
) -> List[dict]:
"""Run Hister crawl job"""
with tempfile.TemporaryDirectory() as tmpdir:
seed_file = Path(tmpdir) / "seeds.txt"
seed_file.write_text("\n".join(seed_urls))
output_file = Path(tmpdir) / "output.jsonl"
cmd = [
"hister", "crawl",
"--input", str(seed_file),
"--job-id", f"prefect-crawl-{asyncio.current_task().get_name()}",
"--depth", str(max_depth),
"--delay", "1.0",
"--output", str(output_file),
"--format", "jsonl"
]
if allowed_domains:
cmd.extend(["--allowed-domain", ",".join(allowed_domains)])
try:
proc = await asyncio.create_subprocess_exec(
*cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE
)
stdout, stderr = await asyncio.wait_for(
proc.communicate(), timeout=job_timeout
)
if proc.returncode != 0:
logger.error(f"Hister failed: {stderr.decode()}")
return []
# Parse output
items = []
if output_file.exists():
import json
with open(output_file) as f:
for line in f:
line = line.strip()
if not line:
continue
try:
data = json.loads(line)
items.append(data)
except json.JSONDecodeError:
continue
return items
except asyncio.TimeoutError:
logger.error(f"Hister job timed out after {job_timeout}s")
return []
except FileNotFoundError:
logger.error("Hister not installed")
return []
@flow(
name="web_crawl",
log_prints=True
)
async def web_crawl_flow(
seed_urls: List[str] = None,
allowed_domains: List[str] = None,
max_depth: int = 2
):
"""Web crawl flow for sites without RSS/API"""
if seed_urls is None:
seed_urls = [
"https://www.coindesk.com",
"https://cointelegraph.com",
"https://www.theblock.co",
"https://decrypt.co",
"https://cryptoslate.com",
]
if allowed_domains is None:
allowed_domains = [
"coindesk.com", "cointelegraph.com", "theblock.co",
"decrypt.co", "cryptoslate.com", "bitcoinmagazine.com"
]
items = await run_hister_crawl(seed_urls, allowed_domains, max_depth)
logger.info(f"Crawled {len(items)} pages")
# Convert to normalized items
normalized = []
for item in items:
normalized.append({
"source_id": f"web:{item.get('url', '').split('/')[2] if item.get('url') else 'unknown'}",
"source_type": "news",
"raw_text": f"{item.get('title', '')}\n\n{item.get('content', item.get('text', ''))}",
"title": item.get("title"),
"url": item.get("url"),
"metadata": {"crawler": "hister", "job": "prefect"}
})
return normalized
if __name__ == "__main__":
asyncio.run(web_crawl_flow())