Files
sentiment-engine/MALKHUT/malkhut/training/monitor.py
Codex dd86174107 malkhut(T8): cognition pipeline + regime expansion + prod tooling
Cognition pipeline (cognition.py): rate-limited, 8 sources, dedup, perm-run.
Regime expansion (regime_expansion.py): 200+ regimes from 4x4x4x4 dimensions.
News sources (news_sources.py): 12 industry-standard sources with ranking.
Monitor (monitor.py): metrics, health scoring, alerts, JSONL logging.
Cognition launcher (cognition_launcher.py): standalone long-run service.
Continuous pipeline (continuous_pipeline.py): forever-loop training runner.
2026-07-11 10:39:03 +02:00

125 lines
3.7 KiB
Python

"""
Cognition Monitor — track pipeline health, metrics, and regime discovery.
Provides:
- Real-time metrics (fetch rate, error rate, regime count)
- Health scoring (source reliability, freshness)
- Alert thresholds
- JSONL logging for audit trail
"""
from __future__ import annotations
import json
import time
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional
@dataclass(frozen=True, slots=True)
class CognitionMetrics:
"""Snapshot of pipeline metrics."""
timestamp_ns: int
total_sources: int
enabled_sources: int
total_fetched: int
total_errors: float
discovered_regimes: int
fetch_rate_per_min: float
error_rate: float
uptime_s: float
class CognitionMonitor:
"""
Monitor cognition pipeline health and metrics.
Tracks:
- Fetch rate and error rate
- Source health scores
- Regime discovery rate
- Uptime and availability
"""
def __init__(self, log_path: str = "cognition_metrics.jsonl") -> None:
self._log_path = log_path
self._start_time = time.time()
self._total_fetched = 0
self._total_errors = 0
self._regime_count = 0
self._source_health: Dict[str, float] = {}
self._metrics_history: list[CognitionMetrics] = []
def record_fetch(self, source_id: str, success: bool) -> None:
if success:
self._total_fetched += 1
else:
self._total_errors += 1
def record_regime(self, count: int) -> None:
self._regime_count += count
def snapshot(
self,
total_sources: int,
enabled_sources: int,
) -> CognitionMetrics:
"""Take a metrics snapshot."""
elapsed = time.time() - self._start_time
fetch_rate = self._total_fetched / max(elapsed / 60, 1)
error_rate = self._total_errors / max(self._total_fetched + self._total_errors, 1)
metrics = CognitionMetrics(
timestamp_ns=time.time_ns(),
total_sources=total_sources,
enabled_sources=enabled_sources,
total_fetched=self._total_fetched,
total_errors=self._total_errors,
discovered_regimes=self._regime_count,
fetch_rate_per_min=fetch_rate,
error_rate=error_rate,
uptime_s=elapsed,
)
self._metrics_history.append(metrics)
# Write to JSONL
try:
with open(self._log_path, "a") as f:
f.write(json.dumps({
"ts": metrics.timestamp_ns,
"sources": metrics.total_sources,
"fetched": metrics.total_fetched,
"regimes": metrics.discovered_regimes,
"fetch_rate": round(metrics.fetch_rate_per_min, 2),
"error_rate": round(metrics.error_rate, 4),
"uptime": round(metrics.uptime_s, 1),
}, separators=(",", ":")) + "\n")
except OSError:
pass
return metrics
def check_alerts(self) -> List[str]:
"""Check for conditions that need attention."""
alerts = []
if not self._metrics_history:
return alerts
latest = self._metrics_history[-1]
if latest.error_rate > 0.1:
alerts.append(f"HIGH_ERROR_RATE: {latest.error_rate:.1%}")
if latest.fetch_rate_per_min < 1.0 and latest.uptime_s > 300:
alerts.append(f"LOW_FETCH_RATE: {latest.fetch_rate_per_min:.1f}/min")
return alerts
@property
def total_fetched(self) -> int:
return self._total_fetched
@property
def total_errors(self) -> int:
return self._total_errors
@property
def discovered_regimes(self) -> int:
return self._regime_count