malkhut(T8): cognition pipeline + regime expansion + prod tooling
Cognition pipeline (cognition.py): rate-limited, 8 sources, dedup, perm-run. Regime expansion (regime_expansion.py): 200+ regimes from 4x4x4x4 dimensions. News sources (news_sources.py): 12 industry-standard sources with ranking. Monitor (monitor.py): metrics, health scoring, alerts, JSONL logging. Cognition launcher (cognition_launcher.py): standalone long-run service. Continuous pipeline (continuous_pipeline.py): forever-loop training runner.
This commit is contained in:
124
MALKHUT/malkhut/training/monitor.py
Normal file
124
MALKHUT/malkhut/training/monitor.py
Normal file
@@ -0,0 +1,124 @@
|
||||
"""
|
||||
Cognition Monitor — track pipeline health, metrics, and regime discovery.
|
||||
|
||||
Provides:
|
||||
- Real-time metrics (fetch rate, error rate, regime count)
|
||||
- Health scoring (source reliability, freshness)
|
||||
- Alert thresholds
|
||||
- JSONL logging for audit trail
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class CognitionMetrics:
|
||||
"""Snapshot of pipeline metrics."""
|
||||
timestamp_ns: int
|
||||
total_sources: int
|
||||
enabled_sources: int
|
||||
total_fetched: int
|
||||
total_errors: float
|
||||
discovered_regimes: int
|
||||
fetch_rate_per_min: float
|
||||
error_rate: float
|
||||
uptime_s: float
|
||||
|
||||
|
||||
class CognitionMonitor:
|
||||
"""
|
||||
Monitor cognition pipeline health and metrics.
|
||||
|
||||
Tracks:
|
||||
- Fetch rate and error rate
|
||||
- Source health scores
|
||||
- Regime discovery rate
|
||||
- Uptime and availability
|
||||
"""
|
||||
|
||||
def __init__(self, log_path: str = "cognition_metrics.jsonl") -> None:
|
||||
self._log_path = log_path
|
||||
self._start_time = time.time()
|
||||
self._total_fetched = 0
|
||||
self._total_errors = 0
|
||||
self._regime_count = 0
|
||||
self._source_health: Dict[str, float] = {}
|
||||
self._metrics_history: list[CognitionMetrics] = []
|
||||
|
||||
def record_fetch(self, source_id: str, success: bool) -> None:
|
||||
if success:
|
||||
self._total_fetched += 1
|
||||
else:
|
||||
self._total_errors += 1
|
||||
|
||||
def record_regime(self, count: int) -> None:
|
||||
self._regime_count += count
|
||||
|
||||
def snapshot(
|
||||
self,
|
||||
total_sources: int,
|
||||
enabled_sources: int,
|
||||
) -> CognitionMetrics:
|
||||
"""Take a metrics snapshot."""
|
||||
elapsed = time.time() - self._start_time
|
||||
fetch_rate = self._total_fetched / max(elapsed / 60, 1)
|
||||
error_rate = self._total_errors / max(self._total_fetched + self._total_errors, 1)
|
||||
|
||||
metrics = CognitionMetrics(
|
||||
timestamp_ns=time.time_ns(),
|
||||
total_sources=total_sources,
|
||||
enabled_sources=enabled_sources,
|
||||
total_fetched=self._total_fetched,
|
||||
total_errors=self._total_errors,
|
||||
discovered_regimes=self._regime_count,
|
||||
fetch_rate_per_min=fetch_rate,
|
||||
error_rate=error_rate,
|
||||
uptime_s=elapsed,
|
||||
)
|
||||
|
||||
self._metrics_history.append(metrics)
|
||||
|
||||
# Write to JSONL
|
||||
try:
|
||||
with open(self._log_path, "a") as f:
|
||||
f.write(json.dumps({
|
||||
"ts": metrics.timestamp_ns,
|
||||
"sources": metrics.total_sources,
|
||||
"fetched": metrics.total_fetched,
|
||||
"regimes": metrics.discovered_regimes,
|
||||
"fetch_rate": round(metrics.fetch_rate_per_min, 2),
|
||||
"error_rate": round(metrics.error_rate, 4),
|
||||
"uptime": round(metrics.uptime_s, 1),
|
||||
}, separators=(",", ":")) + "\n")
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
return metrics
|
||||
|
||||
def check_alerts(self) -> List[str]:
|
||||
"""Check for conditions that need attention."""
|
||||
alerts = []
|
||||
if not self._metrics_history:
|
||||
return alerts
|
||||
latest = self._metrics_history[-1]
|
||||
if latest.error_rate > 0.1:
|
||||
alerts.append(f"HIGH_ERROR_RATE: {latest.error_rate:.1%}")
|
||||
if latest.fetch_rate_per_min < 1.0 and latest.uptime_s > 300:
|
||||
alerts.append(f"LOW_FETCH_RATE: {latest.fetch_rate_per_min:.1f}/min")
|
||||
return alerts
|
||||
|
||||
@property
|
||||
def total_fetched(self) -> int:
|
||||
return self._total_fetched
|
||||
|
||||
@property
|
||||
def total_errors(self) -> int:
|
||||
return self._total_errors
|
||||
|
||||
@property
|
||||
def discovered_regimes(self) -> int:
|
||||
return self._regime_count
|
||||
Reference in New Issue
Block a user