Cognition pipeline (cognition.py): rate-limited, 8 sources, dedup, perm-run. Regime expansion (regime_expansion.py): 200+ regimes from 4x4x4x4 dimensions. News sources (news_sources.py): 12 industry-standard sources with ranking. Monitor (monitor.py): metrics, health scoring, alerts, JSONL logging. Cognition launcher (cognition_launcher.py): standalone long-run service. Continuous pipeline (continuous_pipeline.py): forever-loop training runner.
125 lines
3.7 KiB
Python
125 lines
3.7 KiB
Python
"""
|
|
Cognition Monitor — track pipeline health, metrics, and regime discovery.
|
|
|
|
Provides:
|
|
- Real-time metrics (fetch rate, error rate, regime count)
|
|
- Health scoring (source reliability, freshness)
|
|
- Alert thresholds
|
|
- JSONL logging for audit trail
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import time
|
|
from dataclasses import dataclass, field
|
|
from typing import Any, Dict, List, Optional
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class CognitionMetrics:
|
|
"""Snapshot of pipeline metrics."""
|
|
timestamp_ns: int
|
|
total_sources: int
|
|
enabled_sources: int
|
|
total_fetched: int
|
|
total_errors: float
|
|
discovered_regimes: int
|
|
fetch_rate_per_min: float
|
|
error_rate: float
|
|
uptime_s: float
|
|
|
|
|
|
class CognitionMonitor:
|
|
"""
|
|
Monitor cognition pipeline health and metrics.
|
|
|
|
Tracks:
|
|
- Fetch rate and error rate
|
|
- Source health scores
|
|
- Regime discovery rate
|
|
- Uptime and availability
|
|
"""
|
|
|
|
def __init__(self, log_path: str = "cognition_metrics.jsonl") -> None:
|
|
self._log_path = log_path
|
|
self._start_time = time.time()
|
|
self._total_fetched = 0
|
|
self._total_errors = 0
|
|
self._regime_count = 0
|
|
self._source_health: Dict[str, float] = {}
|
|
self._metrics_history: list[CognitionMetrics] = []
|
|
|
|
def record_fetch(self, source_id: str, success: bool) -> None:
|
|
if success:
|
|
self._total_fetched += 1
|
|
else:
|
|
self._total_errors += 1
|
|
|
|
def record_regime(self, count: int) -> None:
|
|
self._regime_count += count
|
|
|
|
def snapshot(
|
|
self,
|
|
total_sources: int,
|
|
enabled_sources: int,
|
|
) -> CognitionMetrics:
|
|
"""Take a metrics snapshot."""
|
|
elapsed = time.time() - self._start_time
|
|
fetch_rate = self._total_fetched / max(elapsed / 60, 1)
|
|
error_rate = self._total_errors / max(self._total_fetched + self._total_errors, 1)
|
|
|
|
metrics = CognitionMetrics(
|
|
timestamp_ns=time.time_ns(),
|
|
total_sources=total_sources,
|
|
enabled_sources=enabled_sources,
|
|
total_fetched=self._total_fetched,
|
|
total_errors=self._total_errors,
|
|
discovered_regimes=self._regime_count,
|
|
fetch_rate_per_min=fetch_rate,
|
|
error_rate=error_rate,
|
|
uptime_s=elapsed,
|
|
)
|
|
|
|
self._metrics_history.append(metrics)
|
|
|
|
# Write to JSONL
|
|
try:
|
|
with open(self._log_path, "a") as f:
|
|
f.write(json.dumps({
|
|
"ts": metrics.timestamp_ns,
|
|
"sources": metrics.total_sources,
|
|
"fetched": metrics.total_fetched,
|
|
"regimes": metrics.discovered_regimes,
|
|
"fetch_rate": round(metrics.fetch_rate_per_min, 2),
|
|
"error_rate": round(metrics.error_rate, 4),
|
|
"uptime": round(metrics.uptime_s, 1),
|
|
}, separators=(",", ":")) + "\n")
|
|
except OSError:
|
|
pass
|
|
|
|
return metrics
|
|
|
|
def check_alerts(self) -> List[str]:
|
|
"""Check for conditions that need attention."""
|
|
alerts = []
|
|
if not self._metrics_history:
|
|
return alerts
|
|
latest = self._metrics_history[-1]
|
|
if latest.error_rate > 0.1:
|
|
alerts.append(f"HIGH_ERROR_RATE: {latest.error_rate:.1%}")
|
|
if latest.fetch_rate_per_min < 1.0 and latest.uptime_s > 300:
|
|
alerts.append(f"LOW_FETCH_RATE: {latest.fetch_rate_per_min:.1f}/min")
|
|
return alerts
|
|
|
|
@property
|
|
def total_fetched(self) -> int:
|
|
return self._total_fetched
|
|
|
|
@property
|
|
def total_errors(self) -> int:
|
|
return self._total_errors
|
|
|
|
@property
|
|
def discovered_regimes(self) -> int:
|
|
return self._regime_count
|