#!/usr/bin/env python3 """ MALKHUT 60-Minute Smoke Test — comprehensive system validation. Runs the full pipeline for 60 minutes with: - Training pipeline (CMA-ES + genetic programming) - Strategy generator (evolving strategies) - All 9 planner types cycling - Performance metrics tracking - Resource usage monitoring - Strategy development tracking - Improvement metrics Usage: python -m malkhut.smoke_test_60min """ from __future__ import annotations import json import os import resource import sys import threading import time from dataclasses import dataclass, field from typing import Any, Dict, List, Optional # Setup paths _HERE = os.path.dirname(os.path.abspath(__file__)) if _HERE not in sys.path: sys.path.insert(0, _HERE) from malkhut.state import FulfilmentPolicyParams from malkhut.training.pipeline import TrainingPipeline, PipelineConfig from malkhut.training.generator import StrategyGenerator, GeneratorConfig from malkhut.training.registry import PolicyRegistry from malkhut.training.cma_trainer import ScenarioFactory from malkhut.planner.alternatives import PLANNER_REGISTRY, create_planner from malkhut.cwm.core import MinimalCryptoLOBCWM from malkhut.counterparties import default_counterparty_ecology from malkhut.storage.ch_store import MalkhutCHStore def _baseline() -> FulfilmentPolicyParams: return FulfilmentPolicyParams( version="baseline", ucb_c=1.414, max_sims=256, max_depth=3, rollout_depth=3, root_temperature=0.5, min_root_entropy=0.25, quote_offsets_ticks=(0, 1, 2), quote_size_fractions=(0.1, 0.25, 0.5), passive_ttl_ms=200, aggressive_ttl_ms=50, maker_edge_min_bps=0.5, cross_spread_edge_min_bps=5.0, adverse_toxicity_cancel_threshold=0.5, queue_churn_cancel_threshold=0.5, mae_tail_cut_bps=50.0, mfe_giveback_cut_fraction=0.5, max_time_in_loss_s=300.0, failed_recovery_cut_count=3, recovery_velocity_min_bps_per_s=0.0, max_symbol_notional_fraction=0.20, max_single_order_notional_fraction=0.05, reduce_when_global_up_fraction=0.30, session_profit_lock_fraction=0.02, w_expected_pnl=1.0, w_fill_probability=0.5, w_adverse_selection=2.0, w_queue_priority=0.5, w_inventory_risk=1.5, w_tail_loss=5.0, w_fee_quality=0.5, w_time_decay=0.3, w_policy_entropy=0.5, robust_tail_weight=2.0, toxic_counterparty_weight=3.0, low_liquidity_weight=2.0, latency_stress_weight=1.0, ) # ── Resource Monitor ───────────────────────────────────────────────────────── class ResourceMonitor: """Track CPU and RAM usage.""" def __init__(self): self._samples: list = [] self._start = time.time() self._peak_ram = 0.0 self._peak_cpu = 0.0 def sample(self): usage = resource.getrusage(resource.RUSAGE_SELF) ram_mb = usage.ru_maxrss / 1024 try: with open("/proc/self/stat") as f: fields = f.read().split() utime = int(fields[13]) stime = int(fields[14]) elapsed = time.time() - self._start cpu = min(100.0, ((utime + stime) * 10.0) / max(elapsed * 1000.0, 1.0) * 100.0) except Exception: cpu = 0.0 self._samples.append({"time": time.time() - self._start, "cpu": cpu, "ram_mb": ram_mb}) self._peak_ram = max(self._peak_ram, ram_mb) self._peak_cpu = max(self._peak_cpu, cpu) @property def avg_cpu(self) -> float: if not self._samples: return 0 return sum(s["cpu"] for s in self._samples) / len(self._samples) @property def avg_ram(self) -> float: if not self._samples: return 0 return sum(s["ram_mb"] for s in self._samples) / len(self._samples) # ── Strategy Tracker ──────────────────────────────────────────────────────── class StrategyTracker: """Track strategies developed and their improvement.""" def __init__(self): self._strategies: list = [] self._scores: list = [] self._planner_types_used: dict = {} self._best_score_history: list = [] def record(self, score: float, planner_type: str, generation: int): self._strategies.append({"score": score, "planner": planner_type, "gen": generation}) self._scores.append(score) self._planner_types_used[planner_type] = self._planner_types_used.get(planner_type, 0) + 1 self._best_score_history.append(max(self._scores) if self._scores else 0) @property def total_strategies(self) -> int: return len(self._strategies) @property def best_score(self) -> float: return max(self._scores) if self._scores else 0 @property def improvement(self) -> float: if len(self._scores) < 2: return 0 return self._best_score_history[-1] - self._best_score_history[0] @property def planner_usage(self) -> dict: return dict(self._planner_types_used) def summary(self) -> dict: return { "total_strategies": self.total_strategies, "best_score": self.best_score, "improvement": self.improvement, "planner_usage": self.planner_usage, "score_history_len": len(self._best_score_history), } # ── Main Smoke Test ────────────────────────────────────────────────────────── def run_60min_smoke(): DURATION_S = 3600 # 60 minutes print("=" * 70) print("MALKHUT 60-MINUTE SMOKE TEST") print(f"Duration: {DURATION_S}s ({DURATION_S // 60} minutes)") print("=" * 70) monitor = ResourceMonitor() tracker = StrategyTracker() t0 = time.time() # Setup print("\n[1/4] Setting up infrastructure...") monitor.sample() store = MalkhutCHStore() store.ensure_tables() registry = PolicyRegistry(store=store) # Training pipeline print("[2/4] Running training pipeline (cycles through ALL 9 planner types)...") pipeline_config = PipelineConfig( max_generations=20, max_evals_per_generation=10, max_time_s=DURATION_S * 0.6, auto_promote=True, ) pipeline = TrainingPipeline( config=pipeline_config, registry=registry, log_path=os.path.join(_HERE, "smoke_60min.log"), ) monitor.sample() # Run training pipeline_result = pipeline.run( incumbent=_baseline(), symbols=("BTCUSDT",), ) monitor.sample() # Track strategies from training for event in pipeline_result.events: if event.event_type == "generation": tracker.record(event.score, "cma_es", event.generation) print(f" Generations: {pipeline_result.generations_run}") print(f" Evals: {pipeline_result.total_evals}") print(f" Best score: {pipeline_result.best_score:.2f}") # Strategy generator print("[3/4] Running strategy generator (genetic programming)...") remaining_time = DURATION_S * 0.3 - pipeline_result.duration_s if remaining_time > 30: gen_config = GeneratorConfig( population_size=15, generations=3, tournament_size=3, elitism_count=2, ) generator = StrategyGenerator(config=gen_config, registry=registry) scenarios = ScenarioFactory().build_suite(symbols=("BTCUSDT",), steps_per_scenario=5) gen_population = generator.evolve(_baseline(), scenarios) genetic_count = len([g for g in gen_population if g.generation > 0]) for genome in gen_population: if genome.generation > 0: tracker.record(genome.fitness, genome.strategy_type.value, genome.generation) generator.add_to_pool(genome) print(f" Population: {len(gen_population)} strategies") print(f" Genetic strategies: {genetic_count}") monitor.sample() # Planner diversity test print("[4/4] Testing all 9 planner types...") planner_scores = {} for name in PLANNER_REGISTRY.keys(): try: cwm = MinimalCryptoLOBCWM() planner = create_planner(name, cwm=cwm, counterparties=default_counterparty_ecology()) from malkhut.state import ExecutionIntent, IntentKind, MarketWorldState, Mode, OrderBookState, AccountState, PriceLevel s = MarketWorldState( ts_ns=1, mode=Mode.REPLAY_NO_IMPACT, venue=_venue(), book=_book(), account=_account(), intent=_intent(), ) result = planner.plan(s, _baseline(), budget_ms=10) planner_scores[name] = len(result.actions) except Exception as e: planner_scores[name] = f"error: {e}" # Final metrics duration = time.time() - t0 monitor.sample() print() print("=" * 70) print("60-MINUTE SMOKE TEST RESULTS") print("=" * 70) print(f"Duration: {duration:.1f}s ({duration/60:.1f} min)") print(f"Generations: {pipeline_result.generations_run}") print(f"Total evals: {pipeline_result.total_evals}") print(f"Best score: {pipeline_result.best_score:.2f}") print(f"Strategies dev: {tracker.total_strategies}") print(f"Improvement: {tracker.improvement:.2f}") print() print("RESOURCE USAGE") print(f"Peak CPU: {monitor._peak_cpu:.1f}%") print(f"Avg CPU: {monitor.avg_cpu:.1f}%") print(f"Peak RAM: {monitor._peak_ram:.1f} MB") print(f"Avg RAM: {monitor.avg_ram:.1f} MB") print() print("PLANNER USAGE") for ptype, count in tracker.planner_usage.items(): print(f" {ptype:<20} {count} evaluations") print() print("PLANNER DIVERSITY") for name, score in planner_scores.items(): print(f" {name:<20} {score} actions") print() print("EVENTS LOGGED") print(f" Pipeline events: {len(pipeline_result.events)}") print(f" Registry records: {registry.record_count}") print("=" * 70) # Save results results = { "duration_s": duration, "generations": pipeline_result.generations_run, "total_evals": pipeline_result.total_evals, "best_score": pipeline_result.best_score, "strategies_developed": tracker.total_strategies, "improvement": tracker.improvement, "peak_cpu_pct": monitor._peak_cpu, "avg_cpu_pct": monitor.avg_cpu, "peak_ram_mb": monitor._peak_ram, "avg_ram_mb": monitor.avg_ram, "planner_usage": tracker.planner_usage, "planner_diversity": planner_scores, "events_logged": len(pipeline_result.events), "registry_records": registry.record_count, } with open(os.path.join(_HERE, "smoke_60min_results.json"), "w") as f: json.dump(results, f, indent=2) print(f"\nResults saved to smoke_60min_results.json") def _venue(): from malkhut.state import VenueRules return VenueRules(exchange="bingx", symbol="BTCUSDT", tick_size=0.1, lot_size=0.001, min_qty=0.001, min_notional=5.0, maker_fee_bps=-0.2, taker_fee_bps=0.5, post_only_supported=True, reduce_only_supported=True, max_orders_per_second=100, max_cancels_per_minute=120) def _book(): from malkhut.state import OrderBookState, PriceLevel return OrderBookState(ts_ns=1, symbol="BTCUSDT", bids=(PriceLevel(50000.0, 1.0),), asks=(PriceLevel(50001.0, 1.0),)) def _account(): from malkhut.state import AccountState return AccountState(ts_ns=1, equity=10000.0, wallet_balance=10000.0, available_balance=10000.0, margin_used=0.0, total_notional=0.0) def _intent(): from malkhut.state import ExecutionIntent, IntentKind return ExecutionIntent( intent_id="smoke", ts_ns=1, symbol="BTCUSDT", kind=IntentKind.ENTER_LONG, target_qty=0.01, max_notional=500.0, urgency=0.5, alpha_horizon_s=60.0, alpha_bps=2.0, max_slippage_bps=5.0, prefer_maker=True, reduce_only=False, ttl_s=300.0, reason="smoke_test", ) if __name__ == "__main__": run_60min_smoke()