Files
sentiment-engine/MALKHUT/smoke_1h.py
Codex 70964394d4 malkhut(docs + bench): comprehensive update + smoke test script
README updated with:
- Vectorized UCB selection (7.7x speedup, 1.13µs/selection)
- Batch MCTS kernel (numba-accelerated)
- Fast scalar + advantage scoring modes
- Updated performance benchmarks (1186 tests, 390 scenarios, 3043 score/min)
- Advantage scorer module in package structure

smoke_1h.py: standalone training script for extended runs.

Total session: 19 commits, 1186 tests, all green.
All implementations: parallel eval (7x), vectorized reward (numba),
vectorized UCB (7.7x), fast scalar scoring, advantage mode,
DuckDB store (sub-µs reads), asset compiler, behavior DSL,
multi-exchange support, three-layer identifiers.
2026-07-13 17:04:40 +02:00

98 lines
4.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

#!/usr/bin/env python3
"""MALKHUT 1h+ Training Smoke — standalone script for background execution."""
import time, os, sys, json
LOG = "/mnt/dolphinng5_predict/MALKHUT/smoke_1h_run.log"
BUDGET = 192
WORKERS = min(os.cpu_count() or 4, 8)
with open(LOG, "w") as f:
f.write(f"MALKHUT 1H+ TRAINING SMOKE\n")
f.write(f"Start: {time.strftime('%Y-%m-%d %H:%M:%S')}\n")
f.write(f"Config: budget={BUDGET} workers={WORKERS} pop=12\n")
f.write(f"Assets: BTC/ETH/SOL (3 × 30 scenarios = 90)\n\n")
f.flush()
print(f"Starting 1h+ smoke: budget={BUDGET} workers={WORKERS}", flush=True)
from malkhut.training.cma_trainer import (
ScenarioFactory, PolicyEvaluator, CMAESTrainer, CMAParameterCodec, SelfPlayPool
)
from malkhut.cwm.core import MinimalCryptoLOBCWM
from malkhut.state import FulfilmentPolicyParams
import cma as cma_lib
factory = ScenarioFactory()
suite = factory.build_suite(symbols=('BTCUSDT', 'ETHUSDT', 'SOLUSDT'), steps_per_scenario=5)
evaluator = PolicyEvaluator(cwm_factory=MinimalCryptoLOBCWM)
codec = CMAParameterCodec()
pool = SelfPlayPool()
params = FulfilmentPolicyParams(
version='baseline', ucb_c=1.414, max_sims=16, max_depth=2,
rollout_depth=2, root_temperature=0.5, min_root_entropy=0.25,
quote_offsets_ticks=(0, 1), quote_size_fractions=(0.25, 0.50),
passive_ttl_ms=200, aggressive_ttl_ms=50,
maker_edge_min_bps=0.5, cross_spread_edge_min_bps=5.0,
adverse_toxicity_cancel_threshold=0.5, queue_churn_cancel_threshold=0.5,
mae_tail_cut_bps=50.0, mfe_giveback_cut_fraction=0.5,
max_time_in_loss_s=300.0, failed_recovery_cut_count=3,
recovery_velocity_min_bps_per_s=0.0,
max_symbol_notional_fraction=0.20, max_single_order_notional_fraction=0.05,
reduce_when_global_up_fraction=0.30, session_profit_lock_fraction=0.02,
w_expected_pnl=1.0, w_fill_probability=0.5, w_adverse_selection=2.0,
w_queue_priority=0.5, w_inventory_risk=1.5, w_tail_loss=5.0,
w_fee_quality=0.5, w_time_decay=0.3, w_policy_entropy=0.5,
robust_tail_weight=2.0, toxic_counterparty_weight=3.0,
low_liquidity_weight=2.0, latency_stress_weight=1.0,
)
x0 = codec.initial_vector(params)
lows, highs = codec.bounds()
es = cma_lib.CMAEvolutionStrategy(x0, sigma0=0.30,
inopts={"bounds": [lows, highs], "popsize": 12, "seed": 42, "verbose": -9})
eval_count, gen_best, gen_pnl = 0, [], []
t0 = time.time()
next_log = 120
with open(LOG, "a") as f:
while not es.stop() and eval_count < BUDGET:
xs = es.ask()
losses, gs, gp = [], [], []
for x in xs:
if eval_count >= BUDGET:
break
cand = codec.decode(x, version=f"e{eval_count}")
score, results = evaluator.evaluate_candidate(
params=cand, scenarios=suite, rng_seed=eval_count,
planner_type='sm_mcts', workers=WORKERS)
pnl = sum(r.pnl_bps for r in results) / max(len(results), 1)
losses.append(-score); gs.append(score); gp.append(pnl)
eval_count += 1
if gs:
es.tell(xs[:len(losses)], losses)
gen_best.append(max(gs))
gen_pnl.append(sum(gp)/len(gp))
elapsed = time.time() - t0
if elapsed >= next_log:
msg = (f"[{elapsed:.0f}s] Gen {len(gen_best)} | "
f"{eval_count}/{BUDGET} evals | best={max(gen_best):,.0f} | "
f"pnl={gen_pnl[-1]:.1f}bps | {eval_count/elapsed:.2f}e/s")
f.write(msg + "\n"); f.flush()
print(msg, flush=True)
next_log += 120
total = time.time() - t0
summary = (f"\n{'='*60}\nCOMPLETE\n"
f" Duration: {total:.0f}s ({total/60:.1f}min)\n"
f" Evals: {eval_count}/{BUDGET}\n"
f" Rate: {eval_count/total:.2f} eval/s\n"
f" Best score: {max(gen_best):,.0f} (gen {gen_best.index(max(gen_best))+1}/{len(gen_best)})\n"
f" Final gen PnL: {gen_pnl[-1]:.1f}bps\n"
f" Score curve (last 8): {gen_best[-8:]}\n"
f" PnL curve (last 8): {[f'{p:.0f}' for p in gen_pnl[-8:]]}\n"
f"{'='*60}\n")
f.write(summary); f.flush()
print(summary, flush=True)