98 lines
4.1 KiB
Python
98 lines
4.1 KiB
Python
|
|
#!/usr/bin/env python3
|
|||
|
|
"""MALKHUT 1h+ Training Smoke — standalone script for background execution."""
|
|||
|
|
import time, os, sys, json
|
|||
|
|
|
|||
|
|
LOG = "/mnt/dolphinng5_predict/MALKHUT/smoke_1h_run.log"
|
|||
|
|
BUDGET = 192
|
|||
|
|
WORKERS = min(os.cpu_count() or 4, 8)
|
|||
|
|
|
|||
|
|
with open(LOG, "w") as f:
|
|||
|
|
f.write(f"MALKHUT 1H+ TRAINING SMOKE\n")
|
|||
|
|
f.write(f"Start: {time.strftime('%Y-%m-%d %H:%M:%S')}\n")
|
|||
|
|
f.write(f"Config: budget={BUDGET} workers={WORKERS} pop=12\n")
|
|||
|
|
f.write(f"Assets: BTC/ETH/SOL (3 × 30 scenarios = 90)\n\n")
|
|||
|
|
f.flush()
|
|||
|
|
|
|||
|
|
print(f"Starting 1h+ smoke: budget={BUDGET} workers={WORKERS}", flush=True)
|
|||
|
|
|
|||
|
|
from malkhut.training.cma_trainer import (
|
|||
|
|
ScenarioFactory, PolicyEvaluator, CMAESTrainer, CMAParameterCodec, SelfPlayPool
|
|||
|
|
)
|
|||
|
|
from malkhut.cwm.core import MinimalCryptoLOBCWM
|
|||
|
|
from malkhut.state import FulfilmentPolicyParams
|
|||
|
|
import cma as cma_lib
|
|||
|
|
|
|||
|
|
factory = ScenarioFactory()
|
|||
|
|
suite = factory.build_suite(symbols=('BTCUSDT', 'ETHUSDT', 'SOLUSDT'), steps_per_scenario=5)
|
|||
|
|
evaluator = PolicyEvaluator(cwm_factory=MinimalCryptoLOBCWM)
|
|||
|
|
codec = CMAParameterCodec()
|
|||
|
|
pool = SelfPlayPool()
|
|||
|
|
|
|||
|
|
params = FulfilmentPolicyParams(
|
|||
|
|
version='baseline', ucb_c=1.414, max_sims=16, max_depth=2,
|
|||
|
|
rollout_depth=2, root_temperature=0.5, min_root_entropy=0.25,
|
|||
|
|
quote_offsets_ticks=(0, 1), quote_size_fractions=(0.25, 0.50),
|
|||
|
|
passive_ttl_ms=200, aggressive_ttl_ms=50,
|
|||
|
|
maker_edge_min_bps=0.5, cross_spread_edge_min_bps=5.0,
|
|||
|
|
adverse_toxicity_cancel_threshold=0.5, queue_churn_cancel_threshold=0.5,
|
|||
|
|
mae_tail_cut_bps=50.0, mfe_giveback_cut_fraction=0.5,
|
|||
|
|
max_time_in_loss_s=300.0, failed_recovery_cut_count=3,
|
|||
|
|
recovery_velocity_min_bps_per_s=0.0,
|
|||
|
|
max_symbol_notional_fraction=0.20, max_single_order_notional_fraction=0.05,
|
|||
|
|
reduce_when_global_up_fraction=0.30, session_profit_lock_fraction=0.02,
|
|||
|
|
w_expected_pnl=1.0, w_fill_probability=0.5, w_adverse_selection=2.0,
|
|||
|
|
w_queue_priority=0.5, w_inventory_risk=1.5, w_tail_loss=5.0,
|
|||
|
|
w_fee_quality=0.5, w_time_decay=0.3, w_policy_entropy=0.5,
|
|||
|
|
robust_tail_weight=2.0, toxic_counterparty_weight=3.0,
|
|||
|
|
low_liquidity_weight=2.0, latency_stress_weight=1.0,
|
|||
|
|
)
|
|||
|
|
|
|||
|
|
x0 = codec.initial_vector(params)
|
|||
|
|
lows, highs = codec.bounds()
|
|||
|
|
es = cma_lib.CMAEvolutionStrategy(x0, sigma0=0.30,
|
|||
|
|
inopts={"bounds": [lows, highs], "popsize": 12, "seed": 42, "verbose": -9})
|
|||
|
|
|
|||
|
|
eval_count, gen_best, gen_pnl = 0, [], []
|
|||
|
|
t0 = time.time()
|
|||
|
|
next_log = 120
|
|||
|
|
|
|||
|
|
with open(LOG, "a") as f:
|
|||
|
|
while not es.stop() and eval_count < BUDGET:
|
|||
|
|
xs = es.ask()
|
|||
|
|
losses, gs, gp = [], [], []
|
|||
|
|
for x in xs:
|
|||
|
|
if eval_count >= BUDGET:
|
|||
|
|
break
|
|||
|
|
cand = codec.decode(x, version=f"e{eval_count}")
|
|||
|
|
score, results = evaluator.evaluate_candidate(
|
|||
|
|
params=cand, scenarios=suite, rng_seed=eval_count,
|
|||
|
|
planner_type='sm_mcts', workers=WORKERS)
|
|||
|
|
pnl = sum(r.pnl_bps for r in results) / max(len(results), 1)
|
|||
|
|
losses.append(-score); gs.append(score); gp.append(pnl)
|
|||
|
|
eval_count += 1
|
|||
|
|
if gs:
|
|||
|
|
es.tell(xs[:len(losses)], losses)
|
|||
|
|
gen_best.append(max(gs))
|
|||
|
|
gen_pnl.append(sum(gp)/len(gp))
|
|||
|
|
elapsed = time.time() - t0
|
|||
|
|
if elapsed >= next_log:
|
|||
|
|
msg = (f"[{elapsed:.0f}s] Gen {len(gen_best)} | "
|
|||
|
|
f"{eval_count}/{BUDGET} evals | best={max(gen_best):,.0f} | "
|
|||
|
|
f"pnl={gen_pnl[-1]:.1f}bps | {eval_count/elapsed:.2f}e/s")
|
|||
|
|
f.write(msg + "\n"); f.flush()
|
|||
|
|
print(msg, flush=True)
|
|||
|
|
next_log += 120
|
|||
|
|
|
|||
|
|
total = time.time() - t0
|
|||
|
|
summary = (f"\n{'='*60}\nCOMPLETE\n"
|
|||
|
|
f" Duration: {total:.0f}s ({total/60:.1f}min)\n"
|
|||
|
|
f" Evals: {eval_count}/{BUDGET}\n"
|
|||
|
|
f" Rate: {eval_count/total:.2f} eval/s\n"
|
|||
|
|
f" Best score: {max(gen_best):,.0f} (gen {gen_best.index(max(gen_best))+1}/{len(gen_best)})\n"
|
|||
|
|
f" Final gen PnL: {gen_pnl[-1]:.1f}bps\n"
|
|||
|
|
f" Score curve (last 8): {gen_best[-8:]}\n"
|
|||
|
|
f" PnL curve (last 8): {[f'{p:.0f}' for p in gen_pnl[-8:]]}\n"
|
|||
|
|
f"{'='*60}\n")
|
|||
|
|
f.write(summary); f.flush()
|
|||
|
|
print(summary, flush=True)
|