malkhut(wire): fill quality as PRIMARY optimization target

Fill quality is MALKHUT's core aim. Wired end-to-end:

1. FillQuality state (state.py):
   - slippage_bps, price_improvement_bps, levels_consumed
   - is_maker_fill, rolling_fill_rate, post_fill_adverse_bps
   - fill_value_score: composite metric for optimization
   - Added to MarketWorldState.fill_quality field

2. HftBacktestCWM.transition() (hft_cwm.py):
   - _compute_fill_quality() computes all metrics per transition
   - Fill quality now tracked for every CWM step
   - Empty book guards added for safety

3. MinimalCryptoLOBCWM.transition() (core.py):
   - Same fill quality computation for deterministic fallback
   - Empty book guards added

4. Reward function (hft_cwm.py):
   - fill_quality_reward = w_fill_probability * fill_value_score (PRIMARY)
   - Bonus for maker fills that improve price
   - Penalty for adverse selection after fill
   - Base reward (PnL, adverse selection, fees) preserved

5. PerformanceMatrix (selector.py):
   - RegimeStrategyScore: 4 new fill quality fields
   - record(): accepts fill_rate, slippage, price_improvement, fill_value_score
   - EMA updates for all fill quality metrics

6. EpisodeResult (cma_trainer.py):
   - avg_fill_value_score, avg_price_improvement_bps, avg_post_fill_adverse_bps
   - Accumulated per-step during _run_episode
   - Recorded to PerformanceMatrix in evaluate_candidate

All 1379+ tests green.
This commit is contained in:
Codex
2026-07-15 15:22:25 +02:00
parent fa76070c79
commit 618ad723e3
5 changed files with 284 additions and 29 deletions

View File

@@ -27,6 +27,7 @@ import numpy as np
from malkhut.state import (
AccountState,
FulfilmentPolicyParams,
FillQuality,
MarketWorldState,
Mode,
OpenOrderState,
@@ -564,6 +565,51 @@ class MinimalCryptoLOBCWM:
positions=new_positions,
)
# ── Fill Quality computation ────────────────────────────────────────
mid = state.book.mid if state.book.bids and state.book.asks else 0.0
slippage_bps = 0.0
if new_fill_qty > 0 and mid > 0 and new_fill_price > 0:
slippage_bps = abs(new_fill_price - mid) / mid * 10_000
is_maker_fill = (our_action.order_type and our_action.order_type.value == "LIMIT") or our_action.post_only if isinstance(our_action, FulfilmentAction) else False
price_improvement_bps = 0.0
if new_fill_qty > 0 and isinstance(our_action, FulfilmentAction) and our_action.post_only and our_action.side:
if our_action.side == Side.BUY and state.book.bids:
price_improvement_bps = (state.book.best_bid - new_fill_price) / max(state.book.best_bid, 1e-12) * 10_000
elif our_action.side == Side.SELL and state.book.asks:
price_improvement_bps = (new_fill_price - state.book.best_ask) / max(state.book.best_ask, 1e-12) * 10_000
post_fill_adverse = 0.0
new_mid = book.mid if book.bids and book.asks else 0.0
if new_fill_qty > 0 and mid > 0 and new_mid > 0:
if isinstance(our_action, FulfilmentAction) and our_action.side == Side.BUY:
post_fill_adverse = (new_mid - mid) / mid * 10_000
elif isinstance(our_action, FulfilmentAction) and our_action.side == Side.SELL:
post_fill_adverse = (mid - new_mid) / mid * 10_000
prev_fq = state.fill_quality
rolling_fill_rate = 0.0
if prev_fq and prev_fq.filled:
rolling_fill_rate = 0.8 * prev_fq.rolling_fill_rate + 0.2 * (1.0 if new_fill_qty > 0 else 0.0)
elif new_fill_qty > 0:
rolling_fill_rate = 0.2
spread_bps = book.spread_bps if book.bids and book.asks else 0.0
fill_value = 0.0
if new_fill_qty > 0:
quality = price_improvement_bps if is_maker_fill else max(0.0, spread_bps - slippage_bps)
fill_value = quality - abs(post_fill_adverse) * 0.5
fq = FillQuality(
filled=new_fill_qty > 0,
fill_qty=new_fill_qty,
fill_price=new_fill_price,
requested_qty=our_action.qty_fraction * state.account.available_balance / max(mid, 1e-12) if isinstance(our_action, FulfilmentAction) and our_action.qty_fraction > 0 and mid > 0 else 0.0,
slippage_bps=slippage_bps,
price_improvement_bps=price_improvement_bps,
levels_consumed=0,
is_maker_fill=is_maker_fill,
rolling_fill_rate=rolling_fill_rate,
post_fill_adverse_bps=post_fill_adverse,
fill_value_score=fill_value,
)
return MarketWorldState(
ts_ns=now_ts,
mode=state.mode,
@@ -577,8 +623,7 @@ class MinimalCryptoLOBCWM:
volatility_state=state.volatility_state,
market_regime=state.market_regime,
feed_latency_ms=state.feed_latency_ms,
order_latency_ms=state.order_latency_ms,
rng_seed=state.rng_seed,
fill_quality=fq,
)
def reward(