malkhut(wire): fill quality as PRIMARY optimization target
Fill quality is MALKHUT's core aim. Wired end-to-end: 1. FillQuality state (state.py): - slippage_bps, price_improvement_bps, levels_consumed - is_maker_fill, rolling_fill_rate, post_fill_adverse_bps - fill_value_score: composite metric for optimization - Added to MarketWorldState.fill_quality field 2. HftBacktestCWM.transition() (hft_cwm.py): - _compute_fill_quality() computes all metrics per transition - Fill quality now tracked for every CWM step - Empty book guards added for safety 3. MinimalCryptoLOBCWM.transition() (core.py): - Same fill quality computation for deterministic fallback - Empty book guards added 4. Reward function (hft_cwm.py): - fill_quality_reward = w_fill_probability * fill_value_score (PRIMARY) - Bonus for maker fills that improve price - Penalty for adverse selection after fill - Base reward (PnL, adverse selection, fees) preserved 5. PerformanceMatrix (selector.py): - RegimeStrategyScore: 4 new fill quality fields - record(): accepts fill_rate, slippage, price_improvement, fill_value_score - EMA updates for all fill quality metrics 6. EpisodeResult (cma_trainer.py): - avg_fill_value_score, avg_price_improvement_bps, avg_post_fill_adverse_bps - Accumulated per-step during _run_episode - Recorded to PerformanceMatrix in evaluate_candidate All 1379+ tests green.
This commit is contained in:
@@ -147,8 +147,9 @@ class RegimeClassifier:
|
||||
|
||||
@dataclass
|
||||
class RegimeStrategyScore:
|
||||
"""Performance score for a strategy in a specific regime.
|
||||
"""Performance score for a strategy in a specific regime + venue.
|
||||
|
||||
Fill quality metrics are the PRIMARY optimization target.
|
||||
Manifold fields (for Mode 2 recommendation):
|
||||
confidence: 0.0-1.0, how reliable is this score
|
||||
support_count: how many evaluations produced this score
|
||||
@@ -165,6 +166,11 @@ class RegimeStrategyScore:
|
||||
confidence: float = 1.0
|
||||
support_count: int = 1
|
||||
distance_to_nearest: float = 0.0
|
||||
# Fill quality metrics (PRIMARY)
|
||||
avg_fill_rate: float = 0.0
|
||||
avg_slippage_bps: float = 0.0
|
||||
avg_price_improvement_bps: float = 0.0
|
||||
avg_fill_value_score: float = 0.0
|
||||
|
||||
|
||||
class PerformanceMatrix:
|
||||
@@ -192,6 +198,10 @@ class PerformanceMatrix:
|
||||
drawdown_bps: float = 0.0,
|
||||
adverse_fill_ratio: float = 0.0,
|
||||
venue: str = "bingx",
|
||||
fill_rate: float = 0.0,
|
||||
slippage_bps: float = 0.0,
|
||||
price_improvement_bps: float = 0.0,
|
||||
fill_value_score: float = 0.0,
|
||||
) -> None:
|
||||
"""Record a strategy's performance in a regime on a specific venue."""
|
||||
key = (regime, strategy_id, venue)
|
||||
@@ -204,12 +214,20 @@ class PerformanceMatrix:
|
||||
new_pnl = alpha * pnl_bps + (1 - alpha) * existing.avg_pnl_bps
|
||||
new_dd = alpha * drawdown_bps + (1 - alpha) * existing.avg_drawdown_bps
|
||||
new_adverse = alpha * adverse_fill_ratio + (1 - alpha) * existing.avg_adverse_fill_ratio
|
||||
new_fill_rate = alpha * fill_rate + (1 - alpha) * existing.avg_fill_rate
|
||||
new_slip = alpha * slippage_bps + (1 - alpha) * existing.avg_slippage_bps
|
||||
new_improve = alpha * price_improvement_bps + (1 - alpha) * existing.avg_price_improvement_bps
|
||||
new_fv = alpha * fill_value_score + (1 - alpha) * existing.avg_fill_value_score
|
||||
else:
|
||||
new_score = score
|
||||
new_episodes = 1
|
||||
new_pnl = pnl_bps
|
||||
new_dd = drawdown_bps
|
||||
new_adverse = adverse_fill_ratio
|
||||
new_fill_rate = fill_rate
|
||||
new_slip = slippage_bps
|
||||
new_improve = price_improvement_bps
|
||||
new_fv = fill_value_score
|
||||
|
||||
self._scores[key] = RegimeStrategyScore(
|
||||
strategy_id=strategy_id,
|
||||
@@ -223,6 +241,10 @@ class PerformanceMatrix:
|
||||
confidence=min(1.0, new_episodes / 10.0),
|
||||
support_count=new_episodes,
|
||||
distance_to_nearest=0.0,
|
||||
avg_fill_rate=new_fill_rate,
|
||||
avg_slippage_bps=new_slip,
|
||||
avg_price_improvement_bps=new_improve,
|
||||
avg_fill_value_score=new_fv,
|
||||
)
|
||||
self._strategy_regime_history[strategy_id].append(regime)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user