malkhut(T2): Code World Model — deterministic exchange simulator
CWM core (core.py): price-time priority, sequential level consumption, partial fills, queue position, latency injection, maker/taker fees. Numba acceleration (numba_core.py): JIT hot loops, 1.8x fill speedup. Replay verification (replay_verify.py): binary search, trajectory recording. Supporting: adverse_selection, correlation, latency_model, multi_level, queue_model, spread_dynamics, volatility, hftbacktest_validator.
This commit is contained in:
193
MALKHUT/malkhut/cwm/adverse_selection.py
Normal file
193
MALKHUT/malkhut/cwm/adverse_selection.py
Normal file
@@ -0,0 +1,193 @@
|
||||
"""
|
||||
Adverse Selection Cost Model — quantify the cost of being picked off.
|
||||
|
||||
Measures:
|
||||
- Expected adverse selection cost per quote
|
||||
- Cost of being at the front of a toxic queue
|
||||
- Optimal quote placement to minimize adverse selection
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional, Tuple
|
||||
|
||||
import numpy as np
|
||||
from numba import njit
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class AdverseSelectionCost:
|
||||
"""Components of adverse selection cost."""
|
||||
expected_cost_bps: float # expected cost in basis points
|
||||
pick_off_probability: float # probability of being picked off
|
||||
toxic_flow_fraction: float # fraction of fills that are toxic
|
||||
queue_position_risk: float # risk from queue position
|
||||
|
||||
|
||||
@njit(cache=True)
|
||||
def compute_adverse_selection_cost(
|
||||
spread_bps: float,
|
||||
toxicity: float,
|
||||
queue_position: int,
|
||||
recent_trade_rate: float,
|
||||
quote_size_fraction: float,
|
||||
time_horizon_s: float,
|
||||
) -> float:
|
||||
"""
|
||||
Compute expected adverse selection cost in basis points.
|
||||
|
||||
Model:
|
||||
- Base cost = spread_bps * pick_off_probability
|
||||
- Pick-off probability increases with toxicity and queue position
|
||||
- Cost is proportional to quote size
|
||||
|
||||
Returns expected cost in basis points.
|
||||
"""
|
||||
if spread_bps <= 0 or toxicity <= 0:
|
||||
return 0.0
|
||||
|
||||
# Pick-off probability: higher toxicity = more likely to be picked off
|
||||
pick_off_prob = min(1.0, toxicity * 1.5)
|
||||
|
||||
# Queue position factor: front of queue = higher pick-off risk
|
||||
queue_factor = 1.0 / (1.0 + queue_position * 0.1)
|
||||
|
||||
# Expected adverse selection cost
|
||||
base_cost = spread_bps * pick_off_prob * queue_factor
|
||||
|
||||
# Scale by quote size
|
||||
size_factor = quote_size_fraction
|
||||
|
||||
return base_cost * size_factor
|
||||
|
||||
|
||||
@njit(cache=True)
|
||||
def compute_toxic_fill_ratio(
|
||||
fills: np.ndarray,
|
||||
fill_times: np.ndarray,
|
||||
toxicity_threshold: float,
|
||||
) -> float:
|
||||
"""
|
||||
Compute ratio of toxic fills.
|
||||
|
||||
A fill is "toxic" if the price moves adversely after the fill.
|
||||
Simplified: fill is toxic if toxicity > threshold at time of fill.
|
||||
|
||||
Returns 0.0-1.0 ratio.
|
||||
"""
|
||||
if len(fills) == 0:
|
||||
return 0.0
|
||||
toxic_count = 0
|
||||
for i in range(len(fills)):
|
||||
if fills[i] > toxicity_threshold:
|
||||
toxic_count += 1
|
||||
return toxic_count / len(fills)
|
||||
|
||||
|
||||
@njit(cache=True)
|
||||
def optimal_quote_offset(
|
||||
spread_bps: float,
|
||||
toxicity: float,
|
||||
queue_depth: float,
|
||||
our_qty: float,
|
||||
recent_trade_rate: float,
|
||||
) -> int:
|
||||
"""
|
||||
Compute optimal quote offset (ticks from best) to minimize adverse selection.
|
||||
|
||||
Model:
|
||||
- Offset 0 (best bid/ask): highest fill probability, highest adverse selection
|
||||
- Offset 1+: lower fill probability, lower adverse selection
|
||||
- Optimal offset balances fill probability vs adverse selection cost
|
||||
|
||||
Returns optimal offset in ticks.
|
||||
"""
|
||||
if spread_bps <= 0 or toxicity <= 0:
|
||||
return 0
|
||||
|
||||
best_offset = 0
|
||||
best_score = -float("inf")
|
||||
|
||||
for offset in range(5): # check offsets 0-4
|
||||
# Fill probability decreases with offset
|
||||
fill_prob = max(0.0, 1.0 - offset * 0.2)
|
||||
|
||||
# Adverse selection cost decreases with offset
|
||||
adverse_cost = spread_bps * toxicity * max(0.0, 1.0 - offset * 0.3)
|
||||
|
||||
# Score: maximize fill probability minus adverse cost
|
||||
score = fill_prob - adverse_cost * 0.1
|
||||
|
||||
if score > best_score:
|
||||
best_score = score
|
||||
best_offset = offset
|
||||
|
||||
return best_offset
|
||||
|
||||
|
||||
class AdverseSelectionModel:
|
||||
"""
|
||||
Adverse selection cost model for the CWM.
|
||||
|
||||
Integrates with queue model and spread dynamics to provide:
|
||||
- Expected adverse selection cost per quote
|
||||
- Optimal quote placement
|
||||
- Toxic fill ratio tracking
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._toxic_fills: list[float] = []
|
||||
self._total_fills: int = 0
|
||||
|
||||
def compute_cost(
|
||||
self,
|
||||
spread_bps: float,
|
||||
toxicity: float,
|
||||
queue_position: int,
|
||||
recent_trade_rate: float = 0.5,
|
||||
quote_size_fraction: float = 0.25,
|
||||
time_horizon_s: float = 300.0,
|
||||
) -> AdverseSelectionCost:
|
||||
"""Compute adverse selection cost for a quote."""
|
||||
cost_bps = compute_adverse_selection_cost(
|
||||
spread_bps, toxicity, queue_position, recent_trade_rate,
|
||||
quote_size_fraction, time_horizon_s,
|
||||
)
|
||||
pick_off_prob = min(1.0, toxicity * 1.5) * (1.0 / (1.0 + queue_position * 0.1))
|
||||
toxic_frac = self.toxic_fill_ratio
|
||||
|
||||
return AdverseSelectionCost(
|
||||
expected_cost_bps=cost_bps,
|
||||
pick_off_probability=pick_off_prob,
|
||||
toxic_flow_fraction=toxic_frac,
|
||||
queue_position_risk=1.0 / (1.0 + queue_position * 0.1),
|
||||
)
|
||||
|
||||
def optimal_offset(
|
||||
self,
|
||||
spread_bps: float,
|
||||
toxicity: float,
|
||||
queue_depth: float = 1.0,
|
||||
our_qty: float = 0.001,
|
||||
recent_trade_rate: float = 0.5,
|
||||
) -> int:
|
||||
"""Compute optimal quote offset."""
|
||||
return optimal_quote_offset(spread_bps, toxicity, queue_depth, our_qty, recent_trade_rate)
|
||||
|
||||
def record_fill(self, toxicity: float) -> None:
|
||||
"""Record a fill for toxic fill ratio tracking."""
|
||||
self._toxic_fills.append(toxicity)
|
||||
self._total_fills += 1
|
||||
|
||||
@property
|
||||
def toxic_fill_ratio(self) -> float:
|
||||
if self._total_fills == 0:
|
||||
return 0.0
|
||||
return sum(1 for t in self._toxic_fills if t > 0.5) / self._total_fills
|
||||
|
||||
@property
|
||||
def average_toxicity(self) -> float:
|
||||
if not self._toxic_fills:
|
||||
return 0.0
|
||||
return sum(self._toxic_fills) / len(self._toxic_fills)
|
||||
Reference in New Issue
Block a user