Files
sentiment-engine/MALKHUT/malkhut/cwm/adverse_selection.py

194 lines
5.6 KiB
Python
Raw Normal View History

"""
Adverse Selection Cost Model — quantify the cost of being picked off.
Measures:
- Expected adverse selection cost per quote
- Cost of being at the front of a toxic queue
- Optimal quote placement to minimize adverse selection
"""
from __future__ import annotations
import math
from dataclasses import dataclass
from typing import Optional, Tuple
import numpy as np
from numba import njit
@dataclass(frozen=True, slots=True)
class AdverseSelectionCost:
"""Components of adverse selection cost."""
expected_cost_bps: float # expected cost in basis points
pick_off_probability: float # probability of being picked off
toxic_flow_fraction: float # fraction of fills that are toxic
queue_position_risk: float # risk from queue position
@njit(cache=True)
def compute_adverse_selection_cost(
spread_bps: float,
toxicity: float,
queue_position: int,
recent_trade_rate: float,
quote_size_fraction: float,
time_horizon_s: float,
) -> float:
"""
Compute expected adverse selection cost in basis points.
Model:
- Base cost = spread_bps * pick_off_probability
- Pick-off probability increases with toxicity and queue position
- Cost is proportional to quote size
Returns expected cost in basis points.
"""
if spread_bps <= 0 or toxicity <= 0:
return 0.0
# Pick-off probability: higher toxicity = more likely to be picked off
pick_off_prob = min(1.0, toxicity * 1.5)
# Queue position factor: front of queue = higher pick-off risk
queue_factor = 1.0 / (1.0 + queue_position * 0.1)
# Expected adverse selection cost
base_cost = spread_bps * pick_off_prob * queue_factor
# Scale by quote size
size_factor = quote_size_fraction
return base_cost * size_factor
@njit(cache=True)
def compute_toxic_fill_ratio(
fills: np.ndarray,
fill_times: np.ndarray,
toxicity_threshold: float,
) -> float:
"""
Compute ratio of toxic fills.
A fill is "toxic" if the price moves adversely after the fill.
Simplified: fill is toxic if toxicity > threshold at time of fill.
Returns 0.0-1.0 ratio.
"""
if len(fills) == 0:
return 0.0
toxic_count = 0
for i in range(len(fills)):
if fills[i] > toxicity_threshold:
toxic_count += 1
return toxic_count / len(fills)
@njit(cache=True)
def optimal_quote_offset(
spread_bps: float,
toxicity: float,
queue_depth: float,
our_qty: float,
recent_trade_rate: float,
) -> int:
"""
Compute optimal quote offset (ticks from best) to minimize adverse selection.
Model:
- Offset 0 (best bid/ask): highest fill probability, highest adverse selection
- Offset 1+: lower fill probability, lower adverse selection
- Optimal offset balances fill probability vs adverse selection cost
Returns optimal offset in ticks.
"""
if spread_bps <= 0 or toxicity <= 0:
return 0
best_offset = 0
best_score = -float("inf")
for offset in range(5): # check offsets 0-4
# Fill probability decreases with offset
fill_prob = max(0.0, 1.0 - offset * 0.2)
# Adverse selection cost decreases with offset
adverse_cost = spread_bps * toxicity * max(0.0, 1.0 - offset * 0.3)
# Score: maximize fill probability minus adverse cost
score = fill_prob - adverse_cost * 0.1
if score > best_score:
best_score = score
best_offset = offset
return best_offset
class AdverseSelectionModel:
"""
Adverse selection cost model for the CWM.
Integrates with queue model and spread dynamics to provide:
- Expected adverse selection cost per quote
- Optimal quote placement
- Toxic fill ratio tracking
"""
def __init__(self) -> None:
self._toxic_fills: list[float] = []
self._total_fills: int = 0
def compute_cost(
self,
spread_bps: float,
toxicity: float,
queue_position: int,
recent_trade_rate: float = 0.5,
quote_size_fraction: float = 0.25,
time_horizon_s: float = 300.0,
) -> AdverseSelectionCost:
"""Compute adverse selection cost for a quote."""
cost_bps = compute_adverse_selection_cost(
spread_bps, toxicity, queue_position, recent_trade_rate,
quote_size_fraction, time_horizon_s,
)
pick_off_prob = min(1.0, toxicity * 1.5) * (1.0 / (1.0 + queue_position * 0.1))
toxic_frac = self.toxic_fill_ratio
return AdverseSelectionCost(
expected_cost_bps=cost_bps,
pick_off_probability=pick_off_prob,
toxic_flow_fraction=toxic_frac,
queue_position_risk=1.0 / (1.0 + queue_position * 0.1),
)
def optimal_offset(
self,
spread_bps: float,
toxicity: float,
queue_depth: float = 1.0,
our_qty: float = 0.001,
recent_trade_rate: float = 0.5,
) -> int:
"""Compute optimal quote offset."""
return optimal_quote_offset(spread_bps, toxicity, queue_depth, our_qty, recent_trade_rate)
def record_fill(self, toxicity: float) -> None:
"""Record a fill for toxic fill ratio tracking."""
self._toxic_fills.append(toxicity)
self._total_fills += 1
@property
def toxic_fill_ratio(self) -> float:
if self._total_fills == 0:
return 0.0
return sum(1 for t in self._toxic_fills if t > 0.5) / self._total_fills
@property
def average_toxicity(self) -> float:
if not self._toxic_fills:
return 0.0
return sum(self._toxic_fills) / len(self._toxic_fills)