""" Diagnostic tests — WHY doesn't the system improve? Tests that verify: 1. Counterparty fills actually affect the book 2. System fills are affected by book state 3. Different strategies produce different scores 4. Evaluation has enough variance """ import random import pytest from malkhut.state import ( AccountState, ExecutionIntent, FulfilmentPolicyParams, IntentKind, MarketWorldState, Mode, OrderBookState, PriceLevel, Side, VenueRules, ) from malkhut.actions import ActionKind, AgentRole, CounterpartyAction, FulfilmentAction, OrderType from malkhut.cwm.core import MinimalCryptoLOBCWM from malkhut.counterparties import ToxicTakerPolicy, PassiveMakerPolicy, default_counterparty_ecology def _venue(): return VenueRules(exchange="bingx", symbol="BTCUSDT", tick_size=0.1, lot_size=0.001, min_qty=0.001, min_notional=5.0, maker_fee_bps=-0.2, taker_fee_bps=0.5, post_only_supported=True, reduce_only_supported=True, max_orders_per_second=100, max_cancels_per_minute=120) def _state(bid=50000.0, ask=50001.0, bid_qty=1.0, ask_qty=1.0): return MarketWorldState( ts_ns=1_000_000_000, mode=Mode.ENDOGENOUS_AGENT_SIM, venue=_venue(), book=OrderBookState(ts_ns=1, symbol="BTCUSDT", bids=(PriceLevel(bid, bid_qty),), asks=(PriceLevel(ask, ask_qty),)), account=AccountState(ts_ns=1, equity=10000.0, wallet_balance=10000.0, available_balance=10000.0, margin_used=0.0, total_notional=0.0), ) def _params(): return FulfilmentPolicyParams( version="test", ucb_c=1.414, max_sims=64, max_depth=2, rollout_depth=2, root_temperature=0.5, min_root_entropy=0.25, quote_offsets_ticks=(0, 1, 2), quote_size_fractions=(0.1, 0.25, 0.5), passive_ttl_ms=200, aggressive_ttl_ms=50, maker_edge_min_bps=0.5, cross_spread_edge_min_bps=5.0, adverse_toxicity_cancel_threshold=0.5, queue_churn_cancel_threshold=0.5, mae_tail_cut_bps=50.0, mfe_giveback_cut_fraction=0.5, max_time_in_loss_s=300.0, failed_recovery_cut_count=3, recovery_velocity_min_bps_per_s=0.0, max_symbol_notional_fraction=0.20, max_single_order_notional_fraction=0.05, reduce_when_global_up_fraction=0.30, session_profit_lock_fraction=0.02, w_expected_pnl=1.0, w_fill_probability=0.5, w_adverse_selection=2.0, w_queue_priority=0.5, w_inventory_risk=1.5, w_tail_loss=5.0, w_fee_quality=0.5, w_time_decay=0.3, w_policy_entropy=0.5, robust_tail_weight=2.0, toxic_counterparty_weight=3.0, low_liquidity_weight=2.0, latency_stress_weight=1.0, ) # ══════════════════════════════════════════════════════════════════════════════ # DIAGNOSTIC 1: Do counterparties affect the book? # ══════════════════════════════════════════════════════════════════════════════ class TestCounterpartyImpact: def test_cp_buy_reduces_asks(self): """Counterparty BUY should consume ask liquidity.""" cwm = MinimalCryptoLOBCWM() s = _state(ask_qty=0.5) cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.BUY, 0, 1.0, toxicity=0.8, ) r = cwm.transition(s, (_noop(), cp)) total_ask = sum(l.qty for l in r.book.asks) assert total_ask < 0.5, f"Expected ask reduction, got {total_ask}" def test_cp_sell_reduces_bids(self): """Counterparty SELL should consume bid liquidity.""" cwm = MinimalCryptoLOBCWM() s = _state(bid_qty=0.5) cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.SELL, 0, 1.0, toxicity=0.8, ) r = cwm.transition(s, (_noop(), cp)) total_bid = sum(l.qty for l in r.book.bids) assert total_bid < 0.5, f"Expected bid reduction, got {total_bid}" def test_cp_fill_changes_book_state(self): """Counterparty fill should change the book state.""" cwm = MinimalCryptoLOBCWM() s = _state(ask_qty=1.0) cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.BUY, 0, 1.0, toxicity=0.8, ) r1 = cwm.transition(s, (_noop(),)) r2 = cwm.transition(s, (_noop(), cp)) # Book should be different after counterparty fill assert r2.book.best_ask != r1.book.best_ask or sum(l.qty for l in r2.book.asks) != sum(l.qty for l in r1.book.asks) def test_cp_does_not_affect_our_position(self): """Counterparty fill should NOT change our position.""" cwm = MinimalCryptoLOBCWM() s = _state() cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.BUY, 0, 1.0, toxicity=0.8, ) r = cwm.transition(s, (_noop(), cp)) # Our position should be unchanged (no fill on our side) assert r.account.positions.get("BTCUSDT") is None or r.account.positions.get("BTCUSDT").qty == 0.0 # ══════════════════════════════════════════════════════════════════════════════ # DIAGNOSTIC 2: Do different strategies produce different scores? # ══════════════════════════════════════════════════════════════════════════════ class TestStrategyDifferentiation: def test_noop_vs_cross_different_pnl(self): """NOOP and CROSS_SPREAD should produce different PnL.""" cwm = MinimalCryptoLOBCWM() s = _state() r_noop = cwm.transition(s, (_noop(),)) r_cross = cwm.transition(s, (_cross(Side.BUY, 0.1),)) # Cross should produce different equity than noop assert r_noop.account.equity != r_cross.account.equity or \ r_noop.book.best_ask != r_cross.book.best_ask def test_aggressive_vs_passive_different_pnl(self): """Aggressive and passive strategies should produce different PnL.""" cwm = MinimalCryptoLOBCWM() s = _state() # Aggressive: cross spread r_agg = cwm.transition(s, (_cross(Side.BUY, 0.1),)) # Passive: place limit r_pas = cwm.transition(s, (_place(Side.BUY, offset=0, frac=0.1),)) # Should produce different book states assert r_agg.book.best_ask != r_pas.book.best_ask or \ r_agg.account.equity != r_pas.account.equity def test_toxic_vs_safe_different_pnl(self): """Toxic and safe strategies should produce different PnL.""" cwm = MinimalCryptoLOBCWM() s = _state(ask_qty=0.1) # thin book # Toxic: aggressive cross r_toxic = cwm.transition(s, (_cross(Side.BUY, 0.5),)) # Safe: small passive r_safe = cwm.transition(s, (_place(Side.BUY, offset=2, frac=0.01),)) # Toxic should have different equity than safe assert r_toxic.account.equity != r_safe.account.equity # ══════════════════════════════════════════════════════════════════════════════ # DIAGNOSTIC 3: Is the evaluation environment realistic? # ══════════════════════════════════════════════════════════════════════════════ class TestEvaluationRealism: def test_thin_book_consumed_quickly(self): """With thin book, counterparty should consume it quickly.""" cwm = MinimalCryptoLOBCWM() s = _state(bid_qty=0.1, ask_qty=0.1) cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.BUY, 0, 5.0, toxicity=0.9, ) # Run multiple steps state = s for _ in range(5): state = cwm.transition(state, (_noop(), cp)) # Book should be mostly consumed total_ask = sum(l.qty for l in state.book.asks) assert total_ask < 0.5, f"Expected book consumption, got {total_ask}" def test_thick_book_not_consumed(self): """With thick book, counterparty should not consume it all.""" cwm = MinimalCryptoLOBCWM() s = _state(bid_qty=10.0, ask_qty=10.0) cp = CounterpartyAction( AgentRole.TOXIC_TAKER, ActionKind.CROSS_SPREAD, Side.BUY, 0, 1.0, toxicity=0.9, ) state = cwm.transition(s, (_noop(), cp)) total_ask = sum(l.qty for l in state.book.asks) assert total_ask > 5.0, f"Expected thick book to survive, got {total_ask}" # ══════════════════════════════════════════════════════════════════════════════ # DIAGNOSTIC 4: Does the planner actually produce different actions? # ══════════════════════════════════════════════════════════════════════════════ class TestPlannerDifferentiation: def test_noop_produces_noop(self): """Without intent, planner should return NOOP.""" from malkhut.planner.sm_mcts import DecoupledUCBPlanner cwm = MinimalCryptoLOBCWM() planner = DecoupledUCBPlanner(cwm=cwm, counterparties=default_counterparty_ecology()) s = _state() result = planner.plan(s, _params(), budget_ms=10) assert result.selected_action.kind == ActionKind.NOOP def test_with_intent_produces_action(self): """With intent, planner should produce a non-NOOP action.""" from malkhut.planner.sm_mcts import DecoupledUCBPlanner from malkhut.state import ExecutionIntent, IntentKind cwm = MinimalCryptoLOBCWM() planner = DecoupledUCBPlanner(cwm=cwm, counterparties=default_counterparty_ecology()) intent = ExecutionIntent( intent_id="test", ts_ns=1, symbol="BTCUSDT", kind=IntentKind.ENTER_LONG, target_qty=0.01, max_notional=500.0, urgency=0.5, alpha_horizon_s=60.0, alpha_bps=2.0, max_slippage_bps=5.0, prefer_maker=True, reduce_only=False, ttl_s=300.0, reason="test", ) s = MarketWorldState( ts_ns=1, mode=Mode.REPLAY_NO_IMPACT, venue=_venue(), book=OrderBookState(ts_ns=1, symbol="BTCUSDT", bids=(PriceLevel(50000.0, 1.0),), asks=(PriceLevel(50001.0, 1.0),)), account=AccountState(ts_ns=1, equity=10000.0, wallet_balance=10000.0, available_balance=10000.0, margin_used=0.0, total_notional=0.0), intent=intent, ) result = planner.plan(s, _params(), budget_ms=10) # Should produce a non-NOOP action assert result.selected_action.kind != ActionKind.NOOP or len(result.actions) > 1 def _noop(): return FulfilmentAction(ActionKind.NOOP, None, None, 0, 0.0, 0) def _cross(side, frac): return FulfilmentAction(ActionKind.CROSS_SPREAD, side, OrderType.LIMIT, 0, frac, 50) def _place(side, offset=0, frac=0.1): return FulfilmentAction(ActionKind.PLACE, side, OrderType.LIMIT, offset, frac, 200)