512 lines
18 KiB
Python
512 lines
18 KiB
Python
|
|
"""
|
||
|
|
Strategy Generator — genetic programming for strategy evolution.
|
||
|
|
|
||
|
|
Based on genetic programming (GP) principles:
|
||
|
|
- Strategy = genome (parameter vector + strategy type)
|
||
|
|
- Crossover: combine two strategies to create offspring
|
||
|
|
- Mutation: randomly modify a strategy
|
||
|
|
- Selection: tournament selection based on self-play fitness
|
||
|
|
- Population: diverse pool of strategies, hardcoded baseline always available
|
||
|
|
|
||
|
|
Key design principle:
|
||
|
|
The hardcoded baseline is NEVER replaced. Generated strategies are ADDED
|
||
|
|
to the pool. The system GROWS its strategy repertoire.
|
||
|
|
|
||
|
|
Game theory insight:
|
||
|
|
During self-play, the system might "spontaneously generate" new strategies
|
||
|
|
via crossover/mutation of existing ones. These emergent strategies should
|
||
|
|
be captured, trialed, and added if successful.
|
||
|
|
"""
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import math
|
||
|
|
import random
|
||
|
|
import time
|
||
|
|
from dataclasses import dataclass, field
|
||
|
|
from enum import Enum
|
||
|
|
from typing import Any, Callable, List, Optional, Sequence, Tuple
|
||
|
|
|
||
|
|
from malkhut.state import FulfilmentPolicyParams
|
||
|
|
from malkhut.training.cma_trainer import (
|
||
|
|
CMAParameterCodec, EpisodeResult, PolicyEvaluator,
|
||
|
|
PolicySnapshot, Scenario, SelfPlayPool,
|
||
|
|
)
|
||
|
|
from malkhut.training.registry import PolicyRegistry, PolicyStage
|
||
|
|
from malkhut.counterparties import default_counterparty_ecology
|
||
|
|
|
||
|
|
|
||
|
|
# ==============================================================================
|
||
|
|
# Strategy Types — different planner algorithms
|
||
|
|
# ==============================================================================
|
||
|
|
|
||
|
|
class StrategyType(str, Enum):
|
||
|
|
"""Different strategy structures the system can use."""
|
||
|
|
SM_MCTS = "SM_MCTS" # Decoupled UCB/UCT (default)
|
||
|
|
UCB1 = "UCB1" # Standard UCB1 (simpler)
|
||
|
|
THOMPSON_SAMPLING = "THOMPSON" # Thompson sampling
|
||
|
|
GREEDY = "GREEDY" # Always pick best Q-value
|
||
|
|
RANDOM = "RANDOM" # Random action selection
|
||
|
|
HYBRID = "HYBRID" # Mix of multiple strategies
|
||
|
|
|
||
|
|
|
||
|
|
# ==============================================================================
|
||
|
|
# Strategy Genome
|
||
|
|
# ==============================================================================
|
||
|
|
|
||
|
|
@dataclass(frozen=True, slots=True)
|
||
|
|
class StrategyGenome:
|
||
|
|
"""
|
||
|
|
A strategy encoded as a genome for genetic operations.
|
||
|
|
|
||
|
|
The genome has two parts:
|
||
|
|
1. Structural: strategy type, action menu config
|
||
|
|
2. Parametric: the 29 tunable parameters
|
||
|
|
|
||
|
|
Genetic operations (crossover, mutation) work on this genome.
|
||
|
|
"""
|
||
|
|
strategy_type: StrategyType
|
||
|
|
params: FulfilmentPolicyParams
|
||
|
|
generation: int = 0
|
||
|
|
parent_ids: Tuple[str, ...] = ()
|
||
|
|
fitness: float = 0.0
|
||
|
|
episodes_tested: int = 0
|
||
|
|
creation_ts_ns: int = 0
|
||
|
|
|
||
|
|
@property
|
||
|
|
def genome_id(self) -> str:
|
||
|
|
"""Unique identifier for this genome."""
|
||
|
|
return f"{self.strategy_type.value}_{self.params.version}_{self.generation}"
|
||
|
|
|
||
|
|
|
||
|
|
# ==============================================================================
|
||
|
|
# Genetic Operators
|
||
|
|
# ==============================================================================
|
||
|
|
|
||
|
|
class GeneticOperators:
|
||
|
|
"""
|
||
|
|
Genetic operators for strategy evolution.
|
||
|
|
|
||
|
|
Crossover: combine two strategies to create offspring
|
||
|
|
Mutation: randomly modify a strategy
|
||
|
|
Selection: tournament selection based on fitness
|
||
|
|
"""
|
||
|
|
|
||
|
|
def __init__(self, codec: CMAParameterCodec, mutation_rate: float = 0.15,
|
||
|
|
crossover_rate: float = 0.7) -> None:
|
||
|
|
self.codec = codec
|
||
|
|
self.mutation_rate = mutation_rate
|
||
|
|
self.crossover_rate = crossover_rate
|
||
|
|
|
||
|
|
def crossover(
|
||
|
|
self,
|
||
|
|
parent1: StrategyGenome,
|
||
|
|
parent2: StrategyGenome,
|
||
|
|
rng: random.Random,
|
||
|
|
) -> StrategyGenome:
|
||
|
|
"""
|
||
|
|
Uniform crossover: for each parameter, randomly pick from parent1 or parent2.
|
||
|
|
|
||
|
|
Strategy type is inherited from the fitter parent.
|
||
|
|
"""
|
||
|
|
# Strategy type from fitter parent
|
||
|
|
strategy_type = parent1.strategy_type if parent1.fitness >= parent2.fitness else parent2.strategy_type
|
||
|
|
|
||
|
|
# Crossover parameters
|
||
|
|
p1_vec = self.codec.initial_vector(parent1.params)
|
||
|
|
p2_vec = self.codec.initial_vector(parent2.params)
|
||
|
|
child_vec = []
|
||
|
|
|
||
|
|
for i in range(len(p1_vec)):
|
||
|
|
if rng.random() < 0.5:
|
||
|
|
child_vec.append(p1_vec[i])
|
||
|
|
else:
|
||
|
|
child_vec.append(p2_vec[i])
|
||
|
|
|
||
|
|
# Decode child
|
||
|
|
child_params = self.codec.decode(child_vec, version=f"child_{int(time.time_ns())}")
|
||
|
|
|
||
|
|
return StrategyGenome(
|
||
|
|
strategy_type=strategy_type,
|
||
|
|
params=child_params,
|
||
|
|
generation=max(parent1.generation, parent2.generation) + 1,
|
||
|
|
parent_ids=(parent1.genome_id, parent2.genome_id),
|
||
|
|
creation_ts_ns=time.time_ns(),
|
||
|
|
)
|
||
|
|
|
||
|
|
def mutate(
|
||
|
|
self,
|
||
|
|
genome: StrategyGenome,
|
||
|
|
rng: random.Random,
|
||
|
|
) -> StrategyGenome:
|
||
|
|
"""
|
||
|
|
Gaussian mutation: add noise to each parameter with mutation_rate probability.
|
||
|
|
|
||
|
|
Occasionally mutate strategy type (structural mutation).
|
||
|
|
"""
|
||
|
|
vec = self.codec.initial_vector(genome.params)
|
||
|
|
lows, highs = self.codec.bounds()
|
||
|
|
|
||
|
|
mutated_vec = []
|
||
|
|
for i, (v, lo, hi) in enumerate(zip(vec, lows, highs)):
|
||
|
|
if rng.random() < self.mutation_rate:
|
||
|
|
# Gaussian noise scaled by parameter range
|
||
|
|
range_val = hi - lo
|
||
|
|
noise = rng.gauss(0, range_val * 0.1)
|
||
|
|
mutated_vec.append(max(lo, min(hi, v + noise)))
|
||
|
|
else:
|
||
|
|
mutated_vec.append(v)
|
||
|
|
|
||
|
|
# Decode mutated params
|
||
|
|
mutated_params = self.codec.decode(mutated_vec, version=f"mut_{int(time.time_ns())}")
|
||
|
|
|
||
|
|
# Occasionally mutate strategy type (5% chance)
|
||
|
|
strategy_type = genome.strategy_type
|
||
|
|
if rng.random() < 0.05:
|
||
|
|
strategy_type = rng.choice(list(StrategyType))
|
||
|
|
|
||
|
|
return StrategyGenome(
|
||
|
|
strategy_type=strategy_type,
|
||
|
|
params=mutated_params,
|
||
|
|
generation=genome.generation + 1,
|
||
|
|
parent_ids=(genome.genome_id,),
|
||
|
|
creation_ts_ns=time.time_ns(),
|
||
|
|
)
|
||
|
|
|
||
|
|
def tournament_select(
|
||
|
|
self,
|
||
|
|
population: List[StrategyGenome],
|
||
|
|
tournament_size: int = 3,
|
||
|
|
rng: random.Random = None,
|
||
|
|
) -> StrategyGenome:
|
||
|
|
"""Tournament selection: pick tournament_size random, return the best."""
|
||
|
|
if rng is None:
|
||
|
|
rng = random.Random()
|
||
|
|
tournament = rng.sample(population, min(tournament_size, len(population)))
|
||
|
|
return max(tournament, key=lambda g: g.fitness)
|
||
|
|
|
||
|
|
def random_genome(
|
||
|
|
self,
|
||
|
|
strategy_type: Optional[StrategyType] = None,
|
||
|
|
rng: random.Random = None,
|
||
|
|
) -> StrategyGenome:
|
||
|
|
"""Generate a random genome for initial population."""
|
||
|
|
if rng is None:
|
||
|
|
rng = random.Random()
|
||
|
|
|
||
|
|
if strategy_type is None:
|
||
|
|
strategy_type = rng.choice(list(StrategyType))
|
||
|
|
|
||
|
|
# Random parameters within bounds
|
||
|
|
lows, highs = self.codec.bounds()
|
||
|
|
random_vec = [rng.uniform(lo, hi) for lo, hi in zip(lows, highs)]
|
||
|
|
params = self.codec.decode(random_vec, version=f"rand_{int(time.time_ns())}")
|
||
|
|
|
||
|
|
return StrategyGenome(
|
||
|
|
strategy_type=strategy_type,
|
||
|
|
params=params,
|
||
|
|
generation=0,
|
||
|
|
creation_ts_ns=time.time_ns(),
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
# ==============================================================================
|
||
|
|
# Strategy Evaluator
|
||
|
|
# ==============================================================================
|
||
|
|
|
||
|
|
class StrategyEvaluator:
|
||
|
|
"""
|
||
|
|
Evaluates strategies through self-play episodes.
|
||
|
|
|
||
|
|
Each strategy is tested against the current self-play pool.
|
||
|
|
Fitness = robust score across scenarios.
|
||
|
|
|
||
|
|
CRITICAL: uses the genome's strategy_type to create the correct planner.
|
||
|
|
This is how different planners (EXP3, Regret Matching, etc.) are actually
|
||
|
|
used during self-play discovery.
|
||
|
|
"""
|
||
|
|
|
||
|
|
def __init__(self, evaluator: PolicyEvaluator) -> None:
|
||
|
|
self.evaluator = evaluator
|
||
|
|
|
||
|
|
def evaluate(
|
||
|
|
self,
|
||
|
|
genome: StrategyGenome,
|
||
|
|
scenarios: Sequence[Scenario],
|
||
|
|
pool: SelfPlayPool,
|
||
|
|
rng_seed: int = 42,
|
||
|
|
) -> float:
|
||
|
|
"""Evaluate a genome's fitness through self-play."""
|
||
|
|
score, results = self.evaluator.evaluate_candidate(
|
||
|
|
params=genome.params,
|
||
|
|
scenarios=scenarios,
|
||
|
|
rng_seed=rng_seed,
|
||
|
|
planner_type=genome.strategy_type.value, # PASS STRATEGY TYPE
|
||
|
|
)
|
||
|
|
return score
|
||
|
|
|
||
|
|
def evaluate_population(
|
||
|
|
self,
|
||
|
|
population: List[StrategyGenome],
|
||
|
|
scenarios: Sequence[Scenario],
|
||
|
|
pool: SelfPlayPool,
|
||
|
|
rng_seed: int = 42,
|
||
|
|
) -> List[StrategyGenome]:
|
||
|
|
"""Evaluate entire population and update fitness scores."""
|
||
|
|
evaluated = []
|
||
|
|
for i, genome in enumerate(population):
|
||
|
|
fitness = self.evaluate(genome, scenarios, pool, rng_seed + i)
|
||
|
|
evaluated.append(StrategyGenome(
|
||
|
|
strategy_type=genome.strategy_type,
|
||
|
|
params=genome.params,
|
||
|
|
generation=genome.generation,
|
||
|
|
parent_ids=genome.parent_ids,
|
||
|
|
fitness=fitness,
|
||
|
|
episodes_tested=len(scenarios),
|
||
|
|
creation_ts_ns=genome.creation_ts_ns,
|
||
|
|
))
|
||
|
|
return evaluated
|
||
|
|
|
||
|
|
|
||
|
|
# ==============================================================================
|
||
|
|
# Strategy Generator — the main loop
|
||
|
|
# ==============================================================================
|
||
|
|
|
||
|
|
@dataclass(frozen=True, slots=True)
|
||
|
|
class GeneratorConfig:
|
||
|
|
"""Configuration for strategy generation."""
|
||
|
|
population_size: int = 20
|
||
|
|
generations: int = 5
|
||
|
|
tournament_size: int = 3
|
||
|
|
elitism_count: int = 2 # keep top N unchanged
|
||
|
|
mutation_rate: float = 0.15
|
||
|
|
crossover_rate: float = 0.7
|
||
|
|
max_strategies: int = 50 # max strategies in pool
|
||
|
|
min_fitness_threshold: float = -100.0 # minimum fitness to keep
|
||
|
|
|
||
|
|
|
||
|
|
class StrategyGenerator:
|
||
|
|
"""
|
||
|
|
Genetic programming for strategy evolution.
|
||
|
|
|
||
|
|
Key design:
|
||
|
|
- Hardcoded baseline is NEVER replaced
|
||
|
|
- Generated strategies are ADDED to the pool
|
||
|
|
- System GROWS its strategy repertoire
|
||
|
|
- During self-play, crossover/mutation may "spontaneously generate"
|
||
|
|
new strategies that weren't explicitly programmed
|
||
|
|
|
||
|
|
Flow:
|
||
|
|
1. Initialize population (random + baseline)
|
||
|
|
2. Evaluate fitness (self-play episodes)
|
||
|
|
3. Select parents (tournament selection)
|
||
|
|
4. Create offspring (crossover + mutation)
|
||
|
|
5. Evaluate offspring
|
||
|
|
6. Replace weakest with offspring
|
||
|
|
7. Repeat for N generations
|
||
|
|
8. Add successful strategies to pool
|
||
|
|
"""
|
||
|
|
|
||
|
|
def __init__(
|
||
|
|
self,
|
||
|
|
config: Optional[GeneratorConfig] = None,
|
||
|
|
registry: Optional[PolicyRegistry] = None,
|
||
|
|
pool: Optional[SelfPlayPool] = None,
|
||
|
|
) -> None:
|
||
|
|
self.config = config or GeneratorConfig()
|
||
|
|
self._registry = registry or PolicyRegistry()
|
||
|
|
self._pool = pool or SelfPlayPool(max_size=self.config.max_strategies)
|
||
|
|
|
||
|
|
self._codec = CMAParameterCodec()
|
||
|
|
self._operators = GeneticOperators(
|
||
|
|
codec=self._codec,
|
||
|
|
mutation_rate=self.config.mutation_rate,
|
||
|
|
crossover_rate=self.config.crossover_rate,
|
||
|
|
)
|
||
|
|
self._evaluator = StrategyEvaluator(
|
||
|
|
PolicyEvaluator(
|
||
|
|
cwm_factory=lambda: __import__("malkhut.cwm.core", fromlist=["MinimalCryptoLOBCWM"]).MinimalCryptoLOBCWM(),
|
||
|
|
counterparties=default_counterparty_ecology(),
|
||
|
|
)
|
||
|
|
)
|
||
|
|
|
||
|
|
self._population: List[StrategyGenome] = []
|
||
|
|
self._history: List[StrategyGenome] = []
|
||
|
|
self._rng = random.Random(42)
|
||
|
|
|
||
|
|
def initialize_population(
|
||
|
|
self,
|
||
|
|
baseline: FulfilmentPolicyParams,
|
||
|
|
scenarios: Sequence[Scenario],
|
||
|
|
) -> None:
|
||
|
|
"""Initialize population with baseline + random variants."""
|
||
|
|
self._population = []
|
||
|
|
|
||
|
|
# Add hardcoded baseline (always available)
|
||
|
|
baseline_genome = StrategyGenome(
|
||
|
|
strategy_type=StrategyType.SM_MCTS,
|
||
|
|
params=baseline,
|
||
|
|
generation=0,
|
||
|
|
creation_ts_ns=time.time_ns(),
|
||
|
|
fitness=0.0,
|
||
|
|
)
|
||
|
|
self._population.append(baseline_genome)
|
||
|
|
|
||
|
|
# Add random variants
|
||
|
|
for i in range(self.config.population_size - 1):
|
||
|
|
genome = self._operators.random_genome(rng=self._rng)
|
||
|
|
self._population.append(genome)
|
||
|
|
|
||
|
|
# Evaluate initial population
|
||
|
|
self._population = self._evaluator.evaluate_population(
|
||
|
|
self._population, scenarios, self._pool,
|
||
|
|
)
|
||
|
|
|
||
|
|
def evolve(
|
||
|
|
self,
|
||
|
|
baseline: FulfilmentPolicyParams,
|
||
|
|
scenarios: Sequence[Scenario],
|
||
|
|
) -> List[StrategyGenome]:
|
||
|
|
"""
|
||
|
|
Run genetic evolution for N generations.
|
||
|
|
|
||
|
|
Returns the final population (sorted by fitness).
|
||
|
|
"""
|
||
|
|
# Initialize if empty
|
||
|
|
if not self._population:
|
||
|
|
self.initialize_population(baseline, scenarios)
|
||
|
|
|
||
|
|
for gen in range(self.config.generations):
|
||
|
|
# 1. Select parents
|
||
|
|
parents = []
|
||
|
|
for _ in range(self.config.population_size - self.config.elitism_count):
|
||
|
|
p1 = self._operators.tournament_select(
|
||
|
|
self._population, self.config.tournament_size, self._rng,
|
||
|
|
)
|
||
|
|
p2 = self._operators.tournament_select(
|
||
|
|
self._population, self.config.tournament_size, self._rng,
|
||
|
|
)
|
||
|
|
parents.append((p1, p2))
|
||
|
|
|
||
|
|
# 2. Create offspring
|
||
|
|
offspring = []
|
||
|
|
for p1, p2 in parents:
|
||
|
|
if self._rng.random() < self.config.crossover_rate:
|
||
|
|
child = self._operators.crossover(p1, p2, self._rng)
|
||
|
|
else:
|
||
|
|
child = self._operators.mutate(p1, self._rng)
|
||
|
|
offspring.append(child)
|
||
|
|
|
||
|
|
# 3. Evaluate offspring
|
||
|
|
offspring = self._evaluator.evaluate_population(
|
||
|
|
offspring, scenarios, self._pool,
|
||
|
|
)
|
||
|
|
|
||
|
|
# 4. Elitism: keep top N unchanged, PLUS always keep baseline
|
||
|
|
self._population.sort(key=lambda g: g.fitness, reverse=True)
|
||
|
|
elites = self._population[:self.config.elitism_count]
|
||
|
|
|
||
|
|
# Ensure baseline (SM_MCTS with version "baseline") is always present
|
||
|
|
has_baseline = any(
|
||
|
|
g.strategy_type == StrategyType.SM_MCTS and g.params.version == "baseline"
|
||
|
|
for g in elites
|
||
|
|
)
|
||
|
|
if not has_baseline:
|
||
|
|
baseline = next(
|
||
|
|
(g for g in self._population
|
||
|
|
if g.strategy_type == StrategyType.SM_MCTS and g.params.version == "baseline"),
|
||
|
|
None,
|
||
|
|
)
|
||
|
|
if baseline:
|
||
|
|
elites.append(baseline)
|
||
|
|
|
||
|
|
# 5. Replace weakest with offspring
|
||
|
|
self._population = elites + offspring[:self.config.population_size - self.config.elitism_count]
|
||
|
|
|
||
|
|
# 6. Sort by fitness
|
||
|
|
self._population.sort(key=lambda g: g.fitness, reverse=True)
|
||
|
|
|
||
|
|
# 7. Track history
|
||
|
|
self._history.extend(offspring)
|
||
|
|
|
||
|
|
return self._population
|
||
|
|
|
||
|
|
def get_successful_strategies(
|
||
|
|
self,
|
||
|
|
min_fitness: Optional[float] = None,
|
||
|
|
) -> List[StrategyGenome]:
|
||
|
|
"""
|
||
|
|
Get strategies that meet the fitness threshold.
|
||
|
|
|
||
|
|
These are candidates for adding to the pool.
|
||
|
|
"""
|
||
|
|
threshold = min_fitness or self.config.min_fitness_threshold
|
||
|
|
return [g for g in self._population if g.fitness > threshold]
|
||
|
|
|
||
|
|
def add_to_pool(self, genome: StrategyGenome) -> None:
|
||
|
|
"""Add a successful strategy to the self-play pool."""
|
||
|
|
snapshot = PolicySnapshot(
|
||
|
|
params=genome.params,
|
||
|
|
score=genome.fitness,
|
||
|
|
created_ts_ns=genome.creation_ts_ns,
|
||
|
|
evaluation_summary={
|
||
|
|
"strategy_type": genome.strategy_type.value,
|
||
|
|
"generation": genome.generation,
|
||
|
|
"episodes_tested": genome.episodes_tested,
|
||
|
|
},
|
||
|
|
)
|
||
|
|
self._pool.maybe_add(snapshot)
|
||
|
|
|
||
|
|
# Also register in registry
|
||
|
|
self._registry.register_candidate(
|
||
|
|
genome.params, genome.fitness,
|
||
|
|
{"strategy_type": genome.strategy_type.value},
|
||
|
|
)
|
||
|
|
|
||
|
|
def get_diverse_strategies(
|
||
|
|
self,
|
||
|
|
n: int = 5,
|
||
|
|
) -> List[StrategyGenome]:
|
||
|
|
"""
|
||
|
|
Get N diverse strategies from the population.
|
||
|
|
|
||
|
|
Diversity is measured by:
|
||
|
|
- Different strategy types
|
||
|
|
- Different parameter vectors (cosine distance)
|
||
|
|
"""
|
||
|
|
if len(self._population) <= n:
|
||
|
|
return list(self._population)
|
||
|
|
|
||
|
|
# Group by strategy type
|
||
|
|
by_type: dict[StrategyType, list[StrategyGenome]] = {}
|
||
|
|
for g in self._population:
|
||
|
|
by_type.setdefault(g.strategy_type, []).append(g)
|
||
|
|
|
||
|
|
# Pick one from each type, then fill with best remaining
|
||
|
|
selected = []
|
||
|
|
for stype in StrategyType:
|
||
|
|
if stype in by_type and len(selected) < n:
|
||
|
|
best = max(by_type[stype], key=lambda g: g.fitness)
|
||
|
|
selected.append(best)
|
||
|
|
|
||
|
|
# Fill with best remaining
|
||
|
|
remaining = [g for g in self._population if g not in selected]
|
||
|
|
remaining.sort(key=lambda g: g.fitness, reverse=True)
|
||
|
|
while len(selected) < n and remaining:
|
||
|
|
selected.append(remaining.pop(0))
|
||
|
|
|
||
|
|
return selected
|
||
|
|
|
||
|
|
@property
|
||
|
|
def population_size(self) -> int:
|
||
|
|
return len(self._population)
|
||
|
|
|
||
|
|
@property
|
||
|
|
def best_fitness(self) -> float:
|
||
|
|
if not self._population:
|
||
|
|
return -float("inf")
|
||
|
|
return max(g.fitness for g in self._population)
|
||
|
|
|
||
|
|
@property
|
||
|
|
def population(self) -> List[StrategyGenome]:
|
||
|
|
return list(self._population)
|