From 1b5b2997504f7c0d50982308948eb6a309025595 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Tue, 3 Mar 2026 17:49:37 +0000 Subject: [PATCH 001/115] refactor the input dfs into a dynamic input object for greater container flexibility and less piping changes for pools requiring reserve inputs --- quantammsim/core_simulator/dynamic_inputs.py | 112 +++++++++ quantammsim/core_simulator/forward_pass.py | 102 ++++---- quantammsim/hooks/dynamic_fee_base_hook.py | 19 +- quantammsim/pools/ECLP/gyroscope.py | 11 +- quantammsim/pools/FM_AMM/cow_pool.py | 18 +- quantammsim/pools/G3M/balancer/balancer.py | 22 +- .../pools/G3M/quantamm/TFMM_base_pool.py | 12 +- quantammsim/pools/base_pool.py | 6 +- quantammsim/pools/hodl_pool.py | 12 +- quantammsim/pools/reCLAMM/reclamm.py | 18 +- quantammsim/runners/__init__.py | 4 +- quantammsim/runners/jax_runner_utils.py | 127 ++++++---- quantammsim/runners/jax_runners.py | 127 ++++------ quantammsim/runners/multi_period_sgd.py | 2 + quantammsim/runners/training_evaluator.py | 1 + .../finance/param_financial_calculator.py | 15 +- scripts/demo_run_chunks_from_chain_data.py | 11 +- scripts/demo_run_from_chain_data.py | 13 +- scripts/reclamm/sim_vs_world_comparison.py | 35 ++- tests/integration/test_dynamic_gas_fees.py | 31 ++- .../pools/reCLAMM/test_reclamm_fee_revenue.py | 18 +- tests/scripts/dynamic_gas_test.py | 20 +- tests/unit/test_jax_runner_utils.py | 236 ++++++++++++++++++ tests/unit/test_jax_runners_comprehensive.py | 217 ++++++++++++++++ tests/unit/test_lint_bugs.py | 9 +- 25 files changed, 908 insertions(+), 290 deletions(-) create mode 100644 quantammsim/core_simulator/dynamic_inputs.py diff --git a/quantammsim/core_simulator/dynamic_inputs.py b/quantammsim/core_simulator/dynamic_inputs.py new file mode 100644 index 00000000..b1eedacf --- /dev/null +++ b/quantammsim/core_simulator/dynamic_inputs.py @@ -0,0 +1,112 @@ +from dataclasses import dataclass +from typing import Any, NamedTuple, Optional + +import jax.numpy as jnp + + +@dataclass(frozen=True) +class DynamicInputFrames: + """Outer-layer container for optional pandas-backed dynamic inputs.""" + + trades: Optional[Any] = None + fees: Optional[Any] = None + gas_cost: Optional[Any] = None + arb_fees: Optional[Any] = None + lp_supply: Optional[Any] = None + + +class DynamicInputArrays(NamedTuple): + """Fixed-structure JAX pytree for dynamic simulation inputs.""" + + trades: jnp.ndarray + fees: jnp.ndarray + gas_cost: jnp.ndarray + arb_fees: jnp.ndarray + lp_supply: jnp.ndarray + + +def default_dynamic_input_flags() -> dict: + """Static dispatch flags for forward-pass path selection.""" + return { + "use_dynamic_inputs": False, + "has_trades": False, + "has_dynamic_fees": False, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + } + + +def dynamic_input_flags_from_frames(dynamic_input_frames: Optional[DynamicInputFrames]) -> dict: + """Build stable dispatch flags from the outer-layer frame container.""" + if dynamic_input_frames is None: + return default_dynamic_input_flags() + + flags = { + "use_dynamic_inputs": False, + "has_trades": dynamic_input_frames.trades is not None, + "has_dynamic_fees": dynamic_input_frames.fees is not None, + "has_dynamic_gas_cost": dynamic_input_frames.gas_cost is not None, + "has_dynamic_arb_fees": dynamic_input_frames.arb_fees is not None, + "has_lp_supply": dynamic_input_frames.lp_supply is not None, + } + flags["use_dynamic_inputs"] = any(flags.values()) + return flags + + +def resolve_dynamic_input_flags( + dynamic_inputs: Optional[DynamicInputArrays], + dynamic_input_flags: Optional[dict] = None, +) -> dict: + """Return a safe dispatch flag set for the provided hot-path bundle.""" + flags = ( + default_dynamic_input_flags() + if dynamic_input_flags is None + else dict(dynamic_input_flags) + ) + if dynamic_inputs is not None: + flags["use_dynamic_inputs"] = True + return flags + + +def empty_dynamic_input_arrays() -> DynamicInputArrays: + """Create a canonical empty bundle with stable pytree structure.""" + return DynamicInputArrays( + trades=jnp.zeros((1, 3), dtype=jnp.float64), + fees=jnp.zeros((1,), dtype=jnp.float64), + gas_cost=jnp.zeros((1,), dtype=jnp.float64), + arb_fees=jnp.zeros((1,), dtype=jnp.float64), + lp_supply=jnp.ones((1,), dtype=jnp.float64), + ) + + +def resolve_dynamic_input_components( + dynamic_inputs: Optional[DynamicInputArrays], + dynamic_input_flags: dict, + static_dict: dict, +) -> dict: + """Resolve dynamic-input leaves against static scalar defaults.""" + arrays = empty_dynamic_input_arrays() if dynamic_inputs is None else dynamic_inputs + return { + "trades": arrays.trades if dynamic_input_flags["has_trades"] else None, + "fees": ( + arrays.fees + if dynamic_input_flags["has_dynamic_fees"] + else jnp.asarray([static_dict["fees"]], dtype=jnp.float64) + ), + "gas_cost": ( + arrays.gas_cost + if dynamic_input_flags["has_dynamic_gas_cost"] + else jnp.asarray([static_dict["gas_cost"]], dtype=jnp.float64) + ), + "arb_fees": ( + arrays.arb_fees + if dynamic_input_flags["has_dynamic_arb_fees"] + else jnp.asarray([static_dict["arb_fees"]], dtype=jnp.float64) + ), + "lp_supply": ( + arrays.lp_supply + if dynamic_input_flags["has_lp_supply"] + else jnp.ones((1,), dtype=jnp.float64) + ), + } diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index 205feed2..4dd74f62 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -54,11 +54,28 @@ import numpy as np from functools import partial +from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputArrays, + default_dynamic_input_flags, + empty_dynamic_input_arrays, + resolve_dynamic_input_flags, +) np.seterr(all="raise") np.seterr(under="print") +def _resolve_dynamic_inputs(dynamic_inputs, static_dict): + """Return a stable hot-path bundle plus static dispatch flags.""" + dynamic_input_flags = resolve_dynamic_input_flags( + dynamic_inputs, + static_dict.get("dynamic_input_flags"), + ) + if dynamic_inputs is None: + dynamic_inputs = empty_dynamic_input_arrays() + return dynamic_inputs, dynamic_input_flags + + def _apply_price_noise(prices, sigma, seed_int): """Apply multiplicative log-normal noise to prices. @@ -703,15 +720,12 @@ def _calculate_return_value( return return_metrics[return_val]() -@partial(jit, static_argnums=(7, 8)) +@partial(jit, static_argnums=(4, 5)) def forward_pass( params, start_index, prices, - trades_array=None, - fees_array=None, - gas_cost_array=None, - arb_fees_array=None, + dynamic_inputs=None, pool=None, static_dict=None, ): @@ -734,17 +748,8 @@ def forward_pass( prices : array-like A 2D array of market prices for the assets involved in the simulation. - trades_array : array-like, optional - An array of trades to be considered in the simulation. Defaults to None. - - fees_array : array-like, optional - An array of fees to be applied during the simulation. Defaults to None. - - gas_cost_array : array-like, optional - An array of gas costs to be considered in the simulation. Defaults to None. - - arb_fees_array : array-like, optional - An array of arbitrage fees to be applied during the simulation. Defaults to None. + dynamic_inputs : DynamicInputArrays, optional + Fixed-structure bundle of dynamic trades/fees/gas/arb/LP arrays. pool : object An instance of a pool object that provides methods @@ -794,8 +799,8 @@ def forward_pass( - The function handles different cases for fees and trades, adjusting the calculation method accordingly: - 1. If any of `fees_array`, `gas_cost_array`, `arb_fees_array`, - or `trades_array` is provided, it uses `pool.calculate_reserves_with_dynamic_inputs`. + 1. If any dynamic-input flags are enabled, it uses + `pool.calculate_reserves_with_dynamic_inputs`. 2. If any of `fees`, `gas_cost`, or `arb_fees` in `static_dict` is a nonzero scalar value, it uses `pool.calculate_reserves_with_fees`. @@ -836,6 +841,7 @@ def forward_pass( "training_data_kind": "historic", "arb_frequency": 1, "do_trades": False, + "dynamic_input_flags": default_dynamic_input_flags(), } # 'pool' has default of None only to handle how partial function @@ -864,28 +870,20 @@ def forward_pass( # 1. Any of Fees, gas costs, and arb fees are provided as arrays, or trades are provided # 2. Any of Fees, gas costs, and arb fees are nonzero scalar values, with no trades provided # 3. Fees, gas costs, and arb fees are all zero, with no trades provided + dynamic_inputs, dynamic_input_flags = _resolve_dynamic_inputs( + dynamic_inputs, static_dict + ) + fee_revenue = None - if any( - ele is not None - for ele in [fees_array, gas_cost_array, arb_fees_array, trades_array] - ): - # Case 1, at least one of fees, gas costs, or arb fees is not None - if fees_array is None: - fees_array = jnp.array([static_dict["fees"]]) - if gas_cost_array is None: - gas_cost_array = jnp.array([static_dict["gas_cost"]]) - if arb_fees_array is None: - arb_fees_array = jnp.array([static_dict["arb_fees"]]) + if dynamic_input_flags["use_dynamic_inputs"]: + # Case 1, at least one dynamic input is enabled if hasattr(pool, "calculate_reserves_and_fee_revenue_with_dynamic_inputs"): reserves, fee_revenue = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( params, static_dict, prices, start_index, - fees_array=fees_array, - arb_thresh_array=gas_cost_array, - arb_fees_array=arb_fees_array, - trade_array=trades_array, + dynamic_inputs=dynamic_inputs, ) else: reserves = pool.calculate_reserves_with_dynamic_inputs( @@ -893,10 +891,7 @@ def forward_pass( static_dict, prices, start_index, - fees_array=fees_array, - arb_thresh_array=gas_cost_array, - arb_fees_array=arb_fees_array, - trade_array=trades_array, + dynamic_inputs=dynamic_inputs, ) elif True in ( ele > 0.0 @@ -1003,15 +998,12 @@ def forward_pass( return base_metric -@partial(jit, static_argnums=(7, 8)) +@partial(jit, static_argnums=(4, 5)) def forward_pass_nograd( params, start_index, prices, - trades_array=None, - fees_array=None, - gas_cost_array=None, - arb_fees_array=None, + dynamic_inputs=None, pool=None, static_dict=None, ): @@ -1036,17 +1028,8 @@ def forward_pass_nograd( prices : array-like A 2D array of market prices for the assets involved in the simulation. - trades_array : array-like, optional - An array of trades to be considered in the simulation. Defaults to None. - - fees_array : array-like, optional - An array of fees to be applied during the simulation. Defaults to None. - - gas_cost_array : array-like, optional - An array of gas costs to be considered in the simulation. Defaults to None. - - arb_fees_array : array-like, optional - An array of arbitrage fees to be applied during the simulation. Defaults to None. + dynamic_inputs : DynamicInputArrays, optional + Fixed-structure bundle of dynamic trades/fees/gas/arb/LP arrays. pool : object An instance of a pool object that provides methods @@ -1096,8 +1079,8 @@ def forward_pass_nograd( - The function handles different cases for fees and trades, adjusting the calculation method accordingly: - 1. If any of `fees_array`, `gas_cost_array`, `arb_fees_array`, - or `trades_array` is provided, it uses `pool.calculate_reserves_with_dynamic_inputs`. + 1. If any dynamic-input flags are enabled, it uses + `pool.calculate_reserves_with_dynamic_inputs`. 2. If any of `fees`, `gas_cost`, or `arb_fees` in `static_dict` is a nonzero scalar value, it uses `pool.calculate_reserves_with_fees`. @@ -1122,14 +1105,15 @@ def forward_pass_nograd( params = {k: stop_gradient(v) for k, v in params.items()} start_index = stop_gradient(start_index) prices = stop_gradient(prices) + if dynamic_inputs is not None: + dynamic_inputs = DynamicInputArrays( + *(stop_gradient(arr) for arr in dynamic_inputs) + ) return forward_pass( params, start_index, prices, - trades_array, - fees_array, - gas_cost_array, - arb_fees_array, + dynamic_inputs, pool, static_dict, ) diff --git a/quantammsim/hooks/dynamic_fee_base_hook.py b/quantammsim/hooks/dynamic_fee_base_hook.py index 40a97140..9e00af64 100644 --- a/quantammsim/hooks/dynamic_fee_base_hook.py +++ b/quantammsim/hooks/dynamic_fee_base_hook.py @@ -3,6 +3,10 @@ import jax.numpy as jnp from jax.lax import dynamic_slice +from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputArrays, + empty_dynamic_input_arrays, +) class BaseDynamicFeeHook(ABC): """Mixin class to add dynamic fee calculation capabilities to pools. @@ -113,16 +117,21 @@ def calculate_reserves_with_fees( (int((bout_length) / chunk_period), 1), ) dynamic_fees = raw_dynamic_fees.repeat(chunk_period, axis=0).squeeze() - # Use existing dynamic inputs infrastructure + empty_inputs = empty_dynamic_input_arrays() + dynamic_inputs = DynamicInputArrays( + trades=empty_inputs.trades, + fees=dynamic_fees, + gas_cost=jnp.asarray(run_fingerprint["gas_cost"], dtype=jnp.float64), + arb_fees=jnp.asarray(run_fingerprint["arb_fees"], dtype=jnp.float64), + lp_supply=empty_inputs.lp_supply, + ) + return self.calculate_reserves_with_dynamic_inputs( params, run_fingerprint, prices, start_index, - dynamic_fees, - run_fingerprint["gas_cost"], - run_fingerprint["arb_fees"], - dynamic_fees, + dynamic_inputs, additional_oracle_input, ) diff --git a/quantammsim/pools/ECLP/gyroscope.py b/quantammsim/pools/ECLP/gyroscope.py index ad7884a9..5aebcef5 100644 --- a/quantammsim/pools/ECLP/gyroscope.py +++ b/quantammsim/pools/ECLP/gyroscope.py @@ -322,11 +322,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: # Gyroscope ECLP pools are only defined for 2 assets @@ -346,6 +342,11 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees + trade_array = dynamic_inputs.trades + # calculate initial reserves initial_pool_value = run_fingerprint["initial_pool_value"] initial_reserves = initialise_gyroscope_reserves_given_value( diff --git a/quantammsim/pools/FM_AMM/cow_pool.py b/quantammsim/pools/FM_AMM/cow_pool.py index 5a9536fd..96299ce8 100644 --- a/quantammsim/pools/FM_AMM/cow_pool.py +++ b/quantammsim/pools/FM_AMM/cow_pool.py @@ -58,10 +58,9 @@ class CowPool(AbstractPool): start_index, additional_oracle_input=None) -> jnp.ndarray: Calculates the reserves of the pool without considering fees. - calculate_reserves_with_dynamic_inputs(params, run_fingerprint, prices, - start_index, fees_array, arb_thresh_array, arb_fees_array, trade_array, - additional_oracle_input=None) -> jnp.ndarray: - Calculates the reserves of the pool with dynamic inputs for fees, + calculate_reserves_with_dynamic_inputs(params, run_fingerprint, prices, + start_index, dynamic_inputs, additional_oracle_input=None) -> jnp.ndarray: + Calculates the reserves of the pool with dynamic inputs for fees, arbitrage thresholds, arbitrage fees, and trades. init_base_parameters(initial_values_dict, run_fingerprint, n_assets, @@ -196,11 +195,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: bout_length = run_fingerprint["bout_length"] @@ -216,6 +211,11 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees + trade_array = dynamic_inputs.trades + initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = weights * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] diff --git a/quantammsim/pools/G3M/balancer/balancer.py b/quantammsim/pools/G3M/balancer/balancer.py index 0d7ec30f..986d49c3 100644 --- a/quantammsim/pools/G3M/balancer/balancer.py +++ b/quantammsim/pools/G3M/balancer/balancer.py @@ -257,11 +257,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: """ @@ -289,14 +285,8 @@ def calculate_reserves_with_dynamic_inputs( Price history array start_index : jnp.ndarray Starting index for the calculation window - fees_array : jnp.ndarray - Time-varying trading fees - arb_thresh_array : jnp.ndarray - Time-varying arbitrage thresholds - arb_fees_array : jnp.ndarray - Time-varying arbitrage fees - trade_array : jnp.ndarray - Custom trade sequence + dynamic_inputs : DynamicInputArrays + Fixed-structure bundle of dynamic inputs. Returns ------- @@ -316,6 +306,12 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees + trade_array = dynamic_inputs.trades + lp_supply_array = dynamic_inputs.lp_supply + initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = weights * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] diff --git a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py index e15b3610..b087b5a8 100644 --- a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py +++ b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py @@ -254,11 +254,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: bout_length = run_fingerprint["bout_length"] @@ -278,6 +274,12 @@ def calculate_reserves_with_dynamic_inputs( arb_acted_upon_weights = weights arb_acted_upon_local_prices = local_prices + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees + trade_array = dynamic_inputs.trades + lp_supply_array = dynamic_inputs.lp_supply + initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = arb_acted_upon_weights[0] * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] diff --git a/quantammsim/pools/base_pool.py b/quantammsim/pools/base_pool.py index cc2d7b3c..d8c6d759 100644 --- a/quantammsim/pools/base_pool.py +++ b/quantammsim/pools/base_pool.py @@ -93,11 +93,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs: Any, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: pass diff --git a/quantammsim/pools/hodl_pool.py b/quantammsim/pools/hodl_pool.py index 4c1085e3..c0740fb8 100644 --- a/quantammsim/pools/hodl_pool.py +++ b/quantammsim/pools/hodl_pool.py @@ -39,9 +39,9 @@ class HODLPool(AbstractPool): additional_oracle_input=None): Calculates the reserves without fees, assuming no trading activity. - calculate_reserves_with_dynamic_inputs(params, run_fingerprint, prices, start_index, - fees_array, arb_thresh_array, arb_fees_array, trade_array, additional_oracle_input=None): - Calculates the reserves with dynamic inputs, which in this case is + calculate_reserves_with_dynamic_inputs(params, run_fingerprint, prices, start_index, + dynamic_inputs, additional_oracle_input=None): + Calculates the reserves with dynamic inputs, which in this case is the same as reserves without fees due to no activity. init_base_parameters(initial_values_dict, run_fingerprint, n_assets, @@ -126,11 +126,7 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: # hodl means no activity, so reserves are just the initial reserves diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 152a264b..c73b6c4c 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -275,11 +275,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ): """Calculate reserves and LP fee revenue with time-varying inputs. @@ -291,6 +287,9 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( LP fee revenue per timestep in USD. """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 @@ -367,14 +366,13 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - fees_array: jnp.ndarray, - arb_thresh_array: jnp.ndarray, - arb_fees_array: jnp.ndarray, - trade_array: jnp.ndarray, - lp_supply_array: jnp.ndarray = None, + dynamic_inputs, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) + fees_array = dynamic_inputs.fees + arb_thresh_array = dynamic_inputs.gas_cost + arb_fees_array = dynamic_inputs.arb_fees bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 diff --git a/quantammsim/runners/__init__.py b/quantammsim/runners/__init__.py index 987a800c..511800b2 100644 --- a/quantammsim/runners/__init__.py +++ b/quantammsim/runners/__init__.py @@ -28,7 +28,7 @@ from .jax_runner_utils import ( nan_rollback, Hashabledict, - get_trades_and_fees, + prepare_dynamic_inputs, get_unique_tokens, OptunaManager, generate_evaluation_points, @@ -80,7 +80,7 @@ # Utilities "nan_rollback", "Hashabledict", - "get_trades_and_fees", + "prepare_dynamic_inputs", "get_unique_tokens", "OptunaManager", "generate_evaluation_points", diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index 399cd37f..64c916dd 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -14,6 +14,12 @@ raw_fee_like_amounts_to_fee_like_array, raw_trades_to_trade_array, ) +from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputArrays, + DynamicInputFrames, + dynamic_input_flags_from_frames, + empty_dynamic_input_arrays, +) from quantammsim.apis.rest_apis.simulator_dtos.simulation_run_dto import ( LiquidityPoolCoinDto, @@ -1215,39 +1221,40 @@ def unpermute_list_of_params(list_of_params): return list_of_params_to_return -def get_trades_and_fees( - run_fingerprint, raw_trades, fees_df, gas_cost_df, arb_fees_df, lp_supply_df, do_test_period=False -): - """ - Process trade and fee data for a simulation run. +def _to_dynamic_input_arrays( + trades_array, + fees_array, + gas_cost_array, + arb_fees_array, + lp_supply_array, +) -> DynamicInputArrays: + """Normalize optional numpy arrays into the fixed hot-path container.""" + empty = empty_dynamic_input_arrays() + return DynamicInputArrays( + trades=empty.trades if trades_array is None else jnp.asarray(trades_array, dtype=jnp.float64), + fees=empty.fees if fees_array is None else jnp.asarray(fees_array, dtype=jnp.float64), + gas_cost=empty.gas_cost if gas_cost_array is None else jnp.asarray(gas_cost_array, dtype=jnp.float64), + arb_fees=empty.arb_fees if arb_fees_array is None else jnp.asarray(arb_fees_array, dtype=jnp.float64), + lp_supply=empty.lp_supply if lp_supply_array is None else jnp.asarray(lp_supply_array, dtype=jnp.float64), + ) - Takes raw trades, fees, gas costs and arbitrage fees and converts them into arrays - suitable for simulation. Handles both training and test periods if specified. - Parameters - ---------- - run_fingerprint : dict - Dictionary containing run configuration including start/end dates and tokens - raw_trades : pd.DataFrame, optional - DataFrame containing raw trade data - fees_df : pd.DataFrame, optional - DataFrame containing fee data - gas_cost_df : pd.DataFrame, optional - DataFrame containing gas cost data - arb_fees_df : pd.DataFrame, optional - DataFrame containing arbitrage fee data - lp_supply_df : pd.DataFrame, optional - DataFrame containing LP supply data - do_test_period : bool, optional - Whether to process data for a test period after training period (default False) +def prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames: Optional[DynamicInputFrames] = None, + do_test_period: bool = False, +): + """Convert optional pandas inputs into fixed-structure dynamic input bundles.""" + if dynamic_input_frames is None: + dynamic_input_frames = DynamicInputFrames() + + raw_trades = dynamic_input_frames.trades + fees_df = dynamic_input_frames.fees + gas_cost_df = dynamic_input_frames.gas_cost + arb_fees_df = dynamic_input_frames.arb_fees + lp_supply_df = dynamic_input_frames.lp_supply + dynamic_input_flags = dynamic_input_flags_from_frames(dynamic_input_frames) - Returns - ------- - dict - Contains processed arrays for trades, fees, gas costs and arb fees for both - training and test periods as applicable - """ - # Process raw trades if provided if raw_trades is not None: train_period_trades = raw_trades_to_trade_array( raw_trades, @@ -1265,7 +1272,7 @@ def get_trades_and_fees( else: train_period_trades = None test_period_trades = None - # Process fees, gas costs, and arb fees if provided + fees_array = ( raw_fee_like_amounts_to_fee_like_array( fees_df, @@ -1281,8 +1288,8 @@ def get_trades_and_fees( test_fees_array = ( raw_fee_like_amounts_to_fee_like_array( fees_df, - run_fingerprint["startDateString"], run_fingerprint["endDateString"], + run_fingerprint["endTestDateString"], names=["fees"], fill_method="ffill", ) @@ -1361,25 +1368,32 @@ def get_trades_and_fees( else None ) return { - "train_period_trades": train_period_trades, - "test_period_trades": test_period_trades, - "fees_array": fees_array, - "gas_cost_array": gas_cost_array, - "arb_fees_array": arb_fees_array, - "lp_supply_array": lp_supply_array, - "test_fees_array": test_fees_array, - "test_gas_cost_array": test_gas_cost_array, - "test_arb_fees_array": test_arb_fees_array, - "test_lp_supply_array": test_lp_supply_array, - } - else: - return { - "train_period_trades": train_period_trades, - "fees_array": fees_array, - "gas_cost_array": gas_cost_array, - "arb_fees_array": arb_fees_array, - "lp_supply_array": lp_supply_array, + "train_dynamic_inputs": _to_dynamic_input_arrays( + train_period_trades, + fees_array, + gas_cost_array, + arb_fees_array, + lp_supply_array, + ), + "test_dynamic_inputs": _to_dynamic_input_arrays( + test_period_trades, + test_fees_array, + test_gas_cost_array, + test_arb_fees_array, + test_lp_supply_array, + ), + "dynamic_input_flags": dynamic_input_flags, } + return { + "train_dynamic_inputs": _to_dynamic_input_arrays( + train_period_trades, + fees_array, + gas_cost_array, + arb_fees_array, + lp_supply_array, + ), + "dynamic_input_flags": dynamic_input_flags, + } def create_daily_unix_array(start_date_str, end_date_str): @@ -1622,24 +1636,33 @@ def try_forward_pass(n_sets: int) -> bool: "n_assets": n_tokens, "training_data_kind": probe_fingerprint["optimisation_settings"]["training_data_kind"], "do_trades": False, + "dynamic_input_flags": { + "use_dynamic_inputs": False, + "has_trades": False, + "has_dynamic_fees": False, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + }, }, ) # Create vmapped forward pass partial_forward = Partial( forward_pass_nograd, + dynamic_inputs=None, prices=data_dict["prices"], static_dict=static_dict, pool=pool, ) vmapped_forward = jit( - vmap(partial_forward, in_axes=[params_in_axes_dict, None, None]) + vmap(partial_forward, in_axes=[params_in_axes_dict, None]) ) # Run forward pass start_index = (data_dict["start_idx"], 0) - _ = vmapped_forward(params, start_index, None) + _ = vmapped_forward(params, start_index) # Force computation to complete jnp.zeros(1).block_until_ready() diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index c403be5b..1a8fdf3b 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -52,6 +52,10 @@ forward_pass_nograd, _calculate_return_value, ) +from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputFrames, + resolve_dynamic_input_components, +) from quantammsim.core_simulator.windowing_utils import get_indices, filter_coarse_weights_by_data_indices import hashlib @@ -79,7 +83,7 @@ from quantammsim.runners.jax_runner_utils import ( Hashabledict, - get_trades_and_fees, + prepare_dynamic_inputs, get_unique_tokens, OptunaManager, generate_evaluation_points, @@ -639,6 +643,14 @@ def train_on_historic_data( "n_assets": n_assets, "training_data_kind": run_fingerprint["optimisation_settings"]["training_data_kind"], "do_trades": False, + "dynamic_input_flags": { + "use_dynamic_inputs": False, + "has_trades": False, + "has_dynamic_fees": False, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + }, }, ) @@ -653,6 +665,7 @@ def train_on_historic_data( continuous_static_dict["bout_length"] = original_bout_length + data_dict["bout_length_test"] partial_forward_pass_nograd_batch_continuous = Partial( forward_pass_nograd, + dynamic_inputs=None, static_dict=Hashabledict(continuous_static_dict), pool=pool, ) @@ -798,6 +811,7 @@ def init_optimizer(params): # Build scan-compatible update (prices as explicit arg, not closure) partial_step_no_prices = Partial( forward_pass, + dynamic_inputs=None, static_dict=Hashabledict(base_static_dict), pool=pool, ) @@ -1798,14 +1812,10 @@ def do_run_on_historic_data( root=None, price_data=None, verbose=False, - raw_trades=None, fees=None, gas_cost=None, arb_fees=None, - fees_df=None, - gas_cost_df=None, - arb_fees_df=None, - lp_supply_df=None, + dynamic_input_frames: DynamicInputFrames = None, do_test_period=False, low_data_mode=False, preslice_burnin=True, @@ -1831,23 +1841,14 @@ def do_run_on_historic_data( Pre-loaded price data. When None, loaded from parquet files. verbose : bool, optional Print progress information (default False). - raw_trades : DataFrame, optional - Real trade data to inject. Columns: unix timestamp (minute), - token_in, token_out, amount_in. fees : float, optional Swap fee override (e.g. 0.003 for 30 bps). gas_cost : float, optional Gas cost override per transaction. arb_fees : float, optional Arbitrageur fee override. - fees_df : DataFrame, optional - Time-varying swap fees (columns: unix, fee). - gas_cost_df : DataFrame, optional - Time-varying gas costs (columns: unix, gas_cost). - arb_fees_df : DataFrame, optional - Time-varying arb fees (columns: unix, arb_fee). - lp_supply_df : DataFrame, optional - Time-varying LP supply changes. + dynamic_input_frames : DynamicInputFrames, optional + Optional container of trades / fee / gas / arb / LP supply DataFrames. do_test_period : bool, optional If True, also run the OOS test period defined by ``endDateString`` to ``endTestDateString`` (default False). @@ -1892,15 +1893,21 @@ def do_run_on_historic_data( np.random.seed(0) - dynamic_inputs_dict = get_trades_and_fees( + dynamic_inputs_dict = prepare_dynamic_inputs( run_fingerprint, - raw_trades, - fees_df, - gas_cost_df, - arb_fees_df, - lp_supply_df, + dynamic_input_frames=dynamic_input_frames, do_test_period=do_test_period, ) + train_dynamic_inputs = ( + dynamic_inputs_dict["train_dynamic_inputs"] + if dynamic_inputs_dict["dynamic_input_flags"]["use_dynamic_inputs"] + else None + ) + test_dynamic_inputs = ( + dynamic_inputs_dict.get("test_dynamic_inputs") + if dynamic_inputs_dict["dynamic_input_flags"]["use_dynamic_inputs"] + else None + ) # Load price data if not provided if price_data is None: @@ -1944,7 +1951,8 @@ def do_run_on_historic_data( "fees": fees if fees is not None else run_fingerprint["fees"], "arb_fees": arb_fees if arb_fees is not None else run_fingerprint["arb_fees"], "gas_cost": gas_cost if gas_cost is not None else run_fingerprint["gas_cost"], - "do_trades": False if raw_trades is None else run_fingerprint["do_trades"], + "do_trades": dynamic_inputs_dict["dynamic_input_flags"]["has_trades"], + "dynamic_input_flags": dynamic_inputs_dict["dynamic_input_flags"], # Include date strings for run-time use "startDateString": run_fingerprint["startDateString"], "endDateString": run_fingerprint["endDateString"], @@ -2000,10 +2008,7 @@ def do_run_on_historic_data( param, (data_dict["start_idx"], 0), data_dict["prices"], - dynamic_inputs_dict["train_period_trades"], - dynamic_inputs_dict["fees_array"], - dynamic_inputs_dict["gas_cost_array"], - dynamic_inputs_dict["arb_fees_array"], + train_dynamic_inputs, ) if low_data_mode: output_dict["final_prices"] = output_dict["prices"][-1] @@ -2019,10 +2024,7 @@ def do_run_on_historic_data( param, (data_dict["start_idx_test"], 0), data_dict["prices"], - dynamic_inputs_dict["test_period_trades"], - dynamic_inputs_dict["test_fees_array"], - dynamic_inputs_dict["test_gas_cost_array"], - dynamic_inputs_dict["test_arb_fees_array"], + test_dynamic_inputs, ) if low_data_mode: output_dict_test["final_prices"] = output_dict_test["prices"][-1] @@ -2061,14 +2063,10 @@ def do_run_on_historic_data_with_provided_coarse_weights( root=None, price_data=None, verbose=False, - raw_trades=None, fees=None, gas_cost=None, arb_fees=None, - fees_df=None, - gas_cost_df=None, - arb_fees_df=None, - lp_supply_df=None, + dynamic_input_frames: DynamicInputFrames = None, do_test_period=False, low_data_mode=False, ): @@ -2098,22 +2096,14 @@ def do_run_on_historic_data_with_provided_coarse_weights( Pre-loaded price data. verbose : bool, optional Print progress (default False). - raw_trades : DataFrame, optional - Real trade data to inject. fees : float, optional Swap fee override. gas_cost : float, optional Gas cost override. arb_fees : float, optional Arbitrageur fee override. - fees_df : DataFrame, optional - Time-varying swap fees. - gas_cost_df : DataFrame, optional - Time-varying gas costs. - arb_fees_df : DataFrame, optional - Time-varying arb fees. - lp_supply_df : DataFrame, optional - Time-varying LP supply changes. + dynamic_input_frames : DynamicInputFrames, optional + Optional container of trades / fee / gas / arb / LP supply DataFrames. do_test_period : bool, optional Run OOS test period (default False). low_data_mode : bool, optional @@ -2152,13 +2142,9 @@ def do_run_on_historic_data_with_provided_coarse_weights( np.random.seed(0) - dynamic_inputs_dict = get_trades_and_fees( + dynamic_inputs_dict = prepare_dynamic_inputs( run_fingerprint, - raw_trades, - fees_df, - gas_cost_df, - arb_fees_df, - lp_supply_df, + dynamic_input_frames=dynamic_input_frames, do_test_period=do_test_period, ) @@ -2201,7 +2187,8 @@ def do_run_on_historic_data_with_provided_coarse_weights( "fees": fees if fees is not None else run_fingerprint["fees"], "arb_fees": arb_fees if arb_fees is not None else run_fingerprint["arb_fees"], "gas_cost": gas_cost if gas_cost is not None else run_fingerprint["gas_cost"], - "do_trades": False if raw_trades is None else run_fingerprint["do_trades"], + "do_trades": dynamic_inputs_dict["dynamic_input_flags"]["has_trades"], + "dynamic_input_flags": dynamic_inputs_dict["dynamic_input_flags"], # Include date strings for run-time use "startDateString": run_fingerprint["startDateString"], "endDateString": run_fingerprint["endDateString"], @@ -2268,18 +2255,18 @@ def do_run_on_historic_data_with_provided_coarse_weights( # weights=HashableArrayWrapper(weights), # initial_reserves=HashableArrayWrapper(params["initial_reserves"]), # ) - fees_array = dynamic_inputs_dict.get("fees_array") - arb_thresh_array = dynamic_inputs_dict.get("gas_cost_array") - arb_fees_array = dynamic_inputs_dict.get("arb_fees_array") - trade_array = dynamic_inputs_dict.get("trades") - lp_supply_array = dynamic_inputs_dict.get("lp_supply_array") - - if fees_array is None: - fees_array = jnp.array([static_dict["fees"]]) - if arb_thresh_array is None: - arb_thresh_array = jnp.array([static_dict["gas_cost"]]) - if arb_fees_array is None: - arb_fees_array = jnp.array([static_dict["arb_fees"]]) + dynamic_input_flags = dynamic_inputs_dict["dynamic_input_flags"] + dynamic_inputs = dynamic_inputs_dict["train_dynamic_inputs"] + resolved_dynamic_inputs = resolve_dynamic_input_components( + dynamic_inputs, + dynamic_input_flags, + static_dict, + ) + fees_array = resolved_dynamic_inputs["fees"] + arb_thresh_array = resolved_dynamic_inputs["gas_cost"] + arb_fees_array = resolved_dynamic_inputs["arb_fees"] + trade_array = resolved_dynamic_inputs["trades"] + lp_supply_array = resolved_dynamic_inputs["lp_supply"] # initial_pool_value = run_fingerprint["initial_pool_value"] # initial_value_per_token = arb_acted_upon_weights[0] * initial_pool_value @@ -2298,10 +2285,8 @@ def do_run_on_historic_data_with_provided_coarse_weights( fees_array = fees_array[:max_len] arb_thresh_array = arb_thresh_array[:max_len] - arb_thresh_array = arb_thresh_array * 0.0 arb_fees_array = arb_fees_array[:max_len] - if lp_supply_array is not None: - lp_supply_array = lp_supply_array[:max_len] + lp_supply_array = lp_supply_array[:max_len] if trade_array is not None: trade_array = trade_array[:max_len] # Broadcast input arrays to match the maximum leading dimension. @@ -2316,10 +2301,6 @@ def do_run_on_historic_data_with_provided_coarse_weights( arb_fees_array_broadcast = jnp.broadcast_to( arb_fees_array, (max_len,) + arb_fees_array.shape[1:] ) - # if lp_supply_array is not provided, we set it to a constant of 1.0 - if lp_supply_array is None: - lp_supply_array = jnp.array(1.0) - lp_supply_array_broadcast = jnp.broadcast_to( lp_supply_array, (max_len,) + lp_supply_array.shape[1:] ) diff --git a/quantammsim/runners/multi_period_sgd.py b/quantammsim/runners/multi_period_sgd.py index 78029689..c81e00a1 100644 --- a/quantammsim/runners/multi_period_sgd.py +++ b/quantammsim/runners/multi_period_sgd.py @@ -430,6 +430,7 @@ def multi_period_sgd_training( # Create base forward pass base_forward_pass = Partial( forward_pass, + dynamic_inputs=None, prices=data_dict["prices"], static_dict=Hashabledict(static_dict), pool=pool, @@ -506,6 +507,7 @@ def multi_period_sgd_training( partial_nograd = jit(Partial( forward_pass_nograd, + dynamic_inputs=None, prices=data_dict["prices"], static_dict=Hashabledict(static_dict), pool=pool, diff --git a/quantammsim/runners/training_evaluator.py b/quantammsim/runners/training_evaluator.py index e79a071b..011b3f71 100644 --- a/quantammsim/runners/training_evaluator.py +++ b/quantammsim/runners/training_evaluator.py @@ -742,6 +742,7 @@ def _compute_metrics( eval_fn = jit(Partial( forward_pass_nograd, + dynamic_inputs=None, prices=data_dict["prices"], static_dict=Hashabledict(static_dict), pool=pool, diff --git a/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py b/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py index f92f7da3..aa315199 100644 --- a/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py +++ b/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py @@ -20,6 +20,7 @@ from quantammsim.runners.jax_runners import do_run_on_historic_data from quantammsim.runners.jax_runner_utils import optimized_output_conversion +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames import quantammsim.simulator_analysis_tools.finance.financial_analysis_calculator as fac import quantammsim.simulator_analysis_tools.finance.financial_analysis_functions as faf import quantammsim.simulator_analysis_tools.finance.financial_analysis_utils as fau @@ -238,6 +239,12 @@ def run_pool_simulation(simulationRunDto): run_fingerprint["fees"] = static_fee fee_steps_df = None + dynamic_input_frames = DynamicInputFrames( + trades=raw_trades, + fees=fee_steps_df, + gas_cost=gas_cost_df, + ) + print("run fingerprint-------------------", run_fingerprint) print("update rule parameter dict converted-------------------", update_rule_parameter_dict_converted) outputDict = do_run_on_historic_data( @@ -247,9 +254,7 @@ def run_pool_simulation(simulationRunDto): price_data=price_data_local, verbose=True, do_test_period=False, - raw_trades=raw_trades, - gas_cost_df=gas_cost_df, - fees_df=fee_steps_df + dynamic_input_frames=dynamic_input_frames, ) print("outputDict: ", outputDict.keys()) resultTimeSteps = optimized_output_conversion(simulationRunDto, outputDict, tokens) @@ -293,9 +298,7 @@ def run_pool_simulation(simulationRunDto): price_data=price_data_local, verbose=False, do_test_period=False, - raw_trades=raw_trades, - gas_cost_df=gas_cost_df, - fees_df=fee_steps_df, + dynamic_input_frames=dynamic_input_frames, ) # Extract final weights from the result. diff --git a/scripts/demo_run_chunks_from_chain_data.py b/scripts/demo_run_chunks_from_chain_data.py index fbd9cf9c..09c19d14 100644 --- a/scripts/demo_run_chunks_from_chain_data.py +++ b/scripts/demo_run_chunks_from_chain_data.py @@ -28,6 +28,7 @@ import numpy as np import pandas as pd import matplotlib as mpl +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames import matplotlib.pyplot as plt import jax.numpy as jnp @@ -474,10 +475,12 @@ def _df_meta_and_head(df, name, n=3): run_fingerprint=fingerprint, coarse_weights=cw_window, params=params, - fees_df=scraped["fees_df"], - gas_cost_df=scraped["gas_cost_df"], - lp_supply_df=scraped["lp_supply_df"], - arb_fees_df=scraped["arb_fees_df"], + dynamic_input_frames=DynamicInputFrames( + fees=scraped["fees_df"], + gas_cost=scraped["gas_cost_df"], + lp_supply=scraped["lp_supply_df"], + arb_fees=scraped["arb_fees_df"], + ), ) # ---------------- Correct, window-aligned plotting block (time-aware + plain y) ---------------- diff --git a/scripts/demo_run_from_chain_data.py b/scripts/demo_run_from_chain_data.py index 78ecbc4d..b1f41398 100644 --- a/scripts/demo_run_from_chain_data.py +++ b/scripts/demo_run_from_chain_data.py @@ -1,4 +1,5 @@ import jax.numpy as jnp +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.core_simulator.param_utils import ( memory_days_to_logit_lamb, ) @@ -997,10 +998,12 @@ def generate_daily_variations(start_date_str, end_date_str): run_fingerprint=config["fingerprint"], coarse_weights=config["coarse_weights"], params=config["params"], - fees_df=config["fees_df"], - gas_cost_df=config["gas_cost_df"], - lp_supply_df=config["lp_supply_df"], - arb_fees_df=config["arb_fees_df"], + dynamic_input_frames=DynamicInputFrames( + fees=config["fees_df"], + gas_cost=config["gas_cost_df"], + lp_supply=config["lp_supply_df"], + arb_fees=config["arb_fees_df"], + ), ) print("-" * 80) print(f"Pool Type: {config['fingerprint']['rule']}") @@ -1191,4 +1194,4 @@ def generate_daily_variations(start_date_str, end_date_str): # actual_reserves_np=local_reserves, # actual_unix_values=datetime_array, # ) - # raise Exception("Stop here") \ No newline at end of file + # raise Exception("Stop here") diff --git a/scripts/reclamm/sim_vs_world_comparison.py b/scripts/reclamm/sim_vs_world_comparison.py index 0c754ea1..bf7d19d6 100644 --- a/scripts/reclamm/sim_vs_world_comparison.py +++ b/scripts/reclamm/sim_vs_world_comparison.py @@ -34,6 +34,7 @@ from pathlib import Path from datetime import datetime, timezone +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.runners.jax_runners import do_run_on_historic_data # ── On-chain reClAMM params ─────────────────────────────────────────────────── @@ -187,9 +188,18 @@ def sample_at_timestamps(minute_vals, start_unix_sec, timestamps_sec): return minute_vals[indices] -def run_pool(tokens, start, end, rule, fees, params, gas_cost=0.0, - protocol_fee_split=0.0, gas_cost_df=None, - onchain_initial_state=None): +def run_pool( + tokens, + start, + end, + rule, + fees, + params, + gas_cost=0.0, + protocol_fee_split=0.0, + dynamic_input_frames=None, + onchain_initial_state=None, +): """Run a quantammsim pool and return minute-level results. Returns (val_eth, price_ratio, start_unix_sec) where val_eth and @@ -220,7 +230,9 @@ def run_pool(tokens, start, end, rule, fees, params, gas_cost=0.0, fp["reclamm_initial_state"] = onchain_initial_state result = do_run_on_historic_data( - run_fingerprint=fp, params=params, gas_cost_df=gas_cost_df, + run_fingerprint=fp, + params=params, + dynamic_input_frames=dynamic_input_frames, ) # Prices: sorted tokens → [AAVE, ETH] in USD @@ -291,7 +303,8 @@ def run_gas_experiment(args): gas_df = load_gas_csv(pct) val_eth_min, _, _ = run_pool( tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, - protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df, + protocol_fee_split=PROTOCOL_FEE_SPLIT, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df), ) gas_results_min[pct] = val_eth_min @@ -433,7 +446,8 @@ def run_gas_scale_experiment(args): gas_df["trade_gas_cost_usd"] = gas_df_raw["trade_gas_cost_usd"] * scale val_eth_min, pr_min, start_sec = run_pool( tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, - protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df, + protocol_fee_split=PROTOCOL_FEE_SPLIT, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df), onchain_initial_state=onchain_state, ) results_min[(pct, scale)] = (val_eth_min, start_sec) @@ -662,7 +676,8 @@ def run_best_gas_experiment(args): gas_df_50p = load_gas_csv("50p") g50_min, _, _ = run_pool( tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, - protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_50p, + protocol_fee_split=PROTOCOL_FEE_SPLIT, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df_50p), onchain_initial_state=onchain_state, ) @@ -674,7 +689,8 @@ def run_best_gas_experiment(args): gas_df_75p_scaled["trade_gas_cost_usd"] *= 0.75 g75_min, _, _ = run_pool( tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, - protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_75p_scaled, + protocol_fee_split=PROTOCOL_FEE_SPLIT, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df_75p_scaled), onchain_initial_state=onchain_state, ) @@ -686,7 +702,8 @@ def run_best_gas_experiment(args): gas_df_90p_scaled["trade_gas_cost_usd"] *= 0.25 g90_min, _, _ = run_pool( tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, - protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_90p_scaled, + protocol_fee_split=PROTOCOL_FEE_SPLIT, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df_90p_scaled), onchain_initial_state=onchain_state, ) diff --git a/tests/integration/test_dynamic_gas_fees.py b/tests/integration/test_dynamic_gas_fees.py index 5870414a..8eea4cd9 100644 --- a/tests/integration/test_dynamic_gas_fees.py +++ b/tests/integration/test_dynamic_gas_fees.py @@ -9,6 +9,7 @@ import jax.numpy as jnp from pathlib import Path +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.runners.jax_runners import do_run_on_historic_data @@ -78,8 +79,12 @@ class TestDynamicGasAndFees: @pytest.mark.requires_data def test_run_with_gas_and_fees(self, base_fingerprint, base_params, gas_df, fees_df, data_root): """Test simulation with both gas costs and dynamic fees.""" + dynamic_input_frames = DynamicInputFrames(gas_cost=gas_df, fees=fees_df) result = do_run_on_historic_data( - base_fingerprint, base_params, root=data_root, gas_cost_df=gas_df, fees_df=fees_df + base_fingerprint, + base_params, + root=data_root, + dynamic_input_frames=dynamic_input_frames, ) assert result is not None @@ -92,8 +97,12 @@ def test_run_with_gas_and_fees(self, base_fingerprint, base_params, gas_df, fees @pytest.mark.requires_data def test_run_with_gas_only(self, base_fingerprint, base_params, gas_df, data_root): """Test simulation with gas costs only.""" + dynamic_input_frames = DynamicInputFrames(gas_cost=gas_df) result = do_run_on_historic_data( - base_fingerprint, base_params, root=data_root, gas_cost_df=gas_df + base_fingerprint, + base_params, + root=data_root, + dynamic_input_frames=dynamic_input_frames, ) assert result is not None @@ -104,8 +113,12 @@ def test_run_with_gas_only(self, base_fingerprint, base_params, gas_df, data_roo @pytest.mark.requires_data def test_run_with_fees_only(self, base_fingerprint, base_params, fees_df, data_root): """Test simulation with dynamic fees only.""" + dynamic_input_frames = DynamicInputFrames(fees=fees_df) result = do_run_on_historic_data( - base_fingerprint, base_params, root=data_root, fees_df=fees_df + base_fingerprint, + base_params, + root=data_root, + dynamic_input_frames=dynamic_input_frames, ) assert result is not None @@ -120,8 +133,12 @@ def test_gas_reduces_final_value(self, base_fingerprint, base_params, gas_df, da result_no_gas = do_run_on_historic_data(base_fingerprint, base_params, root=data_root) # Run with gas + dynamic_input_frames = DynamicInputFrames(gas_cost=gas_df) result_with_gas = do_run_on_historic_data( - base_fingerprint, base_params, root=data_root, gas_cost_df=gas_df + base_fingerprint, + base_params, + root=data_root, + dynamic_input_frames=dynamic_input_frames, ) if "final_value" in result_no_gas and "final_value" in result_with_gas: @@ -139,8 +156,12 @@ def test_fees_reduce_final_value(self, base_fingerprint, base_params, fees_df, d result_no_fees = do_run_on_historic_data(base_fingerprint, base_params, root=data_root) # Run with fees + dynamic_input_frames = DynamicInputFrames(fees=fees_df) result_with_fees = do_run_on_historic_data( - base_fingerprint, base_params, root=data_root, fees_df=fees_df + base_fingerprint, + base_params, + root=data_root, + dynamic_input_frames=dynamic_input_frames, ) if "final_value" in result_no_fees and "final_value" in result_with_fees: diff --git a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py index 9406a967..0e58726e 100644 --- a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py +++ b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py @@ -267,6 +267,7 @@ class TestPoolMethodWithFees: """pool.calculate_reserves_and_fee_revenue_with_fees returns correct tuple.""" def test_pool_method_with_fees(self): + from quantammsim.core_simulator.dynamic_inputs import DynamicInputArrays from quantammsim.pools.creator import create_pool from quantammsim.runners.jax_runner_utils import Hashabledict @@ -347,13 +348,20 @@ def test_pool_method_with_dynamic_inputs(self): fees_array = jnp.array([0.003]) arb_thresh_array = jnp.array([0.0]) arb_fees_array = jnp.array([0.0]) + dynamic_inputs = DynamicInputArrays( + trades=jnp.zeros((1, 3)), + fees=fees_array, + gas_cost=arb_thresh_array, + arb_fees=arb_fees_array, + lp_supply=jnp.ones((1,)), + ) reserves, fee_revenue = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( - params, run_fingerprint, prices, start_index, - fees_array=fees_array, - arb_thresh_array=arb_thresh_array, - arb_fees_array=arb_fees_array, - trade_array=None, + params, + run_fingerprint, + prices, + start_index, + dynamic_inputs=dynamic_inputs, ) assert reserves.shape == (n_steps, 2) diff --git a/tests/scripts/dynamic_gas_test.py b/tests/scripts/dynamic_gas_test.py index d656b6a8..920a7e3a 100644 --- a/tests/scripts/dynamic_gas_test.py +++ b/tests/scripts/dynamic_gas_test.py @@ -1,7 +1,9 @@ -from quantammsim.runners.jax_runners import do_run_on_historic_data import jax.numpy as jnp import pandas as pd +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames +from quantammsim.runners.jax_runners import do_run_on_historic_data + # Print the results print("=" * 100) print("Simulation Results:") @@ -32,11 +34,21 @@ run_fingerprint["do_trades"] = False result_w_gas_and_fees = do_run_on_historic_data( - run_fingerprint, params, gas_cost_df=gas_df, fees_df=fees_df + run_fingerprint, + params, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df, fees=fees_df), +) +result_w_gas_only = do_run_on_historic_data( + run_fingerprint, + params, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_df), ) -result_w_gas_only = do_run_on_historic_data(run_fingerprint, params, gas_cost_df=gas_df) -result_w_fees_only = do_run_on_historic_data(run_fingerprint, params, fees_df=fees_df) +result_w_fees_only = do_run_on_historic_data( + run_fingerprint, + params, + dynamic_input_frames=DynamicInputFrames(fees=fees_df), +) print(result_w_gas_and_fees["value"][-1440+1]) print(result_w_gas_only["value"][-1440+1]) diff --git a/tests/unit/test_jax_runner_utils.py b/tests/unit/test_jax_runner_utils.py index 9cd95f60..70dbf828 100644 --- a/tests/unit/test_jax_runner_utils.py +++ b/tests/unit/test_jax_runner_utils.py @@ -13,6 +13,7 @@ import pytest import numpy as np import jax.numpy as jnp +import pandas as pd class TestHashabledict: @@ -159,6 +160,21 @@ def test_static_dict_is_hashable_with_real_fingerprint(self): h = hash(hd) assert isinstance(h, int) + def test_static_dict_accepts_dynamic_input_flags(self): + """Nested dynamic-input flags must remain hashable for JIT cache keys.""" + from quantammsim.core_simulator.dynamic_inputs import default_dynamic_input_flags + from quantammsim.runners.jax_runner_utils import create_static_dict, Hashabledict + + fp = self._make_fingerprint() + static = create_static_dict( + fp, + bout_length=10080, + overrides={"dynamic_input_flags": default_dynamic_input_flags()}, + ) + + assert static["dynamic_input_flags"]["use_dynamic_inputs"] is False + assert isinstance(hash(Hashabledict(static)), int) + def test_unknown_array_fields_dropped_with_warning(self): """Arrays not in _TRAINING_ONLY_FIELDS are dropped with a warning. @@ -237,6 +253,226 @@ def test_equality_with_non_dict_returns_false(self): assert d != [1, 2, 3] +class TestDynamicInputPreparation: + """Tests for dynamic input container construction and normalization.""" + + def test_empty_dynamic_input_arrays_have_stable_shapes(self): + """The empty hot-path bundle should have canonical placeholder arrays.""" + from quantammsim.core_simulator.dynamic_inputs import empty_dynamic_input_arrays + + dynamic_inputs = empty_dynamic_input_arrays() + + assert dynamic_inputs.trades.shape == (1, 3) + assert dynamic_inputs.fees.shape == (1,) + assert dynamic_inputs.gas_cost.shape == (1,) + assert dynamic_inputs.arb_fees.shape == (1,) + assert dynamic_inputs.lp_supply.shape == (1,) + + def test_dynamic_input_flags_reflect_present_frames(self): + """Frame-presence flags should drive static dynamic-input dispatch.""" + from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputFrames, + dynamic_input_flags_from_frames, + ) + + flags = dynamic_input_flags_from_frames( + DynamicInputFrames( + trades=pd.DataFrame({"unix": [1], "token_in": ["ETH"], "token_out": ["USDC"], "amount_in": [1.0]}), + fees=pd.DataFrame({"unix": [1], "fees": [0.003]}), + gas_cost=pd.DataFrame({"unix": [1], "trade_gas_cost_usd": [2.0]}), + ) + ) + + assert flags["use_dynamic_inputs"] is True + assert flags["has_trades"] is True + assert flags["has_dynamic_fees"] is True + assert flags["has_dynamic_gas_cost"] is True + assert flags["has_dynamic_arb_fees"] is False + assert flags["has_lp_supply"] is False + + def test_prepare_dynamic_inputs_preserves_fixed_hot_path_structure(self): + """Normalization should return fixed bundles plus static dispatch flags.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:02:00", + "endTestDateString": "2023-01-01 00:04:00", + } + + dynamic_input_frames = DynamicInputFrames( + trades=pd.DataFrame( + { + "unix": [1672531200000, 1672531320000], + "token_in": ["ETH", "USDC"], + "token_out": ["USDC", "ETH"], + "amount_in": [1.5, 2.0], + } + ), + fees=pd.DataFrame({"unix": [1672531200000], "fees": [0.003]}), + gas_cost=pd.DataFrame({"unix": [1672531200000], "trade_gas_cost_usd": [3.25]}), + arb_fees=pd.DataFrame({"unix": [1672531200000], "arb_fees": [0.0005]}), + lp_supply=pd.DataFrame({"unix": [1672531200000], "lp_supply": [1250.0]}), + ) + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=dynamic_input_frames, + do_test_period=True, + ) + + train_inputs = prepared["train_dynamic_inputs"] + test_inputs = prepared["test_dynamic_inputs"] + flags = prepared["dynamic_input_flags"] + + assert flags["use_dynamic_inputs"] is True + assert flags["has_trades"] is True + assert flags["has_dynamic_fees"] is True + assert flags["has_dynamic_gas_cost"] is True + assert flags["has_dynamic_arb_fees"] is True + assert flags["has_lp_supply"] is True + assert train_inputs.trades.shape == (2, 3) + assert train_inputs.fees.shape == (2,) + assert train_inputs.gas_cost.shape == (2,) + assert train_inputs.arb_fees.shape == (2,) + assert train_inputs.lp_supply.shape == (2,) + assert test_inputs.trades.shape == (2, 3) + assert test_inputs.fees.shape == (2,) + assert test_inputs.gas_cost.shape == (2,) + assert test_inputs.arb_fees.shape == (2,) + assert test_inputs.lp_supply.shape == (2,) + np.testing.assert_allclose(np.asarray(train_inputs.fees), np.array([0.003, 0.003])) + + def test_prepare_dynamic_inputs_uses_correct_test_period_values(self): + """Test-period arrays should use values effective from the test window onward.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:02:00", + "endTestDateString": "2023-01-01 00:04:00", + } + end_unix = pd.Timestamp(run_fingerprint["endDateString"]).value // 10**6 + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + fees=pd.DataFrame( + {"unix": [1672531200000, end_unix], "fees": [0.003, 0.004]} + ), + gas_cost=pd.DataFrame( + { + "unix": [1672531200000, end_unix], + "trade_gas_cost_usd": [1.5, 2.5], + } + ), + arb_fees=pd.DataFrame( + {"unix": [1672531200000, end_unix], "arb_fees": [0.0001, 0.0002]} + ), + lp_supply=pd.DataFrame( + {"unix": [1672531200000, end_unix], "lp_supply": [1000.0, 2000.0]} + ), + ), + do_test_period=True, + ) + + np.testing.assert_allclose( + np.asarray(prepared["test_dynamic_inputs"].fees), + np.array([0.004, 0.004]), + ) + np.testing.assert_allclose( + np.asarray(prepared["test_dynamic_inputs"].gas_cost), + np.array([2.5, 2.5]), + ) + np.testing.assert_allclose( + np.asarray(prepared["test_dynamic_inputs"].arb_fees), + np.array([0.0002, 0.0002]), + ) + np.testing.assert_allclose( + np.asarray(prepared["test_dynamic_inputs"].lp_supply), + np.array([2000.0, 2000.0]), + ) + + def test_resolve_dynamic_input_flags_promotes_explicit_bundle(self): + """Passing a bundle directly should force dynamic-path dispatch.""" + from quantammsim.core_simulator.dynamic_inputs import ( + empty_dynamic_input_arrays, + resolve_dynamic_input_flags, + ) + + flags = resolve_dynamic_input_flags( + empty_dynamic_input_arrays(), + { + "use_dynamic_inputs": False, + "has_trades": False, + "has_dynamic_fees": False, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + }, + ) + + assert flags["use_dynamic_inputs"] is True + + def test_resolve_dynamic_input_components_falls_back_to_static_scalars(self): + """Static scalar config should materialize as singleton arrays when no frames are present.""" + from quantammsim.core_simulator.dynamic_inputs import ( + default_dynamic_input_flags, + resolve_dynamic_input_components, + ) + + resolved = resolve_dynamic_input_components( + dynamic_inputs=None, + dynamic_input_flags=default_dynamic_input_flags(), + static_dict={"fees": 0.003, "gas_cost": 2.5, "arb_fees": 0.0001}, + ) + + assert resolved["trades"] is None + np.testing.assert_allclose(np.asarray(resolved["fees"]), np.array([0.003])) + np.testing.assert_allclose(np.asarray(resolved["gas_cost"]), np.array([2.5])) + np.testing.assert_allclose(np.asarray(resolved["arb_fees"]), np.array([0.0001])) + np.testing.assert_allclose(np.asarray(resolved["lp_supply"]), np.array([1.0])) + + def test_resolve_dynamic_input_components_prefers_dynamic_values(self): + """Dynamic arrays should override static scalar defaults for enabled fields.""" + from quantammsim.core_simulator.dynamic_inputs import ( + DynamicInputArrays, + resolve_dynamic_input_components, + ) + + dynamic_inputs = DynamicInputArrays( + trades=jnp.array([[0.0, 1.0, 5.0]]), + fees=jnp.array([0.004]), + gas_cost=jnp.array([3.0]), + arb_fees=jnp.array([0.0003]), + lp_supply=jnp.array([1500.0]), + ) + flags = { + "use_dynamic_inputs": True, + "has_trades": True, + "has_dynamic_fees": True, + "has_dynamic_gas_cost": True, + "has_dynamic_arb_fees": True, + "has_lp_supply": True, + } + + resolved = resolve_dynamic_input_components( + dynamic_inputs=dynamic_inputs, + dynamic_input_flags=flags, + static_dict={"fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0}, + ) + + np.testing.assert_allclose(np.asarray(resolved["trades"]), np.array([[0.0, 1.0, 5.0]])) + np.testing.assert_allclose(np.asarray(resolved["fees"]), np.array([0.004])) + np.testing.assert_allclose(np.asarray(resolved["gas_cost"]), np.array([3.0])) + np.testing.assert_allclose(np.asarray(resolved["arb_fees"]), np.array([0.0003])) + np.testing.assert_allclose(np.asarray(resolved["lp_supply"]), np.array([1500.0])) + + class TestGetSigVariations: """Tests for get_sig_variations function.""" diff --git a/tests/unit/test_jax_runners_comprehensive.py b/tests/unit/test_jax_runners_comprehensive.py index dcdd9ddd..995e6d87 100644 --- a/tests/unit/test_jax_runners_comprehensive.py +++ b/tests/unit/test_jax_runners_comprehensive.py @@ -10,6 +10,7 @@ """ import pytest import numpy as np +import pandas as pd import jax.numpy as jnp import jax from copy import deepcopy @@ -18,6 +19,7 @@ from quantammsim.runners.jax_runners import ( train_on_historic_data, do_run_on_historic_data, + do_run_on_historic_data_with_provided_coarse_weights, ) from quantammsim.runners.jax_runner_utils import ( NestedHashabledict, @@ -31,8 +33,10 @@ create_static_dict, ) from quantammsim.runners.default_run_fingerprint import run_fingerprint_defaults +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.core_simulator.param_utils import recursive_default_set, check_run_fingerprint from quantammsim.pools.creator import create_pool +from quantammsim.utils.data_processing.historic_data_utils import get_data_dict from tests.conftest import TEST_DATA_DIR @@ -389,6 +393,219 @@ def test_multiple_param_sets(self, defaulted_run_fingerprint, sample_params): assert isinstance(results, list) assert len(results) == 2 + def test_dynamic_trades_change_balancer_reserves(self, defaulted_run_fingerprint): + """Dynamic trade input should change the reserve path in the runner.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["rule"] = "balancer" + fp["do_arb"] = False + fp["fees"] = 0.0 + fp["gas_cost"] = 0.0 + fp["arb_fees"] = 0.0 + params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + + trade_unix = pd.Timestamp(fp["startDateString"]).value // 10**6 + trades_df = pd.DataFrame( + { + "unix": [trade_unix], + "token_in": ["ETH"], + "token_out": ["USDC"], + "amount_in": [100.0], + } + ) + + result_without_trades = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + ) + result_with_trades = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames(trades=trades_df), + ) + + assert not np.allclose( + np.asarray(result_without_trades["reserves"]), + np.asarray(result_with_trades["reserves"]), + ) + assert result_with_trades["reserves"][0, 0] > result_without_trades["reserves"][0, 0] + assert result_with_trades["reserves"][0, 1] < result_without_trades["reserves"][0, 1] + + def test_dynamic_arb_fees_match_scalar_arb_fees(self, defaulted_run_fingerprint): + """Constant dynamic arb fees should match the scalar arb-fee path.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["rule"] = "balancer" + fp["fees"] = 0.003 + fp["do_arb"] = True + params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + + arb_fee = 0.002 + arb_fees_df = pd.DataFrame( + { + "unix": [pd.Timestamp(fp["startDateString"]).value // 10**6], + "arb_fees": [arb_fee], + } + ) + + result_scalar = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + arb_fees=arb_fee, + ) + result_dynamic = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames(arb_fees=arb_fees_df), + ) + + np.testing.assert_allclose( + np.asarray(result_dynamic["value"]), + np.asarray(result_scalar["value"]), + rtol=1e-6, + atol=1e-6, + ) + + def test_dynamic_lp_supply_changes_momentum_runner_path(self, defaulted_run_fingerprint, sample_params): + """LP supply changes should affect the main momentum runner path.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["protocol_fee_split"] = 0.25 + + start_unix = pd.Timestamp(fp["startDateString"]).value // 10**6 + midpoint_unix = start_unix + 3 * 1440 * 60 * 1000 + constant_lp_supply_df = pd.DataFrame( + { + "unix": [start_unix], + "lp_supply": [1.0], + } + ) + stepped_lp_supply_df = pd.DataFrame( + { + "unix": [start_unix, midpoint_unix], + "lp_supply": [1.0, 2.0], + } + ) + + result_without_lp_supply = do_run_on_historic_data( + fp, + params=sample_params, + root=TEST_DATA_DIR, + verbose=False, + ) + result_with_constant_lp_supply = do_run_on_historic_data( + fp, + params=sample_params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames(lp_supply=constant_lp_supply_df), + ) + result_with_stepped_lp_supply = do_run_on_historic_data( + fp, + params=sample_params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames(lp_supply=stepped_lp_supply_df), + ) + + np.testing.assert_allclose( + np.asarray(result_with_constant_lp_supply["value"]), + np.asarray(result_without_lp_supply["value"]), + rtol=1e-6, + atol=1e-6, + ) + assert not np.allclose( + np.asarray(result_with_stepped_lp_supply["reserves"][-1]), + np.asarray(result_without_lp_supply["reserves"][-1]), + ) + assert float(result_with_stepped_lp_supply["final_value"]) != pytest.approx( + float(result_without_lp_supply["final_value"]) + ) + + def test_provided_coarse_weights_respect_scalar_and_dynamic_gas(self, defaulted_run_fingerprint, sample_params): + """Provided-coarse-weight path should honor both scalar gas and dynamic gas arrays.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["protocol_fee_split"] = 0.0 + + data_dict = get_data_dict( + list_of_tickers=fp["tokens"], + run_fingerprint=fp, + data_kind=fp["optimisation_settings"]["training_data_kind"], + root=TEST_DATA_DIR, + max_memory_days=fp["max_memory_days"], + start_date_string=fp["startDateString"], + end_time_string=fp["endDateString"], + start_time_test_string=fp["endDateString"], + end_time_test_string=fp["endTestDateString"], + max_mc_version=fp["optimisation_settings"]["max_mc_version"], + do_test_period=False, + ) + + coarse_unix_values = ( + pd.date_range( + start=pd.Timestamp(fp["startDateString"]), + end=pd.Timestamp(fp["endDateString"]), + freq=f"{fp['chunk_period']}min", + ) + .astype(np.int64) + // 10**6 + ) + coarse_weights = { + "weights": jnp.tile(jnp.array([[0.5, 0.5]]), (len(coarse_unix_values), 1)), + "unix_values": jnp.asarray(coarse_unix_values), + } + + params = deepcopy(sample_params) + initial_prices = jnp.asarray(data_dict["prices"][data_dict["start_idx"]], dtype=jnp.float64) + params["initial_reserves"] = (jnp.array([0.5, 0.5]) * fp["initial_pool_value"]) / initial_prices + + gas_cost = 50.0 + gas_cost_df = pd.DataFrame( + { + "unix": [pd.Timestamp(fp["startDateString"]).value // 10**6], + "trade_gas_cost_usd": [gas_cost], + } + ) + + result_no_gas = do_run_on_historic_data_with_provided_coarse_weights( + fp, + coarse_weights=coarse_weights, + params=params, + root=TEST_DATA_DIR, + verbose=False, + ) + result_scalar_gas = do_run_on_historic_data_with_provided_coarse_weights( + fp, + coarse_weights=coarse_weights, + params=params, + root=TEST_DATA_DIR, + verbose=False, + gas_cost=gas_cost, + ) + result_dynamic_gas = do_run_on_historic_data_with_provided_coarse_weights( + fp, + coarse_weights=coarse_weights, + params=params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames(gas_cost=gas_cost_df), + ) + + assert float(result_scalar_gas["final_value"]) != pytest.approx( + float(result_no_gas["final_value"]) + ) + np.testing.assert_allclose( + np.asarray(result_dynamic_gas["value"]), + np.asarray(result_scalar_gas["value"]), + rtol=1e-6, + atol=1e-6, + ) + # ============================================================================ # Validation and Early Stopping Tests diff --git a/tests/unit/test_lint_bugs.py b/tests/unit/test_lint_bugs.py index 95f7ee6a..bbb49e60 100644 --- a/tests/unit/test_lint_bugs.py +++ b/tests/unit/test_lint_bugs.py @@ -117,16 +117,13 @@ def calculate_reserves_zero_fees(self, params, static_dict, prices, start_index) prices = jnp.ones((20, 2)) start_index = jnp.array([0, 0]) - # __wrapped__ arg order: params, start_index, prices, trades, fees, - # gas_cost, arb_fees, pool, static_dict + # __wrapped__ arg order: params, start_index, prices, dynamic_inputs, + # pool, static_dict result = forward_pass.__wrapped__( {}, # params start_index, # start_index prices, # prices - None, # trades_array - None, # fees_array - None, # gas_cost_array - None, # arb_fees_array + None, # dynamic_inputs _MockPool(), # pool static_dict, # static_dict ) From 0dcb04d9eb949490a7d054e2e7a3cf18d39a548d Mon Sep 17 00:00:00 2001 From: christian harrington Date: Wed, 4 Mar 2026 15:57:53 +0000 Subject: [PATCH 002/115] add fix for test directory data instead of main data source --- tests/pools/reCLAMM/test_reclamm_reserves.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index 3887e47f..cf402174 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -9,6 +9,7 @@ import numpy as np import numpy.testing as npt +from tests.conftest import TEST_DATA_DIR from quantammsim.pools.reCLAMM.reclamm_reserves import ( compute_invariant, compute_price_ratio, @@ -734,8 +735,8 @@ def test_shift_exponent_equivalent_to_base(self): fp_common = { "rule": "reclamm", "tokens": ["ETH", "USDC"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2024-06-15 00:00:00", + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-15 00:00:00", "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": 0.0, @@ -748,6 +749,7 @@ def test_shift_exponent_equivalent_to_base(self): "centeredness_margin": jnp.array(0.2), "daily_price_shift_base": jnp.array(base), }, + root=str(TEST_DATA_DIR), ) result_exp = do_run_on_historic_data( run_fingerprint={**fp_common, "reclamm_use_shift_exponent": True}, @@ -756,6 +758,7 @@ def test_shift_exponent_equivalent_to_base(self): "centeredness_margin": jnp.array(0.2), "shift_exponent": jnp.array(shift_exp), }, + root=str(TEST_DATA_DIR), ) np.testing.assert_allclose( @@ -772,10 +775,9 @@ def test_train_on_historic_data_optuna(self): fp = { "rule": "reclamm", "tokens": ["ETH", "USDC"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2024-06-15 00:00:00", - "endTestDateString": "2024-07-01 00:00:00", - "endTestDateString": "2024-08-01 00:00:00", + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-15 00:00:00", + "endTestDateString": "2023-02-01 00:00:00", "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": 0.0025, @@ -810,5 +812,5 @@ def test_train_on_historic_data_optuna(self): }, }, } - result = train_on_historic_data(fp, verbose=False) + result = train_on_historic_data(fp, root=str(TEST_DATA_DIR), verbose=False) assert result is not None From ffda35c26a8bebeed9bef9bd4f93606c849e4dc2 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Wed, 4 Mar 2026 15:58:06 +0000 Subject: [PATCH 003/115] import fix --- tests/pools/reCLAMM/test_reclamm_fee_revenue.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py index 0e58726e..4a31f1fa 100644 --- a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py +++ b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py @@ -9,6 +9,7 @@ import numpy as np import numpy.testing as npt +from quantammsim.core_simulator.dynamic_inputs import DynamicInputArrays from quantammsim.pools.reCLAMM.reclamm_reserves import ( initialise_reclamm_reserves, _jax_calc_reclamm_reserves_with_fees, From 961b796c1c2d8b9ae46496102bbc5ad1b3e1c536 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Wed, 4 Mar 2026 16:28:17 +0000 Subject: [PATCH 004/115] refactor the dynamic inputs given the runtime errors with lax scan. trades is the wrong shape and optional. Centralised in the new materialized function that is used by all dynamic input reserve calcs --- quantammsim/core_simulator/dynamic_inputs.py | 73 +++++++++++++++- quantammsim/core_simulator/forward_pass.py | 15 ++-- quantammsim/core_simulator/windowing_utils.py | 41 ++++----- quantammsim/hooks/dynamic_fee_base_hook.py | 2 +- quantammsim/pools/ECLP/gyroscope.py | 47 ++++------ quantammsim/pools/ECLP/gyroscope_reserves.py | 24 ++--- quantammsim/pools/FM_AMM/cow_pool.py | 39 +++------ quantammsim/pools/FM_AMM/cow_reserves.py | 11 ++- quantammsim/pools/G3M/balancer/balancer.py | 40 +++------ .../pools/G3M/balancer/balancer_reserves.py | 30 +++---- .../pools/G3M/quantamm/TFMM_base_pool.py | 54 ++++-------- .../pools/G3M/quantamm/quantamm_reserves.py | 50 ++++++----- quantammsim/pools/reCLAMM/reclamm.py | 53 +++++------ quantammsim/runners/jax_runner_utils.py | 22 ++++- quantammsim/runners/jax_runners.py | 87 +++++++------------ 15 files changed, 289 insertions(+), 299 deletions(-) diff --git a/quantammsim/core_simulator/dynamic_inputs.py b/quantammsim/core_simulator/dynamic_inputs.py index b1eedacf..c2490888 100644 --- a/quantammsim/core_simulator/dynamic_inputs.py +++ b/quantammsim/core_simulator/dynamic_inputs.py @@ -16,9 +16,9 @@ class DynamicInputFrames: class DynamicInputArrays(NamedTuple): - """Fixed-structure JAX pytree for dynamic simulation inputs.""" + """JAX pytree for dynamic simulation inputs with optional trade data.""" - trades: jnp.ndarray + trades: Optional[jnp.ndarray] fees: jnp.ndarray gas_cost: jnp.ndarray arb_fees: jnp.ndarray @@ -70,9 +70,9 @@ def resolve_dynamic_input_flags( def empty_dynamic_input_arrays() -> DynamicInputArrays: - """Create a canonical empty bundle with stable pytree structure.""" + """Create a canonical empty bundle.""" return DynamicInputArrays( - trades=jnp.zeros((1, 3), dtype=jnp.float64), + trades=None, fees=jnp.zeros((1,), dtype=jnp.float64), gas_cost=jnp.zeros((1,), dtype=jnp.float64), arb_fees=jnp.zeros((1,), dtype=jnp.float64), @@ -110,3 +110,68 @@ def resolve_dynamic_input_components( else jnp.ones((1,), dtype=jnp.float64) ), } + + +def _broadcast_dynamic_input_leaf( + input_name: str, + values: jnp.ndarray, + scan_len: int, + dtype, +) -> jnp.ndarray: + """Broadcast a singleton dynamic-input leaf to the scan length.""" + values = jnp.asarray(values, dtype=dtype) + if values.ndim == 0: + values = values.reshape((1,)) + if values.shape[0] == scan_len: + return values + if values.shape[0] == 1: + return jnp.broadcast_to(values, (scan_len,) + values.shape[1:]) + raise ValueError( + f"{input_name} has leading axis {values.shape[0]}, expected 1 or {scan_len}" + ) + + +def materialize_dynamic_inputs( + dynamic_inputs: Optional[DynamicInputArrays], + dynamic_input_flags: Optional[dict], + static_dict: dict, + scan_len: int, + do_trades: bool, + dtype=jnp.float64, +) -> DynamicInputArrays: + """Resolve and broadcast dynamic inputs for a specific scan length.""" + if dynamic_input_flags is None and dynamic_inputs is not None: + flags = { + "use_dynamic_inputs": True, + "has_trades": do_trades, + "has_dynamic_fees": True, + "has_dynamic_gas_cost": True, + "has_dynamic_arb_fees": True, + "has_lp_supply": True, + } + else: + flags = resolve_dynamic_input_flags(dynamic_inputs, dynamic_input_flags) + + resolved = resolve_dynamic_input_components(dynamic_inputs, flags, static_dict) + + trades = None + if do_trades: + if resolved["trades"] is None: + raise ValueError("Trades must be provided when do_trades=True.") + trades = _broadcast_dynamic_input_leaf( + "trades", resolved["trades"], scan_len, dtype + ) + + return DynamicInputArrays( + trades=trades, + fees=_broadcast_dynamic_input_leaf("fees", resolved["fees"], scan_len, dtype), + gas_cost=_broadcast_dynamic_input_leaf( + "gas_cost", resolved["gas_cost"], scan_len, dtype + ), + arb_fees=_broadcast_dynamic_input_leaf( + "arb_fees", resolved["arb_fees"], scan_len, dtype + ), + lp_supply=_broadcast_dynamic_input_leaf( + "lp_supply", resolved["lp_supply"], scan_len, dtype + ), + ) diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index 4dd74f62..d2a32cd8 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -57,7 +57,6 @@ from quantammsim.core_simulator.dynamic_inputs import ( DynamicInputArrays, default_dynamic_input_flags, - empty_dynamic_input_arrays, resolve_dynamic_input_flags, ) @@ -66,13 +65,11 @@ def _resolve_dynamic_inputs(dynamic_inputs, static_dict): - """Return a stable hot-path bundle plus static dispatch flags.""" + """Return the incoming bundle plus static dispatch flags.""" dynamic_input_flags = resolve_dynamic_input_flags( dynamic_inputs, static_dict.get("dynamic_input_flags"), ) - if dynamic_inputs is None: - dynamic_inputs = empty_dynamic_input_arrays() return dynamic_inputs, dynamic_input_flags @@ -1107,7 +1104,15 @@ def forward_pass_nograd( prices = stop_gradient(prices) if dynamic_inputs is not None: dynamic_inputs = DynamicInputArrays( - *(stop_gradient(arr) for arr in dynamic_inputs) + trades=( + None + if dynamic_inputs.trades is None + else stop_gradient(dynamic_inputs.trades) + ), + fees=stop_gradient(dynamic_inputs.fees), + gas_cost=stop_gradient(dynamic_inputs.gas_cost), + arb_fees=stop_gradient(dynamic_inputs.arb_fees), + lp_supply=stop_gradient(dynamic_inputs.lp_supply), ) return forward_pass( params, diff --git a/quantammsim/core_simulator/windowing_utils.py b/quantammsim/core_simulator/windowing_utils.py index 28d75244..c6cde8cf 100644 --- a/quantammsim/core_simulator/windowing_utils.py +++ b/quantammsim/core_simulator/windowing_utils.py @@ -206,11 +206,12 @@ def raw_fee_like_amounts_to_fee_like_array( ).astype(int) // 10**6 )[:-1] + fill_value = np.nan if fill_method == "ffill" else 0.0 full_index_df = pd.DataFrame( - index=full_index, - columns=names, - data=0, - dtype=np.float64 + index=full_index, + columns=names, + data=fill_value, + dtype=np.float64, ) # Map raw data to the full index DataFrame @@ -236,15 +237,16 @@ def raw_fee_like_amounts_to_fee_like_array( # Ensure unix values are valid valid_unix = pd.to_numeric(raw_inputs['unix'], errors='coerce') valid_mask = valid_unix.notna() + valid_inputs = raw_inputs.loc[valid_mask].copy() + valid_inputs["unix"] = valid_unix.loc[valid_mask].astype(np.int64) + valid_inputs = valid_inputs.sort_values("unix") for name in names: initial_value = None - if valid_mask.any(): + if not valid_inputs.empty: # Try to get the last value before our start date - previous_values = raw_inputs[ - valid_mask & (valid_unix < start_unix) - ] + previous_values = valid_inputs[valid_inputs["unix"] < start_unix] if not previous_values.empty: try: @@ -254,9 +256,7 @@ def raw_fee_like_amounts_to_fee_like_array( if initial_value is None or pd.isna(initial_value): # Try to get first value in our date range - in_range_values = raw_inputs[ - valid_mask & (valid_unix >= start_unix) - ] + in_range_values = valid_inputs[valid_inputs["unix"] >= start_unix] if not in_range_values.empty: try: initial_value = pd.to_numeric(in_range_values[name].iloc[0]) @@ -264,17 +264,12 @@ def raw_fee_like_amounts_to_fee_like_array( initial_value = None if initial_value is not None and pd.notna(initial_value): - # this more complex logic is because of how we have started with prior-to-start values - # filled in, and then we want to ffill the rest - # Fill initial values - full_index_df[name] = full_index_df[name].mask( - full_index_df[name] == 0, - initial_value - ) - # Use ffill() - full_index_df[name] = full_index_df[name].where( - full_index_df[name] != 0 - ).ffill() + # Seed only the leading gap; explicit in-range updates must remain intact. + first_row = full_index_df.index[0] + if pd.isna(full_index_df.at[first_row, name]): + full_index_df.at[first_row, name] = initial_value + + full_index_df[name] = full_index_df[name].ffill().fillna(0.0) except (ValueError, KeyError, TypeError) as e: print(f"Warning: Error during ffill processing: {str(e)}") # On any error, return the original zero-filled DataFrame @@ -387,4 +382,4 @@ def filter_reserves_by_given_timestamp(reserves, unix_values, timestamp): unix_values == timestamp )[0][0] - return reserves[reserves_index].copy() \ No newline at end of file + return reserves[reserves_index].copy() diff --git a/quantammsim/hooks/dynamic_fee_base_hook.py b/quantammsim/hooks/dynamic_fee_base_hook.py index 9e00af64..ad64a5f5 100644 --- a/quantammsim/hooks/dynamic_fee_base_hook.py +++ b/quantammsim/hooks/dynamic_fee_base_hook.py @@ -119,7 +119,7 @@ def calculate_reserves_with_fees( dynamic_fees = raw_dynamic_fees.repeat(chunk_period, axis=0).squeeze() empty_inputs = empty_dynamic_input_arrays() dynamic_inputs = DynamicInputArrays( - trades=empty_inputs.trades, + trades=None, fees=dynamic_fees, gas_cost=jnp.asarray(run_fingerprint["gas_cost"], dtype=jnp.float64), arb_fees=jnp.asarray(run_fingerprint["arb_fees"], dtype=jnp.float64), diff --git a/quantammsim/pools/ECLP/gyroscope.py b/quantammsim/pools/ECLP/gyroscope.py index 5aebcef5..83a7dc28 100644 --- a/quantammsim/pools/ECLP/gyroscope.py +++ b/quantammsim/pools/ECLP/gyroscope.py @@ -32,6 +32,7 @@ from typing import Dict, Any, Optional, Tuple import numpy as np +from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool from quantammsim.pools.ECLP.gyroscope_reserves import ( @@ -342,11 +343,6 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - trade_array = dynamic_inputs.trades - # calculate initial reserves initial_pool_value = run_fingerprint["initial_pool_value"] initial_reserves = initialise_gyroscope_reserves_given_value( @@ -358,34 +354,27 @@ def calculate_reserves_with_dynamic_inputs( sin=jnp.sin(phi), cos=jnp.cos(phi), ) - # any of fees_array, arb_thresh_array, arb_fees_array, trade_array - # can be singletons, in which case we repeat them for the length of the bout - - # Determine the maximum leading dimension max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=run_fingerprint["do_trades"], + dtype=arb_acted_upon_local_prices.dtype, + ) # Handle trade array reordering if needed if run_fingerprint["do_trades"]: - # if we are doing trades, the trades array must be of the same length as the other arrays - assert trade_array.shape[0] == max_len if needs_swap: # Swap trade indices (0->1, 1->0) but keep amounts unchanged - trade_array = trade_array.at[:, :2].set(1 - trade_array[:, :2]) - - # Broadcast input arrays to match the maximum leading dimension. - # If they are singletons, this will just repeat them for the length of the bout. - # If they are arrays of length bout_length, this will cause no change. - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] - ) + materialized_inputs = materialized_inputs._replace( + trades=materialized_inputs.trades.at[:, :2].set( + 1 - materialized_inputs.trades[:, :2] + ) + ) # Calculate reserves reserves = _jax_calc_gyroscope_reserves_with_dynamic_inputs( @@ -396,10 +385,10 @@ def calculate_reserves_with_dynamic_inputs( sin=jnp.sin(phi), cos=jnp.cos(phi), lam=lam, - fees=fees_array_broadcast, - arb_thresh=arb_thresh_array_broadcast, - arb_fees=arb_fees_array_broadcast, - trades=trade_array, + fees=materialized_inputs.fees, + arb_thresh=materialized_inputs.gas_cost, + arb_fees=materialized_inputs.arb_fees, + trades=materialized_inputs.trades, do_trades=run_fingerprint["do_trades"], ) # Restore original order if we swapped diff --git a/quantammsim/pools/ECLP/gyroscope_reserves.py b/quantammsim/pools/ECLP/gyroscope_reserves.py index da09072e..68073109 100644 --- a/quantammsim/pools/ECLP/gyroscope_reserves.py +++ b/quantammsim/pools/ECLP/gyroscope_reserves.py @@ -605,7 +605,7 @@ def _jax_calc_gyroscope_reserves_with_dynamic_fees_and_trades_scan_function_usin gamma = input_list[1] arb_thresh = input_list[2] arb_fees = input_list[3] - trade = input_list[4] + trade = input_list[4] if do_trades else None @@ -727,6 +727,8 @@ def _jax_calc_gyroscope_reserves_with_dynamic_inputs( arb_fees = jnp.where( arb_fees.size == 1, jnp.full(prices.shape[0], arb_fees), arb_fees ) + if do_trades and trades is None: + raise ValueError("Trades must be provided when do_trades=True.") scan_fn = Partial( _jax_calc_gyroscope_reserves_with_dynamic_fees_and_trades_scan_function_using_precalcs, @@ -745,17 +747,15 @@ def _jax_calc_gyroscope_reserves_with_dynamic_inputs( initial_reserves, 0 ] - carry_list_end, reserves = scan( - scan_fn, - carry_list_init, - [ - prices, - gamma, - arb_thresh, - arb_fees, - trades, - ], - ) + scan_inputs = [ + prices, + gamma, + arb_thresh, + arb_fees, + ] + if do_trades: + scan_inputs.append(trades) + carry_list_end, reserves = scan(scan_fn, carry_list_init, scan_inputs) return reserves diff --git a/quantammsim/pools/FM_AMM/cow_pool.py b/quantammsim/pools/FM_AMM/cow_pool.py index 96299ce8..b581c616 100644 --- a/quantammsim/pools/FM_AMM/cow_pool.py +++ b/quantammsim/pools/FM_AMM/cow_pool.py @@ -32,6 +32,7 @@ from functools import partial import numpy as np +from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool from quantammsim.pools.FM_AMM.cow_reserves import ( _jax_calc_cowamm_reserves_with_fees, @@ -211,47 +212,31 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - trade_array = dynamic_inputs.trades - initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = weights * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] - # any of fees_array, arb_thresh_array, arb_fees_array, trade_array - # can be singletons, in which case we repeat them for the length of the bout - - # Determine the maximum leading dimension max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - # Broadcast input arrays to match the maximum leading dimension. - # If they are singletons, this will just repeat them for the length of the bout. - # If they are arrays of length bout_length, this will cause no change. - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=run_fingerprint["do_trades"], + dtype=arb_acted_upon_local_prices.dtype, ) - # if we are doing trades, the trades array must be of the same length as the other arrays - if run_fingerprint["do_trades"]: - assert trade_array.shape[0] == max_len reserves = _jax_calc_cowamm_reserves_with_dynamic_inputs( initial_reserves, arb_acted_upon_local_prices, - fees_array_broadcast, - arb_thresh_array_broadcast, - arb_fees_array_broadcast, + materialized_inputs.fees, + materialized_inputs.gas_cost, + materialized_inputs.arb_fees, weights, run_fingerprint["arb_quality"], - trade_array, + materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], noise_trader_ratio=run_fingerprint["noise_trader_ratio"], diff --git a/quantammsim/pools/FM_AMM/cow_reserves.py b/quantammsim/pools/FM_AMM/cow_reserves.py index f876d51f..ab340a09 100644 --- a/quantammsim/pools/FM_AMM/cow_reserves.py +++ b/quantammsim/pools/FM_AMM/cow_reserves.py @@ -729,7 +729,7 @@ def _jax_calc_cowamm_reserves_with_dynamic_fees_and_trades_scan_function( gamma = input_list[1] arb_thresh = input_list[2] arb_fees = input_list[3] - trade = input_list[4] + trade = input_list[4] if do_trades else None if do_arb: reserves_with_perfect_arb = _jax_calc_cowamm_reserves_with_fees_scan_function( @@ -827,6 +827,8 @@ def _jax_calc_cowamm_reserves_with_dynamic_inputs( initial_prices = prices[0] gamma = 1.0 - fees + if do_trades and trades is None: + raise ValueError("Trades must be provided when do_trades=True.") scan_fn = Partial( _jax_calc_cowamm_reserves_with_dynamic_fees_and_trades_scan_function, @@ -838,8 +840,9 @@ def _jax_calc_cowamm_reserves_with_dynamic_inputs( ) carry_list_init = [initial_prices, initial_reserves] - _, reserves = scan( - scan_fn, carry_list_init, [prices, gamma, arb_thresh, arb_fees, trades] - ) + scan_inputs = [prices, gamma, arb_thresh, arb_fees] + if do_trades: + scan_inputs.append(trades) + _, reserves = scan(scan_fn, carry_list_init, scan_inputs) return reserves diff --git a/quantammsim/pools/G3M/balancer/balancer.py b/quantammsim/pools/G3M/balancer/balancer.py index 986d49c3..49bd5cab 100644 --- a/quantammsim/pools/G3M/balancer/balancer.py +++ b/quantammsim/pools/G3M/balancer/balancer.py @@ -8,6 +8,7 @@ import jax.numpy as jnp from jax.lax import dynamic_slice +from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool from quantammsim.pools.G3M.balancer.balancer_reserves import ( _jax_calc_balancer_reserve_ratios, @@ -306,47 +307,30 @@ def calculate_reserves_with_dynamic_inputs( else: arb_acted_upon_local_prices = local_prices - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - trade_array = dynamic_inputs.trades - lp_supply_array = dynamic_inputs.lp_supply - initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = weights * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] - # any of fees_array, arb_thresh_array, arb_fees_array, trade_array - # can be singletons, in which case we repeat them for the length of the bout - - # Determine the maximum leading dimension max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - # Broadcast input arrays to match the maximum leading dimension. - # If they are singletons, this will just repeat them for the length of the bout. - # If they are arrays of length bout_length, this will cause no change. - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=run_fingerprint["do_trades"], + dtype=arb_acted_upon_local_prices.dtype, ) - # if we are doing trades, the trades array must be of the same length as the other arrays - if run_fingerprint["do_trades"]: - assert trade_array.shape[0] == max_len reserves = _jax_calc_balancer_reserves_with_dynamic_inputs( initial_reserves, weights, arb_acted_upon_local_prices, - fees_array_broadcast, - arb_thresh_array_broadcast, - arb_fees_array_broadcast, + materialized_inputs.fees, + materialized_inputs.gas_cost, + materialized_inputs.arb_fees, jnp.array(run_fingerprint["all_sig_variations"]), - trade_array, + materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], ) diff --git a/quantammsim/pools/G3M/balancer/balancer_reserves.py b/quantammsim/pools/G3M/balancer/balancer_reserves.py index 240ae6aa..d6fd68cc 100644 --- a/quantammsim/pools/G3M/balancer/balancer_reserves.py +++ b/quantammsim/pools/G3M/balancer/balancer_reserves.py @@ -364,7 +364,7 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using gamma = input_list[4] arb_thresh = input_list[5] arb_fees = input_list[6] - trade = input_list[7] + trade = input_list[7] if do_trades else None fees_are_being_charged = gamma != 1.0 @@ -499,6 +499,8 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( arb_fees = jnp.where( arb_fees.size == 1, jnp.full(prices.shape[0], arb_fees), arb_fees ) + if do_trades and trades is None: + raise ValueError("Trades must be provided when do_trades=True.") # pre-calculate some values that are repeatedly used in optimal arb calculations _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( @@ -533,19 +535,17 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( initial_reserves, 0, ] - _, reserves = scan( - scan_fn, - carry_list_init, - [ - prices, - active_initial_weights, - per_asset_ratios, - all_other_assets_ratios, - gamma, - arb_thresh, - arb_fees, - trades, - ], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + ] + if do_trades: + scan_inputs.append(trades) + _, reserves = scan(scan_fn, carry_list_init, scan_inputs) return reserves diff --git a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py index b087b5a8..d5a1654a 100644 --- a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py +++ b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py @@ -20,6 +20,7 @@ from jax.lax import dynamic_slice, scan, fori_loop from jax.tree_util import Partial +from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool from quantammsim.pools.G3M.quantamm.quantamm_reserves import ( _jax_calc_quantAMM_reserve_ratios, @@ -274,59 +275,35 @@ def calculate_reserves_with_dynamic_inputs( arb_acted_upon_weights = weights arb_acted_upon_local_prices = local_prices - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - trade_array = dynamic_inputs.trades - lp_supply_array = dynamic_inputs.lp_supply - initial_pool_value = run_fingerprint["initial_pool_value"] initial_value_per_token = arb_acted_upon_weights[0] * initial_pool_value initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] - # any of fees_array, arb_thresh_array, arb_fees_array, trade_array, and lp_supply_array - # can be singletons, in which case we repeat them for the length of the bout. - - # Determine the maximum leading dimension max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - # Broadcast input arrays to match the maximum leading dimension. - # If they are singletons, this will just repeat them for the length of the bout. - # If they are arrays of length bout_length, this will cause no change. - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] - ) - # if lp_supply_array is not provided, we set it to a constant of 1.0 - if lp_supply_array is None: - lp_supply_array = jnp.array(1.0) - - lp_supply_array_broadcast = jnp.broadcast_to( - lp_supply_array, (max_len,) + lp_supply_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=run_fingerprint["do_trades"], + dtype=arb_acted_upon_local_prices.dtype, ) - # if we are doing trades, the trades array must be of the same length as the other arrays - if run_fingerprint["do_trades"]: - assert trade_array.shape[0] == max_len protocol_fee_split = run_fingerprint.get("protocol_fee_split", 0.0) reserves = _jax_calc_quantAMM_reserves_with_dynamic_inputs( initial_reserves, arb_acted_upon_weights, arb_acted_upon_local_prices, - fees_array_broadcast, - arb_thresh_array_broadcast, - arb_fees_array_broadcast, + materialized_inputs.fees, + materialized_inputs.gas_cost, + materialized_inputs.arb_fees, jnp.array(run_fingerprint["all_sig_variations"]), - trade_array, + materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], run_fingerprint["noise_trader_ratio"], - lp_supply_array_broadcast, + materialized_inputs.lp_supply, protocol_fee_split=protocol_fee_split, ) return reserves @@ -1461,11 +1438,16 @@ def calculate_weights_direct( initial_weights, minimum_weight, params, + jnp.zeros_like(initial_weights), + jnp.ones_like(initial_weights), local_fingerprint["max_memory_days"], local_fingerprint["chunk_period"], local_fingerprint["weight_interpolation_period"], maximum_change, False, + False, + False, + False, ) return target_weights_cpu diff --git a/quantammsim/pools/G3M/quantamm/quantamm_reserves.py b/quantammsim/pools/G3M/quantamm/quantamm_reserves.py index 351a7886..77bbe11e 100644 --- a/quantammsim/pools/G3M/quantamm/quantamm_reserves.py +++ b/quantammsim/pools/G3M/quantamm/quantamm_reserves.py @@ -545,9 +545,14 @@ def _jax_calc_quantAMM_reserves_with_dynamic_fees_and_trades_scan_function_using gamma = input_list[8] arb_thresh = input_list[9] arb_fees = input_list[10] - trade = input_list[11] - do_arb = input_list[12] - lp_supply = input_list[13] + if do_trades: + trade = input_list[11] + do_arb = input_list[12] + lp_supply = input_list[13] + else: + trade = None + do_arb = input_list[11] + lp_supply = input_list[12] fees_are_being_charged = gamma != 1.0 protocol_fee_amount_step = jnp.zeros_like(prev_reserves) @@ -831,6 +836,8 @@ def _jax_calc_quantAMM_reserves_with_dynamic_inputs( arb_fees = jnp.where( arb_fees.size == 1, jnp.full(weights.shape[0], arb_fees), arb_fees ) + if do_trades and trades is None: + raise ValueError("Trades must be provided when do_trades=True.") if lp_supply_array is None: lp_supply_array = jnp.array(1.0) @@ -904,25 +911,22 @@ def _jax_calc_quantAMM_reserves_with_dynamic_inputs( ] # carry_list_init = [initial_weights, initial_i] # nojit_scan = jax.disable_jit()(jax.lax.scan) - carry_list_end, reserves = scan( - scan_fn, - carry_list_init, - [ - weights, - prices, - active_initial_weights, - per_asset_ratios, - all_other_assets_ratios, - lagged_active_initial_weights, - lagged_per_asset_ratios, - lagged_all_other_assets_ratios, - gamma, - arb_thresh, - arb_fees, - trades, - do_arb, - lp_supply_array, - ], - ) + scan_inputs = [ + weights, + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + lagged_active_initial_weights, + lagged_per_asset_ratios, + lagged_all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + ] + if do_trades: + scan_inputs.append(trades) + scan_inputs.extend([do_arb, lp_supply_array]) + carry_list_end, reserves = scan(scan_fn, carry_list_init, scan_inputs) return reserves diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index c73b6c4c..da36710b 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -16,6 +16,7 @@ from typing import Dict, Any, Optional, NamedTuple import numpy as np +from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool from quantammsim.pools.reCLAMM.reclamm_reserves import ( initialise_reclamm_reserves, @@ -287,23 +288,17 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( LP fee revenue per timestep in USD. """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=False, + dtype=s.arb_prices.dtype, ) return _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( @@ -312,9 +307,9 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( s.centeredness_margin, s.daily_price_shift_base, s.seconds_per_step, - fees=fees_array_broadcast, - arb_thresh=arb_thresh_array_broadcast, - arb_fees=arb_fees_array_broadcast, + fees=materialized_inputs.fees, + arb_thresh=materialized_inputs.gas_cost, + arb_fees=materialized_inputs.arb_fees, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), @@ -370,23 +365,17 @@ def calculate_reserves_with_dynamic_inputs( additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) - fees_array = dynamic_inputs.fees - arb_thresh_array = dynamic_inputs.gas_cost - arb_fees_array = dynamic_inputs.arb_fees - bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + run_fingerprint.get("dynamic_input_flags"), + run_fingerprint, + scan_len=max_len, + do_trades=False, + dtype=s.arb_prices.dtype, ) return _jax_calc_reclamm_reserves_with_dynamic_inputs( @@ -395,9 +384,9 @@ def calculate_reserves_with_dynamic_inputs( s.centeredness_margin, s.daily_price_shift_base, s.seconds_per_step, - fees=fees_array_broadcast, - arb_thresh=arb_thresh_array_broadcast, - arb_fees=arb_fees_array_broadcast, + fees=materialized_inputs.fees, + arb_thresh=materialized_inputs.gas_cost, + arb_fees=materialized_inputs.arb_fees, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index 64c916dd..e9c74f9f 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -1060,8 +1060,9 @@ def get_unique_tokens(run_fingerprint): >>> get_unique_tokens(fingerprint) ['BTC', 'DAI', 'ETH'] """ + subsidary_pools = run_fingerprint.get("subsidary_pools", []) all_tokens = [run_fingerprint["tokens"]] + [ - cprd["tokens"] for cprd in run_fingerprint["subsidary_pools"] + cprd["tokens"] for cprd in subsidary_pools ] all_tokens = [item for sublist in all_tokens for item in sublist] unique_tokens = list(set(all_tokens)) @@ -1228,10 +1229,10 @@ def _to_dynamic_input_arrays( arb_fees_array, lp_supply_array, ) -> DynamicInputArrays: - """Normalize optional numpy arrays into the fixed hot-path container.""" + """Normalize optional numpy arrays into the hot-path container.""" empty = empty_dynamic_input_arrays() return DynamicInputArrays( - trades=empty.trades if trades_array is None else jnp.asarray(trades_array, dtype=jnp.float64), + trades=None if trades_array is None else jnp.asarray(trades_array, dtype=jnp.float64), fees=empty.fees if fees_array is None else jnp.asarray(fees_array, dtype=jnp.float64), gas_cost=empty.gas_cost if gas_cost_array is None else jnp.asarray(gas_cost_array, dtype=jnp.float64), arb_fees=empty.arb_fees if arb_fees_array is None else jnp.asarray(arb_fees_array, dtype=jnp.float64), @@ -1244,7 +1245,7 @@ def prepare_dynamic_inputs( dynamic_input_frames: Optional[DynamicInputFrames] = None, do_test_period: bool = False, ): - """Convert optional pandas inputs into fixed-structure dynamic input bundles.""" + """Convert optional pandas inputs into dynamic input bundles.""" if dynamic_input_frames is None: dynamic_input_frames = DynamicInputFrames() @@ -1367,6 +1368,19 @@ def prepare_dynamic_inputs( if lp_supply_df is not None else None ) + + # Unit LP supply is the neutral case; keep it on the static hot path. + if lp_supply_array is not None and np.allclose(lp_supply_array, 1.0): + lp_supply_array = None + if not do_test_period or test_lp_supply_array is None or np.allclose(test_lp_supply_array, 1.0): + dynamic_input_flags["has_lp_supply"] = False + dynamic_input_flags["use_dynamic_inputs"] = any( + value for key, value in dynamic_input_flags.items() if key != "use_dynamic_inputs" + ) + + if do_test_period and test_lp_supply_array is not None and np.allclose(test_lp_supply_array, 1.0): + test_lp_supply_array = None + if do_test_period: return { "train_dynamic_inputs": _to_dynamic_input_arrays( train_period_trades, diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index 1a8fdf3b..aad11bdd 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -54,7 +54,7 @@ ) from quantammsim.core_simulator.dynamic_inputs import ( DynamicInputFrames, - resolve_dynamic_input_components, + materialize_dynamic_inputs, ) from quantammsim.core_simulator.windowing_utils import get_indices, filter_coarse_weights_by_data_indices @@ -160,7 +160,10 @@ def _build_scan_infrastructure( run_scan_chunk : callable ``@jit`` wrapped ``lax.scan(scan_body, carry, None, length=chunk_size)``. scan_body : callable - The raw scan body (for partial-chunk Python fallback). + The raw scan body. + run_scan_step : callable + ``@jit`` wrapped single-step execution used for remainder iterations so + partial chunks follow the same numerics as the full scan path. """ # Local aliases for closed-over constants _start_idx = start_idx @@ -310,7 +313,11 @@ def scan_body(carry, _): def _run_scan_chunk(carry): return lax.scan(scan_body, carry, None, length=chunk_size) - return _run_scan_chunk, scan_body + @jit + def _run_scan_step(carry): + return scan_body(carry, None) + + return _run_scan_chunk, scan_body, _run_scan_step def train_on_historic_data( @@ -806,7 +813,7 @@ def init_optimizer(params): ) if config_key in _scan_infra_cache: - _run_scan_chunk, scan_body = _scan_infra_cache[config_key] + _run_scan_chunk, scan_body, _run_scan_step = _scan_infra_cache[config_key] else: # Build scan-compatible update (prices as explicit arg, not closure) partial_step_no_prices = Partial( @@ -825,7 +832,7 @@ def init_optimizer(params): partial_step_no_prices, params_in_axes_dict, ) - _run_scan_chunk, scan_body = _build_scan_infrastructure( + _run_scan_chunk, scan_body, _run_scan_step = _build_scan_infrastructure( chunk_size, partial_step_no_prices=partial_step_no_prices, forward_nograd_continuous=partial_forward_pass_nograd_continuous, @@ -851,7 +858,7 @@ def init_optimizer(params): swa_freq=swa_freq, n_parameter_sets=n_parameter_sets, ) - _scan_infra_cache[config_key] = (_run_scan_chunk, scan_body) + _scan_infra_cache[config_key] = (_run_scan_chunk, scan_body, _run_scan_step) # ── Initialize carry (prices & nan_bank in carry, not closures) ── carry = { @@ -906,7 +913,7 @@ def init_optimizer(params): "params": {k: [] for k in carry["params"]}, } for _ in range(actual): - carry, step_out = scan_body(carry, None) + carry, step_out = _run_scan_step(carry) all_per_steps["objective"].append(step_out["objective"]) all_per_steps["train_metrics"].append(step_out["train_metrics"]) all_per_steps["test_metrics"].append(step_out["test_metrics"]) @@ -2219,11 +2226,16 @@ def do_run_on_historic_data_with_provided_coarse_weights( initial_weights, minimum_weight, params, + jnp.zeros_like(initial_weights), + jnp.ones_like(initial_weights), run_fingerprint["max_memory_days"], chunk_period, chunk_period, 1.0, False, + False, + False, + False, ) weights = _jax_fine_weights_from_actual_starts_and_diffs( @@ -2257,70 +2269,33 @@ def do_run_on_historic_data_with_provided_coarse_weights( # ) dynamic_input_flags = dynamic_inputs_dict["dynamic_input_flags"] dynamic_inputs = dynamic_inputs_dict["train_dynamic_inputs"] - resolved_dynamic_inputs = resolve_dynamic_input_components( - dynamic_inputs, - dynamic_input_flags, - static_dict, - ) - fees_array = resolved_dynamic_inputs["fees"] - arb_thresh_array = resolved_dynamic_inputs["gas_cost"] - arb_fees_array = resolved_dynamic_inputs["arb_fees"] - trade_array = resolved_dynamic_inputs["trades"] - lp_supply_array = resolved_dynamic_inputs["lp_supply"] - - # initial_pool_value = run_fingerprint["initial_pool_value"] - # initial_value_per_token = arb_acted_upon_weights[0] * initial_pool_value - # initial_reserves = initial_value_per_token / arb_acted_upon_local_prices[0] - initial_reserves = params["initial_reserves"] - - # any of fees_array, arb_thresh_array, arb_fees_array, trade_array, and lp_supply_array - # can be singletons, in which case we repeat them for the length of the bout. - - # Determine the maximum leading dimension max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: max_len = max_len // run_fingerprint["arb_frequency"] - - fees_array = fees_array[:max_len] - arb_thresh_array = arb_thresh_array[:max_len] - arb_fees_array = arb_fees_array[:max_len] - lp_supply_array = lp_supply_array[:max_len] - if trade_array is not None: - trade_array = trade_array[:max_len] - # Broadcast input arrays to match the maximum leading dimension. - # If they are singletons, this will just repeat them for the length of the bout. - # If they are arrays of length bout_length, this will cause no change. - fees_array_broadcast = jnp.broadcast_to( - fees_array, (max_len,) + fees_array.shape[1:] - ) - arb_thresh_array_broadcast = jnp.broadcast_to( - arb_thresh_array, (max_len,) + arb_thresh_array.shape[1:] - ) - arb_fees_array_broadcast = jnp.broadcast_to( - arb_fees_array, (max_len,) + arb_fees_array.shape[1:] - ) - lp_supply_array_broadcast = jnp.broadcast_to( - lp_supply_array, (max_len,) + lp_supply_array.shape[1:] + materialized_inputs = materialize_dynamic_inputs( + dynamic_inputs, + dynamic_input_flags, + static_dict, + scan_len=max_len, + do_trades=run_fingerprint["do_trades"], + dtype=local_prices.dtype, ) - # if we are doing trades, the trades array must be of the same length as the other arrays - if run_fingerprint["do_trades"]: - assert trade_array.shape[0] == max_len protocol_fee_split = run_fingerprint.get("protocol_fee_split", 0.0) reserves = _jax_calc_quantAMM_reserves_with_dynamic_inputs( initial_reserves, weights, local_prices, - fees_array_broadcast, - arb_thresh_array_broadcast, - arb_fees_array_broadcast, + materialized_inputs.fees, + materialized_inputs.gas_cost, + materialized_inputs.arb_fees, jnp.array(static_dict["all_sig_variations"]), - None, + materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], run_fingerprint["noise_trader_ratio"], - lp_supply_array_broadcast, + materialized_inputs.lp_supply, protocol_fee_split=protocol_fee_split, ) From 42d07679efd15ae106048cfcb5520802680841e5 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Wed, 4 Mar 2026 16:28:31 +0000 Subject: [PATCH 005/115] missing file commit --- tests/unit/test_jax_runner_utils.py | 55 +++++++++++++++++++++++++++-- 1 file changed, 53 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_jax_runner_utils.py b/tests/unit/test_jax_runner_utils.py index 70dbf828..620ecb84 100644 --- a/tests/unit/test_jax_runner_utils.py +++ b/tests/unit/test_jax_runner_utils.py @@ -257,12 +257,12 @@ class TestDynamicInputPreparation: """Tests for dynamic input container construction and normalization.""" def test_empty_dynamic_input_arrays_have_stable_shapes(self): - """The empty hot-path bundle should have canonical placeholder arrays.""" + """The empty hot-path bundle should use singleton fee-like placeholders only.""" from quantammsim.core_simulator.dynamic_inputs import empty_dynamic_input_arrays dynamic_inputs = empty_dynamic_input_arrays() - assert dynamic_inputs.trades.shape == (1, 3) + assert dynamic_inputs.trades is None assert dynamic_inputs.fees.shape == (1,) assert dynamic_inputs.gas_cost.shape == (1,) assert dynamic_inputs.arb_fees.shape == (1,) @@ -472,6 +472,57 @@ def test_resolve_dynamic_input_components_prefers_dynamic_values(self): np.testing.assert_allclose(np.asarray(resolved["arb_fees"]), np.array([0.0003])) np.testing.assert_allclose(np.asarray(resolved["lp_supply"]), np.array([1500.0])) + def test_materialize_dynamic_inputs_leaves_trades_optional(self): + """No-trade paths should not expand placeholder trades into the scan inputs.""" + from quantammsim.core_simulator.dynamic_inputs import ( + empty_dynamic_input_arrays, + materialize_dynamic_inputs, + ) + + materialized = materialize_dynamic_inputs( + empty_dynamic_input_arrays(), + { + "use_dynamic_inputs": True, + "has_trades": False, + "has_dynamic_fees": False, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + }, + static_dict={"fees": 0.003, "gas_cost": 2.5, "arb_fees": 0.0001}, + scan_len=4, + do_trades=False, + ) + + assert materialized.trades is None + np.testing.assert_allclose(np.asarray(materialized.fees), np.full(4, 0.003)) + np.testing.assert_allclose(np.asarray(materialized.gas_cost), np.full(4, 2.5)) + np.testing.assert_allclose(np.asarray(materialized.arb_fees), np.full(4, 0.0001)) + np.testing.assert_allclose(np.asarray(materialized.lp_supply), np.ones(4)) + + def test_materialize_dynamic_inputs_requires_trades_when_enabled(self): + """Trade-enabled scans should fail fast if no trade path is available.""" + from quantammsim.core_simulator.dynamic_inputs import ( + empty_dynamic_input_arrays, + materialize_dynamic_inputs, + ) + + with pytest.raises(ValueError, match="Trades must be provided"): + materialize_dynamic_inputs( + empty_dynamic_input_arrays(), + { + "use_dynamic_inputs": True, + "has_trades": False, + "has_dynamic_fees": True, + "has_dynamic_gas_cost": False, + "has_dynamic_arb_fees": False, + "has_lp_supply": False, + }, + static_dict={"fees": 0.003, "gas_cost": 0.0, "arb_fees": 0.0}, + scan_len=2, + do_trades=True, + ) + class TestGetSigVariations: """Tests for get_sig_variations function.""" From 19ee13e0de06f5ff14e1ed8a0c0f73ee7473bd54 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Thu, 5 Mar 2026 14:40:36 +0000 Subject: [PATCH 006/115] initial implementation --- quantammsim/core_simulator/dynamic_inputs.py | 30 +- quantammsim/core_simulator/forward_pass.py | 3 + quantammsim/hooks/dynamic_fee_base_hook.py | 1 + quantammsim/pools/reCLAMM/reclamm.py | 2 + quantammsim/pools/reCLAMM/reclamm_reserves.py | 360 +++++++++++++++++- quantammsim/runners/jax_runner_utils.py | 198 ++++++++++ quantammsim/runners/jax_runners.py | 1 + .../finance/param_financial_calculator.py | 3 + tests/pools/reCLAMM/helpers.py | 15 + tests/pools/reCLAMM/test_reclamm_e2e.py | 2 +- .../pools/reCLAMM/test_reclamm_fee_revenue.py | 1 + tests/pools/reCLAMM/test_reclamm_math.py | 4 +- .../test_reclamm_price_ratio_updates.py | 221 +++++++++++ tests/unit/test_jax_runner_utils.py | 208 ++++++++++ tests/unit/test_jax_runners_comprehensive.py | 109 ++++++ 15 files changed, 1134 insertions(+), 24 deletions(-) create mode 100644 tests/pools/reCLAMM/helpers.py create mode 100644 tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py diff --git a/quantammsim/core_simulator/dynamic_inputs.py b/quantammsim/core_simulator/dynamic_inputs.py index c2490888..598d3f71 100644 --- a/quantammsim/core_simulator/dynamic_inputs.py +++ b/quantammsim/core_simulator/dynamic_inputs.py @@ -13,6 +13,7 @@ class DynamicInputFrames: gas_cost: Optional[Any] = None arb_fees: Optional[Any] = None lp_supply: Optional[Any] = None + reclamm_price_ratio_updates: Optional[Any] = None class DynamicInputArrays(NamedTuple): @@ -23,6 +24,7 @@ class DynamicInputArrays(NamedTuple): gas_cost: jnp.ndarray arb_fees: jnp.ndarray lp_supply: jnp.ndarray + reclamm_price_ratio_updates: jnp.ndarray def default_dynamic_input_flags() -> dict: @@ -34,6 +36,7 @@ def default_dynamic_input_flags() -> dict: "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, } @@ -49,6 +52,9 @@ def dynamic_input_flags_from_frames(dynamic_input_frames: Optional[DynamicInputF "has_dynamic_gas_cost": dynamic_input_frames.gas_cost is not None, "has_dynamic_arb_fees": dynamic_input_frames.arb_fees is not None, "has_lp_supply": dynamic_input_frames.lp_supply is not None, + "has_reclamm_price_ratio_updates": ( + dynamic_input_frames.reclamm_price_ratio_updates is not None + ), } flags["use_dynamic_inputs"] = any(flags.values()) return flags @@ -59,11 +65,9 @@ def resolve_dynamic_input_flags( dynamic_input_flags: Optional[dict] = None, ) -> dict: """Return a safe dispatch flag set for the provided hot-path bundle.""" - flags = ( - default_dynamic_input_flags() - if dynamic_input_flags is None - else dict(dynamic_input_flags) - ) + flags = default_dynamic_input_flags() + if dynamic_input_flags is not None: + flags.update(dict(dynamic_input_flags)) if dynamic_inputs is not None: flags["use_dynamic_inputs"] = True return flags @@ -77,6 +81,10 @@ def empty_dynamic_input_arrays() -> DynamicInputArrays: gas_cost=jnp.zeros((1,), dtype=jnp.float64), arb_fees=jnp.zeros((1,), dtype=jnp.float64), lp_supply=jnp.ones((1,), dtype=jnp.float64), + # Columns: has_event, target_price_ratio, end_step, start_price_ratio_override + reclamm_price_ratio_updates=jnp.array( + [[0.0, 0.0, 0.0, jnp.nan]], dtype=jnp.float64 + ), ) @@ -109,6 +117,11 @@ def resolve_dynamic_input_components( if dynamic_input_flags["has_lp_supply"] else jnp.ones((1,), dtype=jnp.float64) ), + "reclamm_price_ratio_updates": ( + arrays.reclamm_price_ratio_updates + if dynamic_input_flags["has_reclamm_price_ratio_updates"] + else empty_dynamic_input_arrays().reclamm_price_ratio_updates + ), } @@ -148,6 +161,7 @@ def materialize_dynamic_inputs( "has_dynamic_gas_cost": True, "has_dynamic_arb_fees": True, "has_lp_supply": True, + "has_reclamm_price_ratio_updates": True, } else: flags = resolve_dynamic_input_flags(dynamic_inputs, dynamic_input_flags) @@ -174,4 +188,10 @@ def materialize_dynamic_inputs( lp_supply=_broadcast_dynamic_input_leaf( "lp_supply", resolved["lp_supply"], scan_len, dtype ), + reclamm_price_ratio_updates=_broadcast_dynamic_input_leaf( + "reclamm_price_ratio_updates", + resolved["reclamm_price_ratio_updates"], + scan_len, + dtype, + ), ) diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index d2a32cd8..2125ff1f 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -1113,6 +1113,9 @@ def forward_pass_nograd( gas_cost=stop_gradient(dynamic_inputs.gas_cost), arb_fees=stop_gradient(dynamic_inputs.arb_fees), lp_supply=stop_gradient(dynamic_inputs.lp_supply), + reclamm_price_ratio_updates=stop_gradient( + dynamic_inputs.reclamm_price_ratio_updates + ), ) return forward_pass( params, diff --git a/quantammsim/hooks/dynamic_fee_base_hook.py b/quantammsim/hooks/dynamic_fee_base_hook.py index ad64a5f5..1ab02662 100644 --- a/quantammsim/hooks/dynamic_fee_base_hook.py +++ b/quantammsim/hooks/dynamic_fee_base_hook.py @@ -124,6 +124,7 @@ def calculate_reserves_with_fees( gas_cost=jnp.asarray(run_fingerprint["gas_cost"], dtype=jnp.float64), arb_fees=jnp.asarray(run_fingerprint["arb_fees"], dtype=jnp.float64), lp_supply=empty_inputs.lp_supply, + reclamm_price_ratio_updates=empty_inputs.reclamm_price_ratio_updates, ) return self.calculate_reserves_with_dynamic_inputs( diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index da36710b..762301c8 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -310,6 +310,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( fees=materialized_inputs.fees, arb_thresh=materialized_inputs.gas_cost, arb_fees=materialized_inputs.arb_fees, + price_ratio_updates=materialized_inputs.reclamm_price_ratio_updates, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), @@ -387,6 +388,7 @@ def calculate_reserves_with_dynamic_inputs( fees=materialized_inputs.fees, arb_thresh=materialized_inputs.gas_cost, arb_fees=materialized_inputs.arb_fees, + price_ratio_updates=materialized_inputs.reclamm_price_ratio_updates, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 81ad48e2..b098eb12 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -17,7 +17,7 @@ import jax.numpy as jnp from jax import jit -from jax.lax import scan +from jax.lax import scan, cond from jax.tree_util import Partial from functools import partial @@ -556,6 +556,40 @@ def initialise_reclamm_reserves(initial_pool_value, initial_prices, price_ratio) # Scan-based reserve calculations # --------------------------------------------------------------------------- +def apply_target_price_ratio_to_virtual_balances(Ra, Rb, Va, Vb, target_price_ratio): + """Retarget virtual balances to a desired price ratio while preserving orientation. + + The overvalued-side virtual balance is preserved (subject to floor), and the + undervalued-side virtual balance is solved from the reCLAMM ratio constraint. + """ + safe_ratio = jnp.maximum(target_price_ratio, 1.0 + 1e-12) + sqrt_ratio = jnp.sqrt(safe_ratio) + fourth_root_ratio = jnp.sqrt(sqrt_ratio) + centeredness, is_above = compute_centeredness(Ra, Rb, Va, Vb) + + # Above center => B overvalued, so keep Vb and solve Va. + v_over_b_floor = Rb / jnp.maximum(fourth_root_ratio - 1.0, 1e-30) + Vb_kept = jnp.maximum(Vb, v_over_b_floor) + Va_from_b = Ra * (Vb_kept + Rb) / jnp.maximum( + (sqrt_ratio - 1.0) * Vb_kept - Rb, 1e-30 + ) + + # Below center => A overvalued, so keep Va and solve Vb. + v_over_a_floor = Ra / jnp.maximum(fourth_root_ratio - 1.0, 1e-30) + Va_kept = jnp.maximum(Va, v_over_a_floor) + Vb_from_a = Rb * (Va_kept + Ra) / jnp.maximum( + (sqrt_ratio - 1.0) * Va_kept - Ra, 1e-30 + ) + + Va_new = jnp.where(is_above, Va_from_b, Va_kept) + Vb_new = jnp.where(is_above, Vb_kept, Vb_from_a) + + # When centeredness is degenerate (e.g. both sides zero), preserve current virtuals. + invalid_centeredness = ~jnp.isfinite(centeredness) + Va_new = jnp.where(invalid_centeredness, Va, Va_new) + Vb_new = jnp.where(invalid_centeredness, Vb, Vb_new) + return Va_new, Vb_new + def _reclamm_scan_step_zero_fees( carry_list, prices, @@ -653,6 +687,13 @@ def _reclamm_scan_step_zero_fees( return [new_reserves, Va, Vb], new_reserves +# --------------------------------------------------------------------------- +# Test-only diagnostic helpers (virtual-balance history) +# --------------------------------------------------------------------------- +# These helpers mirror production kernels but additionally return Va/Vb +# trajectories for assertions in tests. Production pool paths should use the +# reserve-only kernels above. + def _reclamm_scan_step_zero_fees_full_state( carry_list, prices, @@ -662,7 +703,7 @@ def _reclamm_scan_step_zero_fees_full_state( arc_length_speed=0.0, centeredness_scaling=False, ): - """Like _reclamm_scan_step_zero_fees but outputs (reserves, Va, Vb).""" + """TEST-ONLY: scan step that outputs (reserves, Va, Vb).""" new_carry, new_reserves = _reclamm_scan_step_zero_fees( carry_list, prices, centeredness_margin, daily_price_shift_base, seconds_per_step, arc_length_speed=arc_length_speed, @@ -689,9 +730,10 @@ def _reclamm_scan_step_with_fees_and_revenue( Primary implementation — ``_reclamm_scan_step_with_fees`` wraps this. - Carry: [real_reserves (2,), Va (0-d), Vb (0-d)] + Carry: [real_reserves (2,), Va, Vb, step_idx, active_start_ratio, + active_target_ratio, active_start_step, active_end_step, active_enabled] Input: [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees] + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_update] Returns ------- @@ -702,6 +744,12 @@ def _reclamm_scan_step_with_fees_and_revenue( prev_reserves = carry_list[0] Va = carry_list[1] Vb = carry_list[2] + step_idx = carry_list[3] + active_start_ratio = carry_list[4] + active_target_ratio = carry_list[5] + active_start_step = carry_list[6] + active_end_step = carry_list[7] + active_enabled = carry_list[8] Ra = prev_reserves[0] Rb = prev_reserves[1] @@ -713,6 +761,86 @@ def _reclamm_scan_step_with_fees_and_revenue( gamma = input_list[4] arb_thresh = input_list[5] arb_fees = input_list[6] + price_ratio_update = input_list[7] + + event_has = price_ratio_update[0] > 0.5 + event_target_ratio = jnp.maximum( + jnp.where(jnp.isfinite(price_ratio_update[1]), price_ratio_update[1], 1.0), + 1.0 + 1e-12, + ) + event_end_step = jnp.where( + jnp.isfinite(price_ratio_update[2]), price_ratio_update[2], step_idx + ) + event_start_override = price_ratio_update[3] + + def _apply_schedule_state(_): + current_price_ratio = compute_price_ratio(Ra, Rb, Va, Vb) + start_ratio_from_event = jnp.where( + jnp.isfinite(event_start_override), + event_start_override, + current_price_ratio, + ) + next_active_start_ratio = jnp.where( + event_has, start_ratio_from_event, active_start_ratio + ) + next_active_target_ratio = jnp.where( + event_has, event_target_ratio, active_target_ratio + ) + next_active_start_step = jnp.where(event_has, step_idx, active_start_step) + next_active_end_step = jnp.where( + event_has, jnp.maximum(event_end_step, step_idx), active_end_step + ) + next_active_enabled = jnp.where(event_has, True, active_enabled) + + schedule_duration = next_active_end_step - next_active_start_step + schedule_progress = jnp.where( + schedule_duration <= 0.0, + 1.0, + jnp.clip((step_idx - next_active_start_step) / schedule_duration, 0.0, 1.0), + ) + scheduled_price_ratio = ( + next_active_start_ratio + + (next_active_target_ratio - next_active_start_ratio) * schedule_progress + ) + Va_scheduled, Vb_scheduled = apply_target_price_ratio_to_virtual_balances( + Ra, Rb, Va, Vb, scheduled_price_ratio + ) + return ( + Va_scheduled, + Vb_scheduled, + next_active_start_ratio, + next_active_target_ratio, + next_active_start_step, + next_active_end_step, + next_active_enabled, + ) + + def _skip_schedule_state(_): + return ( + Va, + Vb, + active_start_ratio, + active_target_ratio, + active_start_step, + active_end_step, + active_enabled, + ) + + schedule_active = jnp.logical_or(event_has, active_enabled) + ( + Va, + Vb, + active_start_ratio, + active_target_ratio, + active_start_step, + active_end_step, + active_enabled, + ) = cond( + schedule_active, + _apply_schedule_state, + _skip_schedule_state, + operand=None, + ) # Step 1: Update virtual balances if out of range centeredness, is_above = compute_centeredness(Ra, Rb, Va, Vb) @@ -827,7 +955,17 @@ def _reclamm_scan_step_with_fees_and_revenue( lp_fee_revenue_usd = (lp_fee_income * prices).sum() new_reserves = jnp.array([Ra_new, Rb_new]) - return [new_reserves, Va, Vb], (new_reserves, lp_fee_revenue_usd) + return [ + new_reserves, + Va, + Vb, + step_idx + 1.0, + active_start_ratio, + active_target_ratio, + active_start_step, + active_end_step, + active_enabled, + ], (new_reserves, lp_fee_revenue_usd) def _reclamm_scan_step_with_fees( @@ -865,6 +1003,37 @@ def _reclamm_scan_step_with_fees( return new_carry, new_reserves +def _reclamm_scan_step_with_fees_full_state( + carry_list, + input_list, + weights, + tokens_to_drop, + active_trade_directions, + n, + centeredness_margin, + daily_price_shift_base, + seconds_per_step, + arc_length_speed=0.0, + centeredness_scaling=False, + protocol_fee_split=0.0, +): + """TEST-ONLY: fee scan step that also outputs virtual balances.""" + new_carry, (new_reserves, _fee_rev) = _reclamm_scan_step_with_fees_and_revenue( + carry_list, input_list, + weights=weights, + tokens_to_drop=tokens_to_drop, + active_trade_directions=active_trade_directions, + n=n, + centeredness_margin=centeredness_margin, + daily_price_shift_base=daily_price_shift_base, + seconds_per_step=seconds_per_step, + arc_length_speed=arc_length_speed, + centeredness_scaling=centeredness_scaling, + protocol_fee_split=protocol_fee_split, + ) + return new_carry, (new_reserves, new_carry[1], new_carry[2]) + + @jit def _jax_calc_reclamm_reserves_zero_fees( initial_reserves, @@ -929,7 +1098,7 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( arc_length_speed=0.0, centeredness_scaling=False, ): - """Like _jax_calc_reclamm_reserves_zero_fees but also returns virtual balances. + """TEST-ONLY: Like _jax_calc_reclamm_reserves_zero_fees but returns Va/Vb. Returns ------- @@ -992,6 +1161,8 @@ def _jax_calc_reclamm_reserves_with_fees( gamma_array = jnp.full(prices.shape[0], gamma) arb_thresh_array = jnp.full(prices.shape[0], arb_thresh) arb_fees_array = jnp.full(prices.shape[0], arb_fees) + price_ratio_updates = jnp.zeros((prices.shape[0], 4), dtype=prices.dtype) + price_ratio_updates = price_ratio_updates.at[:, 3].set(jnp.nan) scan_fn = Partial( _reclamm_scan_step_with_fees, @@ -1007,17 +1178,27 @@ def _jax_calc_reclamm_reserves_with_fees( protocol_fee_split=protocol_fee_split, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] + carry_init = [ + initial_reserves, + initial_Va, + initial_Vb, + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] _, reserves = scan( scan_fn, carry_init, [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array], + all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], ) return reserves -@partial(jit, static_argnums=(10,)) +@partial(jit, static_argnums=(11,)) def _jax_calc_reclamm_reserves_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1029,6 +1210,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( fees, arb_thresh, arb_fees, + price_ratio_updates=None, do_trades=False, trades=None, all_sig_variations=None, @@ -1048,6 +1230,18 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( arb_fees = jnp.where( arb_fees.size == 1, jnp.full(prices.shape[0], arb_fees), arb_fees ) + if price_ratio_updates is None: + price_ratio_updates = jnp.zeros((prices.shape[0], 4), dtype=prices.dtype) + price_ratio_updates = price_ratio_updates.at[:, 3].set(jnp.nan) + else: + if price_ratio_updates.ndim == 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[0]) + ) + elif price_ratio_updates.shape[0] == 1 and prices.shape[0] != 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[1]) + ) _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( precalc_shared_values_for_all_signatures(all_sig_variations, n_assets) @@ -1074,16 +1268,115 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( protocol_fee_split=protocol_fee_split, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] + carry_init = [ + initial_reserves, + initial_Va, + initial_Vb, + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] _, reserves = scan( scan_fn, carry_init, [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees], + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], ) return reserves +@partial(jit, static_argnums=(11,)) +def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( + initial_reserves, + initial_Va, + initial_Vb, + prices, + centeredness_margin, + daily_price_shift_base, + seconds_per_step, + fees, + arb_thresh, + arb_fees, + price_ratio_updates=None, + do_trades=False, + trades=None, + all_sig_variations=None, + arc_length_speed=0.0, + centeredness_scaling=False, + protocol_fee_split=0.0, +): + """TEST-ONLY: dynamic-input reserve path returning virtual-balance history.""" + n_assets = 2 + weights = jnp.array([0.5, 0.5]) + + gamma = jnp.where(fees.size == 1, jnp.full(prices.shape[0], 1.0 - fees), 1.0 - fees) + arb_thresh = jnp.where( + arb_thresh.size == 1, jnp.full(prices.shape[0], arb_thresh), arb_thresh + ) + arb_fees = jnp.where( + arb_fees.size == 1, jnp.full(prices.shape[0], arb_fees), arb_fees + ) + if price_ratio_updates is None: + price_ratio_updates = jnp.zeros((prices.shape[0], 4), dtype=prices.dtype) + price_ratio_updates = price_ratio_updates.at[:, 3].set(jnp.nan) + else: + if price_ratio_updates.ndim == 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[0]) + ) + elif price_ratio_updates.shape[0] == 1 and prices.shape[0] != 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[1]) + ) + + _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( + precalc_shared_values_for_all_signatures(all_sig_variations, n_assets) + ) + + active_initial_weights, per_asset_ratios, all_other_assets_ratios = ( + precalc_components_of_optimal_trade_across_prices_and_dynamic_fees( + weights, prices, gamma, tokens_to_drop, + active_trade_directions, leave_one_out_idxs, + ) + ) + + scan_fn = Partial( + _reclamm_scan_step_with_fees_full_state, + weights=weights, + tokens_to_drop=tokens_to_drop, + active_trade_directions=active_trade_directions, + n=n_assets, + centeredness_margin=centeredness_margin, + daily_price_shift_base=daily_price_shift_base, + seconds_per_step=seconds_per_step, + arc_length_speed=arc_length_speed, + centeredness_scaling=centeredness_scaling, + protocol_fee_split=protocol_fee_split, + ) + + carry_init = [ + initial_reserves, + initial_Va, + initial_Vb, + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] + _, (reserves, Va_history, Vb_history) = scan( + scan_fn, + carry_init, + [prices, active_initial_weights, per_asset_ratios, + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], + ) + return reserves, Va_history, Vb_history + + @jit def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( initial_reserves, @@ -1127,6 +1420,8 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( gamma_array = jnp.full(prices.shape[0], gamma) arb_thresh_array = jnp.full(prices.shape[0], arb_thresh) arb_fees_array = jnp.full(prices.shape[0], arb_fees) + price_ratio_updates = jnp.zeros((prices.shape[0], 4), dtype=prices.dtype) + price_ratio_updates = price_ratio_updates.at[:, 3].set(jnp.nan) scan_fn = Partial( _reclamm_scan_step_with_fees_and_revenue, @@ -1142,17 +1437,27 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( protocol_fee_split=protocol_fee_split, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] + carry_init = [ + initial_reserves, + initial_Va, + initial_Vb, + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] _, (reserves, fee_revenue) = scan( scan_fn, carry_init, [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array], + all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], ) return reserves, fee_revenue -@partial(jit, static_argnums=(10,)) +@partial(jit, static_argnums=(11,)) def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1164,6 +1469,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( fees, arb_thresh, arb_fees, + price_ratio_updates=None, do_trades=False, trades=None, all_sig_variations=None, @@ -1189,6 +1495,18 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( arb_fees = jnp.where( arb_fees.size == 1, jnp.full(prices.shape[0], arb_fees), arb_fees ) + if price_ratio_updates is None: + price_ratio_updates = jnp.zeros((prices.shape[0], 4), dtype=prices.dtype) + price_ratio_updates = price_ratio_updates.at[:, 3].set(jnp.nan) + else: + if price_ratio_updates.ndim == 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[0]) + ) + elif price_ratio_updates.shape[0] == 1 and prices.shape[0] != 1: + price_ratio_updates = jnp.broadcast_to( + price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[1]) + ) _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( precalc_shared_values_for_all_signatures(all_sig_variations, n_assets) @@ -1215,11 +1533,21 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( protocol_fee_split=protocol_fee_split, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] + carry_init = [ + initial_reserves, + initial_Va, + initial_Vb, + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] _, (reserves, fee_revenue) = scan( scan_fn, carry_init, [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees], + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], ) return reserves, fee_revenue diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index e9c74f9f..a7598e14 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -1228,6 +1228,7 @@ def _to_dynamic_input_arrays( gas_cost_array, arb_fees_array, lp_supply_array, + reclamm_price_ratio_updates_array, ) -> DynamicInputArrays: """Normalize optional numpy arrays into the hot-path container.""" empty = empty_dynamic_input_arrays() @@ -1237,9 +1238,156 @@ def _to_dynamic_input_arrays( gas_cost=empty.gas_cost if gas_cost_array is None else jnp.asarray(gas_cost_array, dtype=jnp.float64), arb_fees=empty.arb_fees if arb_fees_array is None else jnp.asarray(arb_fees_array, dtype=jnp.float64), lp_supply=empty.lp_supply if lp_supply_array is None else jnp.asarray(lp_supply_array, dtype=jnp.float64), + reclamm_price_ratio_updates=( + empty.reclamm_price_ratio_updates + if reclamm_price_ratio_updates_array is None + else jnp.asarray(reclamm_price_ratio_updates_array, dtype=jnp.float64) + ), ) +def _coerce_reclamm_price_ratio_updates_to_frame(raw_updates) -> pd.DataFrame: + """Accept DataFrame / CSV path / list[dict] / dict payload and return a DataFrame.""" + if raw_updates is None: + return pd.DataFrame( + columns=["unix", "end_unix", "price_ratio", "start_price_ratio"] + ) + if isinstance(raw_updates, pd.DataFrame): + return raw_updates.copy() + if isinstance(raw_updates, (str, Path)): + return pd.read_csv(raw_updates) + if isinstance(raw_updates, list): + return pd.DataFrame(raw_updates) + if isinstance(raw_updates, dict): + for key in ( + "updates", + "rows", + "reclamm_price_ratio_updates", + "price_ratio_updates", + ): + value = raw_updates.get(key) + if isinstance(value, list): + return pd.DataFrame(value) + if isinstance(value, (str, Path)): + return pd.read_csv(value) + return pd.DataFrame(raw_updates) + raise TypeError( + "reclamm_price_ratio_updates must be a DataFrame, CSV path, list of dicts, or dict payload" + ) + + +def _ceil_div_nonnegative(delta: int, denom: int) -> int: + """Ceiling division for non-negative integers.""" + if delta <= 0: + return 0 + return (delta + denom - 1) // denom + + +def _normalize_reclamm_price_ratio_updates_for_window( + raw_updates, + start_date_string: str, + end_date_string: str, + arb_frequency: int, +) -> np.ndarray: + """Normalize manual reCLAMM price-ratio updates into per-step event rows. + + Output columns per step: + 0. has_event (0/1) + 1. target_price_ratio + 2. end_step + 3. start_price_ratio_override (NaN when not supplied) + """ + start_unix = pd.to_datetime(start_date_string, format="%Y-%m-%d %H:%M:%S").value // 10**6 + end_unix = pd.to_datetime(end_date_string, format="%Y-%m-%d %H:%M:%S").value // 10**6 + step_ms = int(arb_frequency) * 60 * 1000 + if step_ms <= 0: + raise ValueError("arb_frequency must be >= 1 for reCLAMM price-ratio updates") + scan_len = int(max((end_unix - start_unix) // step_ms, 0)) + default_matrix = np.zeros((scan_len, 4), dtype=np.float64) + if scan_len > 0: + default_matrix[:, 3] = np.nan + + updates_df = _coerce_reclamm_price_ratio_updates_to_frame(raw_updates) + if updates_df.empty: + return default_matrix + + required = {"unix", "end_unix", "price_ratio"} + missing = sorted(required.difference(updates_df.columns)) + if missing: + raise ValueError( + "reclamm_price_ratio_updates missing required columns: " + + ", ".join(missing) + ) + + updates = updates_df.copy() + updates["unix"] = pd.to_numeric(updates["unix"], errors="coerce") + updates["end_unix"] = pd.to_numeric(updates["end_unix"], errors="coerce") + updates["price_ratio"] = pd.to_numeric(updates["price_ratio"], errors="coerce") + if "start_price_ratio" in updates.columns: + updates["start_price_ratio"] = pd.to_numeric( + updates["start_price_ratio"], errors="coerce" + ) + else: + updates["start_price_ratio"] = np.nan + + invalid_required = updates["unix"].isna() | updates["end_unix"].isna() | updates["price_ratio"].isna() + if invalid_required.any(): + raise ValueError( + "reclamm_price_ratio_updates contains non-numeric unix/end_unix/price_ratio values" + ) + if (updates["price_ratio"] <= 1.0).any(): + raise ValueError("reclamm price_ratio values must be > 1.0") + if (updates["end_unix"] < updates["unix"]).any(): + raise ValueError("reclamm end_unix must be >= unix for every update") + + updates = updates.sort_values("unix", kind="stable") + + for _, row in updates.iterrows(): + event_start_unix = int(row["unix"]) + event_end_unix = int(row["end_unix"]) + target_price_ratio = float(row["price_ratio"]) + start_price_ratio_override = row["start_price_ratio"] + + # Event completes before window start - no effect. + if event_end_unix <= start_unix: + continue + + in_progress_pre_window = event_start_unix < start_unix and event_end_unix > start_unix + if in_progress_pre_window and pd.isna(start_price_ratio_override): + raise ValueError( + "reclamm pre-window in-progress event requires start_price_ratio" + ) + + effective_start_unix = max(event_start_unix, start_unix) + start_step = _ceil_div_nonnegative(effective_start_unix - start_unix, step_ms) + end_step = _ceil_div_nonnegative(event_end_unix - start_unix, step_ms) + + # Starts after current window. + if start_step >= scan_len: + continue + + end_step = min(max(end_step, start_step), scan_len - 1) + default_matrix[start_step, 0] = 1.0 + default_matrix[start_step, 1] = target_price_ratio + default_matrix[start_step, 2] = float(end_step) + default_matrix[start_step, 3] = ( + float(start_price_ratio_override) + if not pd.isna(start_price_ratio_override) + else np.nan + ) + + return default_matrix + + +def _has_reclamm_schedule_events(schedule_array: Optional[np.ndarray]) -> bool: + """Return True when a normalized schedule contains at least one event row.""" + if schedule_array is None: + return False + if schedule_array.size == 0: + return False + return bool(np.any(np.asarray(schedule_array)[:, 0] > 0.5)) + + def prepare_dynamic_inputs( run_fingerprint, dynamic_input_frames: Optional[DynamicInputFrames] = None, @@ -1254,6 +1402,7 @@ def prepare_dynamic_inputs( gas_cost_df = dynamic_input_frames.gas_cost arb_fees_df = dynamic_input_frames.arb_fees lp_supply_df = dynamic_input_frames.lp_supply + reclamm_price_ratio_updates = dynamic_input_frames.reclamm_price_ratio_updates dynamic_input_flags = dynamic_input_flags_from_frames(dynamic_input_frames) if raw_trades is not None: @@ -1369,6 +1518,51 @@ def prepare_dynamic_inputs( else None ) + reclamm_price_ratio_updates_array = ( + _normalize_reclamm_price_ratio_updates_for_window( + reclamm_price_ratio_updates, + run_fingerprint["startDateString"], + run_fingerprint["endDateString"], + run_fingerprint["arb_frequency"], + ) + if reclamm_price_ratio_updates is not None + else None + ) + if do_test_period: + test_reclamm_price_ratio_updates_array = ( + _normalize_reclamm_price_ratio_updates_for_window( + reclamm_price_ratio_updates, + run_fingerprint["endDateString"], + run_fingerprint["endTestDateString"], + run_fingerprint["arb_frequency"], + ) + if reclamm_price_ratio_updates is not None + else None + ) + + train_has_reclamm_schedule = _has_reclamm_schedule_events( + reclamm_price_ratio_updates_array + ) + test_has_reclamm_schedule = False + if do_test_period: + test_has_reclamm_schedule = _has_reclamm_schedule_events( + test_reclamm_price_ratio_updates_array + ) + + if not train_has_reclamm_schedule: + reclamm_price_ratio_updates_array = None + if do_test_period and not test_has_reclamm_schedule: + test_reclamm_price_ratio_updates_array = None + if not train_has_reclamm_schedule and ( + not do_test_period or not test_has_reclamm_schedule + ): + dynamic_input_flags["has_reclamm_price_ratio_updates"] = False + dynamic_input_flags["use_dynamic_inputs"] = any( + value + for key, value in dynamic_input_flags.items() + if key != "use_dynamic_inputs" + ) + # Unit LP supply is the neutral case; keep it on the static hot path. if lp_supply_array is not None and np.allclose(lp_supply_array, 1.0): lp_supply_array = None @@ -1388,6 +1582,7 @@ def prepare_dynamic_inputs( gas_cost_array, arb_fees_array, lp_supply_array, + reclamm_price_ratio_updates_array, ), "test_dynamic_inputs": _to_dynamic_input_arrays( test_period_trades, @@ -1395,6 +1590,7 @@ def prepare_dynamic_inputs( test_gas_cost_array, test_arb_fees_array, test_lp_supply_array, + test_reclamm_price_ratio_updates_array, ), "dynamic_input_flags": dynamic_input_flags, } @@ -1405,6 +1601,7 @@ def prepare_dynamic_inputs( gas_cost_array, arb_fees_array, lp_supply_array, + reclamm_price_ratio_updates_array, ), "dynamic_input_flags": dynamic_input_flags, } @@ -1657,6 +1854,7 @@ def try_forward_pass(n_sets: int) -> bool: "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, }, }, ) diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index aad11bdd..cbc8cdca 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -657,6 +657,7 @@ def train_on_historic_data( "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, }, }, ) diff --git a/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py b/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py index aa315199..cc000b01 100644 --- a/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py +++ b/quantammsim/simulator_analysis_tools/finance/param_financial_calculator.py @@ -243,6 +243,9 @@ def run_pool_simulation(simulationRunDto): trades=raw_trades, fees=fee_steps_df, gas_cost=gas_cost_df, + reclamm_price_ratio_updates=run_fingerprint.get( + "reclamm_price_ratio_updates" + ), ) print("run fingerprint-------------------", run_fingerprint) diff --git a/tests/pools/reCLAMM/helpers.py b/tests/pools/reCLAMM/helpers.py new file mode 100644 index 00000000..4d06f707 --- /dev/null +++ b/tests/pools/reCLAMM/helpers.py @@ -0,0 +1,15 @@ +"""Test-only exports for reCLAMM diagnostic kernels. + +This module centralizes access to kernels that return virtual-balance history. +Production paths should use reserve-only kernels in ``reclamm_reserves``. +""" + +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state, + _jax_calc_reclamm_reserves_zero_fees_full_state, +) + +__all__ = [ + "_jax_calc_reclamm_reserves_zero_fees_full_state", + "_jax_calc_reclamm_reserves_with_dynamic_inputs_full_state", +] diff --git a/tests/pools/reCLAMM/test_reclamm_e2e.py b/tests/pools/reCLAMM/test_reclamm_e2e.py index 25edc982..80eeef6c 100644 --- a/tests/pools/reCLAMM/test_reclamm_e2e.py +++ b/tests/pools/reCLAMM/test_reclamm_e2e.py @@ -30,8 +30,8 @@ initialise_reclamm_reserves, _jax_calc_reclamm_reserves_zero_fees, _jax_calc_reclamm_reserves_with_fees, - _jax_calc_reclamm_reserves_zero_fees_full_state, ) +from tests.pools.reCLAMM.helpers import _jax_calc_reclamm_reserves_zero_fees_full_state ALL_SIG_VARIATIONS_2 = jnp.array([[1, -1], [-1, 1]]) diff --git a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py index 4a31f1fa..bfbd21b1 100644 --- a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py +++ b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py @@ -355,6 +355,7 @@ def test_pool_method_with_dynamic_inputs(self): gas_cost=arb_thresh_array, arb_fees=arb_fees_array, lp_supply=jnp.ones((1,)), + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), ) reserves, fee_revenue = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( diff --git a/tests/pools/reCLAMM/test_reclamm_math.py b/tests/pools/reCLAMM/test_reclamm_math.py index 6f3870d2..b1ce8eea 100644 --- a/tests/pools/reCLAMM/test_reclamm_math.py +++ b/tests/pools/reCLAMM/test_reclamm_math.py @@ -729,9 +729,9 @@ def test_arc_length_single_step_exact(self): def test_arc_length_constant_through_scan(self): """Through the scan, per-step Δs should be approximately constant.""" - from quantammsim.pools.reCLAMM.reclamm_reserves import ( + from quantammsim.pools.reCLAMM.reclamm_reserves import calibrate_arc_length_speed + from tests.pools.reCLAMM.helpers import ( _jax_calc_reclamm_reserves_zero_fees_full_state, - calibrate_arc_length_speed, ) Ra, Rb, Va, Vb, Q = _centered_pool(P=2.0, price_ratio=4.0) diff --git a/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py new file mode 100644 index 00000000..5ee98bcd --- /dev/null +++ b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py @@ -0,0 +1,221 @@ +"""Tests for manual reCLAMM price-ratio schedule updates.""" + +import numpy as np +import numpy.testing as npt +import jax.numpy as jnp +import pytest + +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + compute_price_ratio, + initialise_reclamm_reserves, + _jax_calc_reclamm_reserves_with_dynamic_inputs, + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs, +) +from tests.pools.reCLAMM.helpers import ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state, +) + + +DEFAULT_INITIAL_POOL_VALUE = 1_000_000.0 +DEFAULT_INITIAL_PRICES = jnp.array([2500.0, 1.0], dtype=jnp.float64) +DEFAULT_PRICE_RATIO = 4.0 +DEFAULT_DAILY_PRICE_SHIFT_BASE = 1.0 - 1.0 / 124000.0 +DEFAULT_SECONDS_PER_STEP = 60.0 +ALL_SIG_VARIATIONS_2 = tuple(map(tuple, [[1, -1], [-1, 1]])) + + +def _init_pool(price_ratio=DEFAULT_PRICE_RATIO): + reserves, Va, Vb = initialise_reclamm_reserves( + DEFAULT_INITIAL_POOL_VALUE, + DEFAULT_INITIAL_PRICES, + price_ratio, + ) + return reserves, Va, Vb + + +def _flat_prices(n_steps): + return jnp.stack( + [jnp.full((n_steps,), DEFAULT_INITIAL_PRICES[0]), jnp.ones((n_steps,))], + axis=1, + ) + + +def _empty_schedule(n_steps): + schedule = np.zeros((n_steps, 4), dtype=np.float64) + schedule[:, 3] = np.nan + return jnp.asarray(schedule) + + +def _single_event_schedule( + n_steps, + start_step, + end_step, + target_price_ratio, + start_price_ratio_override=np.nan, +): + schedule = np.zeros((n_steps, 4), dtype=np.float64) + schedule[:, 3] = np.nan + schedule[start_step, 0] = 1.0 + schedule[start_step, 1] = target_price_ratio + schedule[start_step, 2] = float(end_step) + schedule[start_step, 3] = start_price_ratio_override + return jnp.asarray(schedule) + + +class TestReclammPriceRatioUpdates: + def test_schedule_off_matches_baseline_dynamic_kernel(self): + reserves, Va, Vb = _init_pool() + n_steps = 8 + prices = _flat_prices(n_steps) + fees = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + baseline = _jax_calc_reclamm_reserves_with_dynamic_inputs( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.2, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + with_schedule = _jax_calc_reclamm_reserves_with_dynamic_inputs( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.2, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=_empty_schedule(n_steps), + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + npt.assert_allclose(with_schedule, baseline, rtol=1e-10, atol=1e-10) + + def test_single_schedule_reaches_target_ratio_at_end_step(self): + reserves, Va, Vb = _init_pool() + n_steps = 8 + prices = _flat_prices(n_steps) + fees = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + end_step = 4 + schedule = _single_event_schedule( + n_steps, + start_step=1, + end_step=end_step, + target_price_ratio=9.0, + start_price_ratio_override=DEFAULT_PRICE_RATIO, + ) + reserves_out, Va_history, Vb_history = ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.0, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + ) + + ratio_at_end = float( + compute_price_ratio( + reserves_out[end_step, 0], + reserves_out[end_step, 1], + Va_history[end_step], + Vb_history[end_step], + ) + ) + assert ratio_at_end == pytest.approx(9.0, rel=1e-5, abs=1e-5) + + def test_replacement_event_supersedes_active_event(self): + reserves, Va, Vb = _init_pool() + n_steps = 9 + prices = _flat_prices(n_steps) + fees = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + schedule = np.zeros((n_steps, 4), dtype=np.float64) + schedule[:, 3] = np.nan + # Event 1: interpolate toward 8.0 until step 6. + schedule[1] = np.array([1.0, 8.0, 6.0, DEFAULT_PRICE_RATIO], dtype=np.float64) + # Event 2 replaces at step 3 and targets 2.0 by step 4. + schedule[3] = np.array([1.0, 2.0, 4.0, np.nan], dtype=np.float64) + + reserves_out, Va_history, Vb_history = ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.0, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=jnp.asarray(schedule), + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + ) + + ratio_after_replacement = float( + compute_price_ratio( + reserves_out[4, 0], + reserves_out[4, 1], + Va_history[4], + Vb_history[4], + ) + ) + assert ratio_after_replacement == pytest.approx(2.0, rel=1e-4, abs=1e-4) + + def test_dynamic_fee_revenue_path_with_schedule(self): + reserves, Va, Vb = _init_pool() + n_steps = 10 + prices = _flat_prices(n_steps) + fees = jnp.full((n_steps,), 0.003, dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + schedule = _single_event_schedule( + n_steps, + start_step=2, + end_step=6, + target_price_ratio=5.5, + start_price_ratio_override=DEFAULT_PRICE_RATIO, + ) + reserves_out, fee_revenue = ( + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.2, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + ) + assert reserves_out.shape == (n_steps, 2) + assert fee_revenue.shape == (n_steps,) + assert jnp.all(fee_revenue >= 0.0) diff --git a/tests/unit/test_jax_runner_utils.py b/tests/unit/test_jax_runner_utils.py index 620ecb84..2e014f57 100644 --- a/tests/unit/test_jax_runner_utils.py +++ b/tests/unit/test_jax_runner_utils.py @@ -289,6 +289,7 @@ def test_dynamic_input_flags_reflect_present_frames(self): assert flags["has_dynamic_gas_cost"] is True assert flags["has_dynamic_arb_fees"] is False assert flags["has_lp_supply"] is False + assert flags["has_reclamm_price_ratio_updates"] is False def test_prepare_dynamic_inputs_preserves_fixed_hot_path_structure(self): """Normalization should return fixed bundles plus static dispatch flags.""" @@ -333,18 +334,203 @@ def test_prepare_dynamic_inputs_preserves_fixed_hot_path_structure(self): assert flags["has_dynamic_gas_cost"] is True assert flags["has_dynamic_arb_fees"] is True assert flags["has_lp_supply"] is True + assert flags["has_reclamm_price_ratio_updates"] is False assert train_inputs.trades.shape == (2, 3) assert train_inputs.fees.shape == (2,) assert train_inputs.gas_cost.shape == (2,) assert train_inputs.arb_fees.shape == (2,) assert train_inputs.lp_supply.shape == (2,) + assert train_inputs.reclamm_price_ratio_updates.shape == (1, 4) assert test_inputs.trades.shape == (2, 3) assert test_inputs.fees.shape == (2,) assert test_inputs.gas_cost.shape == (2,) assert test_inputs.arb_fees.shape == (2,) assert test_inputs.lp_supply.shape == (2,) + assert test_inputs.reclamm_price_ratio_updates.shape == (1, 4) np.testing.assert_allclose(np.asarray(train_inputs.fees), np.array([0.003, 0.003])) + def test_prepare_dynamic_inputs_normalizes_reclamm_price_ratio_updates(self): + """Manual reCLAMM update schedules should map to per-step event rows.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "arb_frequency": 1, + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:05:00", + "endTestDateString": "2023-01-01 00:06:00", + } + start_unix = pd.Timestamp(run_fingerprint["startDateString"]).value // 10**6 + updates = pd.DataFrame( + { + "unix": [start_unix + 30_000, start_unix + 40_000], + "end_unix": [start_unix + 150_000, start_unix + 220_000], + "price_ratio": [4.0, 6.0], + } + ) + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=updates + ), + ) + flags = prepared["dynamic_input_flags"] + schedule = np.asarray( + prepared["train_dynamic_inputs"].reclamm_price_ratio_updates + ) + + assert flags["has_reclamm_price_ratio_updates"] is True + assert flags["use_dynamic_inputs"] is True + assert schedule.shape == (5, 4) + # Same-step collision should keep the later event. + assert schedule[1, 0] == pytest.approx(1.0) + assert schedule[1, 1] == pytest.approx(6.0) + assert schedule[1, 2] == pytest.approx(4.0) + assert np.isnan(schedule[1, 3]) + assert np.all(schedule[[0, 2, 3, 4], 0] == 0.0) + + def test_prepare_dynamic_inputs_accepts_reclamm_updates_as_list(self): + """List payloads should be accepted for reCLAMM update schedules.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "arb_frequency": 1, + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:03:00", + "endTestDateString": "2023-01-01 00:04:00", + } + start_unix = pd.Timestamp(run_fingerprint["startDateString"]).value // 10**6 + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=[ + { + "unix": start_unix, + "end_unix": start_unix + 120_000, + "price_ratio": 5.0, + "start_price_ratio": 4.25, + } + ] + ), + ) + schedule = np.asarray( + prepared["train_dynamic_inputs"].reclamm_price_ratio_updates + ) + assert schedule.shape == (3, 4) + assert schedule[0, 0] == pytest.approx(1.0) + assert schedule[0, 1] == pytest.approx(5.0) + assert schedule[0, 2] == pytest.approx(2.0) + assert schedule[0, 3] == pytest.approx(4.25) + + def test_prepare_dynamic_inputs_accepts_reclamm_updates_as_csv_path(self, tmp_path): + """CSV payloads should be accepted for reCLAMM update schedules.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "arb_frequency": 1, + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:03:00", + "endTestDateString": "2023-01-01 00:04:00", + } + start_unix = pd.Timestamp(run_fingerprint["startDateString"]).value // 10**6 + csv_path = tmp_path / "reclamm_updates.csv" + pd.DataFrame( + [ + { + "unix": start_unix + 60_000, + "end_unix": start_unix + 180_000, + "price_ratio": 6.0, + "start_price_ratio": 4.1, + } + ] + ).to_csv(csv_path, index=False) + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=str(csv_path) + ), + ) + schedule = np.asarray( + prepared["train_dynamic_inputs"].reclamm_price_ratio_updates + ) + assert schedule.shape == (3, 4) + assert schedule[1, 0] == pytest.approx(1.0) + assert schedule[1, 1] == pytest.approx(6.0) + assert schedule[1, 2] == pytest.approx(2.0) + assert schedule[1, 3] == pytest.approx(4.1) + + def test_prepare_dynamic_inputs_rejects_prewindow_reclamm_events_without_start_ratio(self): + """In-progress pre-window events must provide start_price_ratio.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "arb_frequency": 1, + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:05:00", + "endTestDateString": "2023-01-01 00:06:00", + } + start_unix = pd.Timestamp(run_fingerprint["startDateString"]).value // 10**6 + + with pytest.raises(ValueError, match="start_price_ratio"): + prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=pd.DataFrame( + { + "unix": [start_unix - 120_000], + "end_unix": [start_unix + 120_000], + "price_ratio": [4.5], + } + ) + ), + ) + + def test_prepare_dynamic_inputs_disables_reclamm_schedule_flag_when_window_has_no_events(self): + """Schedule flag should be disabled when no event lands in train/test windows.""" + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames + from quantammsim.runners.jax_runner_utils import prepare_dynamic_inputs + + run_fingerprint = { + "tokens": ["ETH", "USDC"], + "arb_frequency": 1, + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-01 00:05:00", + "endTestDateString": "2023-01-01 00:10:00", + } + start_unix = pd.Timestamp(run_fingerprint["startDateString"]).value // 10**6 + updates = pd.DataFrame( + { + "unix": [start_unix - 300_000], + "end_unix": [start_unix - 60_000], + "price_ratio": [5.0], + "start_price_ratio": [4.0], + } + ) + + prepared = prepare_dynamic_inputs( + run_fingerprint, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=updates + ), + do_test_period=True, + ) + + flags = prepared["dynamic_input_flags"] + assert flags["has_reclamm_price_ratio_updates"] is False + assert flags["use_dynamic_inputs"] is False + assert prepared["train_dynamic_inputs"].reclamm_price_ratio_updates.shape == (1, 4) + assert prepared["test_dynamic_inputs"].reclamm_price_ratio_updates.shape == (1, 4) + def test_prepare_dynamic_inputs_uses_correct_test_period_values(self): """Test-period arrays should use values effective from the test window onward.""" from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames @@ -413,6 +599,7 @@ def test_resolve_dynamic_input_flags_promotes_explicit_bundle(self): "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, }, ) @@ -450,6 +637,7 @@ def test_resolve_dynamic_input_components_prefers_dynamic_values(self): gas_cost=jnp.array([3.0]), arb_fees=jnp.array([0.0003]), lp_supply=jnp.array([1500.0]), + reclamm_price_ratio_updates=jnp.array([[1.0, 4.0, 3.0, jnp.nan]]), ) flags = { "use_dynamic_inputs": True, @@ -458,6 +646,7 @@ def test_resolve_dynamic_input_components_prefers_dynamic_values(self): "has_dynamic_gas_cost": True, "has_dynamic_arb_fees": True, "has_lp_supply": True, + "has_reclamm_price_ratio_updates": True, } resolved = resolve_dynamic_input_components( @@ -471,6 +660,11 @@ def test_resolve_dynamic_input_components_prefers_dynamic_values(self): np.testing.assert_allclose(np.asarray(resolved["gas_cost"]), np.array([3.0])) np.testing.assert_allclose(np.asarray(resolved["arb_fees"]), np.array([0.0003])) np.testing.assert_allclose(np.asarray(resolved["lp_supply"]), np.array([1500.0])) + np.testing.assert_allclose( + np.asarray(resolved["reclamm_price_ratio_updates"]), + np.array([[1.0, 4.0, 3.0, np.nan]]), + equal_nan=True, + ) def test_materialize_dynamic_inputs_leaves_trades_optional(self): """No-trade paths should not expand placeholder trades into the scan inputs.""" @@ -488,6 +682,7 @@ def test_materialize_dynamic_inputs_leaves_trades_optional(self): "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, }, static_dict={"fees": 0.003, "gas_cost": 2.5, "arb_fees": 0.0001}, scan_len=4, @@ -499,6 +694,18 @@ def test_materialize_dynamic_inputs_leaves_trades_optional(self): np.testing.assert_allclose(np.asarray(materialized.gas_cost), np.full(4, 2.5)) np.testing.assert_allclose(np.asarray(materialized.arb_fees), np.full(4, 0.0001)) np.testing.assert_allclose(np.asarray(materialized.lp_supply), np.ones(4)) + np.testing.assert_allclose( + np.asarray(materialized.reclamm_price_ratio_updates), + np.array( + [ + [0.0, 0.0, 0.0, np.nan], + [0.0, 0.0, 0.0, np.nan], + [0.0, 0.0, 0.0, np.nan], + [0.0, 0.0, 0.0, np.nan], + ] + ), + equal_nan=True, + ) def test_materialize_dynamic_inputs_requires_trades_when_enabled(self): """Trade-enabled scans should fail fast if no trade path is available.""" @@ -517,6 +724,7 @@ def test_materialize_dynamic_inputs_requires_trades_when_enabled(self): "has_dynamic_gas_cost": False, "has_dynamic_arb_fees": False, "has_lp_supply": False, + "has_reclamm_price_ratio_updates": False, }, static_dict={"fees": 0.003, "gas_cost": 0.0, "arb_fees": 0.0}, scan_len=2, diff --git a/tests/unit/test_jax_runners_comprehensive.py b/tests/unit/test_jax_runners_comprehensive.py index 995e6d87..52690221 100644 --- a/tests/unit/test_jax_runners_comprehensive.py +++ b/tests/unit/test_jax_runners_comprehensive.py @@ -606,6 +606,115 @@ def test_provided_coarse_weights_respect_scalar_and_dynamic_gas(self, defaulted_ atol=1e-6, ) + def test_reclamm_schedule_only_dynamic_input_changes_path(self, defaulted_run_fingerprint): + """Schedule-only reCLAMM dynamic inputs should alter reserve trajectory.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["rule"] = "reclamm" + fp["do_arb"] = True + fp["fees"] = 0.0 + fp["gas_cost"] = 0.0 + fp["arb_fees"] = 0.0 + fp["reclamm_interpolation_method"] = "geometric" + params = { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array(1.0 - 1.0 / 124000.0), + } + + start_unix = pd.Timestamp(fp["startDateString"]).value // 10**6 + schedule_df = pd.DataFrame( + { + "unix": [start_unix + 2 * 60_000], + "end_unix": [start_unix + 8 * 60_000], + "price_ratio": [7.5], + } + ) + + baseline = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + ) + with_schedule = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=schedule_df + ), + ) + + assert not np.allclose( + np.asarray(baseline["reserves"]), + np.asarray(with_schedule["reserves"]), + ) + + def test_reclamm_schedule_test_period_does_not_leak_into_train(self, defaulted_run_fingerprint): + """Test-only schedule updates should affect test path only.""" + fp = deepcopy(defaulted_run_fingerprint) + fp["rule"] = "reclamm" + fp["do_arb"] = True + fp["fees"] = 0.0 + fp["gas_cost"] = 0.0 + fp["arb_fees"] = 0.0 + params = { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array(1.0 - 1.0 / 124000.0), + } + + test_start_unix = pd.Timestamp(fp["endDateString"]).value // 10**6 + baseline_updates = pd.DataFrame( + { + "unix": [test_start_unix + 60_000], + "end_unix": [test_start_unix + 6 * 60_000], + # Keep baseline on the initial ratio so it is a no-op schedule. + "price_ratio": [4.0], + "start_price_ratio": [4.0], + } + ) + test_only_updates = pd.DataFrame( + { + "unix": [test_start_unix + 60_000], + "end_unix": [test_start_unix + 6 * 60_000], + "price_ratio": [8.0], + } + ) + + train_base, test_base = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + do_test_period=True, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=baseline_updates + ), + ) + train_sched, test_sched = do_run_on_historic_data( + fp, + params=params, + root=TEST_DATA_DIR, + verbose=False, + do_test_period=True, + dynamic_input_frames=DynamicInputFrames( + reclamm_price_ratio_updates=test_only_updates + ), + ) + + np.testing.assert_allclose( + np.asarray(train_sched["value"]), + np.asarray(train_base["value"]), + rtol=1e-6, + atol=1e-6, + ) + assert not np.allclose( + np.asarray(test_sched["value"]), + np.asarray(test_base["value"]), + ) + # ============================================================================ # Validation and Early Stopping Tests From 955dca3055868178d04f3d47caf2bac1002921c0 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Thu, 5 Mar 2026 15:51:50 +0000 Subject: [PATCH 007/115] add fixes for linear interpolation bug and tests --- quantammsim/pools/reCLAMM/reclamm_reserves.py | 27 +++-- .../test_reclamm_price_ratio_updates.py | 107 +++++++++++++++++- 2 files changed, 126 insertions(+), 8 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index b098eb12..5f1ac85d 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -791,6 +791,9 @@ def _apply_schedule_state(_): event_has, jnp.maximum(event_end_step, step_idx), active_end_step ) next_active_enabled = jnp.where(event_has, True, active_enabled) + next_active_enabled = jnp.logical_and( + next_active_enabled, step_idx <= next_active_end_step + ) schedule_duration = next_active_end_step - next_active_start_step schedule_progress = jnp.where( @@ -798,16 +801,22 @@ def _apply_schedule_state(_): 1.0, jnp.clip((step_idx - next_active_start_step) / schedule_duration, 0.0, 1.0), ) - scheduled_price_ratio = ( - next_active_start_ratio - + (next_active_target_ratio - next_active_start_ratio) * schedule_progress + safe_start_ratio = jnp.maximum(next_active_start_ratio, 1.0 + 1e-12) + safe_target_ratio = jnp.maximum(next_active_target_ratio, 1.0 + 1e-12) + scheduled_price_ratio = safe_start_ratio * ( + safe_target_ratio / safe_start_ratio + ) ** schedule_progress + scheduled_price_ratio = jnp.where( + next_active_enabled, scheduled_price_ratio, current_price_ratio ) Va_scheduled, Vb_scheduled = apply_target_price_ratio_to_virtual_balances( Ra, Rb, Va, Vb, scheduled_price_ratio ) + next_Va = jnp.where(next_active_enabled, Va_scheduled, Va) + next_Vb = jnp.where(next_active_enabled, Vb_scheduled, Vb) return ( - Va_scheduled, - Vb_scheduled, + next_Va, + next_Vb, next_active_start_ratio, next_active_target_ratio, next_active_start_step, @@ -816,6 +825,9 @@ def _apply_schedule_state(_): ) def _skip_schedule_state(_): + retained_active_enabled = jnp.logical_and( + active_enabled, step_idx <= active_end_step + ) return ( Va, Vb, @@ -823,10 +835,11 @@ def _skip_schedule_state(_): active_target_ratio, active_start_step, active_end_step, - active_enabled, + retained_active_enabled, ) - schedule_active = jnp.logical_or(event_has, active_enabled) + active_not_expired = jnp.logical_and(active_enabled, step_idx <= active_end_step) + schedule_active = jnp.logical_or(event_has, active_not_expired) ( Va, Vb, diff --git a/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py index 5ee98bcd..db24aca8 100644 --- a/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py +++ b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py @@ -21,7 +21,7 @@ DEFAULT_PRICE_RATIO = 4.0 DEFAULT_DAILY_PRICE_SHIFT_BASE = 1.0 - 1.0 / 124000.0 DEFAULT_SECONDS_PER_STEP = 60.0 -ALL_SIG_VARIATIONS_2 = tuple(map(tuple, [[1, -1], [-1, 1]])) +ALL_SIG_VARIATIONS_2 = jnp.array([[1, -1], [-1, 1]]) def _init_pool(price_ratio=DEFAULT_PRICE_RATIO): @@ -143,6 +143,111 @@ def test_single_schedule_reaches_target_ratio_at_end_step(self): ) assert ratio_at_end == pytest.approx(9.0, rel=1e-5, abs=1e-5) + def test_schedule_interpolates_geometrically_in_ratio_space(self): + reserves, Va, Vb = _init_pool() + n_steps = 8 + prices = _flat_prices(n_steps) + fees = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + start_step = 1 + end_step = 5 + start_ratio = 4.0 + target_ratio = 16.0 + schedule = _single_event_schedule( + n_steps, + start_step=start_step, + end_step=end_step, + target_price_ratio=target_ratio, + start_price_ratio_override=start_ratio, + ) + + reserves_out, Va_history, Vb_history = ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.0, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + ) + + mid_step = 3 + ratio_at_mid = float( + compute_price_ratio( + reserves_out[mid_step, 0], + reserves_out[mid_step, 1], + Va_history[mid_step], + Vb_history[mid_step], + ) + ) + progress = (mid_step - start_step) / (end_step - start_step) + expected_geometric = start_ratio * (target_ratio / start_ratio) ** progress + assert ratio_at_mid == pytest.approx(expected_geometric, rel=1e-5, abs=1e-5) + + def test_schedule_stops_applying_after_end_step(self): + reserves, Va, Vb = _init_pool() + n_steps = 10 + prices = jnp.stack( + [jnp.linspace(DEFAULT_INITIAL_PRICES[0], 5000.0, n_steps), jnp.ones((n_steps,))], + axis=1, + ) + fees = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.zeros((n_steps,), dtype=jnp.float64) + + end_step = 3 + schedule = _single_event_schedule( + n_steps, + start_step=1, + end_step=end_step, + target_price_ratio=9.0, + start_price_ratio_override=DEFAULT_PRICE_RATIO, + ) + reserves_out, Va_history, Vb_history = ( + _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.0, # disable thermostat to isolate schedule behavior + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + ) + + # Reserves continue evolving under changing market prices... + assert not np.allclose( + np.asarray(reserves_out[end_step + 1]), + np.asarray(reserves_out[end_step + 2]), + ) + # ...but virtual balances should be frozen once the schedule has ended. + npt.assert_allclose( + np.asarray(Va_history[end_step + 1 :]), + np.full((n_steps - (end_step + 1),), float(Va_history[end_step])), + rtol=1e-9, + atol=1e-9, + ) + npt.assert_allclose( + np.asarray(Vb_history[end_step + 1 :]), + np.full((n_steps - (end_step + 1),), float(Vb_history[end_step])), + rtol=1e-9, + atol=1e-9, + ) + def test_replacement_event_supersedes_active_event(self): reserves, Va, Vb = _init_pool() n_steps = 9 From d9471a4bea73ab4f8d9e3847fb897b71c63058d1 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Thu, 5 Mar 2026 16:53:14 +0000 Subject: [PATCH 008/115] enable straight through estimation given boolean flag properties of ratio changes etc for gradient based optimisation methods --- quantammsim/pools/reCLAMM/reclamm.py | 15 ++ quantammsim/pools/reCLAMM/reclamm_reserves.py | 85 +++++++-- .../runners/default_run_fingerprint.py | 1 + .../reCLAMM/test_reclamm_differentiability.py | 167 ++++++++++++++++++ 4 files changed, 256 insertions(+), 12 deletions(-) create mode 100644 tests/pools/reCLAMM/test_reclamm_differentiability.py diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 762301c8..0b1e2d4c 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -198,6 +198,11 @@ def _resolve_fees(params, run_fingerprint): return jnp.squeeze(params["fees"]) return run_fingerprint["fees"] + @staticmethod + def _resolve_ste_temperature(run_fingerprint): + """Resolve STE gate temperature for differentiable reCLAMM transitions.""" + return run_fingerprint.get("ste_temperature", 10.0) + @partial(jit, static_argnums=(2,)) def calculate_reserves_with_fees( self, @@ -208,6 +213,7 @@ def calculate_reserves_with_fees( additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) + ste_temperature = self._resolve_ste_temperature(run_fingerprint) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_with_fees( @@ -225,6 +231,7 @@ def calculate_reserves_with_fees( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + ste_temperature=ste_temperature, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -246,6 +253,7 @@ def calculate_reserves_and_fee_revenue_with_fees( LP fee revenue per timestep in USD. """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) + ste_temperature = self._resolve_ste_temperature(run_fingerprint) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -263,6 +271,7 @@ def calculate_reserves_and_fee_revenue_with_fees( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + ste_temperature=ste_temperature, ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), @@ -288,6 +297,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( LP fee revenue per timestep in USD. """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) + ste_temperature = self._resolve_ste_temperature(run_fingerprint) bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: @@ -317,6 +327,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + ste_temperature=ste_temperature, ) @partial(jit, static_argnums=(2,)) @@ -330,6 +341,7 @@ def _calculate_reserves_zero_fees( ) -> jnp.ndarray: """Protected zero-fee implementation for hooks and weight calculation.""" s = self._init_pool_state(params, run_fingerprint, prices, start_index) + ste_temperature = self._resolve_ste_temperature(run_fingerprint) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_zero_fees( @@ -340,6 +352,7 @@ def _calculate_reserves_zero_fees( s.seconds_per_step, arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, + ste_temperature=ste_temperature, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -366,6 +379,7 @@ def calculate_reserves_with_dynamic_inputs( additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) + ste_temperature = self._resolve_ste_temperature(run_fingerprint) bout_length = run_fingerprint["bout_length"] max_len = bout_length - 1 if run_fingerprint["arb_frequency"] != 1: @@ -395,6 +409,7 @@ def calculate_reserves_with_dynamic_inputs( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + ste_temperature=ste_temperature, ) def init_base_parameters( diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 5f1ac85d..7e0bab8c 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -17,7 +17,8 @@ import jax.numpy as jnp from jax import jit -from jax.lax import scan, cond +from jax.lax import scan, cond, stop_gradient +from jax.nn import sigmoid from jax.tree_util import Partial from functools import partial @@ -48,6 +49,35 @@ # Pure math functions # --------------------------------------------------------------------------- +def _ste_gate(hard_bool, soft_value): + """Hard forward / soft backward gate.""" + hard_value = hard_bool.astype(soft_value.dtype) + return soft_value + stop_gradient(hard_value - soft_value) + + +def _ste_greater_than(x, threshold, temperature=10.0): + hard = x > threshold + soft = sigmoid(temperature * (x - threshold)) + return _ste_gate(hard, soft) + + +def _ste_less_than(x, threshold, temperature=10.0): + hard = x < threshold + soft = sigmoid(temperature * (threshold - x)) + return _ste_gate(hard, soft) + + +def _ste_greater_equal(x, threshold, temperature=10.0): + hard = x >= threshold + soft = sigmoid(temperature * (x - threshold)) + return _ste_gate(hard, soft) + + +def _ste_select(mask, when_true, when_false): + """Select between two values using a 0/1 gate that can carry STE gradients.""" + return mask * when_true + (1.0 - mask) * when_false + + def compute_invariant(Ra, Rb, Va, Vb): """Compute constant-product invariant L = (Ra + Va) * (Rb + Vb).""" return (Ra + Va) * (Rb + Vb) @@ -598,6 +628,7 @@ def _reclamm_scan_step_zero_fees( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + ste_temperature=10.0, ): """Single scan step for zero-fee reClAMM pool. @@ -617,7 +648,6 @@ def _reclamm_scan_step_zero_fees( # Step 1: Update virtual balances if out of range centeredness, is_above = compute_centeredness(Ra, Rb, Va, Vb) sqrt_Q = jnp.sqrt(compute_price_ratio(Ra, Rb, Va, Vb)) - out_of_range = centeredness < centeredness_margin market_price = prices[0] / prices[1] # Centeredness-proportional scaling: margin/centeredness multiplier @@ -648,8 +678,11 @@ def _reclamm_scan_step_zero_fees( Va_updated = jnp.where(use_cal, Va_cal, Va_geo) Vb_updated = jnp.where(use_cal, Vb_cal, Vb_geo) - Va = jnp.where(out_of_range, Va_updated, Va) - Vb = jnp.where(out_of_range, Vb_updated, Vb) + out_of_range_gate = _ste_less_than( + centeredness, centeredness_margin, ste_temperature + ) + Va = _ste_select(out_of_range_gate, Va_updated, Va) + Vb = _ste_select(out_of_range_gate, Vb_updated, Vb) # Step 2: Analytical zero-fee arb on effective reserves L = compute_invariant(Ra, Rb, Va, Vb) @@ -702,12 +735,14 @@ def _reclamm_scan_step_zero_fees_full_state( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + ste_temperature=10.0, ): """TEST-ONLY: scan step that outputs (reserves, Va, Vb).""" new_carry, new_reserves = _reclamm_scan_step_zero_fees( carry_list, prices, centeredness_margin, daily_price_shift_base, seconds_per_step, arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, + ste_temperature=ste_temperature, ) return new_carry, (new_reserves, new_carry[1], new_carry[2]) @@ -725,6 +760,7 @@ def _reclamm_scan_step_with_fees_and_revenue( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Single scan step for reClAMM pool with fees, returning LP fee revenue. @@ -858,7 +894,6 @@ def _skip_schedule_state(_): # Step 1: Update virtual balances if out of range centeredness, is_above = compute_centeredness(Ra, Rb, Va, Vb) sqrt_Q = jnp.sqrt(compute_price_ratio(Ra, Rb, Va, Vb)) - out_of_range = centeredness < centeredness_margin market_price = prices[0] / prices[1] # Centeredness-proportional scaling: margin/centeredness multiplier @@ -888,14 +923,15 @@ def _skip_schedule_state(_): Va_updated = jnp.where(use_cal, Va_cal, Va_geo) Vb_updated = jnp.where(use_cal, Vb_cal, Vb_geo) - Va = jnp.where(out_of_range, Va_updated, Va) - Vb = jnp.where(out_of_range, Vb_updated, Vb) + out_of_range_gate = _ste_less_than( + centeredness, centeredness_margin, ste_temperature + ) + Va = _ste_select(out_of_range_gate, Va_updated, Va) + Vb = _ste_select(out_of_range_gate, Vb_updated, Vb) # Step 2: Compute arb trade using G3M machinery on effective reserves effective_reserves = jnp.array([Ra + Va, Rb + Vb]) - fees_are_being_charged = gamma != 1.0 - # Zero-fee analytical arb L = compute_invariant(Ra, Rb, Va, Vb) market_price = prices[0] / prices[1] @@ -918,15 +954,22 @@ def _skip_schedule_state(_): 0, ) - optimal_arb_trade = jnp.where(fees_are_being_charged, fee_trade, zero_fee_trade) + fees_gate = _ste_greater_than( + jnp.abs(gamma - 1.0), jnp.asarray(1e-12, dtype=gamma.dtype), ste_temperature + ) + optimal_arb_trade = _ste_select(fees_gate, fee_trade, zero_fee_trade) # Check profitability for arb profit_to_arb = -(optimal_arb_trade * prices).sum() - arb_thresh arb_external_cost = 0.5 * arb_fees * (jnp.abs(optimal_arb_trade) * prices).sum() - do_trade = profit_to_arb >= arb_external_cost # Apply trade to REAL reserves only - applied_trade = jnp.where(do_trade, optimal_arb_trade, 0.0) + trade_gate = _ste_greater_equal( + profit_to_arb, arb_external_cost, ste_temperature + ) + applied_trade = _ste_select( + trade_gate, optimal_arb_trade, jnp.zeros_like(optimal_arb_trade) + ) Ra_new = Ra + applied_trade[0] Rb_new = Rb + applied_trade[1] @@ -994,6 +1037,7 @@ def _reclamm_scan_step_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Single scan step for reClAMM pool with fees (reserves only). @@ -1012,6 +1056,7 @@ def _reclamm_scan_step_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) return new_carry, new_reserves @@ -1029,6 +1074,7 @@ def _reclamm_scan_step_with_fees_full_state( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """TEST-ONLY: fee scan step that also outputs virtual balances.""" new_carry, (new_reserves, _fee_rev) = _reclamm_scan_step_with_fees_and_revenue( @@ -1043,6 +1089,7 @@ def _reclamm_scan_step_with_fees_full_state( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) return new_carry, (new_reserves, new_carry[1], new_carry[2]) @@ -1058,6 +1105,7 @@ def _jax_calc_reclamm_reserves_zero_fees( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + ste_temperature=10.0, ): """Calculate reClAMM reserves over time with zero fees. @@ -1092,6 +1140,7 @@ def _jax_calc_reclamm_reserves_zero_fees( seconds_per_step=seconds_per_step, arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, + ste_temperature=ste_temperature, ) carry_init = [initial_reserves, initial_Va, initial_Vb] @@ -1110,6 +1159,7 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + ste_temperature=10.0, ): """TEST-ONLY: Like _jax_calc_reclamm_reserves_zero_fees but returns Va/Vb. @@ -1126,6 +1176,7 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( seconds_per_step=seconds_per_step, arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, + ste_temperature=ste_temperature, ) carry_init = [initial_reserves, initial_Va, initial_Vb] @@ -1149,6 +1200,7 @@ def _jax_calc_reclamm_reserves_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Calculate reClAMM reserves over time with fees. @@ -1189,6 +1241,7 @@ def _jax_calc_reclamm_reserves_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) carry_init = [ @@ -1230,6 +1283,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" n_assets = 2 @@ -1279,6 +1333,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) carry_init = [ @@ -1320,6 +1375,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """TEST-ONLY: dynamic-input reserve path returning virtual-balance history.""" n_assets = 2 @@ -1368,6 +1424,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) carry_init = [ @@ -1406,6 +1463,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1448,6 +1506,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) carry_init = [ @@ -1489,6 +1548,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + ste_temperature=10.0, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1544,6 +1604,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + ste_temperature=ste_temperature, ) carry_init = [ diff --git a/quantammsim/runners/default_run_fingerprint.py b/quantammsim/runners/default_run_fingerprint.py index 81bc806b..b02de680 100644 --- a/quantammsim/runners/default_run_fingerprint.py +++ b/quantammsim/runners/default_run_fingerprint.py @@ -98,6 +98,7 @@ "reclamm_interpolation_method": "geometric", # "geometric" or "constant_arc_length" "reclamm_arc_length_speed": None, # auto-calibrate from geometric onset if None "reclamm_centeredness_scaling": False, # scale speed by margin/centeredness + "ste_temperature": 10.0, # STE gate sharpness; higher is closer to hard threshold "reclamm_learn_arc_length_speed": False, # include arc_length_speed in trainable params "reclamm_use_shift_exponent": False, # parametrise shift rate as shift_exponent (log-friendly) "reclamm_learn_fees": False, # include fees in trainable params (Optuna search over fee level) diff --git a/tests/pools/reCLAMM/test_reclamm_differentiability.py b/tests/pools/reCLAMM/test_reclamm_differentiability.py new file mode 100644 index 00000000..29fdcbe2 --- /dev/null +++ b/tests/pools/reCLAMM/test_reclamm_differentiability.py @@ -0,0 +1,167 @@ +"""Differentiability tests for reCLAMM STE-gated training path behavior.""" + +import jax +import jax.numpy as jnp +import numpy as np +import numpy.testing as npt + +from quantammsim.pools.creator import create_pool +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + initialise_reclamm_reserves, + _jax_calc_reclamm_reserves_zero_fees, + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs, +) +from quantammsim.runners.jax_runner_utils import Hashabledict + + +ALL_SIG_VARIATIONS_2 = jnp.array([[1, -1], [-1, 1]]) +DEFAULT_POOL_VALUE = 1_000_000.0 +DEFAULT_INITIAL_PRICES = jnp.array([2500.0, 1.0], dtype=jnp.float64) +DEFAULT_PRICE_RATIO = 4.0 +DEFAULT_SHIFT_BASE = 1.0 - 1.0 / 124000.0 +DEFAULT_SECONDS_PER_STEP = 60.0 + + +def _init_pool_state(): + return initialise_reclamm_reserves( + DEFAULT_POOL_VALUE, + DEFAULT_INITIAL_PRICES, + DEFAULT_PRICE_RATIO, + ) + + +def _trending_prices(n_steps): + return jnp.stack( + [jnp.linspace(DEFAULT_INITIAL_PRICES[0], 4200.0, n_steps), jnp.ones((n_steps,))], + axis=1, + ) + + +def test_ste_forward_outputs_are_temperature_invariant(): + """STE hard-forward path should be invariant to STE temperature.""" + reserves, Va, Vb = _init_pool_state() + n_steps = 12 + prices = _trending_prices(n_steps) + fees = jnp.full((n_steps,), 0.003, dtype=jnp.float64) + arb_thresh = jnp.zeros((n_steps,), dtype=jnp.float64) + arb_fees = jnp.full((n_steps,), 0.0005, dtype=jnp.float64) + + schedule = np.zeros((n_steps, 4), dtype=np.float64) + schedule[:, 3] = np.nan + schedule[2] = np.array([1.0, 6.0, 7.0, DEFAULT_PRICE_RATIO], dtype=np.float64) + schedule = jnp.asarray(schedule) + + low_temp_reserves, low_temp_fee_revenue = ( + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.2, + daily_price_shift_base=DEFAULT_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ste_temperature=3.0, + ) + ) + high_temp_reserves, high_temp_fee_revenue = ( + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( + reserves, + Va, + Vb, + prices, + centeredness_margin=0.2, + daily_price_shift_base=DEFAULT_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=arb_thresh, + arb_fees=arb_fees, + price_ratio_updates=schedule, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ste_temperature=50.0, + ) + ) + npt.assert_allclose(high_temp_reserves, low_temp_reserves, rtol=1e-10, atol=1e-10) + npt.assert_allclose( + high_temp_fee_revenue, low_temp_fee_revenue, rtol=1e-10, atol=1e-10 + ) + + +def test_margin_gradient_is_finite_and_nonzero_in_zero_fee_kernel(): + """Centeredness-margin gradient should flow through always-on STE gates.""" + reserves, Va, Vb = _init_pool_state() + n_steps = 6 + prices = jnp.tile(DEFAULT_INITIAL_PRICES, (n_steps, 1)) + margin = jnp.float64(1.0) + + def _loss(centeredness_margin): + reserves_out = _jax_calc_reclamm_reserves_zero_fees( + reserves, + Va, + Vb, + prices, + centeredness_margin=centeredness_margin, + daily_price_shift_base=DEFAULT_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + ste_temperature=25.0, + ) + return jnp.sum(reserves_out[-1]) + + grad_val = jax.grad(_loss)(margin) + + assert jnp.isfinite(grad_val) + assert jnp.abs(grad_val) > 1e-9 + + +def test_pool_zero_fee_path_uses_configured_ste_temperature(): + """Pool-level path should pass STE temperature through to kernel gradients.""" + pool = create_pool("reclamm") + n_steps = 6 + prices = jnp.tile(DEFAULT_INITIAL_PRICES, (n_steps, 1)) + start_index = jnp.array([0, 0], dtype=jnp.int32) + + run_fp_low_temp = Hashabledict( + { + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": DEFAULT_POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "ste_temperature": 2.0, + } + ) + run_fp_high_temp = Hashabledict( + { + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": DEFAULT_POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "ste_temperature": 50.0, + } + ) + + def _loss(centeredness_margin, run_fingerprint): + params = { + "price_ratio": jnp.float64(DEFAULT_PRICE_RATIO), + "centeredness_margin": centeredness_margin, + "daily_price_shift_base": jnp.float64(DEFAULT_SHIFT_BASE), + } + reserves_out = pool.calculate_reserves_zero_fees( + params, run_fingerprint, prices, start_index + ) + return jnp.sum(reserves_out[-1]) + + margin = jnp.float64(1.0) + low_temp_grad = jax.grad(lambda m: _loss(m, run_fp_low_temp))(margin) + high_temp_grad = jax.grad(lambda m: _loss(m, run_fp_high_temp))(margin) + + assert jnp.isfinite(low_temp_grad) + assert jnp.isfinite(high_temp_grad) + assert jnp.abs(low_temp_grad) > 1e-9 + assert jnp.abs(high_temp_grad) > 1e-9 + assert jnp.abs(high_temp_grad) > jnp.abs(low_temp_grad) * 1.5 From 35719bbd168db179ad53f6b861ae5fc7e19f92a7 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 6 Mar 2026 00:46:13 +0000 Subject: [PATCH 009/115] fix: replace stale fees_array references with dynamic_inputs check The merge of dev into reclamm-phase-1 reintroduced a reference to the old fees_array/gas_cost_array/arb_fees_array/trades_array parameters in the fused reserves guard. These were replaced by the DynamicInputArrays container in the dynamic inputs refactor. Replace the stale check with `dynamic_inputs is None`, which is the correct guard under the new API. --- quantammsim/core_simulator/forward_pass.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index 5d018fee..c1509e98 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -1011,10 +1011,7 @@ def forward_pass( and static_dict["arb_frequency"] == 1 and static_dict.get("turnover_penalty", 0.0) == 0.0 and static_dict.get("price_noise_sigma", 0.0) == 0.0 - and all( - ele is None - for ele in [fees_array, gas_cost_array, arb_fees_array, trades_array] - ) + and dynamic_inputs is None and 1440 % static_dict["chunk_period"] == 0 # chunk_period divides metric_period and not pool._rule_outputs_are_weights # only delta-based pools validated and static_dict["bout_length"] > 1440 * 2 # need ≥2 metric periods From b6aab22d616654018abd660bb41ef91c3f858717 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 11:39:08 +0000 Subject: [PATCH 010/115] remove default get --- quantammsim/pools/reCLAMM/reclamm.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 0b1e2d4c..fbfecab4 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -201,7 +201,7 @@ def _resolve_fees(params, run_fingerprint): @staticmethod def _resolve_ste_temperature(run_fingerprint): """Resolve STE gate temperature for differentiable reCLAMM transitions.""" - return run_fingerprint.get("ste_temperature", 10.0) + return run_fingerprint.get("ste_temperature") @partial(jit, static_argnums=(2,)) def calculate_reserves_with_fees( From 4e86c4df1c6d6c9256f2d85720b5a1c6e945571c Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:43:53 +0000 Subject: [PATCH 011/115] noise calibration port from private repo --- quantammsim/noise_calibration/__init__.py | 39 + quantammsim/noise_calibration/cli.py | 417 +++++++++++ quantammsim/noise_calibration/constants.py | 60 ++ .../noise_calibration/covariate_encoding.py | 228 ++++++ .../noise_calibration/data_pipeline.py | 516 ++++++++++++++ .../noise_calibration/data_validation.py | 41 ++ quantammsim/noise_calibration/formula_arb.py | 36 + quantammsim/noise_calibration/inference.py | 270 +++++++ quantammsim/noise_calibration/model.py | 444 ++++++++++++ quantammsim/noise_calibration/output.py | 306 ++++++++ quantammsim/noise_calibration/plotting.py | 335 +++++++++ .../noise_calibration/postprocessing.py | 667 ++++++++++++++++++ .../noise_calibration/token_classification.py | 23 + 13 files changed, 3382 insertions(+) create mode 100644 quantammsim/noise_calibration/__init__.py create mode 100644 quantammsim/noise_calibration/cli.py create mode 100644 quantammsim/noise_calibration/constants.py create mode 100644 quantammsim/noise_calibration/covariate_encoding.py create mode 100644 quantammsim/noise_calibration/data_pipeline.py create mode 100644 quantammsim/noise_calibration/data_validation.py create mode 100644 quantammsim/noise_calibration/formula_arb.py create mode 100644 quantammsim/noise_calibration/inference.py create mode 100644 quantammsim/noise_calibration/model.py create mode 100644 quantammsim/noise_calibration/output.py create mode 100644 quantammsim/noise_calibration/plotting.py create mode 100644 quantammsim/noise_calibration/postprocessing.py create mode 100644 quantammsim/noise_calibration/token_classification.py diff --git a/quantammsim/noise_calibration/__init__.py b/quantammsim/noise_calibration/__init__.py new file mode 100644 index 00000000..f1fddde2 --- /dev/null +++ b/quantammsim/noise_calibration/__init__.py @@ -0,0 +1,39 @@ +"""Noise calibration package for Balancer pool volume models. + +Public API re-exports from submodules. +""" + +# scipy.signal patch (must run before arviz import) +try: + from scipy.signal import gaussian as _ # noqa: F401 +except ImportError: + from scipy.signal.windows import gaussian as _gauss + import scipy.signal + scipy.signal.gaussian = _gauss + +from .constants import ( + K_COEFF, COEFF_NAMES, BALANCER_API_URL, BALANCER_API_CHAINS, CACHE_DIR, + K_CLUSTERS_DEFAULT, K_FEATURES_DEFAULT, +) +from .token_classification import classify_token_tier, _normalise_symbol +from .data_pipeline import ( + _graphql_request, enumerate_balancer_pools, fetch_pool_snapshots, + fetch_all_snapshots, fetch_token_prices, compute_pair_volatility, + assemble_panel, +) +from .data_validation import validate_panel +from .covariate_encoding import encode_covariates, encode_covariates_structural +from .model import noise_model, noise_model_dp_sigma, noise_model_ibp, noise_model_ibp_dp, stick_breaking_weights, structural_noise_model +from .formula_arb import formula_arb_volume_daily_jax +from .inference import ( + _get_theta_samples, _build_model_kwargs, run_svi, run_nuts, + run_svi_then_nuts, +) +from .postprocessing import ( + extract_noise_params, predict_new_pool, check_convergence, + run_prior_predictive, assign_dp_clusters, assign_ibp_dp_joint, + extract_structural_params, predict_new_pool_structural, +) +from .plotting import plot_diagnostics +from .output import generate_output_json, _save_sample_cache +from .cli import main diff --git a/quantammsim/noise_calibration/cli.py b/quantammsim/noise_calibration/cli.py new file mode 100644 index 00000000..3e169ac3 --- /dev/null +++ b/quantammsim/noise_calibration/cli.py @@ -0,0 +1,417 @@ +"""CLI entry point for noise calibration.""" + +import argparse +import json +import os +import sys +from datetime import date, timedelta + +import numpy as np +import pandas as pd + +from .constants import CACHE_DIR +from .data_pipeline import ( + enumerate_balancer_pools, fetch_all_snapshots, + fetch_token_prices, assemble_panel, +) +from .data_validation import validate_panel +from .covariate_encoding import encode_covariates +from .inference import run_svi, run_nuts, run_svi_then_nuts +from .postprocessing import ( + extract_noise_params, predict_new_pool, + check_convergence, run_prior_predictive, +) +from .plotting import plot_diagnostics +from .output import generate_output_json, _save_sample_cache + + +def _parse_args(): + parser = argparse.ArgumentParser( + description="Unified Bayesian hierarchical noise volume model " + "for Balancer pools (gold standard)" + ) + + # Actions + parser.add_argument("--fetch", action="store_true", + help="Fetch data from Balancer API") + parser.add_argument("--fit", action="store_true", + help="Run inference (SVI default)") + parser.add_argument("--nuts", action="store_true", + help="Use NUTS instead of SVI") + parser.add_argument("--svi-init-nuts", action="store_true", + help="SVI-initialized NUTS (fast warmup)") + parser.add_argument("--plot", action="store_true", + help="Generate diagnostic plots") + parser.add_argument("--prior-predictive", action="store_true", + help="Include prior predictive check") + parser.add_argument("--validate", action="store_true", + help="Run data validation pass") + parser.add_argument("--predict", action="store_true", + help="Predict for unseen pool") + + # Output + parser.add_argument("--output", default=None, + help="Output JSON path") + parser.add_argument("--output-dir", default="results", + help="Plot output directory (default: results)") + + # Predict args + parser.add_argument("--chain", default=None, + help="Chain for --predict") + parser.add_argument("--tokens", nargs="+", default=None, + help="Tokens for --predict") + parser.add_argument("--fee", type=float, default=0.003, + help="Fee for --predict") + + # NUTS hyperparameters + parser.add_argument("--num-warmup", type=int, default=1000, + help="NUTS warmup iterations (default: 1000)") + parser.add_argument("--num-samples", type=int, default=2000, + help="NUTS/SVI samples (default: 2000)") + parser.add_argument("--num-chains", type=int, default=4, + help="NUTS chains (default: 4)") + parser.add_argument("--target-accept", type=float, default=0.85, + help="NUTS target accept prob (default: 0.85)") + parser.add_argument("--max-tree-depth", type=int, default=10, + help="NUTS max tree depth (default: 10)") + parser.add_argument("--seed", type=int, default=42, + help="Random seed (default: 42)") + + # SVI hyperparameters + parser.add_argument("--svi-steps", type=int, default=20000, + help="SVI optimization steps (default: 20000)") + parser.add_argument("--svi-lr", type=float, default=1e-3, + help="SVI learning rate (default: 1e-3)") + + # Model variant + parser.add_argument("--model", choices=["tier", "dp_sigma", "ibp", "ibp_dp", + "structural"], + default="tier", + help="Noise model variant: 'tier' (per-tier sigma_eps), " + "'dp_sigma' (DP mixture on sigma_eps), " + "'ibp' (IBP latent features), " + "'ibp_dp' (IBP features + DP noise clusters), or " + "'structural' (structural mixture: arb + MoE noise)") + parser.add_argument("--k-clusters", type=int, default=6, + help="Number of DP mixture components " + "(capacity ceiling, default: 6)") + parser.add_argument("--k-features", type=int, default=6, + help="Number of IBP latent features " + "(default: 6)") + + # Data + parser.add_argument("--train-days", type=int, default=90, + help="Use only the last N days of data for fitting " + "(default: 90). Aligns with Balancer API hourly price " + "coverage window. Set to 0 to use all data.") + parser.add_argument("--min-tvl", type=float, default=10000.0, + help="Pool enumeration TVL filter") + parser.add_argument("--cache-dir", default=None, + help="Cache directory") + parser.add_argument("--device", choices=["cpu", "gpu", "auto"], + default="auto", + help="JAX device (default: auto)") + + return parser.parse_args() + + +def main(): + args = _parse_args() + + if not any([args.fetch, args.fit, args.predict, args.validate]): + print("ERROR: At least one of --fetch, --fit, --predict, --validate " + "is required", file=sys.stderr) + sys.exit(1) + + cache_dir = args.cache_dir or CACHE_DIR + + # --- JAX setup (BEFORE any JAX ops / imports) --- + if args.fit or args.predict or args.prior_predictive: + # Set device before importing JAX + if args.device == "cpu": + os.environ.setdefault("JAX_PLATFORMS", "cpu") + elif args.device == "gpu": + os.environ.setdefault("JAX_PLATFORMS", "cuda") + # auto: don't touch JAX_PLATFORMS, let JAX pick + + # Set host device count for NUTS multi-chain BEFORE JAX init + if args.nuts or args.svi_init_nuts: + import numpyro as _np_pre + _np_pre.set_host_device_count( + min(args.num_chains, os.cpu_count() or 4) + ) + + import jax + import numpyro + numpyro.enable_x64() + + # --- File paths --- + pools_cache = os.path.join(cache_dir, "pools.parquet") + snaps_cache = os.path.join(cache_dir, "pool_snapshots.parquet") + prices_cache = os.path.join(cache_dir, "token_prices") + panel_cache = os.path.join(cache_dir, "panel.parquet") + + # --- Fetch --- + if args.fetch: + print("Phase 1: Fetching data from Balancer API") + print("=" * 60) + + print("\n1. Enumerating pools...") + pools_df = enumerate_balancer_pools(min_tvl=args.min_tvl) + os.makedirs(cache_dir, exist_ok=True) + pools_df.to_parquet(pools_cache, index=False) + print(f" Saved {len(pools_df)} pools -> {pools_cache}") + + print("\n2. Fetching daily snapshots...") + snapshots_df = fetch_all_snapshots(pools_df, cache_path=snaps_cache) + + print("\n3. Fetching token prices...") + token_addr_by_chain = {} + for _, pool in pools_df.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + print("\n4. Assembling panel (with lagged TVL)...") + panel = assemble_panel(pools_df, snapshots_df, token_prices) + panel.to_parquet(panel_cache, index=False) + print(f" Saved panel -> {panel_cache}") + + print(f"\nFetch complete. Panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # --- Validate --- + if args.validate: + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + panel = pd.read_parquet(panel_cache) + validate_panel(panel) + + # --- Fit --- + if args.fit: + print("\nUnified Noise Volume Model") + print("=" * 60) + + # Load panel + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_cache) + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + # Filter to recent window for training + if args.train_days > 0: + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=args.train_days) + n_before = len(panel) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + print(f" Filtered to last {args.train_days} days " + f"(>= {cutoff}): {len(panel)} obs " + f"(dropped {n_before - len(panel)})") + + # Ensure lagged TVL exists (in case loaded from old cache) + if "log_tvl_lag1" not in panel.columns: + print(" Adding lagged TVL to cached panel...") + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + # Filter: need at least 10 days per pool + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid_pools)].copy() + print(f" After filtering (>= 10 days): {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # Select model variant + if args.model == "structural": + from .model import structural_noise_model + from .covariate_encoding import encode_covariates_structural + model_fn = structural_noise_model + data = encode_covariates_structural(panel) + print(f" Model: structural mixture (arb + MoE noise)") + elif args.model == "ibp_dp": + from .model import noise_model_ibp_dp + model_fn = noise_model_ibp_dp + data = encode_covariates(panel, include_tiers=False) + data["K_features"] = args.k_features + data["K_clusters"] = args.k_clusters + print(f" Model: IBP+DP hybrid " + f"(K_features={args.k_features}, " + f"K_clusters={args.k_clusters})") + elif args.model == "ibp": + from .model import noise_model_ibp + model_fn = noise_model_ibp + data = encode_covariates(panel, include_tiers=False) + data["K_features"] = args.k_features + print(f" Model: IBP latent features " + f"(K_features={args.k_features})") + elif args.model == "dp_sigma": + from .model import noise_model_dp_sigma + model_fn = noise_model_dp_sigma + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = args.k_clusters + print(f" Model: DP mixture on sigma_eps " + f"(K_clusters={args.k_clusters})") + else: + model_fn = None # default = noise_model + data = encode_covariates(panel) + + # Prior predictive + prior_samples = None + if args.prior_predictive: + print("\n Running prior predictive check...") + prior_samples = run_prior_predictive(data, model_fn=model_fn) + + # Inference + mcmc_obj = None + elbo_losses = None + inference_config = {"seed": args.seed} + + if args.svi_init_nuts: + inference_config["method"] = "svi_init_nuts" + inference_config["svi_steps"] = args.svi_steps + inference_config["svi_lr"] = args.svi_lr + inference_config["num_warmup"] = args.num_warmup + inference_config["num_samples"] = args.num_samples + inference_config["num_chains"] = args.num_chains + inference_config["target_accept"] = args.target_accept + inference_config["max_tree_depth"] = args.max_tree_depth + + mcmc_obj, elbo_losses = run_svi_then_nuts( + data, + svi_steps=args.svi_steps, + svi_lr=args.svi_lr, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + model_fn=model_fn, + ) + samples = mcmc_obj + convergence = check_convergence(mcmc_obj, method="nuts") + + elif args.nuts: + inference_config["method"] = "nuts" + inference_config["num_warmup"] = args.num_warmup + inference_config["num_samples"] = args.num_samples + inference_config["num_chains"] = args.num_chains + inference_config["target_accept"] = args.target_accept + inference_config["max_tree_depth"] = args.max_tree_depth + + mcmc_obj = run_nuts( + data, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + model_fn=model_fn, + ) + samples = mcmc_obj + convergence = check_convergence(mcmc_obj, method="nuts") + + else: + inference_config["method"] = "svi" + inference_config["svi_steps"] = args.svi_steps + inference_config["svi_lr"] = args.svi_lr + inference_config["num_samples"] = args.num_samples + + samples, elbo_losses = run_svi( + data, + num_steps=args.svi_steps, + lr=args.svi_lr, + seed=args.seed, + num_samples=args.num_samples, + model_fn=model_fn, + ) + convergence = check_convergence(elbo_losses, method="svi") + + if args.model == "structural": + from .postprocessing import extract_structural_params + pool_params = extract_structural_params(samples, data) + arb_freqs = [p["arb_frequency"] for p in pool_params] + print(f"\n Per-pool arb_frequency: " + f"mean={np.mean(arb_freqs):.1f}, " + f"range=[{np.min(arb_freqs)}, {np.max(arb_freqs)}]") + else: + pool_params = extract_noise_params(samples, data) + b_c_vals = [p["noise_params"]["b_c"] for p in pool_params] + b_0_vals = [p["noise_params"]["b_0"] for p in pool_params] + print(f"\n Per-pool b_c: mean={np.mean(b_c_vals):.3f}, " + f"std={np.std(b_c_vals):.3f}, " + f"range=[{np.min(b_c_vals):.3f}, {np.max(b_c_vals):.3f}]") + print(f" Per-pool b_0: mean={np.mean(b_0_vals):.3f}, " + f"std={np.std(b_0_vals):.3f}") + + if args.output: + generate_output_json( + pool_params, samples, data, convergence, + args.output, inference_config, + ) + + if args.plot: + print("\nGenerating diagnostic plots...") + plot_diagnostics( + samples, data, output_dir=args.output_dir, + elbo_losses=elbo_losses, mcmc=mcmc_obj, + prior_samples=prior_samples, + ) + + # Cache samples for --predict + _save_sample_cache(samples, data, cache_dir) + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + print("ERROR: --predict requires --chain and --tokens", + file=sys.stderr) + sys.exit(1) + + # Load cached samples + sample_cache = os.path.join(cache_dir, "unified_samples.npz") + data_cache = os.path.join(cache_dir, "unified_data.json") + + if not os.path.exists(sample_cache): + print(f"ERROR: Sample cache not found at {sample_cache}", + file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + cached = np.load(sample_cache) + sample_dict = {k: cached[k] for k in cached.files} + + with open(data_cache) as f: + data_meta = json.load(f) + + result = predict_new_pool( + sample_dict, data_meta, args.chain, args.tokens, args.fee + ) + print(json.dumps(result, indent=2)) diff --git a/quantammsim/noise_calibration/constants.py b/quantammsim/noise_calibration/constants.py new file mode 100644 index 00000000..05a217d4 --- /dev/null +++ b/quantammsim/noise_calibration/constants.py @@ -0,0 +1,60 @@ +"""Constants for noise calibration.""" + +import os + +K_COEFF = 4 +COEFF_NAMES = ["intercept", "b_tvl", "b_sigma", "b_weekend"] + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAINS = [ + "MAINNET", "POLYGON", "ARBITRUM", "GNOSIS", "BASE", "SONIC", "OPTIMISM", + "AVALANCHE", +] + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "local_data", "noise_calibration", +) + +# Tier 0: blue-chip — top by volume, wrapped native, major stables +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", +} + +K_CLUSTERS_DEFAULT = 6 +K_FEATURES_DEFAULT = 6 + +# Structural model: observation-level covariates (expanded from K_COEFF=4) +K_OBS_COEFF = 8 +OBS_COEFF_NAMES = [ + "intercept", "b_tvl", "b_sigma", + "b_tvl_sigma", "b_tvl_fee", "b_sigma_fee", + "b_dow_sin", "b_dow_cos", +] + +# Gas costs per arb transaction (USD) by chain +GAS_COSTS = { + "MAINNET": None, # time-varying, loaded from CSV + "POLYGON": 0.005, + "ARBITRUM": 0.005, + "BASE": 0.005, + "GNOSIS": 0.01, + "OPTIMISM": 0.005, + "SONIC": 0.005, + "AVALANCHE": 0.005, + "MODE": 0.005, + "FRAXTAL": 0.005, +} + +# Tier 1: mid-cap DeFi blue-chips (approx CoinGecko rank < 200) +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} diff --git a/quantammsim/noise_calibration/covariate_encoding.py b/quantammsim/noise_calibration/covariate_encoding.py new file mode 100644 index 00000000..a298a6b4 --- /dev/null +++ b/quantammsim/noise_calibration/covariate_encoding.py @@ -0,0 +1,228 @@ +"""Covariate encoding for the hierarchical noise model.""" + +import numpy as np +import pandas as pd + +from .constants import K_COEFF, K_OBS_COEFF, GAS_COSTS + + +def encode_covariates(panel: pd.DataFrame, include_tiers: bool = True) -> dict: + """Build NumPyro-ready arrays from the panel DataFrame. + + Returns dict with arrays for the model plus metadata for output/prediction. + Key difference from hierarchical script: x_obs uses log_tvl_lag1 not log_tvl. + """ + pool_meta = panel.drop_duplicates("pool_id").reset_index(drop=True) + pool_ids = pool_meta["pool_id"].values + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + N_pools = len(pool_ids) + + pool_idx = panel["pool_id"].map(pool_id_to_idx).values + + # --- Build X_pool (pool-level covariates, data-driven) --- + chains = sorted(panel["chain"].unique()) + ref_chain = chains[0] + chain_cols = [] + chain_names = [] + for c in chains[1:]: + chain_cols.append((pool_meta["chain"] == c).astype(float).values) + chain_names.append(f"chain_{c}") + + tier_a_vals = sorted(pool_meta["tier_A"].astype(str).unique()) + ref_tier_a = tier_a_vals[0] + tier_a_cols = [] + tier_a_names = [] + if include_tiers: + for t in tier_a_vals[1:]: + tier_a_cols.append( + (pool_meta["tier_A"].astype(str) == t).astype(float).values + ) + tier_a_names.append(f"tier_A_{t}") + + tier_b_vals = sorted(pool_meta["tier_B"].astype(str).unique()) + ref_tier_b = tier_b_vals[0] + tier_b_cols = [] + tier_b_names = [] + if include_tiers: + for t in tier_b_vals[1:]: + tier_b_cols.append( + (pool_meta["tier_B"].astype(str) == t).astype(float).values + ) + tier_b_names.append(f"tier_B_{t}") + + columns = [np.ones((N_pools, 1))] + col_names = ["intercept"] + + for arr, name in zip(chain_cols, chain_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_a_cols, tier_a_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_b_cols, tier_b_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + columns.append(pool_meta["log_fee"].values.reshape(-1, 1)) + col_names.append("log_fee") + + X_pool = np.hstack(columns) + K_cov = X_pool.shape[1] + + # --- Observation-level arrays (uses LAGGED TVL) --- + x_obs = np.column_stack([ + np.ones(len(panel)), + panel["log_tvl_lag1"].values, + panel["volatility"].values, + panel["weekend"].values, + ]).astype(np.float64) + + y_obs = panel["log_volume"].values.astype(np.float64) + + # --- Per-pool tier_A index for per-tier sigma_eps --- + tier_A_per_pool = pool_meta["tier_A"].values.astype(np.int32) + + print(f" Encoded: N_obs={len(y_obs)}, N_pools={N_pools}, " + f"K_coeff={K_COEFF}, K_cov={K_cov}") + print(f" Covariates: {col_names}") + print(f" Tier distribution: " + f"T0={np.sum(tier_A_per_pool == 0)}, " + f"T1={np.sum(tier_A_per_pool == 1)}, " + f"T2={np.sum(tier_A_per_pool == 2)}") + + return { + "pool_idx": pool_idx.astype(np.int32), + "X_pool": X_pool.astype(np.float64), + "x_obs": x_obs, + "y_obs": y_obs, + "pool_ids": list(pool_ids), + "pool_meta": pool_meta, + "covariate_names": col_names, + "tier_A_per_pool": tier_A_per_pool, + "N_pools": N_pools, + "K_cov": K_cov, + "ref_chain": ref_chain, + "ref_tier_a": ref_tier_a, + "ref_tier_b": ref_tier_b, + "chains": chains, + } + + +def _tier_pair_idx(a: int, b: int) -> int: + """Encode (tier_A, tier_B) pair as a single index. + + Upper triangle of 3x3 grid: + (0,0)->0, (0,1)->1, (0,2)->2, (1,1)->3, (1,2)->4, (2,2)->5. + """ + return a * (5 - a) // 2 + b - a + + +def encode_covariates_structural( + panel: pd.DataFrame, + gas: np.ndarray = None, +) -> dict: + """Build NumPyro-ready arrays for the structural mixture model. + + Extends encode_covariates with: + - x_obs: 8 columns (intercept, tvl, log_sigma, interactions, DOW harmonics) + - Additional arrays: sigma_daily, fee, gas, chain_idx, tier_idx, lag_log_tvl + - n_chains, n_tiers computed from panel + + Parameters + ---------- + panel : pd.DataFrame + Output of assemble_panel(), must have log_sigma, dow_sin, dow_cos, + tvl_x_sigma, tvl_x_fee, sigma_x_fee columns. + gas : np.ndarray, optional + Per-observation gas costs in USD. If None, uses default (0.01 for all). + """ + # Ensure structural columns exist (compute from base columns if missing) + if "log_sigma" not in panel.columns: + panel = panel.copy() + panel["log_sigma"] = np.log(np.maximum(panel["volatility"].values, 1e-6)) + dow = panel["date"].apply( + lambda d: d.weekday() if hasattr(d, "weekday") + else pd.Timestamp(d).weekday() + ) + panel["dow_sin"] = np.sin(2.0 * np.pi * dow / 7.0) + panel["dow_cos"] = np.cos(2.0 * np.pi * dow / 7.0) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + # Reuse X_pool construction from encode_covariates (with tiers for gating) + base = encode_covariates(panel, include_tiers=True) + + # --- Observation-level x_obs: 8 columns --- + x_obs = np.column_stack([ + np.ones(len(panel)), # intercept + panel["log_tvl_lag1"].values, # lagged TVL + panel["log_sigma"].values, # log(volatility) + panel["tvl_x_sigma"].values, # tvl × sigma interaction + panel["tvl_x_fee"].values, # tvl × fee interaction + panel["sigma_x_fee"].values, # sigma × fee interaction + panel["dow_sin"].values, # DOW harmonic sin + panel["dow_cos"].values, # DOW harmonic cos + ]).astype(np.float64) + + # --- Additional arrays for the structural model --- + sigma_daily = (panel["volatility"] / np.sqrt(365.0)).values.astype(np.float64) + fee_per_obs = np.exp(panel["log_fee"].values).astype(np.float64) + lag_log_tvl = panel["log_tvl_lag1"].values.astype(np.float64) + + # Gas: per-observation + if gas is not None: + gas_arr = np.asarray(gas, dtype=np.float64) + else: + gas_arr = np.full(len(panel), 0.01, dtype=np.float64) + + # Chain index: integer per pool + pool_meta = base["pool_meta"] + chains = base["chains"] + chain_to_idx = {c: i for i, c in enumerate(chains)} + chain_idx_per_pool = np.array( + [chain_to_idx[c] for c in pool_meta["chain"]], dtype=np.int32, + ) + + # Tier pair index: per pool + tier_idx_per_pool = np.array( + [_tier_pair_idx(int(row["tier_A"]), int(row["tier_B"])) + for _, row in pool_meta.iterrows()], + dtype=np.int32, + ) + + # Count unique tier pairs and chains + n_chains = len(chains) + tier_pairs = set() + for _, row in pool_meta.iterrows(): + tier_pairs.add((int(row["tier_A"]), int(row["tier_B"]))) + n_tiers = 6 # fixed: upper triangle of 3x3 + + print(f" Structural encoding: N_obs={len(panel)}, " + f"N_pools={base['N_pools']}, n_chains={n_chains}, n_tiers={n_tiers}") + + return { + # Base arrays (same as encode_covariates) + "pool_idx": base["pool_idx"], + "X_pool": base["X_pool"], + "x_obs": x_obs, + "y_obs": base["y_obs"], + "pool_ids": base["pool_ids"], + "pool_meta": pool_meta, + "covariate_names": base["covariate_names"], + "tier_A_per_pool": base["tier_A_per_pool"], + "N_pools": base["N_pools"], + "K_cov": base["K_cov"], + "ref_chain": base["ref_chain"], + "ref_tier_a": base["ref_tier_a"], + "ref_tier_b": base["ref_tier_b"], + "chains": chains, + # Structural model extras + "sigma_daily": sigma_daily, + "fee": fee_per_obs, + "gas": gas_arr, + "chain_idx": chain_idx_per_pool, + "tier_idx": tier_idx_per_pool, + "lag_log_tvl": lag_log_tvl, + "n_chains": n_chains, + "n_tiers": n_tiers, + } diff --git a/quantammsim/noise_calibration/data_pipeline.py b/quantammsim/noise_calibration/data_pipeline.py new file mode 100644 index 00000000..b824cdc4 --- /dev/null +++ b/quantammsim/noise_calibration/data_pipeline.py @@ -0,0 +1,516 @@ +"""Data pipeline: fetch pools, snapshots, prices, and assemble panel.""" + +import json +import os +import time +import urllib.request +from datetime import datetime + +import numpy as np +import pandas as pd + +from .constants import BALANCER_API_URL, BALANCER_API_CHAINS +from .token_classification import classify_token_tier + + +def _graphql_request(query: dict, base_url: str = BALANCER_API_URL, + timeout: int = 30) -> dict: + """Send a GraphQL request to the Balancer V3 API.""" + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8")) + + +def enumerate_balancer_pools( + chains: list = None, + pool_types: list = None, + min_tvl: float = 10000.0, +) -> pd.DataFrame: + """Enumerate all WEIGHTED + RECLAMM pools across chains from Balancer API.""" + if chains is None: + chains = BALANCER_API_CHAINS + if pool_types is None: + pool_types = ["WEIGHTED", "RECLAMM"] + + all_pools = [] + for chain in chains: + print(f" Querying {chain}...", end=" ", flush=True) + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { + chainIn: [$chain] + poolTypeIn: $types + minTvl: $minTvl + } + ) { + id + chain + type + createTime + protocolVersion + poolTokens { + symbol + weight + address + } + dynamicData { + totalLiquidity + swapFee + } + } + } + """, + "variables": { + "chain": chain, + "types": pool_types, + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f"FAILED ({e})") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + weights = [t.get("weight") for t in p.get("poolTokens", [])] + token_addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "protocol_version": p.get("protocolVersion", 0), + "tokens": tokens, + "token_addresses": token_addresses, + "weights": weights, + "swap_fee": fee, + "create_time": p.get("createTime", 0), + "current_tvl": tvl, + }) + + print(f"{len(pools)} pools") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools across {len(chains)} chains") + return df + + +def fetch_pool_snapshots(pool_id: str, chain: str, + base_url: str = BALANCER_API_URL) -> pd.DataFrame: + """Fetch ALL_TIME daily snapshots for a single pool.""" + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + totalShares + } + } + """, + "variables": { + "poolId": pool_id, + "chain": chain, + "range": "ALL_TIME", + }, + } + + body = _graphql_request(query) + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + + if not snapshots: + return pd.DataFrame(columns=["timestamp", "volume_usd", + "total_liquidity_usd", "total_shares"]) + + records = [] + for snap in snapshots: + records.append({ + "timestamp": int(snap["timestamp"]), + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + "total_shares": float(snap.get("totalShares", 0)), + }) + + df = pd.DataFrame(records) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df + + +def fetch_all_snapshots(pools_df: pd.DataFrame, + cache_path: str = None) -> pd.DataFrame: + """Fetch daily snapshots for all pools, with caching.""" + cached = pd.DataFrame() + cached_pool_ids = set() + if cache_path and os.path.exists(cache_path): + cached = pd.read_parquet(cache_path) + cached_pool_ids = set(cached["pool_id"].unique()) + print(f" Cache has {len(cached_pool_ids)} pools, " + f"{len(cached)} pool-days") + + if len(pools_df) == 0: + print(" No pools to fetch.") + return cached if len(cached) > 0 else pd.DataFrame( + columns=["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd", "total_shares"] + ) + to_fetch = pools_df[~pools_df["pool_id"].isin(cached_pool_ids)] + print(f" Need to fetch {len(to_fetch)} new pools") + + new_records = [] + for i, (_, pool) in enumerate(to_fetch.iterrows()): + if (i + 1) % 10 == 0 or i == 0: + print(f" Fetching {i+1}/{len(to_fetch)}: {pool['pool_id'][:10]}... " + f"({pool['chain']})", flush=True) + try: + snap_df = fetch_pool_snapshots(pool["pool_id"], pool["chain"]) + if len(snap_df) > 0: + snap_df["pool_id"] = pool["pool_id"] + snap_df["chain"] = pool["chain"] + cols = ["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + if "total_shares" in snap_df.columns: + cols.append("total_shares") + new_records.append(snap_df[cols]) + except Exception as e: + print(f" FAILED {pool['pool_id'][:10]}: {e}") + time.sleep(0.5) + + if new_records: + new_df = pd.concat(new_records, ignore_index=True) + combined = pd.concat([cached, new_df], ignore_index=True) + else: + combined = cached + + if cache_path and len(combined) > 0: + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + combined.to_parquet(cache_path, index=False) + print(f" Saved cache: {len(combined)} pool-days -> {cache_path}") + + return combined + + +def fetch_token_prices(token_addresses_by_chain: dict, + cache_dir: str = None) -> dict: + """Fetch hourly token prices from Balancer API.""" + if cache_dir: + os.makedirs(cache_dir, exist_ok=True) + + prices = {} + + for chain, tokens in token_addresses_by_chain.items(): + uncached = {} + for symbol, address in tokens.items(): + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") if cache_dir else None + + if cp and os.path.exists(cp): + prices[(chain, symbol)] = pd.read_parquet(cp) + else: + uncached[symbol] = address + + if not uncached: + continue + + addr_to_symbol = {addr: sym for sym, addr in uncached.items()} + addresses = list(uncached.values()) + + print(f" Fetching {len(addresses)} prices on {chain}...", flush=True) + + batch_size = 20 + for batch_start in range(0, len(addresses), batch_size): + batch_addrs = addresses[batch_start:batch_start + batch_size] + query = { + "query": """ + query GetPrices($chain: GqlChain!, $addresses: [String!]!, + $range: GqlTokenChartDataRange!) { + tokenGetHistoricalPrices( + addresses: $addresses, chain: $chain, range: $range + ) { + address + prices { + timestamp + price + } + } + } + """, + "variables": { + "chain": chain, + "addresses": batch_addrs, + "range": "ONE_YEAR", + }, + } + + try: + body = _graphql_request(query, timeout=60) + results = body.get("data", {}).get( + "tokenGetHistoricalPrices", []) + for result in results: + addr = result.get("address", "") + price_list = result.get("prices", []) + symbol = addr_to_symbol.get(addr) + if symbol and price_list: + pdf = pd.DataFrame(price_list) + pdf["timestamp"] = pdf["timestamp"].astype(int) + pdf["price"] = pdf["price"].astype(float) + prices[(chain, symbol)] = pdf + if cache_dir: + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") + pdf.to_parquet(cp, index=False) + except Exception as e: + print(f" FAILED batch on {chain}: {e}") + + time.sleep(0.5) + + print(f" Got prices for {len(prices)} token-chain pairs") + return prices + + +def compute_pair_volatility( + snapshots_df: pd.DataFrame, + pool_row: pd.Series, + token_prices: dict, +) -> pd.Series: + """Compute daily annualised volatility for a pool's pair ratio.""" + tokens = pool_row["tokens"] + chain = pool_row["chain"] + + if len(tokens) < 2: + return pd.Series(dtype=float) + + def _get_price_df(symbol): + key = (chain, symbol) + if key in token_prices: + return token_prices[key] + for k, v in token_prices.items(): + if k[1] == symbol: + return v + return None + + p0_df = _get_price_df(tokens[0]) + p1_df = _get_price_df(tokens[1]) + + stables = {"USDC", "USDT", "DAI", "LUSD", "GHO", "crvUSD", "sDAI", + "WXDAI", "xDAI", "USDC.e", "USDbC"} + + if tokens[0] in stables and tokens[1] in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.01, index=dates) + + if p0_df is None and tokens[0] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + if p1_df is None and tokens[1] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + + if tokens[0] in stables: + if p1_df is None or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p1_df.copy() + ratio_df["ratio"] = 1.0 / ratio_df["price"] + elif tokens[1] in stables: + if p0_df is None or len(p0_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p0_df.copy() + ratio_df["ratio"] = ratio_df["price"] + else: + if p0_df is None or p1_df is None or len(p0_df) == 0 or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + merged = pd.merge_asof( + p0_df.sort_values("timestamp"), + p1_df.sort_values("timestamp"), + on="timestamp", + suffixes=("_0", "_1"), + tolerance=7200, + ).dropna() + if len(merged) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = merged.copy() + ratio_df["ratio"] = merged["price_0"] / merged["price_1"] + + ratio_df["datetime"] = pd.to_datetime(ratio_df["timestamp"], unit="s") + ratio_df["date"] = ratio_df["datetime"].dt.date + ratio_df = ratio_df.sort_values("timestamp") + + ratio_df["log_return"] = np.log( + ratio_df["ratio"] / ratio_df["ratio"].shift(1) + ) + ratio_df = ratio_df.dropna(subset=["log_return"]) + + daily_vol = ratio_df.groupby("date")["log_return"].std() + daily_vol_ann = daily_vol * np.sqrt(24 * 365) + + return daily_vol_ann + + +def assemble_panel( + pools_df: pd.DataFrame, + snapshots_df: pd.DataFrame, + token_prices: dict, +) -> pd.DataFrame: + """Assemble the full panel DataFrame with lagged TVL. + + Adds log_tvl_lag1 = per-pool shift(1) of log_tvl to break + the TVL-volume simultaneity bias. Drops the first observation + per pool (~1 obs per pool). + """ + records = [] + pool_ids = snapshots_df["pool_id"].unique() + n_pools = len(pool_ids) + + # Track volatility fallback rate + n_obs_total = 0 + n_obs_fallback = 0 + pools_all_fallback = [] # pools where every obs hit fallback + + for i, pool_id in enumerate(pool_ids): + if (i + 1) % 20 == 0 or i == 0: + print(f" Assembling {i+1}/{n_pools}...", flush=True) + + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pool_id] + pool_meta = pools_df[pools_df["pool_id"] == pool_id] + if len(pool_meta) == 0: + continue + pool_row = pool_meta.iloc[0] + + tokens = pool_row["tokens"] + if len(tokens) < 2: + continue + + chain = pool_row["chain"] + swap_fee = pool_row["swap_fee"] + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + vol_series = compute_pair_volatility(pool_snaps, pool_row, token_prices) + + pool_obs = 0 + pool_fallback = 0 + has_shares = "total_shares" in pool_snaps.columns + for _, snap in pool_snaps.iterrows(): + date = snap["date"] + volume = snap["volume_usd"] + tvl = snap["total_liquidity_usd"] + shares = float(snap["total_shares"]) if has_shares else 0.0 + + if tvl <= 0 or volume <= 0: + continue + + used_fallback = False + if isinstance(vol_series, pd.Series) and date in vol_series.index: + vol = vol_series[date] + else: + vol = 0.5 + used_fallback = True + + if not np.isfinite(vol) or vol <= 0: + vol = 0.5 + used_fallback = True + + n_obs_total += 1 + if used_fallback: + n_obs_fallback += 1 + pool_fallback += 1 + pool_obs += 1 + + if isinstance(date, datetime): + is_weekend = date.weekday() >= 5 + else: + is_weekend = pd.Timestamp(date).weekday() >= 5 + + # DOW harmonics (deterministic from date) + if isinstance(date, datetime): + dow = date.weekday() + else: + dow = pd.Timestamp(date).weekday() + dow_sin = np.sin(2.0 * np.pi * dow / 7.0) + dow_cos = np.cos(2.0 * np.pi * dow / 7.0) + + record = { + "pool_id": pool_id, + "chain": chain, + "date": date, + "log_volume": np.log(volume), + "log_tvl": np.log(tvl), + "volatility": vol, + "log_sigma": np.log(max(vol, 1e-6)), + "weekend": 1.0 if is_weekend else 0.0, + "log_fee": np.log(max(swap_fee, 1e-6)), + "dow_sin": dow_sin, + "dow_cos": dow_cos, + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": ",".join(tokens[:2]), + "swap_fee": swap_fee, + } + if shares > 0: + record["total_shares"] = shares + records.append(record) + + if pool_obs > 0 and pool_fallback == pool_obs: + pools_all_fallback.append( + (pool_id[:16], chain, ",".join(tokens[:2])) + ) + + panel = pd.DataFrame(records) + + # Add lagged TVL to break simultaneity bias + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + n_before = len(panel) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + n_dropped = n_before - len(panel) + + # Interaction terms (use lagged TVL to break simultaneity) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + print(f"\n Panel: {len(panel)} observations, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + print(f" Dropped {n_dropped} first-day obs for lagged TVL") + + # Volatility coverage report + if n_obs_total > 0: + pct = 100 * n_obs_fallback / n_obs_total + print(f"\n Volatility coverage:") + print(f" {n_obs_fallback}/{n_obs_total} obs used fallback " + f"vol=0.5 ({pct:.1f}%)") + print(f" {len(pools_all_fallback)} pools had 100% fallback") + if pools_all_fallback: + for pid, ch, toks in pools_all_fallback[:10]: + print(f" {pid}... ({ch}) {toks}") + if len(pools_all_fallback) > 10: + print(f" ... and {len(pools_all_fallback) - 10} more") + + return panel diff --git a/quantammsim/noise_calibration/data_validation.py b/quantammsim/noise_calibration/data_validation.py new file mode 100644 index 00000000..8cb44e72 --- /dev/null +++ b/quantammsim/noise_calibration/data_validation.py @@ -0,0 +1,41 @@ +"""Data validation for noise calibration panels.""" + +import numpy as np +import pandas as pd + + +def validate_panel(panel: pd.DataFrame) -> pd.DataFrame: + """Run data validation checks. Prints warnings but does NOT drop rows.""" + print("\n Data validation:") + + # Pools with constant volume + vol_std = panel.groupby("pool_id")["log_volume"].std() + constant_vol = vol_std[vol_std < 0.01] + if len(constant_vol) > 0: + print(f" WARNING: {len(constant_vol)} pools have near-constant " + f"log(volume) (std < 0.01)") + for pid in constant_vol.index[:5]: + print(f" {pid[:16]}... std={constant_vol[pid]:.4f}") + if len(constant_vol) > 5: + print(f" ... and {len(constant_vol) - 5} more") + + # TVL jumps > 10x between consecutive days + panel_sorted = panel.sort_values(["pool_id", "date"]) + tvl_ratio = panel_sorted.groupby("pool_id")["log_tvl"].diff().abs() + big_jumps = tvl_ratio[tvl_ratio > np.log(10)] + if len(big_jumps) > 0: + affected_pools = panel_sorted.loc[big_jumps.index, "pool_id"].nunique() + print(f" WARNING: {len(big_jumps)} TVL jumps > 10x across " + f"{affected_pools} pools") + + # Days where volume > TVL + high_vol = panel[panel["log_volume"] > panel["log_tvl"]] + if len(high_vol) > 0: + affected_pools = high_vol["pool_id"].nunique() + print(f" WARNING: {len(high_vol)} days where volume > TVL across " + f"{affected_pools} pools (potential wash trading)") + + if len(constant_vol) == 0 and len(big_jumps) == 0 and len(high_vol) == 0: + print(" All checks passed.") + + return panel diff --git a/quantammsim/noise_calibration/formula_arb.py b/quantammsim/noise_calibration/formula_arb.py new file mode 100644 index 00000000..80fe3352 --- /dev/null +++ b/quantammsim/noise_calibration/formula_arb.py @@ -0,0 +1,36 @@ +"""JAX-differentiable LVR formula for arb volume. + +Based on arXiv:2305.14604v2 §6 with gas costs and discrete-time correction. +Reference: scripts/plot_formula_arb_vs_real.py:formula_arb_volume_daily (line 58). +""" + +import jax.numpy as jnp + + +def formula_arb_volume_daily_jax(sigma_daily, tvl, fee, gas_usd, cadence_minutes): + """Analytical arb volume per day for a CPMM with gas costs. + + All inputs are JAX scalars or arrays (must be broadcastable). + + Parameters + ---------- + sigma_daily : float + Daily volatility of the log price ratio (NOT annualised). + tvl : float + Pool TVL in USD. + fee : float + Swap fee as fraction (e.g. 0.003 for 30bp). + gas_usd : float + All-in gas cost per arb tx in USD. + cadence_minutes : float + Effective arb cadence in minutes (= simulator's arb_frequency). + """ + block_time_s = cadence_minutes * 60.0 + delta = 2.0 * jnp.sqrt(2.0 * jnp.maximum(gas_usd, 0.0) / jnp.maximum(tvl, 1e-6)) + bLVR = sigma_daily**2 * tvl / 8.0 + sqrt_term = sigma_daily * jnp.sqrt(block_time_s / (2.0 * 86400.0)) + correction = jnp.maximum( + 1.0 - delta / (2.0 * fee) - sqrt_term / (fee + delta / 2.0), + 0.0, + ) + return bLVR * correction / fee diff --git a/quantammsim/noise_calibration/inference.py b/quantammsim/noise_calibration/inference.py new file mode 100644 index 00000000..e315a5bd --- /dev/null +++ b/quantammsim/noise_calibration/inference.py @@ -0,0 +1,270 @@ +"""Inference runners: SVI, NUTS, SVI-initialized NUTS.""" + +import numpy as np + +from .constants import K_COEFF +from .model import noise_model + + +def _get_theta_samples(sample_dict: dict, X_pool: np.ndarray, + data: dict = None) -> np.ndarray: + """Get theta samples, reconstructing from non-centered params if needed. + + MCMC.get_samples() includes the deterministic "theta" site. + SVI's Predictive(guide, ...) does NOT — the guide only samples latent + variables. In that case, reconstruct theta manually: + mu = X_pool @ B^T + L_Sigma = diag(sigma_theta) @ L_Omega + theta = mu + eta @ L_Sigma^T + + For marginalized IBP (W present, z_logit absent): compute MAP feature + assignments from data, then theta = X_pool @ B.T + Z_MAP @ W. + Requires data dict for pool_idx, x_obs, y_obs. + + For legacy STE IBP (z_logit present): theta = X_pool @ B.T + Z_hard @ W. + """ + if "theta" in sample_dict: + return np.array(sample_dict["theta"]) + + # Marginalized IBP path: compute MAP assignments from data + if "W" in sample_dict and "z_logit" not in sample_dict: + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + W = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + + # MAP assignments: (N_pools, K_features) binary + if "v" in sample_dict: + # Hybrid IBP+DP: joint MAP over (features, clusters) + from .postprocessing import assign_ibp_dp_joint + Z_map, _ = assign_ibp_dp_joint(sample_dict, data) + else: + from .postprocessing import assign_ibp_features + Z_map = assign_ibp_features(sample_dict, data) + + mu = np.einsum("pd,sjd->spj", X_pool, B) + # Z_map doesn't vary across samples — broadcast + feature_effect = np.einsum("pk,skj->spj", Z_map.astype(float), W) + return mu + feature_effect + + # Legacy STE IBP path: theta = X_pool @ B.T + Z_hard @ W + if "z_logit" in sample_dict: + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + W = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + z_logit = np.array(sample_dict["z_logit"]) # (S, N_pools, K_features) + Z_hard = (z_logit > 0).astype(float) + + mu = np.einsum("pd,sjd->spj", X_pool, B) # (S, N_pools, K_coeff) + feature_effect = np.einsum("spk,skj->spj", Z_hard, W) + return mu + feature_effect + + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + sigma_theta = np.array(sample_dict["sigma_theta"]) # (S, K_coeff) + L_Omega = np.array(sample_dict["L_Omega"]) # (S, K_coeff, K_coeff) + eta = np.array(sample_dict["eta"]) # (S, N_pools, K_coeff) + + # mu[s, p, j] = sum_d X_pool[p, d] * B[s, j, d] -> (S, N_pools, K_coeff) + mu = np.einsum("pd,sjd->spj", X_pool, B) + + # L_Sigma = diag(sigma_theta) @ L_Omega -> (S, K_coeff, K_coeff) + L_Sigma = sigma_theta[:, :, None] * L_Omega + + # offset = eta @ L_Sigma^T -> (S, N_pools, K_coeff) + offset = np.einsum("spi,sji->spj", eta, L_Sigma) + + return mu + offset + + +def _build_model_kwargs(data: dict, model_fn=None) -> dict: + """Convert data dict to jnp arrays for the model. + + Uses inspect.signature on model_fn to decide which kwargs to include: + - tier_A_per_pool: only if model_fn accepts it + - K_clusters: only if model_fn accepts it and data has it + """ + import inspect + import jax.numpy as jnp + + if model_fn is None: + model_fn = noise_model + + params = set(inspect.signature(model_fn).parameters.keys()) + + kwargs = dict( + pool_idx=jnp.array(data["pool_idx"]), + X_pool=jnp.array(data["X_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K_coeff=K_COEFF, + K_cov=data["K_cov"], + ) + + if "tier_A_per_pool" in params: + kwargs["tier_A_per_pool"] = jnp.array(data["tier_A_per_pool"]) + + if "K_clusters" in params and "K_clusters" in data: + kwargs["K_clusters"] = data["K_clusters"] + + if "K_features" in params and "K_features" in data: + kwargs["K_features"] = data["K_features"] + + # Structural model parameters + if "sigma_daily" in params and "sigma_daily" in data: + kwargs["sigma_daily"] = jnp.array(data["sigma_daily"]) + if "lag_log_tvl" in params and "lag_log_tvl" in data: + kwargs["lag_log_tvl"] = jnp.array(data["lag_log_tvl"]) + if "fee" in params and "fee" in data: + kwargs["fee"] = jnp.array(data["fee"]) + if "gas" in params and "gas" in data: + kwargs["gas"] = jnp.array(data["gas"]) + if "chain_idx" in params and "chain_idx" in data: + kwargs["chain_idx"] = jnp.array(data["chain_idx"]) + if "tier_idx" in params and "tier_idx" in data: + kwargs["tier_idx"] = jnp.array(data["tier_idx"]) + if "n_chains" in params and "n_chains" in data: + kwargs["n_chains"] = data["n_chains"] + if "n_tiers" in params and "n_tiers" in data: + kwargs["n_tiers"] = data["n_tiers"] + if "K_archetypes" in params and "K_archetypes" in data: + kwargs["K_archetypes"] = data["K_archetypes"] + + return kwargs + + +def run_svi(data, num_steps=20000, lr=1e-3, seed=0, + num_samples=1000, model_fn=None) -> tuple: + """Run SVI with AutoNormal guide. + + Returns (samples_dict, elbo_losses). + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import SVI, Trace_ELBO, Predictive + from numpyro.infer.autoguide import AutoNormal + + if model_fn is None: + model_fn = noise_model + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + + print(f"\n Running SVI: {num_steps} steps, lr={lr}") + guide = AutoNormal(model_fn) + optimizer = numpyro.optim.Adam(lr) + svi = SVI(model_fn, guide, optimizer, loss=Trace_ELBO()) + + rng_key = jax.random.PRNGKey(seed) + svi_result = svi.run(rng_key, num_steps, **model_kwargs) + + elbo_losses = np.array(svi_result.losses) + print(f" SVI complete. Final ELBO: {elbo_losses[-1]:.2f}") + print(f" ELBO last 100 std: {np.std(elbo_losses[-100:]):.2f}") + + # Draw posterior samples + predictive = Predictive( + guide, params=svi_result.params, num_samples=num_samples, + ) + samples = predictive(jax.random.PRNGKey(seed + 1), **model_kwargs) + samples = {k: np.array(v) for k, v in samples.items()} + + print(f" Drew {num_samples} posterior samples.") + return samples, elbo_losses + + +def run_nuts(data, num_warmup=1000, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42, + init_values=None, model_fn=None): + """Run NUTS MCMC. + + Uses init_to_value if init_values provided (for SVI-initialized NUTS). + Returns the MCMC object. + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import MCMC, NUTS, init_to_value + + if model_fn is None: + model_fn = noise_model + + # Note: set_host_device_count must be called before JAX init. + # We handle this in main(). Here we just verify device count. + n_devices = len(jax.devices("cpu")) + if n_devices < num_chains: + print(f" WARNING: Only {n_devices} CPU devices available for " + f"{num_chains} chains. Chains will run sequentially.") + + init_strategy = None + if init_values is not None: + init_strategy = init_to_value( + values={k: jnp.array(v) for k, v in init_values.items()} + ) + + kernel = NUTS( + model_fn, + target_accept_prob=target_accept, + max_tree_depth=max_tree_depth, + init_strategy=init_strategy, + ) + mcmc = MCMC( + kernel, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + progress_bar=True, + ) + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + rng_key = jax.random.PRNGKey(seed) + + print(f"\n Running NUTS: {num_chains} chains x " + f"({num_warmup} warmup + {num_samples} samples)") + print(f" target_accept={target_accept}, max_tree_depth={max_tree_depth}") + if init_values is not None: + print(" Using SVI-initialized starting values.") + + mcmc.run(rng_key, **model_kwargs) + mcmc.print_summary(exclude_deterministic=True) + return mcmc + + +def run_svi_then_nuts(data, svi_steps=5000, svi_lr=1e-3, + num_warmup=500, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42, + model_fn=None): + """Run SVI first, then use posterior means as NUTS init. + + Returns (MCMC, elbo_losses). + """ + if model_fn is None: + model_fn = noise_model + + # Phase 1: SVI + print(" Phase 1: SVI warm-start") + samples, elbo_losses = run_svi( + data, num_steps=svi_steps, lr=svi_lr, seed=seed, num_samples=100, + model_fn=model_fn, + ) + + # Extract posterior means for init + init_values = {} + skip_keys = {"y", "theta", "w"} + for k, v in samples.items(): + if k in skip_keys: + continue + init_values[k] = np.mean(v, axis=0) + + # Phase 2: NUTS from SVI init + print("\n Phase 2: NUTS from SVI-initialized values") + mcmc = run_nuts( + data, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + target_accept=target_accept, + max_tree_depth=max_tree_depth, + seed=seed, + init_values=init_values, + model_fn=model_fn, + ) + + return mcmc, elbo_losses diff --git a/quantammsim/noise_calibration/model.py b/quantammsim/noise_calibration/model.py new file mode 100644 index 00000000..d0d5ebf4 --- /dev/null +++ b/quantammsim/noise_calibration/model.py @@ -0,0 +1,444 @@ +"""NumPyro noise volume models.""" + +import jax +import jax.numpy as jnp + +from .formula_arb import formula_arb_volume_daily_jax + + +def _pad_with_ref(alpha): + """Prepend a zero for the reference category.""" + return jnp.concatenate([jnp.zeros(1), alpha]) + + +def stick_breaking_weights(v): + """Convert Beta stick-breaking fractions to K-simplex weights. + + v: array of shape (K-1,) with values in (0, 1). + Returns weights of shape (K,) summing to 1. + + w_1 = v_1 + w_k = v_k * prod_{j> jnp.arange(K_features)[None, :]) & 1 + ).astype(jnp.float32) # (n_configs, K_features) + + # Log-prior for each config from IBP stick-breaking + log_pi = jnp.log(pi + 1e-30) + log_1mpi = jnp.log(1.0 - pi + 1e-30) + log_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Per-config feature effect + feature_effects = configs @ W # (n_configs, K_coeff) + + # Per-observation means + mu_pop_obs = jnp.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config + log_lik = dist.StudentT(df, mu_obs, sigma_eps).log_prob( + y_obs[:, None] + ) # (N_obs, n_configs) + + # Sum log-likelihoods within each pool + pool_log_liks = jnp.zeros((N_pools, n_configs)) + pool_log_liks = pool_log_liks.at[pool_idx].add(log_lik) + + # Marginal: logsumexp over configs per pool + log_marginal = logsumexp( + log_prior[None, :] + pool_log_liks, axis=1 + ) # (N_pools,) + numpyro.factor("log_lik", log_marginal.sum()) + else: + # === Prior predictive: sample explicit assignments === + with numpyro.plate("pools", N_pools): + z_features = numpyro.sample( + "z_features", + dist.Bernoulli(probs=pi).expand([K_features]).to_event(1), + ) + + theta = mu_pop + z_features @ W # (N_pools, K_coeff) + numpyro.deterministic("theta", theta) + + theta_obs = theta[pool_idx] + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample( + "y", dist.StudentT(df, mu_obs, sigma_eps), + ) + + +def noise_model_ibp_dp(pool_idx, X_pool, x_obs, y_obs=None, + N_pools=None, K_coeff=4, K_cov=None, + K_features=6, K_clusters=6): + """Hybrid IBP+DP noise model. + + IBP latent features for mean heterogeneity (theta = X_pool @ B.T + z @ W), + DP mixture for noise heterogeneity (per-cluster sigma_eps). Joint + marginalization over (2^K_features × K_clusters) configurations when + y_obs is provided. + """ + import numpyro + import numpyro.distributions as dist + from jax.scipy.special import logsumexp + + # --- Population effects --- + B = numpyro.sample( + "B", dist.Normal(0.0, 5.0).expand([K_coeff, K_cov]).to_event(2) + ) + + # Student-t degrees of freedom + df = numpyro.sample("df", dist.Gamma(2.0, 0.1)) + + # --- IBP prior on feature prevalences --- + alpha_ibp = numpyro.sample("alpha_ibp", dist.Gamma(2.0, 1.0)) + with numpyro.plate("features", K_features): + v_ibp = numpyro.sample("v_ibp", dist.Beta(alpha_ibp, 1.0)) + pi = jnp.cumprod(v_ibp) # decreasing prevalences + + # --- Feature effect matrix --- + sigma_w = numpyro.sample("sigma_w", dist.HalfNormal(2.0)) + W = numpyro.sample( + "W", dist.Normal(0.0, sigma_w).expand([K_features, K_coeff]).to_event(2) + ) + + # --- DP mixture on sigma_eps --- + alpha_dp = numpyro.sample("alpha_dp", dist.Gamma(1.0, 1.0)) + with numpyro.plate("sticks", K_clusters - 1): + v = numpyro.sample("v", dist.Beta(1.0, alpha_dp)) + w = numpyro.deterministic("w", stick_breaking_weights(v)) + + sigma_eps = numpyro.sample( + "sigma_eps", + dist.HalfNormal(2.0).expand([K_clusters]).to_event(1), + ) + + # --- Population mean per pool --- + mu_pop = X_pool @ B.T # (N_pools, K_coeff) + + if y_obs is not None: + # === Joint marginalization over IBP configs × DP clusters === + + # Enumerate all 2^K binary feature configurations + n_configs = 2 ** K_features + configs = ( + (jnp.arange(n_configs)[:, None] >> jnp.arange(K_features)[None, :]) & 1 + ).astype(jnp.float32) # (n_configs, K_features) + + # Log-prior for each IBP config + log_pi = jnp.log(pi + 1e-30) + log_1mpi = jnp.log(1.0 - pi + 1e-30) + log_ibp_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Per-config feature effect + feature_effects = configs @ W # (n_configs, K_coeff) + + # Per-observation means for each IBP config + mu_pop_obs = jnp.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per IBP config per DP cluster + log_lik = dist.StudentT( + df, mu_obs[:, :, None], sigma_eps[None, None, :] + ).log_prob(y_obs[:, None, None]) # (N_obs, n_configs, K_clusters) + + # Sum log-likelihoods within each pool + pool_log_liks = jnp.zeros((N_pools, n_configs, K_clusters)) + pool_log_liks = pool_log_liks.at[pool_idx].add(log_lik) + + # Joint prior: IBP config prior × DP cluster weight + log_joint_prior = log_ibp_prior[:, None] + jnp.log(w + 1e-30)[None, :] # (n_configs, K_clusters) + + # Marginal log-likelihood per pool: logsumexp over (configs, clusters) + log_marginal = logsumexp( + log_joint_prior[None, :, :] + pool_log_liks, axis=(1, 2) + ) # (N_pools,) + numpyro.factor("log_lik", log_marginal.sum()) + else: + # === Prior predictive: sample explicit assignments === + with numpyro.plate("pools", N_pools): + z_features = numpyro.sample( + "z_features", + dist.Bernoulli(probs=pi).expand([K_features]).to_event(1), + ) + z_cluster = numpyro.sample("z_cluster", dist.Categorical(probs=w)) + + theta = mu_pop + z_features @ W # (N_pools, K_coeff) + numpyro.deterministic("theta", theta) + + theta_obs = theta[pool_idx] + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) + sigma_obs = sigma_eps[z_cluster[pool_idx]] + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("y", dist.StudentT(df, mu_obs, sigma_obs)) + + +def structural_noise_model(pool_idx, X_pool, x_obs, y_obs=None, + sigma_daily=None, lag_log_tvl=None, + fee=None, gas=None, + chain_idx=None, tier_idx=None, + N_pools=None, K_archetypes=3, + n_chains=8, n_tiers=6, + **kwargs): + """Structural mixture model: LVR arb + mixture-of-experts noise. + + Decomposes observed total volume into arb (structurally restricted to + LVR formula) and noise (flexible MoE). All continuous — no discrete + latent variables, so AutoNormal guide works directly. + """ + import numpyro + import numpyro.distributions as dist + + # --- Arb cadence parameters --- + alpha_0 = numpyro.sample("alpha_0", dist.Normal(1.0, 2.0)) + alpha_chain = numpyro.sample( + "alpha_chain", + dist.Normal(0, 1).expand([n_chains - 1]).to_event(1), + ) + alpha_tier = numpyro.sample( + "alpha_tier", + dist.Normal(0, 1).expand([n_tiers - 1]).to_event(1), + ) + alpha_tvl = numpyro.sample("alpha_tvl", dist.Normal(0, 0.5)) + + # Per-observation cadence (broadcast pool-level indices to obs) + log_cadence = ( + alpha_0 + + _pad_with_ref(alpha_chain)[chain_idx[pool_idx]] + + _pad_with_ref(alpha_tier)[tier_idx[pool_idx]] + + alpha_tvl * lag_log_tvl + ) + cadence = jnp.exp(jnp.clip(log_cadence, -2.0, 6.0)) # 0.1 to 400 min + + # V_arb per obs (deterministic given cadence + observables) + V_arb = formula_arb_volume_daily_jax( + sigma_daily, jnp.exp(lag_log_tvl), fee, gas, cadence, + ) + + # --- Noise MoE parameters --- + n_obs_coeff = x_obs.shape[1] + K_pool_cov = X_pool.shape[1] + + W_gate = numpyro.sample( + "W_gate", + dist.Normal(0, 1).expand([K_pool_cov, K_archetypes]).to_event(2), + ) + beta = numpyro.sample( + "beta", + dist.Normal(0, 2).expand([K_archetypes, n_obs_coeff]).to_event(2), + ) + + # Per-pool soft assignment and coefficient blend + logits = X_pool @ W_gate # (N_pools, K) + w = jax.nn.softmax(logits, axis=-1) # (N_pools, K) + beta_pool = jnp.einsum("pk,kc->pc", w, beta) # (N_pools, n_obs_coeff) + log_V_noise = jnp.sum(beta_pool[pool_idx] * x_obs, axis=1) + V_noise = jnp.exp(log_V_noise) + + # --- Observation model --- + df = numpyro.sample("df", dist.Gamma(2.0, 0.1)) + sigma_eps = numpyro.sample("sigma_eps", dist.HalfNormal(1.0)) + + mu = jnp.log(jnp.maximum(V_arb + V_noise, 1e-6)) + + if y_obs is not None: + numpyro.sample("y", dist.StudentT(df, mu, sigma_eps), obs=y_obs) + else: + numpyro.sample("y", dist.StudentT(df, mu, sigma_eps)) diff --git a/quantammsim/noise_calibration/output.py b/quantammsim/noise_calibration/output.py new file mode 100644 index 00000000..931f0592 --- /dev/null +++ b/quantammsim/noise_calibration/output.py @@ -0,0 +1,306 @@ +"""JSON output and sample caching.""" + +import json +import os + +import numpy as np + +from .constants import K_COEFF, COEFF_NAMES, K_OBS_COEFF, OBS_COEFF_NAMES + + +def generate_output_json(pool_params, samples, data, convergence, + output_path, inference_config): + """Write structured JSON output. + + Dispatches format based on whether samples contain DP mixture parameters + (detected via "v" in sample_dict), or structural model parameters + (detected via "W_gate" in sample_dict). + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Structural mixture model path + is_structural = "W_gate" in sample_dict + if is_structural: + _generate_structural_output( + pool_params, sample_dict, data, convergence, + output_path, inference_config, + ) + return + + # Detection priority: check hybrid first, then pure IBP, then DP + is_ibp_dp = ("W" in sample_dict and "v" in sample_dict + and "z_logit" not in sample_dict) + is_ibp = ("W" in sample_dict and "v" not in sample_dict + and "z_logit" not in sample_dict) + is_ibp_ste = "z_logit" in sample_dict # legacy STE artifacts + is_dp = "v" in sample_dict and "W" not in sample_dict + + B_median = np.median(np.array(sample_dict["B"]), axis=0).tolist() + sigma_eps_median = np.median( + np.array(sample_dict["sigma_eps"]), axis=0 + ) + df_median = float(np.median(np.array(sample_dict["df"]))) + + if is_ibp_dp: + model_name = "hierarchical_student_t_ibp_dp" + sigma_eps_structure = "dp_mixture" + + W_median = np.median(np.array(sample_dict["W"]), axis=0).tolist() + v_ibp_median = np.median(np.array(sample_dict["v_ibp"]), axis=0) + pi = np.cumprod(v_ibp_median).tolist() + + from .model import stick_breaking_weights + import jax.numpy as jnp + v_median = np.median(np.array(sample_dict["v"]), axis=0) + cluster_weights = np.array( + stick_breaking_weights(jnp.array(v_median)) + ).tolist() + + population_effects = { + "B": B_median, + "sigma_eps": sigma_eps_median.tolist() if hasattr(sigma_eps_median, 'tolist') else sigma_eps_median, + "df": df_median, + "W": W_median, + "feature_prevalences": pi, + "alpha_ibp": float( + np.median(np.array(sample_dict["alpha_ibp"])) + ), + "cluster_weights": cluster_weights, + "alpha_dp": float( + np.median(np.array(sample_dict["alpha_dp"])) + ), + } + + # Joint MAP assignments + from .postprocessing import assign_ibp_dp_joint + feat_assignments, cluster_assignments = assign_ibp_dp_joint( + sample_dict, data + ) + + pool_entries = {} + for i, p in enumerate(pool_params): + entry = { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + "feature_assignments": feat_assignments[i].tolist(), + "cluster_assignment": int(cluster_assignments[i]), + } + pool_entries[p["pool_id"]] = entry + + elif is_ibp or is_ibp_ste: + model_name = "hierarchical_student_t_ibp" + sigma_eps_structure = "scalar" + + W_median = np.median(np.array(sample_dict["W"]), axis=0).tolist() + v_ibp_median = np.median(np.array(sample_dict["v_ibp"]), axis=0) + pi = np.cumprod(v_ibp_median).tolist() + + population_effects = { + "B": B_median, + "sigma_eps": float(sigma_eps_median), + "df": df_median, + "W": W_median, + "feature_prevalences": pi, + "alpha_ibp": float( + np.median(np.array(sample_dict["alpha_ibp"])) + ), + } + + # Per-pool feature assignments + if is_ibp_ste: + # Legacy STE path: threshold z_logit + z_logit = np.array(sample_dict["z_logit"]) + z_logit_median = np.median(z_logit, axis=0) + feature_assignments = (z_logit_median > 0).astype(int).tolist() + else: + # Marginalized path: MAP assignments from data + from .postprocessing import assign_ibp_features + feature_assignments = assign_ibp_features( + sample_dict, data + ).tolist() + + pool_entries = {} + for i, p in enumerate(pool_params): + entry = { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + "feature_assignments": feature_assignments[i], + } + pool_entries[p["pool_id"]] = entry + + elif is_dp: + model_name = "hierarchical_student_t_dp_sigma" + sigma_eps_structure = "dp_mixture" + else: + model_name = "unified_hierarchical_student_t" + sigma_eps_structure = "per_tier" + + if not is_ibp and not is_ibp_dp: + sigma_theta_median = np.median( + np.array(sample_dict["sigma_theta"]), axis=0 + ).tolist() + + # Correlation matrix + L_Omega = np.array(sample_dict["L_Omega"]) + Omega = np.einsum("sij,skj->sik", L_Omega, L_Omega) + Omega_median = np.median(Omega, axis=0).tolist() + + population_effects = { + "B": B_median, + "sigma_theta": sigma_theta_median, + "sigma_eps": sigma_eps_median.tolist() if hasattr(sigma_eps_median, 'tolist') else sigma_eps_median, + "df": df_median, + "correlation_matrix": Omega_median, + } + + if is_dp: + from .model import stick_breaking_weights + import jax.numpy as jnp + v_median = np.median(np.array(sample_dict["v"]), axis=0) + w = stick_breaking_weights(jnp.array(v_median)) + population_effects["cluster_weights"] = np.array(w).tolist() + population_effects["alpha_dp"] = float( + np.median(np.array(sample_dict["alpha_dp"])) + ) + + pool_entries = { + p["pool_id"]: { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + } + for p in pool_params + } + + output = { + "model": model_name, + "model_spec": { + "K_coeff": K_COEFF, + "K_cov": data["K_cov"], + "coeff_names": COEFF_NAMES, + "covariate_names": data["covariate_names"], + "likelihood": "StudentT", + "tvl_lag": "log_tvl_lag1", + "sigma_eps_structure": sigma_eps_structure, + }, + "inference": inference_config, + "population_effects": population_effects, + "convergence": convergence, + "n_pools": len(pool_params), + "n_obs": len(data["y_obs"]), + "pools": pool_entries, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +def _generate_structural_output(pool_params, sample_dict, data, convergence, + output_path, inference_config): + """Write structural mixture model JSON output.""" + alpha_0 = float(np.median(np.array(sample_dict["alpha_0"]))) + alpha_chain = np.median(np.array(sample_dict["alpha_chain"]), axis=0).tolist() + alpha_tier = np.median(np.array(sample_dict["alpha_tier"]), axis=0).tolist() + alpha_tvl = float(np.median(np.array(sample_dict["alpha_tvl"]))) + + W_gate = np.median(np.array(sample_dict["W_gate"]), axis=0).tolist() + beta = np.median(np.array(sample_dict["beta"]), axis=0).tolist() + K_archetypes = np.array(sample_dict["beta"]).shape[1] + + df_median = float(np.median(np.array(sample_dict["df"]))) + sigma_eps_median = float(np.median(np.array(sample_dict["sigma_eps"]))) + + population_effects = { + "alpha_0": alpha_0, + "alpha_chain": alpha_chain, + "alpha_tier": alpha_tier, + "alpha_tvl": alpha_tvl, + "W_gate": W_gate, + "beta": beta, + "K_archetypes": K_archetypes, + "df": df_median, + "sigma_eps": sigma_eps_median, + } + + pool_entries = {} + for p in pool_params: + pool_entries[p["pool_id"]] = { + "chain": p["chain"], + "tokens": p["tokens"], + "arb_frequency": p["arb_frequency"], + "noise_params": p["noise_params"], + } + + output = { + "model": "structural_mixture", + "model_spec": { + "K_obs_coeff": K_OBS_COEFF, + "obs_coeff_names": OBS_COEFF_NAMES, + "K_cov": data["K_cov"], + "covariate_names": data["covariate_names"], + "likelihood": "StudentT", + }, + "inference": inference_config, + "population_effects": population_effects, + "convergence": convergence, + "n_pools": len(pool_params), + "n_obs": len(data["y_obs"]), + "pools": pool_entries, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +def _save_sample_cache(samples, data, cache_dir): + """Cache posterior samples and data arrays for --predict reuse.""" + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + os.makedirs(cache_dir, exist_ok=True) + + # Save only the samples needed for prediction and diagnostics. + # Skip "y" (S x N_obs, can be >1GB) and "theta" (S x N_pools x K, + # reconstructible from B, eta, sigma_theta, L_Omega). + skip_keys = {"y", "theta"} + sample_cache = os.path.join(cache_dir, "unified_samples.npz") + np.savez_compressed( + sample_cache, + **{k: np.array(v) for k, v in sample_dict.items() + if k not in skip_keys}, + ) + + # Data arrays for predict + data_cache = os.path.join(cache_dir, "unified_data.json") + cache_data = { + "pool_ids": data["pool_ids"], + "covariate_names": data["covariate_names"], + "K_cov": data["K_cov"], + "N_pools": data["N_pools"], + "ref_chain": data["ref_chain"], + "ref_tier_a": data["ref_tier_a"], + "ref_tier_b": data["ref_tier_b"], + "chains": data["chains"], + } + with open(data_cache, "w") as f: + json.dump(cache_data, f, indent=2) + + print(f" Cached samples -> {sample_cache}") + print(f" Cached data metadata -> {data_cache}") diff --git a/quantammsim/noise_calibration/plotting.py b/quantammsim/noise_calibration/plotting.py new file mode 100644 index 00000000..727443bd --- /dev/null +++ b/quantammsim/noise_calibration/plotting.py @@ -0,0 +1,335 @@ +"""Diagnostic plots for noise calibration.""" + +import os + +import numpy as np +import pandas as pd + +from .constants import K_COEFF, COEFF_NAMES +from .inference import _get_theta_samples + + +def plot_diagnostics(samples, data, output_dir, elbo_losses=None, + mcmc=None, prior_samples=None): + """Generate up to 9 diagnostic plots.""" + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + os.makedirs(output_dir, exist_ok=True) + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) # (S, N_pools, K_coeff) + theta_median = np.median(theta_samples, axis=0) + + pool_idx = data["pool_idx"] + x_obs = data["x_obs"] + y_obs = data["y_obs"] + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + + # --- 1. Prior predictive check --- + if prior_samples is not None: + y_prior = prior_samples.get("y", None) + if y_prior is not None: + fig, ax = plt.subplots(figsize=(10, 5)) + # Flatten a subsample of prior draws + y_prior_flat = y_prior.flatten() + # Clip for display + clip_lo, clip_hi = np.percentile(y_prior_flat, [0.5, 99.5]) + y_prior_clipped = y_prior_flat[ + (y_prior_flat >= clip_lo) & (y_prior_flat <= clip_hi) + ] + ax.hist(y_prior_clipped, bins=100, alpha=0.5, density=True, + color="steelblue", label="Prior predictive") + ax.hist(y_obs, bins=100, alpha=0.5, density=True, + color="coral", label="Observed") + ax.set_xlabel("log(volume)") + ax.set_ylabel("Density") + ax.set_title("Prior predictive check: log-volume") + ax.legend() + plt.tight_layout() + path = os.path.join(output_dir, "prior_predictive.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 2. ELBO loss curve (SVI only) --- + if elbo_losses is not None: + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + + ax = axes[0] + ax.plot(elbo_losses, alpha=0.3, color="steelblue", linewidth=0.5) + # Smoothed + window = min(100, len(elbo_losses) // 10) + if window > 1: + smoothed = pd.Series(elbo_losses).rolling(window).mean().values + ax.plot(smoothed, color="red", linewidth=1.5, label=f"Rolling {window}") + ax.legend() + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence") + + ax = axes[1] + # Last 20% of training + start = len(elbo_losses) * 4 // 5 + ax.plot(range(start, len(elbo_losses)), elbo_losses[start:], + color="steelblue", linewidth=0.8) + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence (last 20%)") + + plt.tight_layout() + path = os.path.join(output_dir, "elbo_convergence.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 3. Trace plots (NUTS only) --- + if mcmc is not None: + try: + import arviz as az + idata = az.from_numpyro(mcmc) + var_names = ["sigma_theta", "sigma_eps", "df"] + available = [v for v in var_names if v in idata.posterior] + if available: + axes = az.plot_trace(idata, var_names=available, compact=True) + fig = axes.ravel()[0].figure + fig.set_size_inches(14, 3 * len(available)) + path = os.path.join(output_dir, "trace_plots.png") + fig.savefig(path, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {path}") + except Exception as e: + print(f" WARNING: Trace plots failed: {e}") + + # --- 4. Posterior predictive: predicted vs observed --- + # Compute y_pred and r2 here; r2 is reused in plot 9 (model summary). + theta_obs = theta_median[pool_idx] + y_pred = np.sum(theta_obs * x_obs, axis=1) + r2 = 1 - np.var(y_obs - y_pred) / np.var(y_obs) + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + ax.scatter(y_obs, y_pred, alpha=0.1, s=4, color="steelblue") + lims = [min(y_obs.min(), y_pred.min()), max(y_obs.max(), y_pred.max())] + ax.plot(lims, lims, "r--", linewidth=1) + ax.set_xlabel("Observed log(volume)") + ax.set_ylabel("Predicted log(volume)") + ax.set_title("Posterior predictive check") + ax.text(0.05, 0.95, f"R² = {r2:.3f}", transform=ax.transAxes, + fontsize=11, verticalalignment="top") + + ax = axes[1] + residuals = y_obs - y_pred + ax.hist(residuals, bins=60, color="steelblue", edgecolor="white", alpha=0.8) + ax.axvline(0, color="red", linestyle="--") + ax.set_xlabel("Residual") + + sigma_eps_samples = np.array(sample_dict.get("sigma_eps", [0])) + if sigma_eps_samples.ndim > 1: + sigma_str = ", ".join(f"{np.median(sigma_eps_samples[:, i]):.2f}" + for i in range(sigma_eps_samples.shape[1])) + else: + sigma_str = f"{np.median(sigma_eps_samples):.2f}" + ax.set_title(f"Residuals (sigma_eps ~ [{sigma_str}])") + + plt.tight_layout() + path = os.path.join(output_dir, "posterior_predictive.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 5. Per-pool b_c by chain/tier --- + b_tvl_all = theta_median[:, 1] + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + chains_present = sorted(pool_meta["chain"].unique()) + chain_data = [] + chain_labels = [] + for c in chains_present: + mask = pool_meta["chain"].values == c + if mask.sum() > 0: + chain_data.append(b_tvl_all[mask]) + chain_labels.append(f"{c}\n(n={mask.sum()})") + if chain_data: + ax.boxplot(chain_data, tick_labels=chain_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by chain") + + ax = axes[1] + tier_a_vals = pool_meta["tier_A"].values.astype(int) + tier_labels_map = {0: "Blue-chip", 1: "Mid-cap", 2: "Long-tail"} + tier_data = [] + tier_labels = [] + for t in [0, 1, 2]: + mask = tier_a_vals == t + if mask.sum() > 0: + tier_data.append(b_tvl_all[mask]) + tier_labels.append(f"{tier_labels_map[t]}\n(n={mask.sum()})") + if tier_data: + ax.boxplot(tier_data, tick_labels=tier_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by token tier (best token)") + + plt.tight_layout() + path = os.path.join(output_dir, "per_pool_b_c.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 6. Correlation matrix posterior --- + L_Omega_samples = np.array(sample_dict["L_Omega"]) # (S, K, K) + Omega_samples = np.einsum("sij,skj->sik", L_Omega_samples, L_Omega_samples) + Omega_median = np.median(Omega_samples, axis=0) + + fig, ax = plt.subplots(figsize=(7, 6)) + im = ax.imshow(Omega_median, vmin=-1, vmax=1, cmap="RdBu_r") + ax.set_xticks(range(K_COEFF)) + ax.set_yticks(range(K_COEFF)) + ax.set_xticklabels(COEFF_NAMES, rotation=45, ha="right") + ax.set_yticklabels(COEFF_NAMES) + for i in range(K_COEFF): + for j in range(K_COEFF): + ax.text(j, i, f"{Omega_median[i, j]:.2f}", ha="center", + va="center", fontsize=10, + color="white" if abs(Omega_median[i, j]) > 0.5 else "black") + plt.colorbar(im, ax=ax, shrink=0.8) + ax.set_title("Posterior median correlation matrix (Omega)") + plt.tight_layout() + path = os.path.join(output_dir, "correlation_matrix.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 7. Shrinkage plot: OLS b_c vs hierarchical b_c --- + ols_b_c = np.zeros(len(pool_ids)) + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + if mask.sum() < 5: + ols_b_c[i] = np.nan + continue + x_i = x_obs[mask] + y_i = y_obs[mask] + try: + beta, _, _, _ = np.linalg.lstsq(x_i, y_i, rcond=None) + ols_b_c[i] = beta[1] # TVL coefficient + except np.linalg.LinAlgError: + ols_b_c[i] = np.nan + + hier_b_c = theta_median[:, 1] + valid = np.isfinite(ols_b_c) + + if valid.sum() > 2: + fig, ax = plt.subplots(figsize=(8, 8)) + ax.scatter(ols_b_c[valid], hier_b_c[valid], alpha=0.6, s=20, + color="steelblue") + + pop_b_c = np.median(hier_b_c) + ax.axhline(pop_b_c, color="red", linestyle="--", linewidth=0.8, + label=f"Population median = {pop_b_c:.3f}") + + lims = [min(np.nanmin(ols_b_c[valid]), hier_b_c[valid].min()) - 0.2, + max(np.nanmax(ols_b_c[valid]), hier_b_c[valid].max()) + 0.2] + ax.plot(lims, lims, "k:", linewidth=0.8, alpha=0.5) + ax.set_xlabel("Per-pool OLS b_c (lagged TVL)") + ax.set_ylabel("Hierarchical posterior median b_c") + ax.set_title("Shrinkage: OLS vs hierarchical TVL elasticity") + ax.legend() + plt.tight_layout() + path = os.path.join(output_dir, "shrinkage_b_c.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 8. beta_tvl vs beta_vol scatter colored by chain --- + fig, ax = plt.subplots(figsize=(10, 7)) + pool_id_to_chain = dict(zip(pool_meta["pool_id"], pool_meta["chain"])) + chain_colors = {} + cmap = plt.cm.tab10 + unique_chains = sorted(pool_meta["chain"].unique()) + for i, c in enumerate(unique_chains): + chain_colors[c] = cmap(i % 10) + + beta_tvl_arr = theta_median[:, 1] + beta_vol_arr = theta_median[:, 2] + for i, pid in enumerate(pool_ids): + c = pool_id_to_chain.get(pid, "?") + ax.scatter(beta_tvl_arr[i], beta_vol_arr[i], + color=chain_colors.get(c, "gray"), alpha=0.6, s=20, + edgecolors="white", linewidths=0.3) + + from matplotlib.lines import Line2D + handles = [Line2D([0], [0], marker="o", color="w", + markerfacecolor=chain_colors[c], markersize=8, + label=c) + for c in unique_chains if c in chain_colors] + ax.legend(handles=handles, fontsize=8, loc="best") + ax.set_xlabel("b_tvl (TVL elasticity)") + ax.set_ylabel("b_sigma (volatility sensitivity)") + ax.set_title("Pool-specific coefficients by chain") + ax.axhline(0, color="gray", linewidth=0.5, linestyle="--") + ax.axvline(0, color="gray", linewidth=0.5, linestyle="--") + plt.tight_layout() + path = os.path.join(output_dir, "beta_tvl_vs_beta_vol.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 9. Model summary panel --- + B_samples = np.array(sample_dict["B"]) + B_median = np.median(B_samples, axis=0) # (K_coeff, K_cov) + sigma_theta_med = np.median(np.array(sample_dict["sigma_theta"]), axis=0) + df_med = np.median(np.array(sample_dict["df"])) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) + + col_names = data["covariate_names"] + + fig, ax = plt.subplots(figsize=(12, 8)) + ax.axis("off") + + summary = "Group-level regression B (posterior median):\n" + header = f" {'covariate':<20s}" + for cn in COEFF_NAMES: + header += f" {cn:>10s}" + summary += header + "\n" + summary += " " + "-" * (20 + 11 * K_COEFF) + "\n" + for j, name in enumerate(col_names): + line = f" {name:<20s}" + for k in range(K_COEFF): + line += f" {B_median[k, j]:>10.3f}" + summary += line + "\n" + + summary += f"\nsigma_theta: [{', '.join(f'{v:.3f}' for v in sigma_theta_med)}]\n" + summary += f"\nCorrelation matrix (Omega):\n" + for i in range(K_COEFF): + row = " [" + " ".join(f"{Omega_median[i, j]:>6.3f}" + for j in range(K_COEFF)) + "]\n" + summary += row + + tier_names = ["blue-chip", "mid-cap", "long-tail"] + sigma_eps_str = ", ".join(f"{tier_names[i]}={sigma_eps_med[i]:.3f}" + for i in range(len(sigma_eps_med))) + summary += f"\nsigma_eps: [{sigma_eps_str}]\n" + summary += f"df (Student-t): {df_med:.1f}\n" + summary += f"R^2: {r2:.3f}\n" + + ax.text(0.02, 0.98, summary, transform=ax.transAxes, + fontsize=7, verticalalignment="top", fontfamily="monospace") + ax.set_title("Model Summary") + plt.tight_layout() + path = os.path.join(output_dir, "model_summary.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") diff --git a/quantammsim/noise_calibration/postprocessing.py b/quantammsim/noise_calibration/postprocessing.py new file mode 100644 index 00000000..4980b448 --- /dev/null +++ b/quantammsim/noise_calibration/postprocessing.py @@ -0,0 +1,667 @@ +"""Post-processing: extract params, predict, convergence, prior predictive.""" + +import numpy as np + +from .constants import K_COEFF, COEFF_NAMES, K_OBS_COEFF, OBS_COEFF_NAMES +from .token_classification import classify_token_tier +from .covariate_encoding import _tier_pair_idx +from .inference import _get_theta_samples, _build_model_kwargs +from .model import noise_model + + +def extract_noise_params(samples, data, use_median=True) -> list: + """Extract per-pool noise params from posterior samples. + + Handles both MCMC.get_samples() and SVI samples dict. + Applies weekend absorption: b_0_eff = b_0_raw + b_weekend * (2/7). + """ + # Get theta samples + if hasattr(samples, "get_samples"): + # MCMC object + sample_dict = samples.get_samples() + else: + sample_dict = samples + + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) # (S, N_pools, K_coeff) + + agg_fn = np.median if use_median else np.mean + theta_agg = agg_fn(theta_samples, axis=0) # (N_pools, K_coeff) + theta_std = np.std(theta_samples, axis=0) + + pool_ids = data["pool_ids"] + pool_meta = data["pool_meta"] + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + b_0_raw, b_tvl, b_sigma, b_weekend = theta_agg[i] + std_vals = theta_std[i] + + # Weekend absorption: simulator has no weekend indicator, + # so fold the expected weekend effect into the intercept. + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "theta_median": [float(x) for x in theta_agg[i]], + "theta_std": [float(x) for x in std_vals], + "b_weekend": float(b_weekend), + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(meta["swap_fee"]), + }, + }) + + return results + + +def predict_new_pool(samples, data, chain: str, tokens: list, + fee: float, feature_assignments=None) -> dict: + """Predict noise params for an unseen pool using population effects. + + Constructs z_new, computes mu_new = B @ z_new across all posterior samples, + returns point estimate + 90% credible intervals with weekend absorption. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Build z_new using data-driven column names + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b}": + z_new[i] = 1.0 + + # mu_new = B @ z_new across all posterior samples + B_samples = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + mu_samples = np.einsum("skd,d->sk", B_samples, z_new) # (S, K_coeff) + + # IBP path: add feature effects + is_ibp = "W" in sample_dict + if is_ibp: + W_samples = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + if feature_assignments is not None: + # User-specified binary features + Z = np.array(feature_assignments) # (K_features,) + feature_effect = np.einsum("skj,k->sj", W_samples, Z) + prediction_source = "ibp_user_features" + else: + # Marginal: weight by prevalences pi = cumprod(v_ibp) + v_ibp = np.array(sample_dict["v_ibp"]) # (S, K_features) + pi = np.cumprod(v_ibp, axis=1) # (S, K_features) + feature_effect = np.einsum("skj,sk->sj", W_samples, pi) + prediction_source = "ibp_marginal" + mu_samples = mu_samples + feature_effect + else: + prediction_source = "population_level" + + mu_median = np.median(mu_samples, axis=0) + mu_q05 = np.percentile(mu_samples, 5, axis=0) + mu_q95 = np.percentile(mu_samples, 95, axis=0) + + # Weekend absorption + b_0_raw, b_tvl, b_sigma, b_weekend = mu_median + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + result = { + "chain": chain, + "tokens": tokens, + "fee": fee, + "prediction_source": prediction_source, + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(fee), + }, + "credible_intervals_90": { + name: { + "median": float(mu_median[k]), + "q05": float(mu_q05[k]), + "q95": float(mu_q95[k]), + } + for k, name in enumerate(COEFF_NAMES) + }, + } + + print(f"\n Predicted noise_params for {chain} {tokens} (fee={fee}):") + for name, ci in result["credible_intervals_90"].items(): + print(f" {name:12s}: {ci['median']:+.3f} " + f"[{ci['q05']:+.3f}, {ci['q95']:+.3f}]") + print(f"\n Effective b_0 (weekend-absorbed): {b_0_effective:.3f}") + + return result + + +def check_convergence(mcmc_or_losses, method="nuts") -> dict: + """Compute convergence diagnostics. + + For NUTS: R-hat, ESS, divergences. + For SVI: final ELBO, ELBO stability. + """ + if method == "svi": + losses = np.array(mcmc_or_losses) + return { + "method": "svi", + "final_elbo": float(losses[-1]), + "elbo_last_100_std": float(np.std(losses[-100:])), + "elbo_last_100_mean": float(np.mean(losses[-100:])), + } + + # NUTS diagnostics + import arviz as az + + mcmc = mcmc_or_losses + idata = az.from_numpyro(mcmc) + + n_chains = idata.posterior.sizes.get("chain", 1) + + rhat_max = float("nan") + if n_chains >= 2: + rhat = az.rhat(idata) + rhat_vals = [] + for var in rhat.data_vars: + if var == "theta": + continue + vals = rhat[var].values + rhat_vals.extend(vals.flatten()) + rhat_max = float(np.nanmax(rhat_vals)) if rhat_vals else float("nan") + + ess = az.ess(idata) + ess_vals = [] + for var in ess.data_vars: + if var == "theta": + continue + vals = ess[var].values + ess_vals.extend(vals.flatten()) + ess_min = float(np.nanmin(ess_vals)) if ess_vals else float("nan") + + divergences = int(idata.sample_stats["diverging"].sum().values) + + print(f"\n Convergence diagnostics:") + if n_chains >= 2: + print(f" R-hat max: {rhat_max:.4f} " + f"{'OK' if rhat_max < 1.05 else 'WARNING'}") + else: + print(f" R-hat max: N/A (need >= 2 chains)") + print(f" ESS min: {ess_min:.0f} " + f"{'OK' if ess_min > 400 else 'WARNING'}") + print(f" Divergences: {divergences} " + f"{'OK' if divergences == 0 else 'WARNING'}") + + return { + "method": "nuts", + "r_hat_max": rhat_max, + "ess_min": ess_min, + "divergences": divergences, + } + + +def assign_dp_clusters(samples, data) -> np.ndarray: + """Compute posterior MAP cluster assignments for DP mixture model. + + Uses median posterior samples for v->w, sigma_eps, df, theta to compute + per-pool-per-cluster log-likelihoods, then returns argmax assignments. + """ + from scipy.special import logsumexp + from .model import stick_breaking_weights + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Median posterior parameters + v_med = np.median(np.array(sample_dict["v"]), axis=0) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) + df_med = float(np.median(np.array(sample_dict["df"]))) + + # Compute w from v via stick-breaking (using numpy) + import jax.numpy as jnp + w = np.array(stick_breaking_weights(jnp.array(v_med))) + + # Reconstruct theta + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) + theta_med = np.median(theta_samples, axis=0) # (N_pools, K_coeff) + + # Per-observation predicted means + pool_idx = np.array(data["pool_idx"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + K_clusters = len(sigma_eps_med) + + theta_obs = theta_med[pool_idx] + mu_obs = np.sum(theta_obs * x_obs, axis=1) # (N_obs,) + + # Log-likelihood per observation per cluster + from scipy.stats import t as t_dist + log_lik_per_k = np.zeros((len(y_obs), K_clusters)) + for k in range(K_clusters): + log_lik_per_k[:, k] = t_dist.logpdf( + y_obs, df_med, loc=mu_obs, scale=sigma_eps_med[k] + ) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, K_clusters)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik_per_k[i] + + # Posterior cluster probabilities: log p(z=k|data) = log w_k + sum log p(y|k) + log_posterior = np.log(w + 1e-30)[None, :] + pool_log_liks + # MAP assignment + assignments = np.argmax(log_posterior, axis=1).astype(np.int64) + return assignments + + +def assign_ibp_features(samples, data) -> np.ndarray: + """Compute MAP feature assignments for marginalized IBP model. + + Enumerates all 2^K binary feature configurations per pool, evaluates + per-pool log-posterior (log-prior + log-likelihood), returns argmax + config as (N_pools, K_features) binary ndarray. + + Uses median posterior parameters for B, W, v_ibp, sigma_eps, df. + """ + from scipy.stats import t as t_dist + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + B_med = np.median(np.array(sample_dict["B"]), axis=0) # (K_coeff, K_cov) + W_med = np.median(np.array(sample_dict["W"]), axis=0) # (K_features, K_coeff) + v_ibp_med = np.median(np.array(sample_dict["v_ibp"]), axis=0) # (K_features,) + sigma_eps_med = float(np.median(np.array(sample_dict["sigma_eps"]))) + df_med = float(np.median(np.array(sample_dict["df"]))) + + pi = np.cumprod(v_ibp_med) # (K_features,) + K_features = len(pi) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + + # Enumerate all 2^K configs + n_configs = 2 ** K_features + configs = ( + (np.arange(n_configs)[:, None] >> np.arange(K_features)[None, :]) & 1 + ).astype(float) # (n_configs, K_features) + + # Log-prior per config + log_pi = np.log(pi + 1e-30) + log_1mpi = np.log(1.0 - pi + 1e-30) + log_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Population mean + mu_pop = X_pool @ B_med.T # (N_pools, K_coeff) + + # Feature effects per config + feature_effects = configs @ W_med # (n_configs, K_coeff) + + # Per-obs means: mu_pop_obs + feature_mu + mu_pop_obs = np.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config + log_lik = t_dist.logpdf( + y_obs[:, None], df_med, loc=mu_obs, scale=sigma_eps_med + ) # (N_obs, n_configs) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, n_configs)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik[i] + + # Posterior = log_prior + pool_log_liks; MAP config per pool + log_posterior = log_prior[None, :] + pool_log_liks # (N_pools, n_configs) + best_config_idx = np.argmax(log_posterior, axis=1) # (N_pools,) + + return configs[best_config_idx].astype(int) # (N_pools, K_features) + + +def assign_ibp_dp_joint(samples, data) -> tuple: + """Compute MAP joint (feature, cluster) assignments for hybrid IBP+DP model. + + Enumerates all (2^K_features × K_clusters) joint configurations per pool, + evaluates joint log-posterior, returns argmax assignments. + + Returns: + (feature_assignments, cluster_assignments): + feature_assignments: (N_pools, K_features) binary ndarray + cluster_assignments: (N_pools,) int ndarray + """ + from scipy.stats import t as t_dist + from .model import stick_breaking_weights + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + B_med = np.median(np.array(sample_dict["B"]), axis=0) # (K_coeff, K_cov) + W_med = np.median(np.array(sample_dict["W"]), axis=0) # (K_features, K_coeff) + v_ibp_med = np.median(np.array(sample_dict["v_ibp"]), axis=0) # (K_features,) + v_med = np.median(np.array(sample_dict["v"]), axis=0) # (K_clusters-1,) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) # (K_clusters,) + df_med = float(np.median(np.array(sample_dict["df"]))) + + pi = np.cumprod(v_ibp_med) # (K_features,) + K_features = len(pi) + K_clusters = len(sigma_eps_med) + + import jax.numpy as jnp + w = np.array(stick_breaking_weights(jnp.array(v_med))) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + + # Enumerate all 2^K configs + n_configs = 2 ** K_features + configs = ( + (np.arange(n_configs)[:, None] >> np.arange(K_features)[None, :]) & 1 + ).astype(float) # (n_configs, K_features) + + # IBP log-prior per config + log_pi = np.log(pi + 1e-30) + log_1mpi = np.log(1.0 - pi + 1e-30) + log_ibp_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Joint log-prior: IBP config × DP cluster + log_joint_prior = log_ibp_prior[:, None] + np.log(w + 1e-30)[None, :] # (n_configs, K_clusters) + + # Population mean + mu_pop = X_pool @ B_med.T # (N_pools, K_coeff) + feature_effects = configs @ W_med # (n_configs, K_coeff) + + # Per-obs means + mu_pop_obs = np.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config per cluster + log_lik = np.zeros((len(y_obs), n_configs, K_clusters)) + for k in range(K_clusters): + log_lik[:, :, k] = t_dist.logpdf( + y_obs[:, None], df_med, loc=mu_obs, scale=sigma_eps_med[k] + ) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, n_configs, K_clusters)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik[i] + + # Joint posterior: log_joint_prior + pool_log_liks + log_posterior = log_joint_prior[None, :, :] + pool_log_liks # (N_pools, n_configs, K_clusters) + + # Flatten to (N_pools, n_configs * K_clusters), argmax, unravel + flat = log_posterior.reshape(N_pools, -1) + best_flat_idx = np.argmax(flat, axis=1) + best_config_idx = best_flat_idx // K_clusters + best_cluster_idx = best_flat_idx % K_clusters + + feature_assignments = configs[best_config_idx].astype(int) # (N_pools, K_features) + cluster_assignments = best_cluster_idx.astype(np.int64) # (N_pools,) + + return feature_assignments, cluster_assignments + + +def extract_structural_params(samples, data, use_median=True) -> list: + """Extract per-pool arb frequency and noise coefficients from structural model. + + Parameters + ---------- + samples : dict + Posterior samples from SVI/NUTS with structural_noise_model. + data : dict + Output of encode_covariates_structural(). + + Returns + ------- + list of dict + Per-pool dicts with: pool_id, chain, tokens, arb_frequency, noise_params. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + import jax + import jax.numpy as jnp + from .model import _pad_with_ref + + agg_fn = np.median if use_median else np.mean + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # MoE parameters + W_gate = agg_fn(np.array(sample_dict["W_gate"]), axis=0) + beta = agg_fn(np.array(sample_dict["beta"]), axis=0) + + chain_idx = np.array(data["chain_idx"]) + tier_idx = np.array(data["tier_idx"]) + X_pool = np.array(data["X_pool"]) + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + + # Per-pool cadence + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + # Per-pool log_tvl (median across observations) + pool_idx_arr = np.array(data["pool_idx"]) + lag_log_tvl = np.array(data["lag_log_tvl"]) + N_pools = data["N_pools"] + + pool_tvl_median = np.zeros(N_pools) + for p in range(N_pools): + mask = pool_idx_arr == p + if mask.any(): + pool_tvl_median[p] = np.median(lag_log_tvl[mask]) + + # Per-pool noise coefficients via MoE gating + logits = X_pool @ W_gate # (N_pools, K_archetypes) + w = np.exp(logits - logits.max(axis=1, keepdims=True)) + w = w / w.sum(axis=1, keepdims=True) # softmax + beta_pool = w @ beta # (N_pools, K_obs_coeff) + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + log_cadence = ( + alpha_0 + + padded_chain[chain_idx[i]] + + padded_tier[tier_idx[i]] + + alpha_tvl * pool_tvl_median[i] + ) + cadence = np.exp(np.clip(log_cadence, -2.0, 6.0)) + arb_freq = int(np.clip(np.round(cadence), 1, 60)) + + noise_coeffs = { + name: float(beta_pool[i, k]) + for k, name in enumerate(OBS_COEFF_NAMES) + } + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "arb_frequency": arb_freq, + "noise_params": noise_coeffs, + }) + + return results + + +def predict_new_pool_structural( + samples, data, chain: str, tokens: list, fee: float, tvl_est: float, +) -> dict: + """Predict cadence and noise coefficients for a hypothetical pool. + + Uses the structural model's arb cadence parameters and MoE gating. + + Parameters + ---------- + samples : dict + Posterior samples from structural_noise_model. + data : dict + Output of encode_covariates_structural(). + chain : str + Chain name. + tokens : list of str + Token symbols. + fee : float + Swap fee (fraction). + tvl_est : float + Estimated TVL in USD. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + agg_fn = np.median + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # MoE parameters + W_gate = agg_fn(np.array(sample_dict["W_gate"]), axis=0) + beta = agg_fn(np.array(sample_dict["beta"]), axis=0) + + # Construct chain and tier indices for the new pool + chains = data["chains"] + chain_to_idx = {c: i for i, c in enumerate(chains)} + c_idx = chain_to_idx.get(chain, 0) # fallback to reference + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tier_a + t_idx = _tier_pair_idx(tier_a, tier_b) + + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + log_tvl = np.log(max(tvl_est, 1.0)) + log_cadence = ( + alpha_0 + + padded_chain[c_idx] + + padded_tier[t_idx] + + alpha_tvl * log_tvl + ) + cadence = np.exp(np.clip(log_cadence, -2.0, 6.0)) + arb_freq = int(np.clip(np.round(cadence), 1, 60)) + + # Construct X_pool_new for gating + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + tier_a_str = str(tier_a) + tier_b_str = str(tier_b) + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a_str}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b_str}": + z_new[i] = 1.0 + + # MoE gating for new pool + logits = z_new @ W_gate # (K_archetypes,) + w = np.exp(logits - logits.max()) + w = w / w.sum() + beta_new = w @ beta # (K_obs_coeff,) + + noise_coeffs = { + name: float(beta_new[k]) + for k, name in enumerate(OBS_COEFF_NAMES) + } + + return { + "chain": chain, + "tokens": tokens, + "fee": fee, + "tvl_est": tvl_est, + "arb_frequency": arb_freq, + "noise_params": noise_coeffs, + "archetype_weights": w.tolist(), + } + + +def run_prior_predictive(data, num_samples=500, model_fn=None) -> dict: + """Run prior predictive check (no observations).""" + import jax + from numpyro.infer import Predictive + + if model_fn is None: + model_fn = noise_model + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + model_kwargs["y_obs"] = None # no observations + + predictive = Predictive(model_fn, num_samples=num_samples) + rng_key = jax.random.PRNGKey(99) + prior_samples = predictive(rng_key, **model_kwargs) + prior_samples = {k: np.array(v) for k, v in prior_samples.items()} + + print(f" Prior predictive: drew {num_samples} samples") + y_prior = prior_samples.get("y", None) + if y_prior is not None: + print(f" Prior log-volume range: " + f"[{np.percentile(y_prior, 1):.1f}, " + f"{np.percentile(y_prior, 99):.1f}]") + print(f" Observed log-volume range: " + f"[{data['y_obs'].min():.1f}, {data['y_obs'].max():.1f}]") + + return prior_samples diff --git a/quantammsim/noise_calibration/token_classification.py b/quantammsim/noise_calibration/token_classification.py new file mode 100644 index 00000000..aad9caa7 --- /dev/null +++ b/quantammsim/noise_calibration/token_classification.py @@ -0,0 +1,23 @@ +"""Token tier classification.""" + +from .constants import _TIER_0, _TIER_1 + + +def _normalise_symbol(symbol: str) -> str: + """Normalise wrapped/bridged variants to canonical form.""" + s = symbol.strip() + mapping = { + "WETH": "WETH", "WBTC": "WBTC", "cbBTC": "cbBTC", + "WMATIC": "WMATIC", "WAVAX": "WAVAX", "WXDAI": "WXDAI", "wS": "wS", + } + return mapping.get(s, s) + + +def classify_token_tier(symbol: str) -> int: + """Classify a token symbol into tier 0/1/2.""" + s = _normalise_symbol(symbol) + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 From 7e70f3312ec8dc1812a5ebe8d45b45614cf0c4dd Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:49:56 +0000 Subject: [PATCH 012/115] add new experiments and scripts related to noise modelling and tuning --- experiments/compare_exposures.py | 141 ++ experiments/diagnose_spikes.py | 216 +++ experiments/diagnostic_lp_supply.py | 134 ++ experiments/pool_registry.py | 512 +++++++ experiments/run_pool_battery.py | 914 ++++++++++++ experiments/tune_reclamm_params.py | 21 +- scripts/benchmark_reclamm_interpolation.py | 648 ++++++++ scripts/calibrate_noise_bayesian.py | 853 +++++++++++ scripts/calibrate_noise_hierarchical.py | 1556 ++++++++++++++++++++ scripts/calibrate_noise_unified.py | 6 + scripts/calibrate_reclamm_noise.py | 842 +++++++++++ scripts/compare_reclamm_thermostats.py | 379 +++++ scripts/demo_run_chunks_from_chain_data.py | 11 +- scripts/demo_run_from_chain_data.py | 13 +- scripts/demo_run_reclamm.py | 207 +++ scripts/plot_predicted_vs_real_volume.py | 151 ++ scripts/plot_reclamm_optuna_result.py | 451 ++++++ scripts/plot_top50_predicted_vs_real.py | 521 +++++++ scripts/run_structural_top50.py | 456 ++++++ scripts/sim_vs_world_comparison.py | 972 ++++++++++++ 20 files changed, 8986 insertions(+), 18 deletions(-) create mode 100644 experiments/compare_exposures.py create mode 100644 experiments/diagnose_spikes.py create mode 100644 experiments/diagnostic_lp_supply.py create mode 100644 experiments/pool_registry.py create mode 100644 experiments/run_pool_battery.py create mode 100644 scripts/benchmark_reclamm_interpolation.py create mode 100644 scripts/calibrate_noise_bayesian.py create mode 100644 scripts/calibrate_noise_hierarchical.py create mode 100644 scripts/calibrate_noise_unified.py create mode 100644 scripts/calibrate_reclamm_noise.py create mode 100644 scripts/compare_reclamm_thermostats.py create mode 100644 scripts/demo_run_reclamm.py create mode 100644 scripts/plot_predicted_vs_real_volume.py create mode 100644 scripts/plot_reclamm_optuna_result.py create mode 100644 scripts/plot_top50_predicted_vs_real.py create mode 100644 scripts/run_structural_top50.py create mode 100644 scripts/sim_vs_world_comparison.py diff --git a/experiments/compare_exposures.py b/experiments/compare_exposures.py new file mode 100644 index 00000000..14fdd06a --- /dev/null +++ b/experiments/compare_exposures.py @@ -0,0 +1,141 @@ +"""Compare effective weight/exposure trajectories: reClAMM vs Balancer 50/50. + +Prints weight stats and saves a plot of weight[AAVE] over time for both pools. +""" + +import jax.numpy as jnp +import numpy as np +import matplotlib.pyplot as plt +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(exponent): + return 1.0 - exponent / 124649.0 + + +TOKENS = ["AAVE", "ETH"] +START = "2024-06-01 00:00:00" +END = "2025-06-01 00:00:00" + +CONFIGS = { + "reClAMM on-chain (pr=1.5)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(0.1)), + }, + }, + "reClAMM wide (pr=4)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(1.0)), + }, + }, + "reClAMM Phase 2 (pr=4, m=0.1)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.1), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(0.001)), + }, + }, + "Balancer 50/50": { + "fingerprint": { + "tokens": TOKENS, "rule": "balancer", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "initial_weights_logits": jnp.zeros(2), + }, + }, +} + +results = {} +for name, cfg in CONFIGS.items(): + print(f"Running {name}...") + r = do_run_on_historic_data( + run_fingerprint=cfg["fingerprint"], params=cfg["params"] + ) + results[name] = r + +# Compute effective weights (value fraction in token 0 = AAVE) +print("\n" + "=" * 90) +print(f" {'Config':<35s} {'w_AAVE mean':>10s} {'w_AAVE std':>10s} " + f"{'w_AAVE min':>10s} {'w_AAVE max':>10s} {'vs HODL':>10s}") +print("-" * 90) + +daily = 1440 # subsample to daily for stats and plotting +fig, axes = plt.subplots(3, 1, figsize=(14, 10), sharex=True) + +for name, r in results.items(): + reserves = np.array(r["reserves"]) + prices = np.array(r["prices"]) + values = reserves * prices # (T, 2) + total = values.sum(axis=1, keepdims=True) + weights = values / np.clip(total, 1e-10, None) # (T, 2) + w_aave = weights[::daily, 0] + + hodl_value = float((reserves[0] * prices[-1]).sum()) + vs_hodl = r["final_value"] / hodl_value - 1.0 + + print(f" {name:<35s} {w_aave.mean():>10.4f} {w_aave.std():>10.4f} " + f"{w_aave.min():>10.4f} {w_aave.max():>10.4f} {vs_hodl * 100:>9.2f}%") + + days = np.arange(len(w_aave)) + axes[0].plot(days, w_aave, label=name, alpha=0.8) + + # Pool value over time + pool_val = np.array(r["value"])[::daily] + axes[1].plot(days[:len(pool_val)], pool_val / 1e6, label=name, alpha=0.8) + +print("=" * 90) + +# HODL line +r0 = results[list(results.keys())[0]] +prices_daily = np.array(r0["prices"])[::daily] +reserves_0 = np.array(r0["reserves"])[0] +hodl_val = (reserves_0 * prices_daily).sum(axis=1) / 1e6 +axes[1].plot(np.arange(len(hodl_val)), hodl_val, label="HODL", ls="--", color="gray", alpha=0.7) + +# Price ratio (AAVE/ETH) on third axis +price_ratio_series = prices_daily[:, 0] / prices_daily[:, 1] +axes[2].plot(np.arange(len(price_ratio_series)), price_ratio_series, color="black", alpha=0.7) +axes[2].set_ylabel("AAVE/ETH price") +axes[2].set_xlabel("Days") + +axes[0].set_ylabel("AAVE weight (value fraction)") +axes[0].axhline(0.5, ls="--", color="gray", alpha=0.5) +axes[0].legend(fontsize=8) +axes[0].set_title("Effective AAVE exposure over time") + +axes[1].set_ylabel("Pool value ($M)") +axes[1].legend(fontsize=8) +axes[1].set_title("Pool value over time") + +plt.tight_layout() +plt.savefig("reclamm_exposure_comparison.png", dpi=150) +print("\nSaved reclamm_exposure_comparison.png") diff --git a/experiments/diagnose_spikes.py b/experiments/diagnose_spikes.py new file mode 100644 index 00000000..b146c6bc --- /dev/null +++ b/experiments/diagnose_spikes.py @@ -0,0 +1,216 @@ +"""Diagnose spikes in sim-vs-world deviation for LP-supply-normalized runs. + +Compares old (no LP supply) vs new (with LP supply + per-LP normalization) +to pinpoint what causes spikes in the deviation time series. +""" + +import os +import numpy as np +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from datetime import datetime, timezone + +from experiments.pool_registry import ( + POOL_REGISTRY, extract_on_chain_state, extract_initial_state, + get_data_end_date, load_world_history, load_bpt_supply_df, +) +from experiments.run_pool_battery import ( + run_sim, sample_at_timestamps, _start_str_from_pool, + _onchain_params_to_sim, PROTOCOL_FEE_SPLIT, +) +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +POOL_LABEL = "WAVAX_USDC" # Change to "cbBTC_WETH" etc. + + +def main(): + pool = POOL_REGISTRY[POOL_LABEL] + extract_on_chain_state(pool) + initial_state = extract_initial_state(pool) + start_str = _start_str_from_pool(pool) + end_str = get_data_end_date(pool.tokens) + lp_supply_df = load_bpt_supply_df(pool, end_date=end_str) + + start_sec = datetime.strptime( + start_str, "%Y-%m-%d %H:%M:%S" + ).replace(tzinfo=timezone.utc).timestamp() + + print(f"Pool: {pool.label}, TVL: ${pool.initial_pool_value_usd:,.0f}") + print(f"Period: {start_str} to {end_str}") + print(f"BPT range: {lp_supply_df['lp_supply'].min():.4f} to {lp_supply_df['lp_supply'].max():.4f}") + + # ---- Run sims ---- + # 1. Old way: no lp_supply at all + result_old = run_sim( + pool, gas_cost=0.0, arb_frequency=1, + initial_state=initial_state, start=start_str, end=end_str, + lp_supply_df=None, + ) + + # 2. New way: lp_supply in scan + per-LP normalization + result_new = run_sim( + pool, gas_cost=0.0, arb_frequency=1, + initial_state=initial_state, start=start_str, end=end_str, + lp_supply_df=lp_supply_df, + ) + + # 3. Raw lp run (scan has lp_supply, but we DON'T divide by it) + params = _onchain_params_to_sim(pool) + fp = { + "tokens": pool.tokens, "rule": "reclamm", + "startDateString": start_str, "endDateString": end_str, + "initial_pool_value": pool.initial_pool_value_usd, + "fees": pool.swap_fee, "gas_cost": 0.0, "arb_fees": 0.0, + "do_arb": True, "arb_frequency": 1, "chunk_period": 1440, + "weight_interpolation_period": 1440, + "reclamm_use_shift_exponent": True, + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "protocol_fee_split": PROTOCOL_FEE_SPLIT, + "reclamm_initial_state": initial_state, + } + result_raw = do_run_on_historic_data( + run_fingerprint=fp, params=params, lp_supply_df=lp_supply_df, + ) + v_lp_raw = np.array(result_raw["value"]) + + v_old = np.array(result_old["value_usd"]) # no LP, no normalization + v_new = np.array(result_new["value_usd"]) # LP scan + divided by lp_supply + + # ---- World ---- + world = load_world_history(pool, end_date=end_str) + world_ts = world["timestamps"] + prices_min = result_old["prices"] + + prices_at_world = np.stack([ + sample_at_timestamps(prices_min[:, i], start_sec, world_ts) + for i in range(prices_min.shape[1]) + ], axis=1) + + # BPT-normalized world value (same in both old and new) + world_bpt_val = ( + world["bal_0"] * prices_at_world[:, 0] + + world["bal_1"] * prices_at_world[:, 1] + ) + world_growth = world_bpt_val / world_bpt_val[0] + + # Raw world value (absolute, un-normalized) + world_raw_val = ( + world["raw_bal_0"] * prices_at_world[:, 0] + + world["raw_bal_1"] * prices_at_world[:, 1] + ) + + # Sample sim at world timestamps + old_at_world = sample_at_timestamps(v_old, start_sec, world_ts) + new_at_world = sample_at_timestamps(v_new, start_sec, world_ts) + raw_at_world = sample_at_timestamps(v_lp_raw, start_sec, world_ts) + + old_growth = old_at_world / old_at_world[0] + new_growth = new_at_world / new_at_world[0] + raw_growth = raw_at_world / raw_at_world[0] + + # Deviations + dev_old = (old_growth / world_growth - 1) * 100 + dev_new = (new_growth / world_growth - 1) * 100 + + # Raw vs raw-world comparison (both absolute) + world_raw_growth = world_raw_val / world_raw_val[0] + dev_raw = (raw_growth / world_raw_growth - 1) * 100 + + days = (world_ts - world_ts[0]) / 86400 + + # LP supply at world timestamps + lp_unix = np.array(lp_supply_df["unix"]) + lp_vals = np.array(lp_supply_df["lp_supply"]) + lp_at_world = np.interp(world_ts, lp_unix / 1000, lp_vals) + + # ---- Print diagnostics ---- + print(f"\n--- Final deviations ---") + print(f"Old (no LP): {dev_old[-1]:+.4f}%") + print(f"New (LP + per-LP): {dev_new[-1]:+.4f}%") + print(f"Raw (LP, absolute): {dev_raw[-1]:+.4f}%") + + # Spike analysis + for label, dev in [("Old", dev_old), ("New", dev_new), ("Raw", dev_raw)]: + diffs = np.abs(np.diff(dev)) + n_spikes_01 = np.sum(diffs > 0.1) + n_spikes_05 = np.sum(diffs > 0.5) + n_spikes_10 = np.sum(diffs > 1.0) + print(f"\n{label} — step-to-step jumps in deviation:") + print(f" >0.1%: {n_spikes_01}, >0.5%: {n_spikes_05}, >1.0%: {n_spikes_10}") + if n_spikes_10 > 0: + spike_idx = np.where(diffs > 1.0)[0] + for si in spike_idx[:5]: + print(f" day {days[si]:.1f}: dev {dev[si]:+.2f}% -> {dev[si+1]:+.2f}% " + f"(Δ={dev[si+1]-dev[si]:+.2f}%, lp={lp_at_world[si]:.4f}->{lp_at_world[si+1]:.4f})") + + # World growth spikes + world_g_diffs = np.diff(world_growth) + n_world_spikes = np.sum(np.abs(world_g_diffs) > 0.01) + print(f"\nWorld BPT-normalized growth jumps > 1%: {n_world_spikes}") + if n_world_spikes > 0: + wsi = np.where(np.abs(world_g_diffs) > 0.01)[0] + for si in wsi[:5]: + print(f" day {days[si]:.1f}: growth {world_growth[si]:.4f} -> {world_growth[si+1]:.4f} " + f"(Δ={world_g_diffs[si]:+.4f}, lp={lp_at_world[si]:.4f}->{lp_at_world[si+1]:.4f})") + + # ---- Plot ---- + fig, axes = plt.subplots(4, 1, figsize=(14, 16), sharex=True) + + ax = axes[0] + ax.plot(days, dev_old, "b-", linewidth=1.5, label=f"Old (no LP) → {dev_old[-1]:+.2f}%") + ax.plot(days, dev_new, "r-", linewidth=1.5, label=f"New (LP + per-LP norm) → {dev_new[-1]:+.2f}%") + ax.axhline(0, color="gray", linestyle=":", alpha=0.5) + ax.set_ylabel("% deviation from world") + ax.set_title(f"{pool.label} — gas=0, arb=1min — old vs new deviation") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + ax = axes[1] + ax.plot(days, dev_raw, "g-", linewidth=1.5, label=f"Raw absolute (LP scan, raw world) → {dev_raw[-1]:+.2f}%") + ax.axhline(0, color="gray", linestyle=":", alpha=0.5) + ax.set_ylabel("% deviation") + ax.set_title("Alternative: raw absolute sim vs raw absolute world") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + ax = axes[2] + ax.plot(days, world_growth, "k-", linewidth=2, label="World (BPT-normalized)") + ax.plot(days, old_growth, "b-", linewidth=1, alpha=0.8, label="Old sim") + ax.plot(days, new_growth, "r-", linewidth=1, alpha=0.8, label="New sim (LP + per-LP)") + ax.set_ylabel("Growth factor") + ax.set_title("Growth factors") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + ax = axes[3] + ax.plot(days, lp_at_world, "g-", linewidth=2, label="LP supply (BPT/BPT₀)") + # Mark large LP changes + lp_diffs = np.abs(np.diff(lp_at_world)) + big_lp = np.where(lp_diffs > 0.05)[0] + if len(big_lp): + ax.scatter(days[big_lp], lp_at_world[big_lp], c="red", s=40, zorder=5, + label=f"Large LP events ({len(big_lp)})") + ax.set_ylabel("BPT / BPT₀") + ax.set_xlabel("Days from start") + ax.set_title("On-chain BPT supply") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + fig.suptitle( + f"{pool.label} ({pool.chain}) — spike diagnosis", + fontsize=13, fontweight="bold", + ) + plt.tight_layout() + + os.makedirs("results", exist_ok=True) + out = f"results/diagnose_spikes_{POOL_LABEL}.png" + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f"\nSaved: {out}") + + +if __name__ == "__main__": + main() diff --git a/experiments/diagnostic_lp_supply.py b/experiments/diagnostic_lp_supply.py new file mode 100644 index 00000000..3f3c3175 --- /dev/null +++ b/experiments/diagnostic_lp_supply.py @@ -0,0 +1,134 @@ +"""Diagnostic: sim vs world absolute pool value for a single (gas=0, arb_freq=1) run. + +Plots raw USD pool value over time for both sim and world, no per-LP normalization. +""" + +import os +import numpy as np +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from datetime import datetime, timezone + +from experiments.pool_registry import ( + POOL_REGISTRY, extract_on_chain_state, extract_initial_state, + get_data_end_date, load_world_history, load_bpt_supply_df, +) +from experiments.run_pool_battery import run_sim, sample_at_timestamps, _start_str_from_pool + + +def main(): + pool = POOL_REGISTRY["cbBTC_WETH"] + extract_on_chain_state(pool) + initial_state = extract_initial_state(pool) + + start_str = _start_str_from_pool(pool) + end_str = get_data_end_date(pool.tokens) + lp_supply_df = load_bpt_supply_df(pool, end_date=end_str) + + print(f"Pool: {pool.label}, TVL: ${pool.initial_pool_value_usd:,.0f}") + print(f"BPT: {lp_supply_df['lp_supply'].iloc[0]:.4f} -> {lp_supply_df['lp_supply'].iloc[-1]:.4f}") + print(f"Period: {start_str} to {end_str}") + + # Run sim WITH lp_supply (gas=0, arb_freq=1) + result_lp = run_sim(pool, gas_cost=0.0, arb_frequency=1, + initial_state=initial_state, + start=start_str, end=end_str, + lp_supply_df=lp_supply_df) + + # Run sim WITHOUT lp_supply (gas=0, arb_freq=1) + result_no_lp = run_sim(pool, gas_cost=0.0, arb_frequency=1, + initial_state=initial_state, + start=start_str, end=end_str, + lp_supply_df=None) + + # World + world = load_world_history(pool, end_date=end_str) + world_ts = world["timestamps"] + raw_bal_0 = world["raw_bal_0"] + raw_bal_1 = world["raw_bal_1"] + + start_sec = result_lp["start_unix_sec"] + prices_min = result_lp["prices"] + + # World value at world timestamps (raw balances × USD prices) + prices_at_world = np.stack([ + sample_at_timestamps(prices_min[:, i], start_sec, world_ts) + for i in range(prices_min.shape[1]) + ], axis=1) + world_value = raw_bal_0 * prices_at_world[:, 0] + raw_bal_1 * prices_at_world[:, 1] + + # Sim values (minute-resolution) + sim_value_lp = np.array(result_lp["value_usd"]) + sim_value_no_lp = np.array(result_no_lp["value_usd"]) + n_minutes = len(sim_value_lp) + sim_times_sec = start_sec + np.arange(n_minutes) * 60 + sim_days = (sim_times_sec - start_sec) / 86400 + world_days = (world_ts - start_sec) / 86400 + + # BPT supply at world timestamps (for annotation) + lp_at_world = np.interp( + world_ts, + np.array(lp_supply_df["unix"]) / 1000, + np.array(lp_supply_df["lp_supply"]), + ) + + # --- Plot --- + fig, axes = plt.subplots(3, 1, figsize=(14, 12), sharex=True) + + # Panel 1: absolute pool value + ax = axes[0] + ax.plot(world_days, world_value, "k-", linewidth=2, label="World (raw balances × prices)") + ax.plot(sim_days, sim_value_lp, "b-", linewidth=1, alpha=0.8, label="Sim (with lp_supply)") + ax.plot(sim_days, sim_value_no_lp, "r--", linewidth=1, alpha=0.8, label="Sim (no lp_supply)") + ax.set_ylabel("Pool value (USD)") + ax.set_title(f"{pool.label} — gas=0, arb_freq=1min — absolute pool value") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + # Panel 2: growth factors + ax = axes[1] + world_growth = world_value / world_value[0] + sim_growth_lp = sim_value_lp / sim_value_lp[0] + sim_growth_no_lp = sim_value_no_lp / sim_value_no_lp[0] + ax.plot(world_days, world_growth, "k-", linewidth=2, label="World growth") + ax.plot(sim_days, sim_growth_lp, "b-", linewidth=1, alpha=0.8, label="Sim growth (with lp_supply)") + ax.plot(sim_days, sim_growth_no_lp, "r--", linewidth=1, alpha=0.8, label="Sim growth (no lp_supply)") + ax.axhline(1.0, color="gray", linestyle=":", alpha=0.5) + ax.set_ylabel("Growth factor") + ax.set_title("Growth factors (value / initial value)") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + # Panel 3: BPT supply + ax = axes[2] + ax.plot(world_days, lp_at_world, "g-", linewidth=2, label="BPT supply (normalized)") + ax.set_ylabel("BPT / BPT₀") + ax.set_xlabel("Days from start") + ax.set_title("On-chain BPT supply") + ax.legend(fontsize=9) + ax.grid(True, alpha=0.2) + + fig.suptitle( + f"{pool.label} ({pool.chain}) — TVL=${pool.initial_pool_value_usd:,.0f} — " + f"PR={pool.on_chain_params['price_ratio']:.4f}", + fontsize=12, fontweight="bold", + ) + plt.tight_layout() + + os.makedirs("results", exist_ok=True) + out = "results/diagnostic_lp_supply_cbBTC_WETH.png" + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f"\nSaved: {out}") + + # Print key numbers + print(f"\nWorld: {world_value[0]:.0f} -> {world_value[-1]:.0f} (growth={world_growth[-1]:.4f})") + print(f"Sim (lp): {sim_value_lp[0]:.0f} -> {sim_value_lp[-1]:.0f} (growth={sim_growth_lp[-1]:.4f})") + print(f"Sim (no lp): {sim_value_no_lp[0]:.0f} -> {sim_value_no_lp[-1]:.0f} (growth={sim_growth_no_lp[-1]:.4f})") + print(f"\nDeviation (lp): {(sim_growth_lp[-1]/world_growth[-1] - 1)*100:+.2f}%") + print(f"Deviation (no lp): {(sim_growth_no_lp[-1]/world_growth[-1] - 1)*100:+.2f}%") + + +if __name__ == "__main__": + main() diff --git a/experiments/pool_registry.py b/experiments/pool_registry.py new file mode 100644 index 00000000..cfdce2f4 --- /dev/null +++ b/experiments/pool_registry.py @@ -0,0 +1,512 @@ +"""Registry of on-chain reClAMM pools for sim-vs-world gas calibration. + +Extracts pool state from reclamm-simulations DB and computes TVL in USD +at each pool's plausible_start date. Maps chain → realistic gas costs. +Also provides initial on-chain state (Ra, Rb, Va, Vb) and world balance +history for comparison. + +Pools excluded: + - EUR_USDC_b, sUSDai_USDT0, WXPL_USDT0: stable/stable pairs + - wstETH_GNO: boosted (wstETH yield-bearing) +""" + +import math +import os +import sqlite3 +from dataclasses import dataclass, field +from datetime import datetime, timezone +from typing import Optional + +import numpy as np +import pandas as pd + + +# --------------------------------------------------------------------------- +# Database path (reclamm-simulations repo) +# --------------------------------------------------------------------------- +DEFAULT_DB_PATH = os.path.expanduser( + "~/Projects/reclamm-simulations/data/pools_history.db" +) + +# --------------------------------------------------------------------------- +# Chain → gas cost batteries (USD) +# --------------------------------------------------------------------------- +# Non-mainnet chains use flat gas costs. +# Ethereum uses time-varying gas from on-chain percentile CSVs. +CHAIN_GAS_COSTS = { + "base": [0.0, 0.01, 0.1, 0.5], + "gnosis": [0.0, 0.01, 0.1, 0.5], + "avalanche": [0.0, 0.01, 0.1, 0.5], +} + +# Ethereum mainnet: time-varying gas percentiles + flat zero baseline. +# CSVs live in gas_csvs/ with columns [unix, USD]. +GAS_CSV_DIR = os.path.join(os.path.dirname(__file__), "..", "gas_csvs") +ETHEREUM_GAS_PERCENTILES = ["50p", "75p", "90p", "95p"] + + +@dataclass +class PoolConfig: + """Static metadata for a simulatable on-chain reClAMM pool.""" + + label: str + tokens: list # quantammsim ticker names, e.g. ['BTC', 'ETH'] + chain: str + swap_fee: float + db_label: str # table name in pools_history.db + plausible_start: str # YYYY-MM-DD + reverse: bool # True if DB token order is reversed vs quantammsim + pool_address: str = "" # on-chain contract address (hex, no 0x prefix) + # Filled by extract_on_chain_state(): + on_chain_params: Optional[dict] = None # price_ratio, margin, shift_rate + initial_pool_value_usd: Optional[float] = None + + +# --------------------------------------------------------------------------- +# Pool definitions (non-stable, non-boosted pools with quantammsim tickers) +# --------------------------------------------------------------------------- +# --------------------------------------------------------------------------- +# Chain → Balancer V3 API chain identifier +# --------------------------------------------------------------------------- +BALANCER_API_CHAIN = { + "base": "BASE", + "ethereum": "MAINNET", + "gnosis": "GNOSIS", + "avalanche": "AVALANCHE", + "arbitrum": "ARBITRUM", + "polygon": "POLYGON", + "optimism": "OPTIMISM", + "sonic": "SONIC", +} + + +POOL_REGISTRY = { + "cbBTC_WETH": PoolConfig( + label="cbBTC_WETH", + tokens=["BTC", "ETH"], + chain="base", + swap_fee=0.0005, + db_label="cbBTC_WETH", + plausible_start="2025-08-01", + reverse=True, + pool_address="19aeb8168d921bb069c6771bbaff7c09116720d0", + ), + "cbBTC_WETH_post_oct": PoolConfig( + label="cbBTC_WETH_post_oct", + tokens=["BTC", "ETH"], + chain="base", + swap_fee=0.0005, + db_label="cbBTC_WETH", + plausible_start="2025-12-01", + reverse=True, + pool_address="19aeb8168d921bb069c6771bbaff7c09116720d0", + ), + "AAVE_WETH": PoolConfig( + label="AAVE_WETH", + tokens=["AAVE", "ETH"], + chain="ethereum", + swap_fee=0.0025, + db_label="AAVE_WETH", + plausible_start="2025-08-15", + reverse=False, + pool_address="9d1fcf346ea1b073de4d5834e25572cc6ad71f4d", + ), + "AAVE_WETH_post_gov": PoolConfig( + label="AAVE_WETH_post_gov", + tokens=["AAVE", "ETH"], + chain="ethereum", + swap_fee=0.0025, + db_label="AAVE_WETH", + plausible_start="2025-12-21", + reverse=False, + pool_address="9d1fcf346ea1b073de4d5834e25572cc6ad71f4d", + ), + "COW_WETH_b": PoolConfig( + label="COW_WETH_b", + tokens=["COW", "ETH"], + chain="base", + swap_fee=0.003, + db_label="COW_WETH_b", + plausible_start="2025-07-18", + reverse=True, + pool_address="ff028c1ec4559d3aa2b0859aa582925b5cc28069", + ), + "COW_WETH_e": PoolConfig( + label="COW_WETH_e", + tokens=["COW", "ETH"], + chain="ethereum", + swap_fee=0.003, + db_label="COW_WETH_e", + plausible_start="2025-09-21", + reverse=True, + pool_address="d321300ef77067d4a868f117d37706eb81368e98", + ), + "WAVAX_USDC": PoolConfig( + label="WAVAX_USDC", + tokens=["AVAX", "USDC"], + chain="avalanche", + swap_fee=0.001, + db_label="WAVAX_USDC", + plausible_start="2025-08-17", + reverse=False, + pool_address="8750ccffcddbff81b63790dbcb1ffd8c7dc4c16d", + ), + "GNO_USDC": PoolConfig( + label="GNO_USDC", + tokens=["GNO", "USDC"], + chain="gnosis", + swap_fee=0.003, + db_label="GNO_USDC", + plausible_start="2025-09-18", + reverse=True, + pool_address="70b3b56773ace43fe86ee1d80cbe03176cbe4c09", + ), +} + + +def _date_to_unix(date_str: str) -> int: + """Convert YYYY-MM-DD or YYYY-MM-DD HH:MM:SS to unix timestamp (seconds).""" + for fmt in ("%Y-%m-%d %H:%M:%S", "%Y-%m-%d"): + try: + dt = datetime.strptime(date_str, fmt).replace(tzinfo=timezone.utc) + return int(dt.timestamp()) + except ValueError: + continue + raise ValueError(f"Cannot parse date: {date_str}") + + +def _get_usd_price_at(ticker: str, unix_ms: int, data_root: str) -> float: + """Get the USD price of a ticker at a given unix timestamp (ms).""" + path = os.path.join(data_root, f"{ticker}_USD.parquet") + df = pd.read_parquet(path) + idx = (df["unix"] - unix_ms).abs().idxmin() + return float(df.iloc[idx]["close"]) + + +def extract_on_chain_state( + pool: PoolConfig, + db_path: str = DEFAULT_DB_PATH, + data_root: str = None, +) -> PoolConfig: + """Query the DB for on-chain state at plausible_start and compute USD TVL. + + Mutates and returns the pool config with on_chain_params and + initial_pool_value_usd filled in. + """ + if data_root is None: + data_root = os.path.join( + os.path.dirname(__file__), "..", "quantammsim", "data" + ) + + conn = sqlite3.connect(db_path) + cur = conn.cursor() + ts = _date_to_unix(pool.plausible_start) + + cur.execute( + f"""SELECT * FROM {pool.db_label} + WHERE timestamp <= ? + ORDER BY timestamp DESC LIMIT 1""", + (ts + 3600,), + ) + row = cur.fetchone() + conn.close() + + if row is None: + raise ValueError( + f"No DB data for {pool.db_label} at {pool.plausible_start}" + ) + + # DB columns: timestamp, block_number, bpt_supply, balance_0, balance_1, + # spot_price, virtual_0, virtual_1, time_last_interaction, + # price_ratio, margin, shift_rate, swap_fee + balance_0, balance_1 = row[3], row[4] + price_ratio = row[9] + margin = row[10] + shift_rate = row[11] + + pool.on_chain_params = { + "price_ratio": price_ratio, + "margin": margin, + "shift_rate": shift_rate, + "swap_fee": row[12], + } + + # Compute TVL in USD from per-token USD prices. + # DB stores balances in contract token order (bring_pool_data.py never + # applies reverse). The reverse flag tells us the mapping: + # reverse=False → balance_0=tokens[0], balance_1=tokens[1] + # reverse=True → balance_0=tokens[1], balance_1=tokens[0] + unix_ms = ts * 1000 + if pool.reverse: + tickers_in_db_order = [pool.tokens[1], pool.tokens[0]] + else: + tickers_in_db_order = [pool.tokens[0], pool.tokens[1]] + + usd_prices = [] + for ticker in tickers_in_db_order: + if ticker == "USDC": + usd_prices.append(1.0) + else: + usd_prices.append( + _get_usd_price_at(ticker, unix_ms, data_root) + ) + + pool.initial_pool_value_usd = ( + balance_0 * usd_prices[0] + balance_1 * usd_prices[1] + ) + return pool + + +def extract_initial_state( + pool: PoolConfig, + db_path: str = DEFAULT_DB_PATH, +) -> dict: + """Extract on-chain Ra, Rb, Va, Vb at plausible_start in quantammsim order. + + quantammsim sorts tokens alphabetically, so token[0] is the + alphabetically-first ticker. The reverse flag maps DB contract + order to this sorted order. + + Returns dict with keys Ra, Rb, Va, Vb (floats). + """ + conn = sqlite3.connect(db_path) + cur = conn.cursor() + ts = _date_to_unix(pool.plausible_start) + + cur.execute( + f"""SELECT balance_0, balance_1, virtual_0, virtual_1 + FROM {pool.db_label} + WHERE timestamp <= ? + ORDER BY timestamp DESC LIMIT 1""", + (ts + 3600,), + ) + row = cur.fetchone() + conn.close() + + if row is None: + raise ValueError( + f"No DB data for {pool.db_label} at {pool.plausible_start}" + ) + + b0, b1, v0, v1 = row + if pool.reverse: + # DB contract order is opposite to quantammsim sorted order + return {"Ra": b1, "Rb": b0, "Va": v1, "Vb": v0} + else: + return {"Ra": b0, "Rb": b1, "Va": v0, "Vb": v1} + + +def load_world_history( + pool: PoolConfig, + end_date: str = None, + db_path: str = DEFAULT_DB_PATH, +) -> dict: + """Load on-chain balance history from the DB. + + Returns dict with: + timestamps: array of unix timestamps (seconds) + bal_0: BPT-normalized balance of quantammsim token[0] + bal_1: BPT-normalized balance of quantammsim token[1] + raw_bal_0: raw (un-normalized) balance of quantammsim token[0] + raw_bal_1: raw (un-normalized) balance of quantammsim token[1] + governance_events: list of (timestamp, field, old_val, new_val) + """ + conn = sqlite3.connect(db_path) + cur = conn.cursor() + + ts_start = _date_to_unix(pool.plausible_start) - 1000 + if end_date: + ts_end = _date_to_unix(end_date) + else: + ts_end = 2_000_000_000 # far future + + cur.execute( + f"""SELECT timestamp, bpt_supply, balance_0, balance_1, + price_ratio, margin, shift_rate, swap_fee + FROM {pool.db_label} + WHERE timestamp BETWEEN ? AND ? + ORDER BY timestamp""", + (ts_start, ts_end), + ) + rows = cur.fetchall() + conn.close() + + if not rows: + raise ValueError(f"No world history for {pool.db_label}") + + initial_bpt = rows[0][1] + timestamps = [] + bal_db_0_norm = [] + bal_db_1_norm = [] + bal_db_0_raw = [] + bal_db_1_raw = [] + governance_events = [] + + for i, row in enumerate(rows): + ts, bpt, b0, b1, pr, margin, shift_rate, swap_fee = row + timestamps.append(ts) + norm = initial_bpt / bpt + bal_db_0_norm.append(b0 * norm) + bal_db_1_norm.append(b1 * norm) + bal_db_0_raw.append(b0) + bal_db_1_raw.append(b1) + + # Detect governance changes. + # price_ratio drifts continuously via the shift mechanism, so + # only flag large discrete jumps (>1% relative change) as governance. + # margin, shift_rate, and swap_fee are set by governance and don't drift. + if i > 0: + prev = rows[i - 1] + if not math.isclose(prev[4], pr, rel_tol=0.01): + governance_events.append((ts, "price_ratio", prev[4], pr)) + if not math.isclose(prev[5], margin, rel_tol=1e-6): + governance_events.append((ts, "margin", prev[5], margin)) + if not math.isclose(prev[6], shift_rate, rel_tol=1e-6): + governance_events.append((ts, "shift_rate", prev[6], shift_rate)) + + bal_db_0_norm = np.array(bal_db_0_norm) + bal_db_1_norm = np.array(bal_db_1_norm) + bal_db_0_raw = np.array(bal_db_0_raw) + bal_db_1_raw = np.array(bal_db_1_raw) + + # Apply reverse: swap to quantammsim sorted token order + if pool.reverse: + bal_sorted_0, bal_sorted_1 = bal_db_1_norm, bal_db_0_norm + raw_sorted_0, raw_sorted_1 = bal_db_1_raw, bal_db_0_raw + else: + bal_sorted_0, bal_sorted_1 = bal_db_0_norm, bal_db_1_norm + raw_sorted_0, raw_sorted_1 = bal_db_0_raw, bal_db_1_raw + + return { + "timestamps": np.array(timestamps), + "bal_0": bal_sorted_0, + "bal_1": bal_sorted_1, + "raw_bal_0": raw_sorted_0, + "raw_bal_1": raw_sorted_1, + "governance_events": governance_events, + } + + +def load_bpt_supply_df( + pool: PoolConfig, + end_date: str = None, + db_path: str = DEFAULT_DB_PATH, +) -> pd.DataFrame: + """Load BPT supply as a DataFrame suitable for do_run_on_historic_data. + + Returns DataFrame with columns: + unix: timestamps in milliseconds + lp_supply: BPT normalized to 1.0 at plausible_start + + The normalization matches the simulator convention: lp_supply=1.0 at the + start of the sim, scaling proportionally as the on-chain pool grows/shrinks. + """ + conn = sqlite3.connect(db_path) + cur = conn.cursor() + + ts_start = _date_to_unix(pool.plausible_start) - 1000 + if end_date: + ts_end = _date_to_unix(end_date) + else: + ts_end = 2_000_000_000 + + cur.execute( + f"""SELECT timestamp, bpt_supply + FROM {pool.db_label} + WHERE timestamp BETWEEN ? AND ? + ORDER BY timestamp""", + (ts_start, ts_end), + ) + rows = cur.fetchall() + conn.close() + + if not rows: + raise ValueError(f"No BPT data for {pool.db_label}") + + initial_bpt = rows[0][1] + return pd.DataFrame({ + # Round to nearest minute boundary so timestamps land on the minute grid + # used by raw_fee_like_amounts_to_fee_like_array. + "unix": [round(r[0] / 60) * 60 * 1000 for r in rows], + "lp_supply": [r[1] / initial_bpt for r in rows], + }) + + +def get_data_end_date(tokens: list, data_root: str = None) -> str: + """Find the latest common date across all token parquets. + + Returns a date string like '2026-02-18 00:00:00'. + """ + if data_root is None: + data_root = os.path.join( + os.path.dirname(__file__), "..", "quantammsim", "data" + ) + + min_end = float("inf") + for ticker in tokens: + path = os.path.join(data_root, f"{ticker}_USD.parquet") + df = pd.read_parquet(path, columns=["unix"]) + last = float(df["unix"].iloc[-1]) + if last < min_end: + min_end = last + + # Convert ms to datetime + dt = datetime.utcfromtimestamp(min_end / 1000) + return dt.strftime("%Y-%m-%d %H:%M:%S") + + +def load_gas_csv(percentile: str) -> pd.DataFrame: + """Load a gas percentile CSV as a DataFrame for do_run_on_historic_data. + + Returns DataFrame with columns [unix, trade_gas_cost_usd], timestamps + floored to minute boundaries. + """ + path = os.path.join(GAS_CSV_DIR, f"Gas_{percentile}.csv") + df = pd.read_csv(path) + df = df.rename(columns={"USD": "trade_gas_cost_usd"}) + df["unix"] = (df["unix"] // 60000) * 60000 # floor to minute boundary + return df + + +def get_gas_costs(pool: PoolConfig, custom: list = None) -> list: + """Return the gas cost battery for a pool's chain. + + For Ethereum, returns a list mixing flat 0.0 with gas percentile labels + (e.g. ["0.0", "50p", "75p", "90p", "95p"]). + For other chains, returns flat USD values. + """ + if custom is not None: + return custom + if pool.chain == "ethereum": + flat = [0.0, 0.1, 0.5, 1.0, 3.0, 5.0, 10.0] + return flat + ETHEREUM_GAS_PERCENTILES + return CHAIN_GAS_COSTS.get(pool.chain, [0.0, 0.1, 1.0]) + + +def print_pool_summary(pool: PoolConfig): + """Print a summary of the pool's on-chain state.""" + print(f"\n{'='*60}") + print(f"Pool: {pool.label}") + print(f" Chain: {pool.chain}") + print(f" Tokens: {pool.tokens[0]}/{pool.tokens[1]}") + print(f" Swap fee: {pool.swap_fee}") + print(f" Start: {pool.plausible_start}") + if pool.on_chain_params: + p = pool.on_chain_params + print(f" On-chain: PR={p['price_ratio']:.4f} " + f"margin={p['margin']} shift_rate={p['shift_rate']} " + f"fee={p['swap_fee']}") + if pool.initial_pool_value_usd: + print(f" TVL: ${pool.initial_pool_value_usd:,.0f} USD") + print(f" Gas battery: {get_gas_costs(pool)}") + print(f"{'='*60}") + + +if __name__ == "__main__": + # Print summary of all pools + for label, pool in POOL_REGISTRY.items(): + try: + extract_on_chain_state(pool) + print_pool_summary(pool) + except Exception as e: + print(f"\n{label}: FAILED — {e}") diff --git a/experiments/run_pool_battery.py b/experiments/run_pool_battery.py new file mode 100644 index 00000000..def2c470 --- /dev/null +++ b/experiments/run_pool_battery.py @@ -0,0 +1,914 @@ +"""Sim-vs-world gas + arb-frequency calibration for on-chain reClAMM pools. + +For each pool in the registry, runs quantammsim forward passes with exact +on-chain parameters across a 2D grid of (gas_cost, arb_frequency), then +compares the simulated pool value trajectory against the actual on-chain +trajectory. + +arb_frequency is the period between arb trades in minutes (1 = every minute, +the most aggressive; higher = sparser arb). + +Generalizes scripts/sim_vs_world_comparison.py to work for any pool. + +Usage: + cd /Users/matthew/Projects/quantammsim-reclamm + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + + # Single pool (default gas + arb_freq grid) + python experiments/run_pool_battery.py cbBTC_WETH + + # All pools with data available + python experiments/run_pool_battery.py --all + + # Custom grids + python experiments/run_pool_battery.py cbBTC_WETH --gas-costs 0.0 0.5 1.0 --arb-freqs 1 5 15 60 + + # Dry run (show config without running) + python experiments/run_pool_battery.py cbBTC_WETH --dry-run + + # List available pools + python experiments/run_pool_battery.py --list +""" + +import argparse +import json +import os +import time + +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +from datetime import datetime, timezone + +from experiments.pool_registry import ( + POOL_REGISTRY, + PoolConfig, + extract_initial_state, + extract_on_chain_state, + get_data_end_date, + get_gas_costs, + load_bpt_supply_df, + load_gas_csv, + load_world_history, + print_pool_summary, +) +from quantammsim.runners.jax_runners import do_run_on_historic_data + +PROTOCOL_FEE_SPLIT = 0.5 +DEFAULT_ARB_FREQS = [1, 2, 3, 5, 10, 15, 20] + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def sample_at_timestamps(minute_vals, start_unix_sec, timestamps_sec): + """Sample a minute-level array at specific Unix timestamps. + + For each target timestamp, finds the nearest minute index in the + sim output and returns the corresponding value. + """ + indices = np.round((timestamps_sec - start_unix_sec) / 60).astype(int) + indices = np.clip(indices, 0, len(minute_vals) - 1) + return minute_vals[indices] + + +def compute_log_rmse(sim_growth, world_growth): + """RMSE of log(sim/world) across all trajectory points. + + Symmetric in over/under-estimation, natural for multiplicative processes. + A score of 0.02 means typical 2% deviation at any point in time. + """ + log_ratio = np.log(sim_growth / world_growth) + return np.sqrt(np.mean(log_ratio ** 2)) + + +def _start_str_from_pool(pool): + """Derive sim start time from pool's plausible_start, rounded to minute.""" + ts = int( + datetime.strptime(pool.plausible_start, "%Y-%m-%d") + .replace(tzinfo=timezone.utc) + .timestamp() + ) + ts_minute = (ts // 60) * 60 + return datetime.utcfromtimestamp(ts_minute).strftime("%Y-%m-%d %H:%M:%S") + + +def _onchain_params_to_sim(pool): + """Map DB param names to quantammsim param dict (jnp arrays).""" + p = pool.on_chain_params + return { + "price_ratio": jnp.array(p["price_ratio"]), + "centeredness_margin": jnp.array(p["margin"]), + "shift_exponent": jnp.array(p["shift_rate"]), + } + + +# --------------------------------------------------------------------------- +# Core sim runner +# --------------------------------------------------------------------------- + +def run_sim(pool, gas_cost, arb_frequency, initial_state, start, end, + protocol_fee_split=PROTOCOL_FEE_SPLIT, lp_supply_df=None, + noise_config=None): + """Run a single forward pass with exact on-chain params. + + gas_cost can be: + - float: flat gas cost in USD (e.g. 0.0, 0.5) + - str: gas percentile label (e.g. "50p", "90p") — loads time-varying + gas from CSV + + noise_config can be: + - None: no noise model (arb-only, default) + - dict with keys 'noise_model' and 'reclamm_noise_params': inject + Tsoukalas noise model into the sim + + Returns dict with minute-level per-LP value (USD), prices (USD per token), + and start_unix_sec. When lp_supply_df is provided, value_usd is divided + by the interpolated LP supply so it is comparable to BPT-normalized world + balances. + """ + params = _onchain_params_to_sim(pool) + + # Resolve gas: percentile string → DataFrame, float → scalar + gas_cost_df = None + if isinstance(gas_cost, str): + gas_cost_df = load_gas_csv(gas_cost) + flat_gas = 0.0 # placeholder; gas_cost_df overrides + else: + flat_gas = gas_cost + + fp = { + "tokens": pool.tokens, + "rule": "reclamm", + "startDateString": start, + "endDateString": end, + "initial_pool_value": pool.initial_pool_value_usd, + "fees": pool.swap_fee, + "gas_cost": flat_gas, + "arb_fees": 0.0, + "do_arb": True, + "arb_frequency": arb_frequency, + "chunk_period": 1440, + "weight_interpolation_period": 1440, + "reclamm_use_shift_exponent": True, + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "protocol_fee_split": protocol_fee_split, + "reclamm_initial_state": initial_state, + } + + if noise_config is not None: + fp["noise_model"] = noise_config["noise_model"] + fp["reclamm_noise_params"] = noise_config["reclamm_noise_params"] + + result = do_run_on_historic_data( + run_fingerprint=fp, params=params, lp_supply_df=lp_supply_df, + gas_cost_df=gas_cost_df, + ) + + start_unix_sec = datetime.strptime( + start, "%Y-%m-%d %H:%M:%S" + ).replace(tzinfo=timezone.utc).timestamp() + + value_usd = np.array(result["value"]) + + # Normalize to per-LP value so comparison with BPT-normalized world is valid. + # + # Subtlety: the scan applies lp_supply every arb_frequency minutes. + # Between scan steps, reserves are constant (no arb), so the pool value + # reflects the lp_supply from the LAST scan step. If a BPT event occurs + # between scan steps (e.g. at minute 3 when arb_frequency=5), the value + # at minutes 3-4 still reflects the old lp_supply. We must divide by + # the scan-step-aligned lp_supply, not the current minute's lp_supply, + # otherwise we get transient spikes. + if lp_supply_df is not None: + n_minutes = len(value_usd) + # Map each minute to its most recent scan-step time + scan_step_minutes = ( + np.arange(n_minutes) // arb_frequency * arb_frequency + ) + scan_step_times_ms = ( + start_unix_sec * 1000 + scan_step_minutes * 60_000 + ) + lp_unix = np.array(lp_supply_df["unix"]) + lp_vals = np.array(lp_supply_df["lp_supply"]) + indices = np.searchsorted(lp_unix, scan_step_times_ms, side="right") - 1 + indices = np.clip(indices, 0, len(lp_vals) - 1) + value_usd = value_usd / lp_vals[indices] + + return { + "value_usd": value_usd, + "prices": np.array(result["prices"]), # (T, n_tokens) in USD + "start_unix_sec": start_unix_sec, + } + + +# --------------------------------------------------------------------------- +# Pool calibration (2D grid: gas_cost × arb_frequency) +# --------------------------------------------------------------------------- + +def run_pool_calibration(pool, gas_costs, arb_freqs, verbose=True, + noise_config=None): + """Run 2D gas × arb_frequency calibration for a single pool. + + Parameters + ---------- + noise_config : dict, optional + If provided, passed through to run_sim to inject noise model. + Keys: 'noise_model', 'reclamm_noise_params'. + + Returns dict with: + world_growth: array of world growth factors + sim_growths: {(gas_cost, arb_freq): growth array} + timestamps: world timestamps (seconds) + governance_idx: index of first governance event (or n_points) + n_points: number of comparison points + days: array of days from start + gas_costs: list of gas costs + arb_freqs: list of arb frequencies + """ + # Extract on-chain state + initial reserves + extract_on_chain_state(pool) + initial_state = extract_initial_state(pool) + + if verbose: + print_pool_summary(pool) + print(f" Initial state: Ra={initial_state['Ra']:.4f}, " + f"Rb={initial_state['Rb']:.4f}, " + f"Va={initial_state['Va']:.4f}, Vb={initial_state['Vb']:.4f}") + + start_str = _start_str_from_pool(pool) + end_str = get_data_end_date(pool.tokens) + + # Load BPT supply history for LP supply scaling + lp_supply_df = load_bpt_supply_df(pool, end_date=end_str) + + if verbose: + bpt_start = lp_supply_df["lp_supply"].iloc[0] + bpt_end = lp_supply_df["lp_supply"].iloc[-1] + print(f" BPT supply: {bpt_start:.4f} → {bpt_end:.4f} " + f"({(bpt_end/bpt_start - 1)*100:+.1f}%)") + print(f" Sim period: {start_str} to {end_str}") + n_runs = len(gas_costs) * len(arb_freqs) + print(f" Grid: {len(gas_costs)} gas × {len(arb_freqs)} arb_freq = {n_runs} runs") + + # Load world history (BPT-normalized balances + governance events) + # BPT-normalized is correct for the growth ratio metric since LP supply + # cancels out of sim_growth/world_growth. Raw balances are available + # in world["raw_bal_0/1"] for absolute trajectory comparison. + world = load_world_history(pool, end_date=end_str) + world_ts = world["timestamps"] + world_bal_0 = world["bal_0"] + world_bal_1 = world["bal_1"] + gov_events = world["governance_events"] + + if verbose: + print(f" World points: {len(world_ts)}") + if gov_events: + for ts, field, old, new in gov_events: + dt = datetime.utcfromtimestamp(ts).strftime("%Y-%m-%d") + print(f" Governance: {field} {old:.6f} -> {new:.6f} on {dt}") + else: + print(" No governance events") + + # Governance cutoff index + if gov_events: + gov_idx = np.searchsorted(world_ts, gov_events[0][0]) + else: + gov_idx = len(world_ts) + + # Run sims across the 2D grid + sim_results = {} + prices_min = None + start_sec = None + + for gc in gas_costs: + for af in arb_freqs: + if verbose: + print(f"\n Running gas=${gc}, arb_freq={af}min...") + t0 = time.time() + result = run_sim(pool, gc, af, initial_state, start_str, end_str, + lp_supply_df=lp_supply_df, + noise_config=noise_config) + elapsed = time.time() - t0 + if verbose: + print(f" Done in {elapsed:.1f}s") + sim_results[(gc, af)] = result + if prices_min is None: + prices_min = result["prices"] + start_sec = result["start_unix_sec"] + + # Truncate at governance + n = min(gov_idx, len(world_ts)) + world_ts_trunc = world_ts[:n] + + # Sample USD prices at world timestamps for world valuation + prices_at_world = np.stack([ + sample_at_timestamps(prices_min[:, i], start_sec, world_ts_trunc) + for i in range(prices_min.shape[1]) + ], axis=1) + + # World value in USD = sum(bal_i * price_usd_i) + world_value = ( + world_bal_0[:n] * prices_at_world[:, 0] + + world_bal_1[:n] * prices_at_world[:, 1] + ) + world_growth = world_value / world_value[0] + + # Sim growths at world timestamps + sim_growths = {} + for key, result in sim_results.items(): + sim_val = sample_at_timestamps( + result["value_usd"], start_sec, world_ts_trunc, + ) + sim_growths[key] = sim_val / sim_val[0] + + days = (world_ts_trunc - world_ts_trunc[0]) / 86400 + + return { + "world_growth": world_growth, + "sim_growths": sim_growths, + "timestamps": world_ts_trunc, + "governance_idx": gov_idx, + "n_points": n, + "days": days, + "gas_costs": list(gas_costs), + "arb_freqs": list(arb_freqs), + } + + +# --------------------------------------------------------------------------- +# Plotting +# --------------------------------------------------------------------------- + +def plot_pool_calibration(pool, calibration, output_dir="results", suffix=""): + """Plot 2D gas × arb_freq calibration as heatmap + time series. + + Left: heatmap of final % deviation (gas_cost × arb_freq). + Right: time series for each arb_freq at best gas cost. + """ + os.makedirs(output_dir, exist_ok=True) + + world_growth = calibration["world_growth"] + sim_growths = calibration["sim_growths"] + days = calibration["days"] + gas_costs = calibration["gas_costs"] + arb_freqs = calibration["arb_freqs"] + + # Build RMSE matrix (log ratio, %) + rmse_matrix = np.zeros((len(arb_freqs), len(gas_costs))) + for i, af in enumerate(arb_freqs): + for j, gc in enumerate(gas_costs): + rmse_matrix[i, j] = compute_log_rmse( + sim_growths[(gc, af)], world_growth + ) * 100 + + fig, (ax_heat, ax_ts) = plt.subplots( + 1, 2, figsize=(18, 7), + gridspec_kw={"width_ratios": [1, 1.5]}, + ) + + # Left: heatmap of trajectory RMSE + im = ax_heat.imshow( + rmse_matrix, aspect="auto", cmap="RdYlGn_r", + vmin=0, vmax=rmse_matrix.max(), origin="lower", + ) + ax_heat.set_xticks(range(len(gas_costs))) + gas_labels = [ + f"gas {gc}" if isinstance(gc, str) else f"${gc}" + for gc in gas_costs + ] + ax_heat.set_xticklabels(gas_labels, fontsize=9) + ax_heat.set_yticks(range(len(arb_freqs))) + ax_heat.set_yticklabels([f"{af}min" for af in arb_freqs], fontsize=9) + ax_heat.set_xlabel("gas cost (USD)") + ax_heat.set_ylabel("arb frequency (minutes)") + ax_heat.set_title("Trajectory RMSE (log ratio, %)") + + # Annotate cells + for i in range(len(arb_freqs)): + for j in range(len(gas_costs)): + val = rmse_matrix[i, j] + color = "white" if val > rmse_matrix.max() * 0.6 else "black" + ax_heat.text(j, i, f"{val:.2f}%", ha="center", va="center", + fontsize=8, color=color) + + # Mark cell with least-negative mean bias (closest to 0 from below). + # If no cell is below world on average, fall back to lowest RMSE. + bias_matrix = np.zeros_like(rmse_matrix) + for i, af in enumerate(arb_freqs): + for j, gc in enumerate(gas_costs): + bias_matrix[i, j] = float(np.mean( + np.log(sim_growths[(gc, af)] / world_growth) + )) + negative_mask = bias_matrix < 0 + if negative_mask.any(): + # Among negative cells, find the one closest to 0 (max value) + masked = np.where(negative_mask, bias_matrix, -np.inf) + best_idx = np.unravel_index(np.argmax(masked), masked.shape) + else: + best_idx = np.unravel_index(np.argmin(rmse_matrix), rmse_matrix.shape) + ax_heat.add_patch(plt.Rectangle( + (best_idx[1] - 0.5, best_idx[0] - 0.5), 1, 1, + fill=False, edgecolor="lime", linewidth=3, + )) + + fig.colorbar(im, ax=ax_heat, label="RMSE (%)", shrink=0.8) + + # Right: 4 closest from below + 1 first above world + # Rationale: sim should underestimate (can't capture organic swaps, MEV + # rebates, etc.), so being below world is expected. The one-above config + # brackets where the sim crosses from conservative to optimistic. + ax_ts.axhline(y=0.0, color="brown", linewidth=2, label="world (on-chain)") + + # Classify configs by mean log ratio (trajectory-average bias). + # Using the mean rather than endpoint avoids a curve that's above + # world for 80% of the trajectory being classified as "below" + # just because it dips at the end. + below = [] # (mean_bias, rmse, gc, af) where sim < world on average + above = [] # (mean_bias, rmse, gc, af) where sim >= world on average + for (gc, af), sg in sim_growths.items(): + mean_bias = float(np.mean(np.log(sg / world_growth))) + rmse = compute_log_rmse(sg, world_growth) + if mean_bias < 0: + below.append((mean_bias, rmse, gc, af)) + else: + above.append((mean_bias, rmse, gc, af)) + + # Sort: below by mean_bias descending (closest to 0 first) + below.sort(key=lambda x: x[0], reverse=True) + # Sort: above by mean_bias ascending (closest to 0 first) + above.sort(key=lambda x: x[0]) + + # Select: up to 4 from below, 1 from above, fill if needed + selected = [] + n_below = min(4, len(below)) + n_above = min(1, len(above)) + selected.extend(below[:n_below]) + selected.extend(above[:n_above]) + remaining = 5 - len(selected) + if remaining > 0 and len(below) > n_below: + selected.extend(below[n_below:n_below + remaining]) + remaining = 5 - len(selected) + if remaining > 0 and len(above) > n_above: + selected.extend(above[n_above:n_above + remaining]) + + colors_below = plt.cm.Blues(np.linspace(0.4, 0.8, n_below)) + colors_above = np.array([[0.8, 0.2, 0.2, 1.0]]) # red for above + plot_colors = list(colors_below) + list(colors_above[:n_above]) + # Fill remaining with grey + while len(plot_colors) < len(selected): + plot_colors.append([0.5, 0.5, 0.5, 1.0]) + + for rank, (mean_bias, rmse, gc, af) in enumerate(selected): + dev = (sim_growths[(gc, af)] / world_growth - 1) * 100 + gc_label = f"gas {gc}" if isinstance(gc, str) else f"gas=${gc}" + marker = "\u25b2" if mean_bias >= 0 else "\u25bc" # ▲ above, ▼ below + ax_ts.plot(days, dev, color=plot_colors[rank], linewidth=2, + label=f"{marker} {gc_label}, arb={af}min " + f"bias={mean_bias*100:+.2f}% RMSE={rmse*100:.2f}%") + + ax_ts.set_xlabel("days") + ax_ts.set_ylabel("% deviation from world") + trunc = " (pre-governance)" if calibration["governance_idx"] < calibration["n_points"] + 1 else "" + ax_ts.set_title(f"Best bracket: {n_below} below + {n_above} above world{trunc}") + ax_ts.legend(fontsize=7, loc="best") + ax_ts.grid(True, alpha=0.2) + + p = pool.on_chain_params + fig.suptitle( + f"{pool.label} ({pool.chain}) — {pool.tokens[0]}/{pool.tokens[1]}\n" + f"PR={p['price_ratio']:.4f} margin={p['margin']} " + f"shift={p['shift_rate']} fee={pool.swap_fee} " + f"TVL=${pool.initial_pool_value_usd:,.0f} " + f"protocol_fee={PROTOCOL_FEE_SPLIT}", + fontsize=10, + ) + plt.tight_layout() + + out = os.path.join(output_dir, f"gas_calibration_{pool.label}{suffix}.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {out}") + return out + + +def plot_cross_pool_summary(all_results, output_dir="results"): + """Plot cross-pool comparison: best (gas, arb_freq) and residual deviation.""" + os.makedirs(output_dir, exist_ok=True) + + pool_labels = [] + best_configs = [] + best_devs = [] + + best_rmses = [] + best_biases = [] + for pool, cal in all_results: + wg = cal["world_growth"] + # Best = least-negative mean bias (closest from below) + below_keys = [ + k for k in cal["sim_growths"] + if np.mean(np.log(cal["sim_growths"][k] / wg)) < 0 + ] + if below_keys: + best_key = max( + below_keys, + key=lambda k: np.mean(np.log(cal["sim_growths"][k] / wg)), + ) + else: + best_key = min( + cal["sim_growths"].keys(), + key=lambda k: compute_log_rmse(cal["sim_growths"][k], wg), + ) + best_rmse = compute_log_rmse(cal["sim_growths"][best_key], wg) * 100 + best_bias = float(np.mean(np.log(cal["sim_growths"][best_key] / wg))) * 100 + pool_labels.append(f"{pool.label}\n({pool.chain})") + best_configs.append(best_key) + best_rmses.append(best_rmse) + best_biases.append(best_bias) + + fig, (ax_cfg, ax_dev) = plt.subplots(1, 2, figsize=(16, 6)) + + x = np.arange(len(pool_labels)) + + # Left: RMSE at best config + config_strs = [ + f"gas {gc}\narb={af}min" if isinstance(gc, str) + else f"gas=${gc}\narb={af}min" + for gc, af in best_configs + ] + ax_cfg.barh(x, best_rmses, color="steelblue") + ax_cfg.set_yticks(x) + ax_cfg.set_yticklabels(pool_labels, fontsize=9) + ax_cfg.set_xlabel("Trajectory RMSE (%)") + ax_cfg.set_title("RMSE at best config") + for i, (cs, rmse) in enumerate(zip(config_strs, best_rmses)): + ax_cfg.text(rmse + 0.05, i, f"{cs} (RMSE={rmse:.2f}%)", va="center", fontsize=8) + ax_cfg.grid(True, alpha=0.2, axis="x") + + # Right: mean bias at best config (negative = conservative) + colors = ["green" if d < 0 else "orange" if d < 1 else "red" + for d in best_biases] + ax_dev.bar(x, best_biases, color=colors) + ax_dev.axhline(y=0, color="brown", linewidth=1) + ax_dev.set_xticks(x) + ax_dev.set_xticklabels(pool_labels, fontsize=8) + ax_dev.set_ylabel("Mean bias (%)") + ax_dev.set_title("Mean trajectory bias at best config") + ax_dev.grid(True, alpha=0.2, axis="y") + + fig.suptitle("Cross-pool gas + arb frequency calibration", fontsize=12) + plt.tight_layout() + + out = os.path.join(output_dir, "gas_calibration_cross_pool.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f"\nSaved cross-pool summary: {out}") + return out + + +# --------------------------------------------------------------------------- +# Summary printing +# --------------------------------------------------------------------------- + +def print_calibration_summary(pool, calibration): + """Print a summary table of the 2D grid results (trajectory RMSE).""" + world_growth = calibration["world_growth"] + sim_growths = calibration["sim_growths"] + n_days = calibration["days"][-1] + gas_costs = calibration["gas_costs"] + arb_freqs = calibration["arb_freqs"] + + print(f"\n {pool.label} ({pool.chain}) — {n_days:.0f} days") + print(f" World growth: {world_growth[-1]:.4f}") + + # Print as table: rows=arb_freq, cols=gas_cost (values = trajectory RMSE %) + col_label = "arb\\gas" + gas_labels = [ + f"gas {gc}" if isinstance(gc, str) else f"${gc}" + for gc in gas_costs + ] + header = f" {col_label:<10}" + "".join(f"{gl:<10}" for gl in gas_labels) + print(header) + print(f" {'-'*len(header)}") + for af in arb_freqs: + row = f" {af:>4}min " + for gc in gas_costs: + rmse = compute_log_rmse( + sim_growths[(gc, af)], world_growth + ) * 100 + row += f"{rmse:>8.2f}% " + print(row) + + +# --------------------------------------------------------------------------- +# Data availability check +# --------------------------------------------------------------------------- + +def check_data_available(pool, data_root=None): + """Check that all required parquet files exist for a pool.""" + if data_root is None: + data_root = os.path.join( + os.path.dirname(__file__), "..", "quantammsim", "data" + ) + for ticker in pool.tokens: + if ticker == "USDC": + continue + path = os.path.join(data_root, f"{ticker}_USD.parquet") + if not os.path.exists(path): + return False + return True + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Sim-vs-world gas + arb-frequency calibration for on-chain reClAMM pools" + ) + parser.add_argument("pool", nargs="?", + help="Pool label (e.g. cbBTC_WETH)") + parser.add_argument("--all", action="store_true", + help="Run all pools with available data") + parser.add_argument("--list", action="store_true", + help="List available pools and exit") + parser.add_argument("--dry-run", action="store_true", + help="Show pool config without running") + parser.add_argument("--gas-costs", nargs="+", type=float, default=None, + help="Override gas cost battery (flat USD values)") + parser.add_argument("--arb-freqs", nargs="+", type=int, + default=DEFAULT_ARB_FREQS, + help="Arb frequency values in minutes (default: 1 2 3 5 10 15 20)") + parser.add_argument("--protocol-fee", type=float, default=PROTOCOL_FEE_SPLIT, + help="Protocol fee split (default 0.5)") + parser.add_argument("--output-dir", default="results", + help="Directory for output plots and JSON") + parser.add_argument("--calibrate-noise", action="store_true", + help="Calibrate Tsoukalas noise model from Balancer API + DB " + "and inject into sim fingerprints") + parser.add_argument("--noise-model", choices=["sqrt", "log", "loglinear"], + default="sqrt", + help="Noise model variant (default: sqrt)") + parser.add_argument("--noise-params-json", default=None, + help="Path to hierarchical noise params JSON " + "(from calibrate_noise_hierarchical.py). " + "Looks up pool by address or uses --predict.") + args = parser.parse_args() + + # --list mode + if args.list: + print("\nAvailable pools:\n") + for label, pool in POOL_REGISTRY.items(): + has_data = check_data_available(pool) + status = "READY" if has_data else "MISSING DATA" + try: + extract_on_chain_state(pool) + print_pool_summary(pool) + except Exception as e: + print(f" {label}: {e}") + print(f" Data: {status}") + return + + # Determine which pools to run + if args.all: + pool_labels = [ + label for label, pool in POOL_REGISTRY.items() + if check_data_available(pool) + ] + if not pool_labels: + print("No pools have all required data files.") + return + elif args.pool: + if args.pool not in POOL_REGISTRY: + print(f"Unknown pool: {args.pool}") + print(f"Available: {list(POOL_REGISTRY.keys())}") + return + if not check_data_available(POOL_REGISTRY[args.pool]): + missing = [ + f"{t}_USD.parquet" for t in POOL_REGISTRY[args.pool].tokens + if t != "USDC" and not os.path.exists( + os.path.join( + os.path.dirname(__file__), "..", "quantammsim", + "data", f"{t}_USD.parquet" + ) + ) + ] + print(f"Missing data for {args.pool}: {missing}") + return + pool_labels = [args.pool] + else: + parser.print_help() + return + + # Collect runs + runs = [] + for label in pool_labels: + pool = POOL_REGISTRY[label] + gas_costs = get_gas_costs(pool, args.gas_costs) + runs.append((pool, gas_costs)) + + arb_freqs = args.arb_freqs + n_total = sum(len(gcs) * len(arb_freqs) for _, gcs in runs) + + print(f"\n{'='*60}") + print(f"GAS + ARB CALIBRATION: {len(runs)} pool(s), {n_total} total runs") + for pool, gcs in runs: + print(f" {pool.label:15s} ({pool.chain:10s}) " + f"gas={gcs} arb_freq={arb_freqs}") + print(f" Protocol fee split: {args.protocol_fee}") + print(f"{'='*60}") + + if args.dry_run: + print("\n--- DRY RUN ---\n") + for pool, gas_costs in runs: + extract_on_chain_state(pool) + initial_state = extract_initial_state(pool) + print_pool_summary(pool) + print(f" Initial state: {initial_state}") + start = _start_str_from_pool(pool) + end = get_data_end_date(pool.tokens) + print(f" Sim period: {start} to {end}") + world = load_world_history(pool, end_date=end) + n_gov = len(world["governance_events"]) + print(f" World points: {len(world['timestamps'])}, " + f"governance events: {n_gov}") + if world["governance_events"]: + for ts, field, old, new in world["governance_events"]: + dt = datetime.utcfromtimestamp(ts).strftime("%Y-%m-%d") + print(f" {field} {old:.6f} -> {new:.6f} on {dt}") + print(f" Gas battery: {gas_costs}") + print(f" Arb freqs: {arb_freqs}") + n_runs = len(gas_costs) * len(arb_freqs) + print(f" Total runs: {n_runs}") + return + + # Execute calibration for each pool + all_results = [] + for pool, gas_costs in runs: + print(f"\n{'#'*60}") + print(f"POOL: {pool.label}") + print(f"{'#'*60}") + + # Noise calibration (if requested) + noise_config = None + if args.noise_params_json and args.noise_model == "loglinear": + # Load hierarchical noise params from JSON + with open(args.noise_params_json) as f: + hier_data = json.load(f) + # Look up pool by address + addr = pool.pool_address.lower() + pool_params = None + for p in hier_data["pools"]: + pid = p["pool_id"].lower().replace("0x", "") + if pid.startswith(addr) or addr.startswith(pid): + pool_params = p["noise_params"] + break + if pool_params is None: + # Fall back to population-level prediction + from scripts.calibrate_noise_hierarchical import predict_new_pool + chain_map = {"ethereum": "MAINNET", "base": "BASE", + "gnosis": "GNOSIS", "arbitrum": "ARBITRUM", + "polygon": "POLYGON", "optimism": "OPTIMISM", + "sonic": "SONIC", "avalanche": "AVALANCHE"} + api_chain = chain_map.get(pool.chain, pool.chain.upper()) + # Reconstruct posteriors + encoding from the JSON + posteriors_from_json = { + "Phi_mean": np.array(hier_data["Phi"]), + } + encoding_from_json = { + "covariate_names": hier_data["covariate_names"], + } + pool_params = predict_new_pool( + posteriors_from_json, encoding_from_json, + api_chain, pool.tokens, pool.swap_fee, + ) + print(f"\n Using population-level loglinear params (pool not in JSON)") + else: + print(f"\n Using hierarchical loglinear params for {pool.label}") + print(f" b_0 = {pool_params['b_0']:.4f}, " + f"b_sigma = {pool_params['b_sigma']:.6f}, " + f"b_c = {pool_params['b_c']:.4f}") + # Strip metadata keys (prefixed with _) — JAX can't trace strings + sim_params = {k: v for k, v in pool_params.items() + if not k.startswith("_")} + noise_config = { + "noise_model": "loglinear", + "reclamm_noise_params": sim_params, + } + elif args.calibrate_noise: + from scripts.calibrate_reclamm_noise import ( + build_calibration_df, + run_ols_calibration, + ) + noise_model_name = ( + "tsoukalas_sqrt" if args.noise_model == "sqrt" + else "tsoukalas_log" + ) + print(f"\n Calibrating noise model ({args.noise_model})...") + cal_df = build_calibration_df(pool) + noise_params, diag = run_ols_calibration( + cal_df, pool.swap_fee, args.noise_model, + ) + print(f" R² = {diag['r_squared']:.4f}, n = {diag['n_obs']}") + for key in ["a_0", "a_sigma", "a_c"]: + param_key = "a_0_base" if key == "a_0" else key + val = noise_params[param_key] + se = diag["se"][key] + t_stat = val / se if se > 0 else float("inf") + print(f" {key:>8} = {val:>10.4f} (SE={se:.4f}, t={t_stat:.2f})") + noise_config = { + "noise_model": noise_model_name, + "reclamm_noise_params": noise_params, + } + + t0 = time.time() + calibration = run_pool_calibration( + pool, gas_costs, arb_freqs, noise_config=noise_config, + ) + elapsed = time.time() - t0 + + print_calibration_summary(pool, calibration) + plot_pool_calibration(pool, calibration, output_dir=args.output_dir) + all_results.append((pool, calibration)) + + print(f"\n Total time for {pool.label}: {elapsed:.0f}s") + + # Cross-pool summary + if len(all_results) > 1: + plot_cross_pool_summary(all_results, output_dir=args.output_dir) + + # Final summary table + print(f"\n{'='*60}") + print("CALIBRATION COMPLETE") + print(f"{'='*60}") + print(f"\n{'Pool':<16} {'Chain':<10} {'Best Gas':>9} {'Best Arb':>9} " + f"{'Bias':>8} {'RMSE':>8} {'Days':>6}") + print("-" * 70) + for pool, cal in all_results: + wg = cal["world_growth"] + # Best = least-negative mean bias (closest from below) + below_keys = [ + k for k in cal["sim_growths"] + if np.mean(np.log(cal["sim_growths"][k] / wg)) < 0 + ] + if below_keys: + best_key = max( + below_keys, + key=lambda k: np.mean(np.log(cal["sim_growths"][k] / wg)), + ) + else: + best_key = min( + cal["sim_growths"].keys(), + key=lambda k: compute_log_rmse(cal["sim_growths"][k], wg), + ) + best_bias = float(np.mean(np.log(cal["sim_growths"][best_key] / wg))) * 100 + best_rmse = compute_log_rmse(cal["sim_growths"][best_key], wg) * 100 + n_days = cal["days"][-1] + gc_label = f"gas {best_key[0]}" if isinstance(best_key[0], str) else f"${best_key[0]}" + print(f"{pool.label:<16} {pool.chain:<10} {gc_label:<9} " + f"{best_key[1]:>4}min {best_bias:>+7.2f}% {best_rmse:>7.2f}% {n_days:>5.0f}d") + + # Save JSON summary + os.makedirs(args.output_dir, exist_ok=True) + summary = [] + for pool, cal in all_results: + wg_arr = cal["world_growth"] + wg_final = float(wg_arr[-1]) + pool_summary = { + "label": pool.label, + "chain": pool.chain, + "tokens": pool.tokens, + "swap_fee": pool.swap_fee, + "tvl_usd": pool.initial_pool_value_usd, + "on_chain_params": pool.on_chain_params, + "n_days": float(cal["days"][-1]), + "n_governance_events": 1 if cal["governance_idx"] < cal["n_points"] else 0, + "world_growth": wg_final, + "grid_results": {}, + } + for (gc, af) in sorted(cal["sim_growths"].keys(), key=lambda k: (str(k[0]), k[1])): + sg_arr = cal["sim_growths"][(gc, af)] + rmse = compute_log_rmse(sg_arr, wg_arr) * 100 + pool_summary["grid_results"][f"gas={gc}_arb={af}"] = { + "gas_cost": gc, + "arb_frequency": af, + "sim_growth": float(sg_arr[-1]), + "pct_deviation": float((sg_arr[-1] / wg_final - 1) * 100), + "trajectory_rmse_pct": float(rmse), + } + summary.append(pool_summary) + + ts = datetime.now().strftime("%Y%m%d_%H%M%S") + json_path = os.path.join(args.output_dir, f"gas_calibration_{ts}.json") + with open(json_path, "w") as f: + json.dump(summary, f, indent=2, default=str) + print(f"\nSummary saved to {json_path}") + + +if __name__ == "__main__": + main() diff --git a/experiments/tune_reclamm_params.py b/experiments/tune_reclamm_params.py index 0951e2c1..49699230 100644 --- a/experiments/tune_reclamm_params.py +++ b/experiments/tune_reclamm_params.py @@ -38,6 +38,17 @@ def main(): parser.add_argument("--interpolation", default="geometric", choices=["geometric", "constant_arc_length"]) parser.add_argument("--centeredness-scaling", action="store_true") + parser.add_argument("--noise-trader-ratio", type=float, default=0.0) + parser.add_argument("--start-date", default="2024-06-01 00:00:00") + parser.add_argument("--end-date", default="2025-01-01 00:00:00", + help="End of training / start of test") + parser.add_argument("--end-test-date", default="2025-06-01 00:00:00") + parser.add_argument("--bout-offset", type=int, default=None, + help="bout_offset in minutes (default: 10080 = 7 days)") + parser.add_argument("--val-fraction", type=float, default=None, + help="Validation holdout fraction (default: 0.2, use 0 to disable)") + parser.add_argument("--overfitting-penalty", type=float, default=None, + help="Overfitting penalty weight (default: 0.2)") args = parser.parse_args() learn_speed = args.interpolation == "constant_arc_length" @@ -48,29 +59,33 @@ def main(): fp = { "rule": "reclamm", "tokens": ["AAVE", "ETH"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-01-01 00:00:00", - "endTestDateString": "2025-06-01 00:00:00", + "startDateString": args.start_date, + "endDateString": args.end_date, + "endTestDateString": args.end_test_date, "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": args.fees, "gas_cost": args.gas_cost, "arb_fees": 0.0, "protocol_fee_split": 0.5, + "noise_trader_ratio": args.noise_trader_ratio, "return_val": args.objective, "reclamm_interpolation_method": args.interpolation, "reclamm_centeredness_scaling": args.centeredness_scaling, "reclamm_learn_arc_length_speed": learn_speed, "reclamm_use_shift_exponent": True, + **({"bout_offset": args.bout_offset} if args.bout_offset is not None else {}), "optimisation_settings": { "method": "optuna", "n_parameter_sets": 1, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), "optuna_settings": { "make_scalar": True, "expand_around": False, "n_trials": args.n_trials, "multi_objective": False, "parameter_config": param_config, + **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), }, }, } diff --git a/scripts/benchmark_reclamm_interpolation.py b/scripts/benchmark_reclamm_interpolation.py new file mode 100644 index 00000000..f462bbf0 --- /dev/null +++ b/scripts/benchmark_reclamm_interpolation.py @@ -0,0 +1,648 @@ +"""Benchmark reClAMM range shift interpolation: current vs optimal midpoint. + +Compares total arb loss during a range shift under different interpolation methods: + Geometric VB -- exponential decay of overvalued virtual (what contracts do) + Linear VB -- uniform steps in VB + Linear Z -- uniform steps in Z = sqrt(P)*VA - VB/sqrt(P) (optimal, from note) + Optimal 2-step -- exact midpoint via quadratic formula (Section 5 of note) + Brute-force optimal -- JAX gradient-optimised Z-target sequence + +Key result: per-step loss ~ (DeltaZ)^2 / (4X). Equal Z-increments minimise +total loss, analogous to TFMM optimal intermediate for G3M weight changes. + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/benchmark_reclamm_interpolation.py +""" + +import numpy as np +import matplotlib.pyplot as plt + +import jax +import jax.numpy as jnp +from scipy.optimize import minimize as scipy_minimize + +jax.config.update("jax_enable_x64", True) + + +# ── Core reClAMM mechanics ───────────────────────────────────────────────── + + +def compute_VA_from_VB(RA, RB, VB, Q): + """Contract rule (eq 15): VA = RA*(VB + RB) / ((Q-1)*VB - RB).""" + return RA * (VB + RB) / ((Q - 1) * VB - RB) + + +def compute_Z(VA, VB, P): + """Z = sqrt(P)*VA - VB/sqrt(P) (eq 12).""" + sqP = np.sqrt(P) + return sqP * VA - VB / sqP + + +def pool_value(RA, RB, P): + """Real pool value: P*RA + RB (eq 3).""" + return P * RA + RB + + +def micro_step(RA, RB, VA_new, VB_new, P): + """Virtual-balance update then arb to equilibrium Y/X = P. + + Returns (RA_new, RB_new, arb_loss). + """ + val_before = pool_value(RA, RB, P) + X = RA + VA_new + Y = RB + VB_new + L = X * Y + X_eq = np.sqrt(L / P) + Y_eq = P * X_eq + RA_new = X_eq - VA_new + RB_new = Y_eq - VB_new + return RA_new, RB_new, val_before - pool_value(RA_new, RB_new, P) + + +def solve_VB_for_Z(RA, RB, Z_star, Q, P): + """Solve quadratic for VB achieving Z(VB) = Z_star. + + Derived by substituting VA = RA*(VB+RB)/((Q-1)*VB-RB) into + Z = sqrt(P)*VA - VB/sqrt(P), then collecting terms in VB. + + NOTE: The research note (eq 28) has a sign error: the RB/sqrt(P) + term in b should be positive, not negative. Re-derived here from + scratch. + + Returns the physically valid root (VB > RB/(Q-1), positive). + Raises ValueError if no valid root exists. + """ + sqP = np.sqrt(P) + a = -(Q - 1) / sqP + b = sqP * RA + RB / sqP - (Q - 1) * Z_star # +RB/sqP, not minus + c = sqP * RA * RB + Z_star * RB + disc = b * b - 4 * a * c + if disc < -1e-6: + raise ValueError(f"negative discriminant: {disc:.4e}") + disc = max(disc, 0.0) + sd = np.sqrt(disc) + r1, r2 = (-b + sd) / (2 * a), (-b - sd) / (2 * a) + floor = RB / (Q - 1) + 1e-12 + ok = [r for r in (r1, r2) if r > floor] + if not ok: + raise ValueError(f"no valid root: r1={r1:.4f}, r2={r2:.4f}, floor={floor:.4f}") + return min(ok) + + +# ── Interpolation methods ────────────────────────────────────────────────── + + +def run_shift(RA, RB, VA_stale, VB_start, VB_end, Q, P, N, schedule): + """Execute N-step range shift (B overvalued, VB decreasing). + + schedule: "geometric" | "linear_VB" | "linear_Z" + + VA_stale: the current (possibly stale) VA -- used only for Z_start + in the linear_Z schedule. All micro-steps compute VA from the + contract rule with current reserves. + """ + # For linear_Z, precompute Z endpoints using contract-rule VA + if schedule == "linear_Z": + VA_start_cr = compute_VA_from_VB(RA, RB, VB_start, Q) + Z0 = compute_Z(VA_start_cr, VB_start, P) + VA_end_approx = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end = compute_Z(VA_end_approx, VB_end, P) + + total_loss = 0.0 + RA_c, RB_c = RA, RB + + for i in range(1, N + 1): + frac = i / N + if schedule == "geometric": + VB_i = VB_start * (VB_end / VB_start) ** frac + elif schedule == "linear_VB": + VB_i = VB_start + frac * (VB_end - VB_start) + elif schedule == "linear_Z": + Z_i = Z0 + frac * (Z_end - Z0) + VB_i = solve_VB_for_Z(RA_c, RB_c, Z_i, Q, P) + else: + raise ValueError(schedule) + + VA_i = compute_VA_from_VB(RA_c, RB_c, VB_i, Q) + RA_c, RB_c, loss = micro_step(RA_c, RB_c, VA_i, VB_i, P) + total_loss += loss + + return total_loss, RA_c, RB_c + + +def run_shift_optimal_2step(RA, RB, VA_stale, VB_start, VB_end, Q, P): + """Exact 2-step optimal midpoint (Section 5 of the note). + + Computes Z* = (Z_start + Z_end) / 2, solves quadratic for VB_mid. + """ + VA_start_cr = compute_VA_from_VB(RA, RB, VB_start, Q) + Z0 = compute_Z(VA_start_cr, VB_start, P) + VA_end_approx = compute_VA_from_VB(RA, RB, VB_end, Q) + Z2 = compute_Z(VA_end_approx, VB_end, P) + Z_star = (Z0 + Z2) / 2.0 + + # Step 1: jump to Z-midpoint + VB_mid = solve_VB_for_Z(RA, RB, Z_star, Q, P) + VA_mid = compute_VA_from_VB(RA, RB, VB_mid, Q) + RA1, RB1, loss1 = micro_step(RA, RB, VA_mid, VB_mid, P) + + # Step 2: jump to endpoint + VA_end = compute_VA_from_VB(RA1, RB1, VB_end, Q) + RA2, RB2, loss2 = micro_step(RA1, RB1, VA_end, VB_end, P) + + return loss1 + loss2, RA2, RB2 + + +# ── Scenario setup ───────────────────────────────────────────────────────── + + +def setup_centered_pool(P, price_ratio, R_scale=10000.0): + """Centered pool at price P with contract-rule-consistent virtuals. + + Returns (RA, RB, VA, VB, Q). + """ + Q = np.sqrt(price_ratio) + q4 = price_ratio ** 0.25 + + RA = R_scale + RB = P * R_scale + VA = RA / (q4 - 1) + VB = RB / (q4 - 1) + + return RA, RB, VA, VB, Q + + +def setup_decentered_pool(P_init, P_final, price_ratio, R_scale=10000.0): + """Centered pool at P_init, arb to P_final, then refresh virtuals. + + The refresh applies the contract rule to get consistent (VA, VB) at + the post-arb reserves, then arbs once more. This gives a decentered + but fully consistent state (equilibrium + contract rule). + + Returns (RA, RB, VA, VB, Q). + """ + Q = np.sqrt(price_ratio) + q4 = price_ratio ** 0.25 + + RA0 = R_scale + RB0 = P_init * R_scale + VA0 = RA0 / (q4 - 1) + VB0 = RB0 / (q4 - 1) + + # Arb to P_final (L preserved, virtuals stale) + X0 = RA0 + VA0 + Y0 = RB0 + VB0 + L = X0 * Y0 + X_new = np.sqrt(L / P_final) + Y_new = np.sqrt(L * P_final) + RA = X_new - VA0 + RB = Y_new - VB0 + + # Refresh: apply contract rule for current VB, then arb + VB = VB0 + VA = compute_VA_from_VB(RA, RB, VB, Q) + RA, RB, _ = micro_step(RA, RB, VA, VB, P_final) + + return RA, RB, VA, VB, Q + + +# ── JAX-differentiable versions for brute-force optimisation ────────────── + + +def _compute_VA_from_VB_jax(RA, RB, VB, Q): + return RA * (VB + RB) / ((Q - 1) * VB - RB) + + +def _compute_Z_jax(VA, VB, P): + sqP = jnp.sqrt(P) + return sqP * VA - VB / sqP + + +def _pool_value_jax(RA, RB, P): + return P * RA + RB + + +def _micro_step_jax(RA, RB, VA, VB, P): + val_before = _pool_value_jax(RA, RB, P) + X = RA + VA + Y = RB + VB + L = X * Y + X_eq = jnp.sqrt(L / P) + Y_eq = P * X_eq + RA_new = X_eq - VA + RB_new = Y_eq - VB + return RA_new, RB_new, val_before - _pool_value_jax(RA_new, RB_new, P) + + +def _solve_VB_for_Z_jax(RA, RB, Z_star, Q, P): + sqP = jnp.sqrt(P) + a = -(Q - 1) / sqP + b = sqP * RA + RB / sqP - (Q - 1) * Z_star + c = sqP * RA * RB + Z_star * RB + disc = jnp.maximum(b * b - 4 * a * c, 1e-30) + sd = jnp.sqrt(disc) + r1 = (-b + sd) / (2 * a) + r2 = (-b - sd) / (2 * a) + floor = RB / (Q - 1) + 1e-8 + return jnp.where(r2 > floor, r2, r1) + + +def _z_targets_from_raw(raw_params, Z_start, Z_end): + """Map unconstrained params -> sorted Z targets via softplus gaps.""" + gaps = jax.nn.softplus(raw_params) + gaps = gaps / jnp.sum(gaps) * (Z_end - Z_start) + return Z_start + jnp.cumsum(gaps) + + +def _make_loss_fn(N): + """Build a JIT-compiled loss function for a given N (unrolled loop).""" + + def total_loss(raw_params, RA, RB, Q, P, Z_start, Z_end): + Z_all = _z_targets_from_raw(raw_params, Z_start, Z_end) + RA_c, RB_c = RA, RB + total = 0.0 + for i in range(N): + VB_i = _solve_VB_for_Z_jax(RA_c, RB_c, Z_all[i], Q, P) + VA_i = _compute_VA_from_VB_jax(RA_c, RB_c, VB_i, Q) + RA_c, RB_c, loss = _micro_step_jax(RA_c, RB_c, VA_i, VB_i, P) + total = total + loss + return total + + return jax.jit(jax.value_and_grad(total_loss)) + + +def optimise_z_targets(RA, RB, Q, P, Z_start, Z_end, N, verbose=False): + """Find the Z-target sequence minimising total arb loss. + + Returns (optimal_loss, optimal_Z_targets_array_of_length_N). + """ + loss_and_grad_fn = _make_loss_fn(N) + RA_j = jnp.float64(RA) + RB_j = jnp.float64(RB) + Q_j = jnp.float64(Q) + P_j = jnp.float64(P) + Zs_j = jnp.float64(Z_start) + Ze_j = jnp.float64(Z_end) + + def objective(x): + val, grad = loss_and_grad_fn( + jnp.array(x, dtype=jnp.float64), RA_j, RB_j, Q_j, P_j, Zs_j, Ze_j + ) + return float(val), np.array(grad, dtype=np.float64) + + x0 = np.zeros(N) # softplus(0) = ln2, uniform gaps → linear Z init + result = scipy_minimize(objective, x0, jac=True, method="L-BFGS-B") + + optimal_Z = np.array( + _z_targets_from_raw(jnp.array(result.x), Zs_j, Ze_j) + ) + if verbose: + print(f" N={N}: loss={result.fun:.6f} " + f"nit={result.nit} success={result.success}") + return result.fun, optimal_Z + + +# ── Experiments ──────────────────────────────────────────────────────────── + + +def main(): + # --- Scenario: centered pool, moderate VB decay --- + P = 2.0 # token A costs 2 units of token B + price_ratio = 4.0 # rho, so Q = sqrt(4) = 2 + R_scale = 10000.0 + decay_fraction = 0.90 # VB_end = 0.90 * VB_start (10% decay) + + RA, RB, VA, VB, Q = setup_centered_pool(P, price_ratio, R_scale) + VB_start = VB + VB_end = VB * decay_fraction + + # Diagnostics + C = min(RA * VB, RB * VA) / max(RA * VB, RB * VA) + is_above = RA * VB > RB * VA + X = RA + VA + print("=" * 72) + print(f"Scenario: centered pool at P={P}, price_ratio={price_ratio}, Q={Q:.4f}") + print(f" RA={RA:.2f} RB={RB:.2f} VA={VA:.2f} VB={VB:.2f}") + print(f" Effective X={X:.2f} Pool value = {pool_value(RA, RB, P):.2f}") + print(f" Centeredness = {C:.4f} is_above = {is_above}") + print(f" VB shift: {VB_start:.2f} -> {VB_end:.2f} ({decay_fraction:.0%})") + VB_floor = RB / (Q - 1) + print(f" VB floor (denominator > 0): {VB_floor:.2f}") + Z_start = compute_Z(VA, VB, P) + VA_end_cr = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end = compute_Z(VA_end_cr, VB_end, P) + print(f" Z_start = {Z_start:.4f} Z_end = {Z_end:.4f}") + print(f" Approx 1-step loss ~ (DeltaZ)^2/(4X) = {(Z_end-Z_start)**2/(4*X):.2f}") + print("=" * 72) + + # ── Experiment 1: Loss vs N ──────────────────────────────────────── + + N_values = [1, 2, 3, 4, 6, 8, 12, 16, 24, 32, 48, 64, 96, 128] + schedules = ["geometric", "linear_VB", "linear_Z"] + results = {s: [] for s in schedules} + + for N in N_values: + for sched in schedules: + try: + loss, _, _ = run_shift( + RA, RB, VA, VB_start, VB_end, Q, P, N, sched + ) + except (ValueError, AssertionError) as e: + loss = np.nan + results[sched].append(loss) + + # Optimal 2-step (single point) + try: + loss_opt2, _, _ = run_shift_optimal_2step( + RA, RB, VA, VB_start, VB_end, Q, P + ) + except (ValueError, AssertionError): + loss_opt2 = np.nan + + # Table + loss_1 = results["geometric"][0] + print(f"\n{'N':>5s} {'Geo VB':>12s} {'Lin VB':>12s} {'Lin Z':>12s}" + f" {'Geo/1step':>9s} {'LinZ/1step':>10s} {'LinZ/Geo':>9s}") + print("-" * 80) + for j, N in enumerate(N_values): + g = results["geometric"][j] + lv = results["linear_VB"][j] + lz = results["linear_Z"][j] + print(f"{N:>5d} {g:>12.6f} {lv:>12.6f} {lz:>12.6f}" + f" {g / loss_1:>9.4f} {lz / loss_1:>10.4f} {lz / g:>9.4f}") + + print(f"\n Optimal 2-step loss: {loss_opt2:.6f}") + print(f" Geometric N=2 loss: {results['geometric'][1]:.6f}" + f" (opt/geo = {loss_opt2 / results['geometric'][1]:.4f})") + print(f" Linear Z N=2 loss: {results['linear_Z'][1]:.6f}" + f" (opt/linZ = {loss_opt2 / results['linear_Z'][1]:.4f})") + + # ── Experiment 2: Z and VB trajectories at N=8 ───────────────────── + + N_viz = 8 + traj_data = {} + for sched in schedules: + VB_traj, Z_traj, loss_traj = [VB_start], [], [] + VA_s = VA # stale + Z_traj.append(compute_Z(VA_s, VB_start, P)) + + RA_c, RB_c = RA, RB + if sched == "linear_Z": + Z0 = Z_traj[0] + VA_end_a = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end_val = compute_Z(VA_end_a, VB_end, P) + + for i in range(1, N_viz + 1): + frac = i / N_viz + if sched == "geometric": + VB_i = VB_start * (VB_end / VB_start) ** frac + elif sched == "linear_VB": + VB_i = VB_start + frac * (VB_end - VB_start) + else: + Z_i = Z0 + frac * (Z_end_val - Z0) + VB_i = solve_VB_for_Z(RA_c, RB_c, Z_i, Q, P) + + try: + VA_i = compute_VA_from_VB(RA_c, RB_c, VB_i, Q) + VB_traj.append(VB_i) + Z_traj.append(compute_Z(VA_i, VB_i, P)) + RA_c, RB_c, loss = micro_step(RA_c, RB_c, VA_i, VB_i, P) + loss_traj.append(loss) + except (ValueError, AssertionError): + break + + traj_data[sched] = { + "VB": np.array(VB_traj), + "Z": np.array(Z_traj), + "loss": np.array(loss_traj), + } + + # ── Experiment 3: sweep shift size at N=2 ────────────────────────── + + decay_sweep = np.linspace(0.80, 0.99, 30) + sweep = {s: [] for s in ["geometric", "linear_Z", "optimal_2step"]} + for df in decay_sweep: + VB_e = VB * df + try: + g, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "geometric") + lz, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "linear_Z") + o2, _, _ = run_shift_optimal_2step(RA, RB, VA, VB_start, VB_e, Q, P) + except (AssertionError, ValueError): + g = lz = o2 = np.nan + sweep["geometric"].append(g) + sweep["linear_Z"].append(lz) + sweep["optimal_2step"].append(o2) + + # ── Plots ────────────────────────────────────────────────────────── + + colours = {"geometric": "C0", "linear_VB": "C1", "linear_Z": "C2"} + labels = { + "geometric": "Geometric VB (contract)", + "linear_VB": "Linear VB", + "linear_Z": "Linear Z (optimal)", + } + + fig, axes = plt.subplots(2, 2, figsize=(13, 10)) + + # (0,0) Loss vs N + ax = axes[0, 0] + for s in schedules: + ax.plot(N_values, results[s], "o-", ms=4, color=colours[s], label=labels[s]) + ax.axhline(loss_opt2, color="C3", ls=":", label=f"Optimal 2-step = {loss_opt2:.4f}") + ax.set_xlabel("Steps N") + ax.set_ylabel("Total arb loss") + ax.set_title("Arb loss vs interpolation steps") + ax.set_xscale("log") + ax.set_yscale("log") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (0,1) Ratio linear_Z / geometric + ax = axes[0, 1] + ratios = np.array(results["linear_Z"]) / np.array(results["geometric"]) + ax.plot(N_values, ratios, "o-", color="C2") + ax.axhline(1.0, color="gray", ls="--", alpha=0.5) + ax.set_xlabel("Steps N") + ax.set_ylabel("Loss(Linear Z) / Loss(Geometric VB)") + ax.set_title("Relative improvement of Z-optimal") + ax.grid(True, alpha=0.3) + + # (1,0) Z trajectories at N=8 + ax = axes[1, 0] + steps = np.arange(N_viz + 1) + for s in schedules: + ax.plot(steps, traj_data[s]["Z"], "o-", ms=4, color=colours[s], label=labels[s]) + ax.set_xlabel("Step") + ax.set_ylabel("Z = sqrt(P)*VA - VB/sqrt(P)") + ax.set_title(f"Z trajectory (N={N_viz})") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (1,1) 2-step loss vs shift size + ax = axes[1, 1] + shift_pct = (1 - decay_sweep) * 100 + ax.plot(shift_pct, sweep["geometric"], color="C0", label="Geometric VB (N=2)") + ax.plot(shift_pct, sweep["linear_Z"], color="C2", label="Linear Z (N=2)") + ax.plot(shift_pct, sweep["optimal_2step"], ":", color="C3", label="Optimal 2-step") + ax.set_xlabel("Shift size (% VB decay)") + ax.set_ylabel("Arb loss") + ax.set_title("2-step loss vs shift magnitude") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_interpolation_benchmark.png", dpi=150) + print("\nSaved reclamm_interpolation_benchmark.png") + + # ── Per-step loss bar chart for N=8 ──────────────────────────────── + + fig2, ax = plt.subplots(figsize=(10, 5)) + x = np.arange(1, N_viz + 1) + w = 0.25 + for i, s in enumerate(schedules): + ax.bar(x + i * w, traj_data[s]["loss"], w, color=colours[s], label=labels[s]) + ax.set_xlabel("Step") + ax.set_ylabel("Per-step arb loss") + ax.set_title(f"Per-step loss distribution (N={N_viz})") + ax.legend(fontsize=8) + ax.set_xticks(x + w) + plt.tight_layout() + plt.savefig("reclamm_interpolation_perstep.png", dpi=150) + print("Saved reclamm_interpolation_perstep.png") + + # ── Experiment 4: small-shift regime (paper's approximation valid) ─── + + print("\n" + "=" * 72) + print("Experiment 4: Optimal 2-step vs Geometric N=2 at small shifts") + print(" (reserves nearly constant → paper's analysis should hold)") + print("-" * 72) + print(f" {'Decay %':>8s} {'Geo N=2':>12s} {'LinZ N=2':>12s} " + f"{'Opt2':>12s} {'Opt2/Geo':>9s} {'Opt2/LinZ':>9s}") + print("-" * 72) + + small_decays = [0.999, 0.998, 0.995, 0.99, 0.98, 0.95, 0.90, 0.80] + for df in small_decays: + VB_e = VB * df + try: + g, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "geometric") + lz, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "linear_Z") + o2, _, _ = run_shift_optimal_2step( + RA, RB, VA, VB_start, VB_e, Q, P + ) + except (ValueError, AssertionError) as e: + print(f" {(1-df)*100:>7.1f}% FAILED: {e}") + continue + print(f" {(1-df)*100:>7.1f}% {g:>12.6f} {lz:>12.6f} " + f"{o2:>12.6f} {o2/g:>9.6f} {o2/lz:>9.6f}") + + print("=" * 72) + + # ── Experiment 5: brute-force JAX-optimised Z targets ──────────────── + + print("\n" + "=" * 72) + print("Experiment 5: Brute-force optimal Z targets (JAX + L-BFGS-B)") + print(" Parameterisation: softplus gaps → sorted Z targets") + print(" Initialised at linear Z (uniform gaps)") + print("-" * 72) + + opt_N_values = [2, 3, 4, 6, 8, 12, 16, 24, 32] + opt_losses = {} + opt_Z_trajs = {} + + for N in opt_N_values: + loss_bf, Z_bf = optimise_z_targets( + RA, RB, Q, P, Z_start, Z_end, N, verbose=True + ) + opt_losses[N] = loss_bf + opt_Z_trajs[N] = Z_bf + + # Comparison table + print(f"\n {'N':>5s} {'Geometric':>12s} {'Linear Z':>12s} " + f"{'BF Optimal':>12s} {'BF/LinZ':>9s} {'BF/Geo':>9s}") + print("-" * 72) + for N in opt_N_values: + idx = N_values.index(N) if N in N_values else None + g = results["geometric"][idx] if idx is not None else np.nan + lz = results["linear_Z"][idx] if idx is not None else np.nan + bf = opt_losses[N] + print(f" {N:>5d} {g:>12.6f} {lz:>12.6f} " + f"{bf:>12.6f} {bf/lz:>9.6f} {bf/g:>9.6f}") + + # ── Plot: overlay brute-force on the main loss-vs-N chart ──────────── + + fig3, axes3 = plt.subplots(1, 2, figsize=(14, 5)) + + # (left) Loss vs N with brute-force overlay + ax = axes3[0] + for s in schedules: + ax.plot(N_values, results[s], "o-", ms=4, color=colours[s], + label=labels[s]) + bf_Ns = sorted(opt_losses.keys()) + bf_vals = [opt_losses[n] for n in bf_Ns] + ax.plot(bf_Ns, bf_vals, "s--", ms=5, color="C3", label="BF Optimal (JAX)") + ax.set_xlabel("Steps N") + ax.set_ylabel("Total arb loss") + ax.set_title("Arb loss vs interpolation steps (with BF optimal)") + ax.set_xscale("log") + ax.set_yscale("log") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (right) Z trajectory comparison at N=8 + ax = axes3[1] + N_cmp = 8 + steps_cmp = np.arange(N_cmp + 1) + + # Geometric: compute Z trajectory from VB + z_geo = [Z_start] + RA_t, RB_t = RA, RB + for i in range(1, N_cmp + 1): + frac = i / N_cmp + VB_i = VB_start * (VB_end / VB_start) ** frac + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_geo.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + # Linear Z + z_linz = [Z_start] + RA_t, RB_t = RA, RB + for i in range(1, N_cmp + 1): + frac = i / N_cmp + Z_i = Z_start + frac * (Z_end - Z_start) + VB_i = solve_VB_for_Z(RA_t, RB_t, Z_i, Q, P) + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_linz.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + # BF optimal + z_bf = [Z_start] + list(opt_Z_trajs[N_cmp]) + # Trace actual Z achieved after arb at each step + z_bf_actual = [Z_start] + RA_t, RB_t = RA, RB + for i in range(N_cmp): + VB_i = solve_VB_for_Z(RA_t, RB_t, opt_Z_trajs[N_cmp][i], Q, P) + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_bf_actual.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + ax.plot(steps_cmp, z_geo, "o-", ms=4, color="C0", label="Geometric VB") + ax.plot(steps_cmp, z_linz, "o-", ms=4, color="C2", label="Linear Z") + ax.plot(steps_cmp, z_bf_actual, "s--", ms=5, color="C3", + label="BF Optimal") + ax.plot(steps_cmp, np.linspace(Z_start, Z_end, N_cmp + 1), + ":", color="gray", alpha=0.5, label="Ideal linear Z") + ax.set_xlabel("Step") + ax.set_ylabel("Z = sqrt(P)*VA - VB/sqrt(P)") + ax.set_title(f"Z trajectory comparison (N={N_cmp})") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_interpolation_bruteforce.png", dpi=150) + print("\nSaved reclamm_interpolation_bruteforce.png") + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_bayesian.py b/scripts/calibrate_noise_bayesian.py new file mode 100644 index 00000000..efdfb4bb --- /dev/null +++ b/scripts/calibrate_noise_bayesian.py @@ -0,0 +1,853 @@ +"""Bayesian hierarchical noise volume model across Balancer pools. + +Full Bayesian version of the noise calibration: ALL K=4 per-pool +coefficients (intercept, TVL elasticity, volatility response, weekend +effect) vary per pool with pool-level covariates modulating their priors, +and an LKJ-decomposed covariance capturing correlations between +coefficients. + +Generative model: + For pool i with pool-level covariates z_i, day t: + + mu_i = B . z_i # K-vector population mean + eta_i ~ N(0, I_K) # non-centered offsets + theta_i = mu_i + diag(sigma) . L . eta_i # per-pool coefficients + + log(V_{i,t}) ~ N(theta_i . x_{i,t}, sigma_eps^2) + + theta_i = [intercept_i, b_tvl_i, b_sigma_i, b_weekend_i] + x_{i,t} = [1, log_tvl, volatility, weekend] + z_i = [1, chain_dummies(6), tier_A_dummies(2), tier_B_dummies(2), log_fee] + +Priors: + B_{k,d} ~ N(0, 5^2) + sigma_k ~ HalfNormal(2.0) + L ~ LKJCholesky(K=4, eta=2) + sigma_eps ~ HalfNormal(3.0) + +Usage: + # Full pipeline: fit + output + diagnostics + python scripts/calibrate_noise_bayesian.py \\ + --fit --output results/bayesian_noise_params.json --plot + + # Predict for an unseen pool + python scripts/calibrate_noise_bayesian.py \\ + --predict --chain BASE --tokens ETH USDC --fee 0.003 + + # Custom NUTS settings + python scripts/calibrate_noise_bayesian.py \\ + --fit --num-warmup 2000 --num-samples 4000 --num-chains 4 +""" + +import argparse +import json +import os +import sys + +import numpy as np +import pandas as pd + +# arviz 0.17.x imports scipy.signal.gaussian which was removed in scipy 1.13+. +# Patch it back from scipy.signal.windows before any arviz import. +try: + from scipy.signal import gaussian as _ # noqa: F401 +except ImportError: + from scipy.signal.windows import gaussian as _gauss + import scipy.signal + scipy.signal.gaussian = _gauss + +# --------------------------------------------------------------------------- +# Reuse constants and helpers from the frequentist script +# --------------------------------------------------------------------------- + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "local_data", "noise_calibration" +) + +# Reference levels for dummy coding (dropped categories) +REF_CHAIN = "ARBITRUM" +REF_TIER = 0 + +# Ordered non-reference chains (alphabetical excluding REF_CHAIN) +CHAIN_ORDER = ["BASE", "GNOSIS", "MAINNET", "OPTIMISM", "POLYGON", "SONIC"] + +K = 4 # number of per-pool coefficients +D = 12 # pool-level covariate dimension: 1 + 6 chains + 2 tier_A + 2 tier_B + 1 log_fee + +COEFF_NAMES = ["intercept", "b_tvl", "b_sigma", "b_weekend"] + + +# --------------------------------------------------------------------------- +# Token tier helpers (duplicated to avoid import fragility) +# --------------------------------------------------------------------------- + +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", +} + +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} + + +def classify_token_tier(symbol: str) -> int: + s = symbol.strip() + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 + + +# --------------------------------------------------------------------------- +# Data preparation +# --------------------------------------------------------------------------- + +def load_panel(cache_dir: str = CACHE_DIR) -> pd.DataFrame: + """Load the cached panel parquet produced by calibrate_noise_hierarchical.py --fetch.""" + panel_path = os.path.join(cache_dir, "panel.parquet") + if not os.path.exists(panel_path): + print(f"ERROR: Panel cache not found at {panel_path}", file=sys.stderr) + print("Run: python scripts/calibrate_noise_hierarchical.py --fetch", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_path) + + # Filter pools with < 10 observations + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid_pools)].copy() + + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + return panel + + +def _build_z_pool(pool_meta: pd.DataFrame) -> np.ndarray: + """Build (N_pools, D) pool-level covariate matrix. + + Columns: [1, chain_BASE, ..., chain_SONIC (6), + tier_A_1, tier_A_2, tier_B_1, tier_B_2, log_fee] + """ + N = len(pool_meta) + z = np.zeros((N, D), dtype=np.float64) + + # Intercept + z[:, 0] = 1.0 + + # Chain dummies (columns 1-6) + for j, chain in enumerate(CHAIN_ORDER): + z[:, 1 + j] = (pool_meta["chain"].values == chain).astype(float) + + # tier_A dummies (columns 7-8): tiers 1 and 2, reference = 0 + tier_a = pool_meta["tier_A"].values.astype(int) + z[:, 7] = (tier_a == 1).astype(float) + z[:, 8] = (tier_a == 2).astype(float) + + # tier_B dummies (columns 9-10): tiers 1 and 2, reference = 0 + tier_b = pool_meta["tier_B"].values.astype(int) + z[:, 9] = (tier_b == 1).astype(float) + z[:, 10] = (tier_b == 2).astype(float) + + # log_fee (column 11) + z[:, 11] = np.log(np.maximum(pool_meta["swap_fee"].values.astype(float), 1e-6)) + + return z + + +def prepare_data(panel: pd.DataFrame) -> dict: + """Construct JAX-ready arrays from the panel DataFrame. + + Returns dict with: + pool_idx : (N_obs,) int32 — pool index per observation + z_pool : (N_pools, D) float64 — pool-level covariates + x_obs : (N_obs, K) float64 — within-day regressors + y_obs : (N_obs,) float64 — log_volume + pool_ids : list — ordered pool IDs + pool_meta : DataFrame — per-pool metadata (indexed same as z_pool rows) + """ + # Stable pool ordering + pool_ids = sorted(panel["pool_id"].unique()) + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + N_pools = len(pool_ids) + + # Pool-level metadata (one row per pool) + pool_meta = panel.drop_duplicates("pool_id").set_index("pool_id").loc[pool_ids].reset_index() + z_pool = _build_z_pool(pool_meta) + + # Observation-level arrays + pool_idx = panel["pool_id"].map(pool_id_to_idx).values.astype(np.int32) + x_obs = np.column_stack([ + np.ones(len(panel)), + panel["log_tvl"].values, + panel["volatility"].values, + panel["weekend"].values, + ]).astype(np.float64) + y_obs = panel["log_volume"].values.astype(np.float64) + + print(f" Prepared: N_obs={len(y_obs)}, N_pools={N_pools}, K={K}, D={D}") + print(f" z_pool range check — log_fee: [{z_pool[:, 11].min():.2f}, {z_pool[:, 11].max():.2f}]") + + return { + "pool_idx": pool_idx, + "z_pool": z_pool, + "x_obs": x_obs, + "y_obs": y_obs, + "pool_ids": pool_ids, + "pool_meta": pool_meta, + "N_pools": N_pools, + } + + +# --------------------------------------------------------------------------- +# NumPyro model +# --------------------------------------------------------------------------- + +def hierarchical_noise_model(pool_idx, z_pool, x_obs, y_obs=None, + N_pools=None, K=4, D=12): + """Bayesian hierarchical noise volume model. + + Non-centered parameterization with LKJ correlation prior. + """ + import jax.numpy as jnp + import numpyro + import numpyro.distributions as dist + + N_obs = pool_idx.shape[0] + + # --- Population coefficient matrix B: (K, D) --- + B = numpyro.sample("B", dist.Normal(0.0, 5.0).expand([K, D]).to_event(2)) + + # --- Per-pool scale and correlation --- + sigma = numpyro.sample("sigma", dist.HalfNormal(2.0).expand([K]).to_event(1)) + L_Omega = numpyro.sample("L_Omega", dist.LKJCholesky(K, concentration=2.0)) + + # Cholesky factor of covariance: diag(sigma) @ L_Omega + L_Sigma = jnp.diag(sigma) @ L_Omega # (K, K) + + # --- Non-centered pool effects --- + with numpyro.plate("pools", N_pools): + eta = numpyro.sample("eta", dist.Normal(0.0, 1.0).expand([K]).to_event(1)) + + # theta_i = B @ z_i + L_Sigma @ eta_i for each pool i + # mu: (N_pools, K) = z_pool @ B^T + mu = z_pool @ B.T # (N_pools, K) + theta = mu + eta @ L_Sigma.T # (N_pools, K) + + # --- Observation model --- + sigma_eps = numpyro.sample("sigma_eps", dist.HalfNormal(3.0)) + + # Predicted log-volume: theta[pool_idx] . x_obs (dot product per obs) + theta_obs = theta[pool_idx] # (N_obs, K) + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) # (N_obs,) + + with numpyro.plate("obs", N_obs): + numpyro.sample("y", dist.Normal(mu_obs, sigma_eps), obs=y_obs) + + # Deterministic: store theta for extraction + numpyro.deterministic("theta", theta) + + +# --------------------------------------------------------------------------- +# Inference +# --------------------------------------------------------------------------- + +def run_inference(data, num_warmup=1000, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42): + """Run NUTS on the hierarchical model. + + Returns the MCMC object with samples. + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import MCMC, NUTS + + # Use all available CPU cores for chains + numpyro.set_host_device_count(min(num_chains, len(jax.devices("cpu")))) + + kernel = NUTS( + hierarchical_noise_model, + target_accept_prob=target_accept, + max_tree_depth=max_tree_depth, + ) + mcmc = MCMC( + kernel, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + progress_bar=True, + ) + + rng_key = jax.random.PRNGKey(seed) + + print(f"\n Running NUTS: {num_chains} chains x " + f"({num_warmup} warmup + {num_samples} samples)") + print(f" target_accept={target_accept}, max_tree_depth={max_tree_depth}") + + mcmc.run( + rng_key, + pool_idx=jnp.array(data["pool_idx"]), + z_pool=jnp.array(data["z_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K=K, + D=D, + ) + + mcmc.print_summary(exclude_deterministic=True) + return mcmc + + +# --------------------------------------------------------------------------- +# Post-processing +# --------------------------------------------------------------------------- + +def extract_noise_params(mcmc, data) -> list: + """Extract per-pool noise params from MCMC posterior. + + Reconstructs theta from the non-centered parameterization, + takes posterior medians, and applies weekend absorption: + b_0_effective = b_0_raw + b_weekend * (2/7) + + Returns list of dicts compatible with reclamm_loglinear_noise_volume. + """ + samples = mcmc.get_samples() + theta_samples = samples["theta"] # (n_samples, N_pools, K) + + # Posterior median per pool + theta_median = np.median(theta_samples, axis=0) # (N_pools, K) + theta_std = np.std(theta_samples, axis=0) # (N_pools, K) + + pool_ids = data["pool_ids"] + pool_meta = data["pool_meta"] + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + b_0_raw, b_tvl, b_sigma, b_weekend = theta_median[i] + std_vals = theta_std[i] + + # Weekend absorption: simulator has no weekend indicator, + # so fold the expected weekend effect into the intercept. + # Weekend days = 2/7 of all days. + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "theta_median": [float(x) for x in theta_median[i]], + "theta_std": [float(x) for x in std_vals], + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(meta["swap_fee"]), + }, + }) + + return results + + +def predict_new_pool(mcmc, data, chain: str, tokens: list, fee: float) -> dict: + """Predict noise params for an unseen pool using population effects. + + Constructs z_new, computes mu_new = B @ z_new across posterior samples, + and returns median + 90% credible intervals. + """ + # Build z_new + z_new = np.zeros(D, dtype=np.float64) + z_new[0] = 1.0 # intercept + + # Chain dummies + if chain in CHAIN_ORDER: + j = CHAIN_ORDER.index(chain) + z_new[1 + j] = 1.0 + + # Tier dummies + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + if tier_a == 1: + z_new[7] = 1.0 + elif tier_a == 2: + z_new[8] = 1.0 + if tier_b == 1: + z_new[9] = 1.0 + elif tier_b == 2: + z_new[10] = 1.0 + + # log_fee + z_new[11] = np.log(max(fee, 1e-6)) + + # Compute mu_new = B @ z_new across all posterior samples + B_samples = np.array(mcmc.get_samples()["B"]) # (n_samples, K, D) + mu_samples = np.einsum("skd,d->sk", B_samples, z_new) # (n_samples, K) + + mu_median = np.median(mu_samples, axis=0) + mu_q05 = np.percentile(mu_samples, 5, axis=0) + mu_q95 = np.percentile(mu_samples, 95, axis=0) + + # Weekend absorption + b_0_raw, b_tvl, b_sigma, b_weekend = mu_median + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + result = { + "chain": chain, + "tokens": tokens, + "fee": fee, + "prediction_source": "population_level", + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(fee), + }, + "credible_intervals_90": { + name: { + "median": float(mu_median[k]), + "q05": float(mu_q05[k]), + "q95": float(mu_q95[k]), + } + for k, name in enumerate(COEFF_NAMES) + }, + } + + print(f"\n Predicted noise_params for {chain} {tokens} (fee={fee}):") + for name, ci in result["credible_intervals_90"].items(): + print(f" {name:12s}: {ci['median']:+.3f} " + f"[{ci['q05']:+.3f}, {ci['q95']:+.3f}]") + print(f"\n Effective b_0 (weekend-absorbed): {b_0_effective:.3f}") + + return result + + +# --------------------------------------------------------------------------- +# Diagnostics +# --------------------------------------------------------------------------- + +def check_convergence(mcmc) -> dict: + """Compute convergence diagnostics: R-hat, ESS, divergences.""" + import arviz as az + + idata = az.from_numpyro(mcmc) + + # R-hat and ESS for non-deterministic parameters + n_chains = idata.posterior.sizes.get("chain", 1) + + rhat_max = float("nan") + if n_chains >= 2: + rhat = az.rhat(idata) + rhat_vals = [] + for var in rhat.data_vars: + if var == "theta": + continue # deterministic + vals = rhat[var].values + rhat_vals.extend(vals.flatten()) + rhat_max = float(np.nanmax(rhat_vals)) if rhat_vals else float("nan") + + ess = az.ess(idata) + ess_vals = [] + for var in ess.data_vars: + if var == "theta": + continue + vals = ess[var].values + ess_vals.extend(vals.flatten()) + ess_min = float(np.nanmin(ess_vals)) if ess_vals else float("nan") + + # Divergences + divergences = int(idata.sample_stats["diverging"].sum().values) + + print(f"\n Convergence diagnostics:") + if n_chains >= 2: + print(f" R-hat max: {rhat_max:.4f} {'OK' if rhat_max < 1.05 else 'WARNING'}") + else: + print(f" R-hat max: N/A (need >= 2 chains)") + print(f" ESS min: {ess_min:.0f} {'OK' if ess_min > 400 else 'WARNING'}") + print(f" Divergences: {divergences} {'OK' if divergences == 0 else 'WARNING'}") + + return { + "r_hat_max": rhat_max, + "ess_min": ess_min, + "divergences": divergences, + } + + +def plot_bayesian_diagnostics(mcmc, data, output_dir="results"): + """Generate ArviZ diagnostic plots.""" + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + import arviz as az + + os.makedirs(output_dir, exist_ok=True) + idata = az.from_numpyro(mcmc) + samples = mcmc.get_samples() + + # --- 1. Trace plots for sigma, sigma_eps --- + axes = az.plot_trace(idata, var_names=["sigma", "sigma_eps"], compact=True) + fig1 = axes.ravel()[0].figure + fig1.set_size_inches(14, 8) + path1 = os.path.join(output_dir, "bayesian_trace_sigma.png") + fig1.savefig(path1, dpi=150, bbox_inches="tight") + plt.close(fig1) + print(f" Saved: {path1}") + + # --- 2. Posterior predictive: predicted vs observed --- + theta_samples = samples["theta"] # (S, N_pools, K) + sigma_eps_samples = np.array(samples["sigma_eps"]) # (S,) + theta_median = np.median(theta_samples, axis=0) # (N_pools, K) + + pool_idx = data["pool_idx"] + x_obs = data["x_obs"] + y_obs = data["y_obs"] + + theta_obs = theta_median[pool_idx] + y_pred = np.sum(theta_obs * x_obs, axis=1) + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + ax.scatter(y_obs, y_pred, alpha=0.1, s=4, color="steelblue") + lims = [min(y_obs.min(), y_pred.min()), max(y_obs.max(), y_pred.max())] + ax.plot(lims, lims, "r--", linewidth=1) + ax.set_xlabel("Observed log(volume)") + ax.set_ylabel("Predicted log(volume)") + ax.set_title("Posterior predictive check") + r2 = 1 - np.var(y_obs - y_pred) / np.var(y_obs) + ax.text(0.05, 0.95, f"R² = {r2:.3f}", transform=ax.transAxes, + fontsize=11, verticalalignment="top") + + ax = axes[1] + residuals = y_obs - y_pred + ax.hist(residuals, bins=60, color="steelblue", edgecolor="white", alpha=0.8) + ax.axvline(0, color="red", linestyle="--") + ax.set_xlabel("Residual") + ax.set_title(f"Residual distribution (σ_ε ≈ {np.median(sigma_eps_samples):.2f})") + + plt.tight_layout() + path2 = os.path.join(output_dir, "bayesian_posterior_predictive.png") + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + # --- 3. Per-pool b_c (TVL elasticity) by chain/tier --- + pool_meta = data["pool_meta"] + b_tvl_all = theta_median[:, 1] # index 1 = b_tvl + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + chains_present = sorted(pool_meta["chain"].unique()) + chain_data = [] + chain_labels = [] + for c in chains_present: + mask = pool_meta["chain"].values == c + if mask.sum() > 0: + chain_data.append(b_tvl_all[mask]) + chain_labels.append(f"{c}\n(n={mask.sum()})") + ax.boxplot(chain_data, tick_labels=chain_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by chain") + + ax = axes[1] + # By tier_A + tier_a_vals = pool_meta["tier_A"].values.astype(int) + tier_labels_map = {0: "Blue-chip", 1: "Mid-cap", 2: "Long-tail"} + tier_data = [] + tier_labels = [] + for t in [0, 1, 2]: + mask = tier_a_vals == t + if mask.sum() > 0: + tier_data.append(b_tvl_all[mask]) + tier_labels.append(f"{tier_labels_map[t]}\n(n={mask.sum()})") + ax.boxplot(tier_data, tick_labels=tier_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by token tier (best token)") + + plt.tight_layout() + path3 = os.path.join(output_dir, "bayesian_per_pool_b_c.png") + plt.savefig(path3, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path3}") + + # --- 4. Correlation matrix posterior --- + L_Omega_samples = np.array(samples["L_Omega"]) # (S, K, K) + # Correlation = L @ L^T + Omega_samples = np.einsum("sij,skj->sik", L_Omega_samples, L_Omega_samples) + Omega_median = np.median(Omega_samples, axis=0) + + fig, ax = plt.subplots(figsize=(7, 6)) + im = ax.imshow(Omega_median, vmin=-1, vmax=1, cmap="RdBu_r") + ax.set_xticks(range(K)) + ax.set_yticks(range(K)) + ax.set_xticklabels(COEFF_NAMES, rotation=45, ha="right") + ax.set_yticklabels(COEFF_NAMES) + for i in range(K): + for j in range(K): + ax.text(j, i, f"{Omega_median[i, j]:.2f}", ha="center", va="center", + fontsize=10, color="white" if abs(Omega_median[i, j]) > 0.5 else "black") + plt.colorbar(im, ax=ax, shrink=0.8) + ax.set_title("Posterior median correlation matrix (Ω)") + plt.tight_layout() + path4 = os.path.join(output_dir, "bayesian_correlation_matrix.png") + plt.savefig(path4, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path4}") + + # --- 5. Shrinkage plot: OLS b_c vs hierarchical b_c --- + # Compute per-pool OLS b_c for comparison + panel_meta = data["pool_meta"] + pool_idx_arr = data["pool_idx"] + pool_ids = data["pool_ids"] + + ols_b_c = np.zeros(len(pool_ids)) + for i, pid in enumerate(pool_ids): + mask = pool_idx_arr == i + if mask.sum() < 5: + ols_b_c[i] = np.nan + continue + x_i = data["x_obs"][mask] + y_i = data["y_obs"][mask] + # Simple OLS: y = X @ beta + try: + beta, _, _, _ = np.linalg.lstsq(x_i, y_i, rcond=None) + ols_b_c[i] = beta[1] # TVL coefficient + except np.linalg.LinAlgError: + ols_b_c[i] = np.nan + + hier_b_c = theta_median[:, 1] + valid = np.isfinite(ols_b_c) + + fig, ax = plt.subplots(figsize=(8, 8)) + ax.scatter(ols_b_c[valid], hier_b_c[valid], alpha=0.6, s=20, color="steelblue") + + # Population mean line + pop_b_c = np.median(hier_b_c) + ax.axhline(pop_b_c, color="red", linestyle="--", linewidth=0.8, + label=f"Population median = {pop_b_c:.3f}") + + # 45-degree line + lims = [min(np.nanmin(ols_b_c[valid]), hier_b_c[valid].min()) - 0.2, + max(np.nanmax(ols_b_c[valid]), hier_b_c[valid].max()) + 0.2] + ax.plot(lims, lims, "k:", linewidth=0.8, alpha=0.5) + ax.set_xlabel("Per-pool OLS b_c") + ax.set_ylabel("Hierarchical posterior median b_c") + ax.set_title("Shrinkage: OLS vs hierarchical TVL elasticity") + ax.legend() + plt.tight_layout() + path5 = os.path.join(output_dir, "bayesian_shrinkage_b_c.png") + plt.savefig(path5, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path5}") + + +# --------------------------------------------------------------------------- +# JSON output +# --------------------------------------------------------------------------- + +def generate_output_json(pool_params, mcmc, data, convergence, output_path, + num_warmup, num_samples, num_chains, target_accept): + """Write structured JSON output with population effects and per-pool params.""" + samples = mcmc.get_samples() + + B_median = np.median(np.array(samples["B"]), axis=0).tolist() + sigma_median = np.median(np.array(samples["sigma"]), axis=0).tolist() + sigma_eps_median = float(np.median(np.array(samples["sigma_eps"]))) + + output = { + "model": "bayesian_hierarchical_loglinear", + "inference": { + "method": "NUTS", + "num_warmup": num_warmup, + "num_samples": num_samples, + "num_chains": num_chains, + "target_accept_prob": target_accept, + }, + "population_effects": { + "B": B_median, + "sigma": sigma_median, + "sigma_eps": sigma_eps_median, + "coeff_names": COEFF_NAMES, + "covariate_names": ( + ["intercept"] + [f"chain_{c}" for c in CHAIN_ORDER] + + ["tier_A_1", "tier_A_2", "tier_B_1", "tier_B_2", "log_fee"] + ), + }, + "convergence": convergence, + "pools": { + p["pool_id"]: { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + } + for p in pool_params + }, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Bayesian hierarchical noise volume model for Balancer pools" + ) + parser.add_argument( + "--fetch", action="store_true", + help="Fetch pool data (delegates to calibrate_noise_hierarchical.py --fetch)", + ) + parser.add_argument("--fit", action="store_true", help="Run NUTS inference") + parser.add_argument("--plot", action="store_true", help="Generate diagnostic plots") + parser.add_argument("--output", default=None, help="Output JSON path") + parser.add_argument("--output-dir", default="results", help="Plot output directory") + parser.add_argument("--predict", action="store_true", help="Predict for a new pool") + parser.add_argument("--chain", default=None, help="Chain for --predict") + parser.add_argument("--tokens", nargs="+", default=None, help="Tokens for --predict") + parser.add_argument("--fee", type=float, default=0.003, help="Fee for --predict") + parser.add_argument("--cache-dir", default=None, help="Cache directory") + + # NUTS hyperparameters + parser.add_argument("--num-warmup", type=int, default=1000) + parser.add_argument("--num-samples", type=int, default=2000) + parser.add_argument("--num-chains", type=int, default=4) + parser.add_argument("--target-accept", type=float, default=0.85) + parser.add_argument("--max-tree-depth", type=int, default=10) + parser.add_argument("--seed", type=int, default=42) + + args = parser.parse_args() + + cache_dir = args.cache_dir or CACHE_DIR + + if not any([args.fetch, args.fit, args.predict]): + parser.error("At least one of --fetch, --fit, --predict is required") + + # --- Fetch (delegate to existing script) --- + if args.fetch: + import subprocess + cmd = [ + sys.executable, "scripts/calibrate_noise_hierarchical.py", + "--fetch", "--cache-dir", cache_dir, + ] + print("Delegating data fetch to calibrate_noise_hierarchical.py...") + subprocess.run(cmd, check=True) + + # --- Fit --- + if args.fit: + print("\nBayesian Hierarchical Noise Volume Model") + print("=" * 60) + + panel = load_panel(cache_dir) + data = prepare_data(panel) + + mcmc = run_inference( + data, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + ) + + convergence = check_convergence(mcmc) + pool_params = extract_noise_params(mcmc, data) + + # Print summary statistics + b_c_vals = [p["noise_params"]["b_c"] for p in pool_params] + b_0_vals = [p["noise_params"]["b_0"] for p in pool_params] + print(f"\n Per-pool b_c: mean={np.mean(b_c_vals):.3f}, " + f"std={np.std(b_c_vals):.3f}, " + f"range=[{np.min(b_c_vals):.3f}, {np.max(b_c_vals):.3f}]") + print(f" Per-pool b_0: mean={np.mean(b_0_vals):.3f}, " + f"std={np.std(b_0_vals):.3f}") + + if args.output: + generate_output_json( + pool_params, mcmc, data, convergence, args.output, + args.num_warmup, args.num_samples, args.num_chains, + args.target_accept, + ) + + if args.plot: + print("\nGenerating diagnostic plots...") + plot_bayesian_diagnostics(mcmc, data, output_dir=args.output_dir) + + # Save MCMC samples for --predict reuse + mcmc_cache = os.path.join(cache_dir, "bayesian_mcmc_samples.npz") + samples = mcmc.get_samples() + np.savez_compressed( + mcmc_cache, + **{k: np.array(v) for k, v in samples.items()}, + ) + # Also save data arrays for predict + data_cache = os.path.join(cache_dir, "bayesian_data.npz") + np.savez_compressed( + data_cache, + pool_idx=data["pool_idx"], + z_pool=data["z_pool"], + x_obs=data["x_obs"], + y_obs=data["y_obs"], + ) + # Save pool_ids list + with open(os.path.join(cache_dir, "bayesian_pool_ids.json"), "w") as f: + json.dump(data["pool_ids"], f) + print(f" Saved MCMC samples -> {mcmc_cache}") + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + parser.error("--predict requires --chain and --tokens") + + # Load cached MCMC samples + mcmc_cache = os.path.join(cache_dir, "bayesian_mcmc_samples.npz") + if not os.path.exists(mcmc_cache): + print(f"ERROR: MCMC cache not found at {mcmc_cache}", file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + # For predict, we only need B samples — create a minimal mock + cached = np.load(mcmc_cache) + + class _MockMCMC: + """Minimal interface to reuse predict_new_pool with cached samples.""" + def __init__(self, samples_dict): + self._samples = samples_dict + def get_samples(self): + return self._samples + + samples_dict = {k: cached[k] for k in cached.files} + mock_mcmc = _MockMCMC(samples_dict) + + result = predict_new_pool(mock_mcmc, None, args.chain, args.tokens, args.fee) + print(json.dumps(result, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_hierarchical.py b/scripts/calibrate_noise_hierarchical.py new file mode 100644 index 00000000..4fef642d --- /dev/null +++ b/scripts/calibrate_noise_hierarchical.py @@ -0,0 +1,1556 @@ +"""Bayesian hierarchical noise volume model across Balancer WEIGHTED + RECLAMM pools. + +Pools data cross-sectionally across all Balancer weighted/reCLAMM pools, +fits a Bayesian hierarchical model where pool covariates (chain, token tier, +fee) modulate all coefficients via group-level regression, with full +posterior inference via NumPyro. + +Model: + Hyperpriors: + Φ ~ Normal(0, 2) (K × 3) group-level regression + σ_θ ~ HalfNormal(2) (3,) per-coefficient scales + L_ω ~ LKJCholesky(3, η=2) correlation structure + β_weekend ~ Normal(0, 2) shared nuisance + σ_ε ~ HalfNormal(3) observation noise + + For each pool i: + x_i = [1, chain_dummies, tier_dummies, log_fee] (K,) covariates + z_i ~ N(0, I₃) non-centered + θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i (α_i, β_tvl_i, β_vol_i) + + For each observation (i, t): + log(V) ~ N(α_i + β_tvl_i·log_tvl + β_vol_i·vol + β_weekend·weekend, σ²_ε) + +Usage: + # Full pipeline: fetch data + fit model + output + python scripts/calibrate_noise_hierarchical.py \\ + --fetch --fit --output results/hierarchical_noise_params.json --plot + + # Use cached data, re-fit only + python scripts/calibrate_noise_hierarchical.py \\ + --fit --output results/hierarchical_noise_params.json + + # Predict for a new pool + python scripts/calibrate_noise_hierarchical.py \\ + --predict --chain BASE --tokens ETH BTC --fee 0.003 + + # Use NUTS instead of SVI + python scripts/calibrate_noise_hierarchical.py \\ + --fit --nuts --output results/hierarchical_noise_params.json +""" + +import argparse +import json +import os +import sys +import time +import urllib.request +from datetime import datetime, timezone + +import numpy as np +import pandas as pd + +import jax +import jax.numpy as jnp +import numpyro +import numpyro.distributions as dist +from numpyro.infer import SVI, MCMC, NUTS, Trace_ELBO, Predictive +from numpyro.infer.autoguide import AutoMultivariateNormal + +numpyro.enable_x64() + + +# --------------------------------------------------------------------------- +# Constants +# --------------------------------------------------------------------------- + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAINS = [ + "MAINNET", "POLYGON", "ARBITRUM", "GNOSIS", "BASE", "SONIC", "OPTIMISM", + "AVALANCHE", +] + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "local_data", "noise_calibration" +) + +# --------------------------------------------------------------------------- +# Token tier classification +# --------------------------------------------------------------------------- + +# Tier 0: blue-chip — top by volume, wrapped native, major stables +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", # Sonic native +} + +# Tier 1: mid-cap DeFi blue-chips (approx CoinGecko rank < 200) +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} + + +def _normalise_symbol(symbol: str) -> str: + """Normalise wrapped/bridged variants to canonical form.""" + s = symbol.strip() + # Common wrapped → unwrapped + mapping = { + "WETH": "WETH", # keep WETH as-is (it's in tier 0) + "WBTC": "WBTC", + "cbBTC": "cbBTC", + "WMATIC": "WMATIC", + "WAVAX": "WAVAX", + "WXDAI": "WXDAI", + "wS": "wS", + } + return mapping.get(s, s) + + +def classify_token_tier(symbol: str) -> int: + """Classify a token symbol into tier 0/1/2. + + Returns + ------- + int + 0 = blue-chip, 1 = mid-cap, 2 = long-tail + """ + s = _normalise_symbol(symbol) + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 + + +# --------------------------------------------------------------------------- +# Phase 1: API data ingestion +# --------------------------------------------------------------------------- + +def _graphql_request(query: dict, base_url: str = BALANCER_API_URL, + timeout: int = 30) -> dict: + """Send a GraphQL request to the Balancer V3 API.""" + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8")) + + +def enumerate_balancer_pools( + chains: list = None, + pool_types: list = None, + min_tvl: float = 10000.0, +) -> pd.DataFrame: + """Enumerate all WEIGHTED + RECLAMM pools across chains from Balancer API. + + Parameters + ---------- + chains : list of str + API chain identifiers (e.g. ["MAINNET", "BASE"]). + pool_types : list of str + Pool type filters (e.g. ["WEIGHTED", "STABLE"]). + min_tvl : float + Minimum TVL in USD to include. + + Returns + ------- + pd.DataFrame + Columns: pool_id, chain, pool_type, tokens (list of symbols), + swap_fee, create_time, dynamic_data_tvl. + """ + if chains is None: + chains = BALANCER_API_CHAINS + if pool_types is None: + pool_types = ["WEIGHTED", "RECLAMM"] + + all_pools = [] + for chain in chains: + print(f" Querying {chain}...", end=" ", flush=True) + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { + chainIn: [$chain] + poolTypeIn: $types + minTvl: $minTvl + } + ) { + id + chain + type + createTime + protocolVersion + poolTokens { + symbol + weight + address + } + dynamicData { + totalLiquidity + swapFee + } + } + } + """, + "variables": { + "chain": chain, + "types": pool_types, + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f"FAILED ({e})") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + weights = [t.get("weight") for t in p.get("poolTokens", [])] + token_addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "protocol_version": p.get("protocolVersion", 0), + "tokens": tokens, + "token_addresses": token_addresses, + "weights": weights, + "swap_fee": fee, + "create_time": p.get("createTime", 0), + "current_tvl": tvl, + }) + + print(f"{len(pools)} pools") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools across {len(chains)} chains") + return df + + +def fetch_pool_snapshots(pool_id: str, chain: str, + base_url: str = BALANCER_API_URL) -> pd.DataFrame: + """Fetch ALL_TIME daily snapshots for a single pool. + + Returns + ------- + pd.DataFrame + Columns: timestamp, volume_usd, total_liquidity_usd. + """ + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + } + } + """, + "variables": { + "poolId": pool_id, + "chain": chain, + "range": "ALL_TIME", + }, + } + + body = _graphql_request(query) + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + + if not snapshots: + return pd.DataFrame(columns=["timestamp", "volume_usd", "total_liquidity_usd"]) + + records = [] + for snap in snapshots: + records.append({ + "timestamp": int(snap["timestamp"]), + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + }) + + df = pd.DataFrame(records) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + # Deduplicate by date (keep last snapshot per day) + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df + + +def fetch_all_snapshots(pools_df: pd.DataFrame, + cache_path: str = None) -> pd.DataFrame: + """Fetch daily snapshots for all pools, with caching. + + Parameters + ---------- + pools_df : pd.DataFrame + Pool enumeration from enumerate_balancer_pools. + cache_path : str, optional + Path to parquet cache. If it exists, only fetch missing pools. + + Returns + ------- + pd.DataFrame + Panel with columns: pool_id, chain, date, volume_usd, + total_liquidity_usd. + """ + # Load cache if exists + cached = pd.DataFrame() + cached_pool_ids = set() + if cache_path and os.path.exists(cache_path): + cached = pd.read_parquet(cache_path) + cached_pool_ids = set(cached["pool_id"].unique()) + print(f" Cache has {len(cached_pool_ids)} pools, " + f"{len(cached)} pool-days") + + # Determine which pools need fetching + if len(pools_df) == 0: + print(" No pools to fetch.") + return cached if len(cached) > 0 else pd.DataFrame( + columns=["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + ) + to_fetch = pools_df[~pools_df["pool_id"].isin(cached_pool_ids)] + print(f" Need to fetch {len(to_fetch)} new pools") + + new_records = [] + for i, (_, pool) in enumerate(to_fetch.iterrows()): + if (i + 1) % 10 == 0 or i == 0: + print(f" Fetching {i+1}/{len(to_fetch)}: {pool['pool_id'][:10]}... " + f"({pool['chain']})", flush=True) + try: + snap_df = fetch_pool_snapshots(pool["pool_id"], pool["chain"]) + if len(snap_df) > 0: + snap_df["pool_id"] = pool["pool_id"] + snap_df["chain"] = pool["chain"] + new_records.append(snap_df[ + ["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + ]) + except Exception as e: + print(f" FAILED {pool['pool_id'][:10]}: {e}") + time.sleep(0.5) # Rate limit + + if new_records: + new_df = pd.concat(new_records, ignore_index=True) + combined = pd.concat([cached, new_df], ignore_index=True) + else: + combined = cached + + # Save cache + if cache_path and len(combined) > 0: + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + combined.to_parquet(cache_path, index=False) + print(f" Saved cache: {len(combined)} pool-days → {cache_path}") + + return combined + + +def fetch_token_prices(token_addresses_by_chain: dict, + cache_dir: str = None) -> dict: + """Fetch hourly token prices from Balancer API. + + Parameters + ---------- + token_addresses_by_chain : dict + {chain: {symbol: address, ...}, ...} + cache_dir : str, optional + Directory for per-token price caches. + + Returns + ------- + dict + {(chain, symbol): pd.DataFrame with columns [timestamp, price], ...} + """ + if cache_dir: + os.makedirs(cache_dir, exist_ok=True) + + prices = {} + + for chain, tokens in token_addresses_by_chain.items(): + # Check cache first, collect uncached addresses + uncached = {} + for symbol, address in tokens.items(): + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") if cache_dir else None + + if cp and os.path.exists(cp): + prices[(chain, symbol)] = pd.read_parquet(cp) + else: + uncached[symbol] = address + + if not uncached: + continue + + # Batch fetch: API supports multiple addresses per request + addr_to_symbol = {addr: sym for sym, addr in uncached.items()} + addresses = list(uncached.values()) + + print(f" Fetching {len(addresses)} prices on {chain}...", + flush=True) + + # Batch in groups of 20 to avoid oversized requests + batch_size = 20 + for batch_start in range(0, len(addresses), batch_size): + batch_addrs = addresses[batch_start:batch_start + batch_size] + query = { + "query": """ + query GetPrices($chain: GqlChain!, $addresses: [String!]!, + $range: GqlTokenChartDataRange!) { + tokenGetHistoricalPrices( + addresses: $addresses, chain: $chain, range: $range + ) { + address + prices { + timestamp + price + } + } + } + """, + "variables": { + "chain": chain, + "addresses": batch_addrs, + "range": "ONE_YEAR", + }, + } + + try: + body = _graphql_request(query, timeout=60) + results = body.get("data", {}).get( + "tokenGetHistoricalPrices", []) + for result in results: + addr = result.get("address", "") + price_list = result.get("prices", []) + symbol = addr_to_symbol.get(addr) + if symbol and price_list: + pdf = pd.DataFrame(price_list) + pdf["timestamp"] = pdf["timestamp"].astype(int) + pdf["price"] = pdf["price"].astype(float) + prices[(chain, symbol)] = pdf + if cache_dir: + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") + pdf.to_parquet(cp, index=False) + except Exception as e: + print(f" FAILED batch on {chain}: {e}") + + time.sleep(0.5) + + print(f" Got prices for {len(prices)} token-chain pairs") + return prices + + +def compute_pair_volatility( + snapshots_df: pd.DataFrame, + pool_row: pd.Series, + token_prices: dict, +) -> pd.Series: + """Compute daily annualised volatility for a pool's pair ratio. + + Uses hourly prices from the API to compute daily realised volatility. + Falls back to a default of 0.5 if price data is insufficient. + + Parameters + ---------- + snapshots_df : pd.DataFrame + Pool's daily snapshots (need dates). + pool_row : pd.Series + Pool metadata row (need tokens, chain). + token_prices : dict + {(chain, symbol): DataFrame, ...} + + Returns + ------- + pd.Series + Indexed by date, values are annualised daily volatility. + """ + tokens = pool_row["tokens"] + chain = pool_row["chain"] + + if len(tokens) < 2: + return pd.Series(dtype=float) + + # Get price series for token[0] and token[1] + # Try chain-specific first, then any chain + def _get_price_df(symbol): + # Exact match + key = (chain, symbol) + if key in token_prices: + return token_prices[key] + # Any chain + for k, v in token_prices.items(): + if k[1] == symbol: + return v + return None + + p0_df = _get_price_df(tokens[0]) + p1_df = _get_price_df(tokens[1]) + + # If either is a stablecoin, use $1 + stables = {"USDC", "USDT", "DAI", "LUSD", "GHO", "crvUSD", "sDAI", + "WXDAI", "xDAI", "USDC.e", "USDbC"} + + if tokens[0] in stables and tokens[1] in stables: + # Stable-stable pair: near-zero vol + dates = snapshots_df["date"].unique() + return pd.Series(0.01, index=dates) + + if p0_df is None and tokens[0] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) # fallback + if p1_df is None and tokens[1] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) # fallback + + # Build hourly price ratio + if tokens[0] in stables: + # ratio = 1 / p1 + if p1_df is None or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p1_df.copy() + ratio_df["ratio"] = 1.0 / ratio_df["price"] + elif tokens[1] in stables: + # ratio = p0 + if p0_df is None or len(p0_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p0_df.copy() + ratio_df["ratio"] = ratio_df["price"] + else: + # Both non-stable: ratio = p0/p1 + if p0_df is None or p1_df is None or len(p0_df) == 0 or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + # Merge on nearest timestamp + merged = pd.merge_asof( + p0_df.sort_values("timestamp"), + p1_df.sort_values("timestamp"), + on="timestamp", + suffixes=("_0", "_1"), + tolerance=7200, # 2 hour tolerance + ).dropna() + if len(merged) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = merged.copy() + ratio_df["ratio"] = merged["price_0"] / merged["price_1"] + + ratio_df["datetime"] = pd.to_datetime(ratio_df["timestamp"], unit="s") + ratio_df["date"] = ratio_df["datetime"].dt.date + ratio_df = ratio_df.sort_values("timestamp") + + # Log returns + ratio_df["log_return"] = np.log( + ratio_df["ratio"] / ratio_df["ratio"].shift(1) + ) + ratio_df = ratio_df.dropna(subset=["log_return"]) + + # Daily vol from hourly returns, annualised + daily_vol = ratio_df.groupby("date")["log_return"].std() + # Hourly data → ~24 returns/day. Annualise: σ_daily * sqrt(365) + # But std() already gives daily std from hourly returns, so: + # σ_annual = σ_hourly * sqrt(24 * 365) + daily_vol_ann = daily_vol * np.sqrt(24 * 365) + + return daily_vol_ann + + +def assemble_panel( + pools_df: pd.DataFrame, + snapshots_df: pd.DataFrame, + token_prices: dict, +) -> pd.DataFrame: + """Assemble the full panel DataFrame for hierarchical estimation. + + Parameters + ---------- + pools_df : pd.DataFrame + Pool enumeration from enumerate_balancer_pools. + snapshots_df : pd.DataFrame + Daily snapshots from fetch_all_snapshots. + token_prices : dict + Token prices from fetch_token_prices. + + Returns + ------- + pd.DataFrame + Panel with columns: pool_id, chain, date, log_volume, log_tvl, + volatility, weekend, log_fee, tier_A, tier_B, tokens. + """ + records = [] + pool_ids = snapshots_df["pool_id"].unique() + n_pools = len(pool_ids) + + for i, pool_id in enumerate(pool_ids): + if (i + 1) % 20 == 0 or i == 0: + print(f" Assembling {i+1}/{n_pools}...", flush=True) + + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pool_id] + pool_meta = pools_df[pools_df["pool_id"] == pool_id] + if len(pool_meta) == 0: + continue + pool_row = pool_meta.iloc[0] + + tokens = pool_row["tokens"] + if len(tokens) < 2: + continue + + chain = pool_row["chain"] + swap_fee = pool_row["swap_fee"] + + # Token tiers: sort by tier (best tier first) + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] # best (lowest) tier + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + # Compute volatility for this pool's pair + vol_series = compute_pair_volatility(pool_snaps, pool_row, token_prices) + + for _, snap in pool_snaps.iterrows(): + date = snap["date"] + volume = snap["volume_usd"] + tvl = snap["total_liquidity_usd"] + + # Skip zero/negative TVL or volume + if tvl <= 0 or volume <= 0: + continue + + # Volatility lookup + if isinstance(vol_series, pd.Series) and date in vol_series.index: + vol = vol_series[date] + else: + vol = 0.5 # fallback + + if not np.isfinite(vol) or vol <= 0: + vol = 0.5 + + # Weekend indicator + if isinstance(date, datetime): + is_weekend = date.weekday() >= 5 + else: + is_weekend = pd.Timestamp(date).weekday() >= 5 + + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": date, + "log_volume": np.log(volume), + "log_tvl": np.log(tvl), + "volatility": vol, + "weekend": 1.0 if is_weekend else 0.0, + "log_fee": np.log(max(swap_fee, 1e-6)), + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": ",".join(tokens[:2]), + "swap_fee": swap_fee, + }) + + panel = pd.DataFrame(records) + print(f"\n Panel: {len(panel)} observations, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + return panel + + +# --------------------------------------------------------------------------- +# Phase 2: Bayesian hierarchical model +# --------------------------------------------------------------------------- + +def _encode_covariates(panel: pd.DataFrame) -> dict: + """Build NumPyro-ready arrays from the panel DataFrame. + + Returns + ------- + dict with keys: + pool_idx : (N_obs,) int array mapping each observation to its pool + X_pool : (N_pools, K) covariate matrix (intercept + dummies + log_fee) + log_tvl, volatility, weekend, log_volume : (N_obs,) float arrays + pool_ids : (N_pools,) pool ID strings + covariate_names : list of str, column names for X_pool + ref_chain, ref_tier_a, ref_tier_b : reference categories + chains : sorted list of all chains + pool_meta : DataFrame of per-pool metadata + """ + pool_meta = panel.drop_duplicates("pool_id").reset_index(drop=True) + pool_ids = pool_meta["pool_id"].values + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + + pool_idx = panel["pool_id"].map(pool_id_to_idx).values + + # Build X_pool columns + chains = sorted(panel["chain"].unique()) + ref_chain = chains[0] + chain_cols = [] + chain_names = [] + for c in chains[1:]: + chain_cols.append((pool_meta["chain"] == c).astype(float).values) + chain_names.append(f"chain_{c}") + + tier_a_vals = sorted(pool_meta["tier_A"].astype(str).unique()) + ref_tier_a = tier_a_vals[0] + tier_a_cols = [] + tier_a_names = [] + for t in tier_a_vals[1:]: + tier_a_cols.append( + (pool_meta["tier_A"].astype(str) == t).astype(float).values + ) + tier_a_names.append(f"tier_A_{t}") + + tier_b_vals = sorted(pool_meta["tier_B"].astype(str).unique()) + ref_tier_b = tier_b_vals[0] + tier_b_cols = [] + tier_b_names = [] + for t in tier_b_vals[1:]: + tier_b_cols.append( + (pool_meta["tier_B"].astype(str) == t).astype(float).values + ) + tier_b_names.append(f"tier_B_{t}") + + N_pools = len(pool_ids) + columns = [np.ones((N_pools, 1))] + col_names = ["intercept"] + + for arr, name in zip(chain_cols, chain_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_a_cols, tier_a_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_b_cols, tier_b_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + columns.append(pool_meta["log_fee"].values.reshape(-1, 1)) + col_names.append("log_fee") + + X_pool = np.hstack(columns) + + return { + "pool_idx": pool_idx.astype(np.int32), + "X_pool": X_pool.astype(np.float64), + "log_tvl": panel["log_tvl"].values.astype(np.float64), + "volatility": panel["volatility"].values.astype(np.float64), + "weekend": panel["weekend"].values.astype(np.float64), + "log_volume": panel["log_volume"].values.astype(np.float64), + "pool_ids": pool_ids, + "covariate_names": col_names, + "ref_chain": ref_chain, + "ref_tier_a": ref_tier_a, + "ref_tier_b": ref_tier_b, + "chains": chains, + "pool_meta": pool_meta, + } + + +def _hierarchical_noise_model( + pool_idx, X_pool, log_tvl, volatility, weekend, log_volume=None, +): + """NumPyro model: Bayesian hierarchical loglinear noise volume. + + All pool covariates modulate all three coefficients (α, β_tvl, β_vol) + through the group-level regression matrix Φ, with correlated random + effects via LKJ-Cholesky. + """ + N_pools = X_pool.shape[0] + K = X_pool.shape[1] + + # Hyperpriors + Phi = numpyro.sample("Phi", dist.Normal(0, 2).expand([K, 3]).to_event(2)) + sigma_theta = numpyro.sample( + "sigma_theta", dist.HalfNormal(2).expand([3]).to_event(1) + ) + L_omega = numpyro.sample("L_omega", dist.LKJCholesky(3, concentration=2)) + beta_weekend = numpyro.sample("beta_weekend", dist.Normal(0, 2)) + sigma_eps = numpyro.sample("sigma_eps", dist.HalfNormal(3)) + + # Non-centered pool random effects + with numpyro.plate("pools", N_pools): + z = numpyro.sample("z", dist.Normal(jnp.zeros(3), 1).to_event(1)) + + # θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i + mu = X_pool @ Phi # (N_pools, 3) + L_Sigma = sigma_theta[:, None] * L_omega # (3, 3) + theta = mu + z @ L_Sigma.T # (N_pools, 3) + + alpha = theta[:, 0] + beta_tvl = theta[:, 1] + beta_vol = theta[:, 2] + + # Observation model + loc = (alpha[pool_idx] + + beta_tvl[pool_idx] * log_tvl + + beta_vol[pool_idx] * volatility + + beta_weekend * weekend) + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("log_volume", dist.Normal(loc, sigma_eps), obs=log_volume) + + +def fit_bayesian_model( + panel: pd.DataFrame, use_nuts: bool = False, +) -> tuple: + """Fit the Bayesian hierarchical model via SVI or NUTS. + + Parameters + ---------- + panel : pd.DataFrame + Panel from assemble_panel. + use_nuts : bool + If True, use NUTS MCMC (slower, exact). Otherwise SVI with + AutoMultivariateNormal guide. + + Returns + ------- + samples : dict + Posterior samples keyed by parameter name. + encoding : dict + From _encode_covariates (needed downstream). + """ + encoding = _encode_covariates(panel) + + model_kwargs = dict( + pool_idx=jnp.array(encoding["pool_idx"]), + X_pool=jnp.array(encoding["X_pool"]), + log_tvl=jnp.array(encoding["log_tvl"]), + volatility=jnp.array(encoding["volatility"]), + weekend=jnp.array(encoding["weekend"]), + log_volume=jnp.array(encoding["log_volume"]), + ) + + N_pools = encoding["X_pool"].shape[0] + K = encoding["X_pool"].shape[1] + print(f" N obs = {len(encoding['pool_idx'])}, " + f"N pools = {N_pools}, K covariates = {K}") + print(f" Covariates: {encoding['covariate_names']}") + + rng_key = jax.random.PRNGKey(0) + + if use_nuts: + print(" Running NUTS (500 warmup + 1000 samples)...") + kernel = NUTS(_hierarchical_noise_model) + mcmc = MCMC(kernel, num_warmup=500, num_samples=1000, num_chains=1) + mcmc.run(rng_key, **model_kwargs) + samples = mcmc.get_samples() + print(" NUTS complete.") + else: + print(" Running SVI with AutoMultivariateNormal (20k steps)...") + guide = AutoMultivariateNormal(_hierarchical_noise_model) + optimizer = numpyro.optim.Adam(1e-3) + svi = SVI(_hierarchical_noise_model, guide, optimizer, + loss=Trace_ELBO()) + svi_result = svi.run(rng_key, 20_000, **model_kwargs) + print(f" SVI complete. Final ELBO loss: {svi_result.losses[-1]:.2f}") + + predictive = Predictive( + guide, params=svi_result.params, num_samples=1000, + ) + samples = predictive(jax.random.PRNGKey(1), **model_kwargs) + print(" Drew 1000 posterior samples.") + + return samples, encoding + + +def extract_posteriors(samples: dict, encoding: dict) -> dict: + """Reconstruct pool-specific coefficients from posterior samples. + + Computes θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i for each posterior draw, + then returns posterior means and variance components. + + Returns + ------- + dict with keys: + pool_effects : {pool_id: {alpha, beta_tvl, beta_vol}} + Phi_mean : (K, 3) array + sigma_theta_mean : (3,) array + correlation_matrix : (3, 3) array + beta_weekend_mean : float + sigma_eps_mean : float + theta_samples : (S, N_pools, 3) array (for diagnostics) + """ + Phi = np.array(samples["Phi"]) # (S, K, 3) + sigma_theta = np.array(samples["sigma_theta"]) # (S, 3) + L_omega = np.array(samples["L_omega"]) # (S, 3, 3) + z = np.array(samples["z"]) # (S, N_pools, 3) + beta_weekend = np.array(samples["beta_weekend"]) # (S,) + sigma_eps = np.array(samples["sigma_eps"]) # (S,) + + X_pool = encoding["X_pool"] # (N_pools, K) + pool_ids = encoding["pool_ids"] + + # mu = X_pool @ Phi for each sample: (S, N_pools, 3) + mu = np.einsum("pk,skj->spj", X_pool, Phi) + + # L_Sigma = diag(sigma_theta) @ L_omega: (S, 3, 3) + L_Sigma = sigma_theta[:, :, None] * L_omega + + # offset = z @ L_Sigma^T: (S, N_pools, 3) + offset = np.einsum("spi,sji->spj", z, L_Sigma) + + theta = mu + offset # (S, N_pools, 3) + theta_mean = theta.mean(axis=0) # (N_pools, 3) + + pool_effects = {} + for i, pid in enumerate(pool_ids): + pool_effects[pid] = { + "alpha": float(theta_mean[i, 0]), + "beta_tvl": float(theta_mean[i, 1]), + "beta_vol": float(theta_mean[i, 2]), + } + + Phi_mean = Phi.mean(axis=0) # (K, 3) + + # Correlation matrix: R = L_omega @ L_omega^T, averaged over samples + R_samples = np.einsum("sij,skj->sik", L_omega, L_omega) + R_mean = R_samples.mean(axis=0) + + return { + "pool_effects": pool_effects, + "Phi_mean": Phi_mean, + "sigma_theta_mean": sigma_theta.mean(axis=0), + "correlation_matrix": R_mean, + "beta_weekend_mean": float(beta_weekend.mean()), + "sigma_eps_mean": float(sigma_eps.mean()), + "theta_samples": theta, + } + + +def compute_noise_params(posteriors: dict, panel: pd.DataFrame) -> list: + """Convert posterior pool effects to per-pool noise_params dicts. + + Each pool now has its own β_tvl and β_vol (from the hierarchical + posterior), rather than sharing global slopes. + + Parameters + ---------- + posteriors : dict + From extract_posteriors. + panel : pd.DataFrame + Panel data. + + Returns + ------- + list of dict + Each dict has: pool_id, chain, tokens, noise_params. + """ + pool_effects = posteriors["pool_effects"] + pool_meta = panel.drop_duplicates("pool_id").set_index("pool_id") + + results = [] + for pool_id, effects in pool_effects.items(): + if pool_id not in pool_meta.index: + continue + meta = pool_meta.loc[pool_id] + swap_fee = float(meta.get("swap_fee", 0.003)) + + results.append({ + "pool_id": pool_id, + "chain": meta["chain"], + "tokens": (meta["tokens"].split(",") + if isinstance(meta["tokens"], str) + else meta["tokens"]), + "noise_params": { + "b_0": effects["alpha"], + "b_sigma": effects["beta_vol"], + "b_c": effects["beta_tvl"], + "base_fee": swap_fee, + }, + }) + + return results + + +def _build_covariate_vector( + encoding: dict, chain: str, tokens: list, fee: float, +) -> np.ndarray: + """Construct a covariate vector x for a new pool. + + Matches the column order of X_pool from _encode_covariates so that + x @ Phi_mean gives population-level predictions for all 3 coefficients. + """ + col_names = encoding["covariate_names"] + x = np.zeros(len(col_names)) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + x[i] = 1.0 + elif name == "log_fee": + x[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + x[i] = 1.0 + elif name == f"tier_A_{tier_a}": + x[i] = 1.0 + elif name == f"tier_B_{tier_b}": + x[i] = 1.0 + + return x + + +def predict_new_pool( + posteriors: dict, + encoding: dict, + chain: str, + tokens: list, + fee: float, +) -> dict: + """Predict noise params for an unseen pool. + + Uses population-level estimates only (x @ Φ, no pool random effect). + + Parameters + ---------- + posteriors : dict + From extract_posteriors. + encoding : dict + From _encode_covariates (or loaded from cache). + chain : str + Chain API identifier (e.g. "BASE"). + tokens : list + Token symbols (e.g. ["ETH", "BTC"]). + fee : float + Swap fee rate. + + Returns + ------- + dict + noise_params dict with pool-predicted coefficients. + """ + x = _build_covariate_vector(encoding, chain, tokens, fee) + Phi_mean = posteriors["Phi_mean"] + + theta_pred = x @ Phi_mean # (3,) — population-level prediction + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + return { + "b_0": float(theta_pred[0]), + "b_sigma": float(theta_pred[2]), + "b_c": float(theta_pred[1]), + "base_fee": float(fee), + "_prediction_source": "population_level", + "_alpha": float(theta_pred[0]), + "_beta_tvl": float(theta_pred[1]), + "_beta_vol": float(theta_pred[2]), + "_tier_a": tier_a, + "_tier_b": tier_b, + } + + +# --------------------------------------------------------------------------- +# Phase 3: Diagnostics and output +# --------------------------------------------------------------------------- + +def plot_hierarchical_diagnostics( + panel: pd.DataFrame, + posteriors: dict, + encoding: dict, + output_dir: str = "results", +): + """Generate diagnostic plots for the Bayesian hierarchical model. + + Figure 1 (2x2): + (0,0) Pool-specific coefficient distributions (α, β_tvl, β_vol) + (0,1) Chain effects on all 3 coefficients + (1,0) Tier effects on all 3 coefficients + (1,1) Model summary (Φ, σ_θ, correlations, σ_ε) + + Figure 2: + β_tvl vs β_vol scatter colored by chain + """ + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + os.makedirs(output_dir, exist_ok=True) + + pool_effects = posteriors["pool_effects"] + Phi_mean = posteriors["Phi_mean"] + col_names = encoding["covariate_names"] + pool_meta = panel.drop_duplicates("pool_id") + + alphas = [e["alpha"] for e in pool_effects.values()] + beta_tvls = [e["beta_tvl"] for e in pool_effects.values()] + beta_vols = [e["beta_vol"] for e in pool_effects.values()] + + # --- Figure 1: Diagnostics 2x2 --- + fig, axes = plt.subplots(2, 2, figsize=(14, 10)) + + # (0,0) Pool-specific coefficient distributions + ax = axes[0, 0] + bins = 25 + ax.hist(alphas, bins=bins, alpha=0.6, label="α (intercept)", + color="steelblue", edgecolor="white") + ax.hist(beta_tvls, bins=bins, alpha=0.6, label="β_tvl", + color="coral", edgecolor="white") + ax.hist(beta_vols, bins=bins, alpha=0.6, label="β_vol", + color="seagreen", edgecolor="white") + ax.set_xlabel("Coefficient value") + ax.set_ylabel("Count") + ax.set_title(f"Pool-specific coefficients (n={len(alphas)})") + ax.legend() + + # (0,1) Chain effects on all 3 coefficients + ax = axes[0, 1] + chain_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("chain_")} + chain_labels = [] + chain_alpha_effects = [] + chain_tvl_effects = [] + chain_vol_effects = [] + # Reference chain (effect = 0) + ref_chain = encoding["ref_chain"] + chain_counts = pool_meta["chain"].value_counts() + chain_labels.append(f"{ref_chain}\n(n={chain_counts.get(ref_chain, 0)})") + chain_alpha_effects.append(0.0) + chain_tvl_effects.append(0.0) + chain_vol_effects.append(0.0) + for name, row_idx in sorted(chain_rows.items()): + chain_name = name.replace("chain_", "") + chain_labels.append( + f"{chain_name}\n(n={chain_counts.get(chain_name, 0)})" + ) + chain_alpha_effects.append(Phi_mean[row_idx, 0]) + chain_tvl_effects.append(Phi_mean[row_idx, 1]) + chain_vol_effects.append(Phi_mean[row_idx, 2]) + + y_pos = np.arange(len(chain_labels)) + bar_h = 0.25 + ax.barh(y_pos - bar_h, chain_alpha_effects, bar_h, label="α", + color="steelblue", alpha=0.8) + ax.barh(y_pos, chain_tvl_effects, bar_h, label="β_tvl", + color="coral", alpha=0.8) + ax.barh(y_pos + bar_h, chain_vol_effects, bar_h, label="β_vol", + color="seagreen", alpha=0.8) + ax.set_yticks(y_pos) + ax.set_yticklabels(chain_labels) + ax.axvline(0, color="red", linestyle="--", linewidth=1) + ax.set_xlabel("Effect (relative to reference)") + ax.set_title("Chain effects on all coefficients") + ax.legend(fontsize=8) + + # (1,0) Tier effects on all 3 coefficients + ax = axes[1, 0] + tier_names = ["0 (blue-chip)", "1 (mid-cap)", "2 (long-tail)"] + coeff_labels = ["α", "β_tvl", "β_vol"] + tier_a_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("tier_A_")} + tier_b_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("tier_B_")} + + # Build effects matrix: (3 tiers) x (3 coefficients) x (A/B) + x_pos = np.arange(len(tier_names)) + width = 0.13 + for coeff_idx, (coeff_name, color) in enumerate( + zip(coeff_labels, ["steelblue", "coral", "seagreen"]) + ): + tier_a_vals = [0.0, 0.0, 0.0] # reference tier gets 0 + tier_b_vals = [0.0, 0.0, 0.0] + for name, row_idx in tier_a_rows.items(): + tier_val = name.replace("tier_A_", "") + if tier_val in ("0", "1", "2"): + tier_a_vals[int(tier_val)] = Phi_mean[row_idx, coeff_idx] + for name, row_idx in tier_b_rows.items(): + tier_val = name.replace("tier_B_", "") + if tier_val in ("0", "1", "2"): + tier_b_vals[int(tier_val)] = Phi_mean[row_idx, coeff_idx] + offset = (coeff_idx - 1) * width * 2 + ax.bar(x_pos + offset - width / 2, tier_a_vals, width, + label=f"{coeff_name} (A)" if coeff_idx == 0 else "", + color=color, alpha=0.7, edgecolor="white") + ax.bar(x_pos + offset + width / 2, tier_b_vals, width, + label=f"{coeff_name} (B)" if coeff_idx == 0 else "", + color=color, alpha=0.4, edgecolor="white", hatch="//") + + ax.set_xticks(x_pos) + ax.set_xticklabels(tier_names) + ax.set_ylabel("Effect on coefficient") + ax.set_title("Token tier effects (solid=A, hatched=B)") + ax.axhline(0, color="black", linewidth=0.5) + # Manual legend for coefficient colors + from matplotlib.patches import Patch + ax.legend(handles=[Patch(color=c, label=l) for c, l in + zip(["steelblue", "coral", "seagreen"], coeff_labels)], + fontsize=8) + + # (1,1) Model summary text + ax = axes[1, 1] + ax.axis("off") + sigma_theta = posteriors["sigma_theta_mean"] + R = posteriors["correlation_matrix"] + summary = "Group-level regression Φ (posterior mean):\n" + summary += f" {'covariate':<20s} {'α':>8s} {'β_tvl':>8s} {'β_vol':>8s}\n" + summary += " " + "-" * 46 + "\n" + for j, name in enumerate(col_names): + summary += (f" {name:<20s} {Phi_mean[j,0]:>8.3f} " + f"{Phi_mean[j,1]:>8.3f} {Phi_mean[j,2]:>8.3f}\n") + summary += f"\nσ_θ: [{sigma_theta[0]:.3f}, {sigma_theta[1]:.3f}, " + summary += f"{sigma_theta[2]:.3f}]\n" + summary += f"Correlation:\n" + for i in range(3): + summary += f" [{R[i,0]:>6.3f} {R[i,1]:>6.3f} {R[i,2]:>6.3f}]\n" + summary += f"β_weekend: {posteriors['beta_weekend_mean']:.4f}\n" + summary += f"σ_ε: {posteriors['sigma_eps_mean']:.4f}\n" + ax.text(0.02, 0.98, summary, transform=ax.transAxes, + fontsize=7, verticalalignment="top", fontfamily="monospace") + ax.set_title("Model Summary") + + fig.suptitle( + "Bayesian Hierarchical Noise Model — Diagnostics", fontsize=13, + ) + plt.tight_layout() + path = os.path.join(output_dir, "hierarchical_diagnostics.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- Figure 2: β_tvl vs β_vol scatter colored by chain --- + fig2, ax2 = plt.subplots(figsize=(10, 7)) + pool_id_to_chain = dict( + zip(pool_meta["pool_id"], pool_meta["chain"]) + ) + chain_colors = {} + cmap = plt.cm.tab10 + unique_chains = sorted(pool_meta["chain"].unique()) + for i, c in enumerate(unique_chains): + chain_colors[c] = cmap(i % 10) + + for pid, effects in pool_effects.items(): + c = pool_id_to_chain.get(pid, "?") + ax2.scatter(effects["beta_tvl"], effects["beta_vol"], + color=chain_colors.get(c, "gray"), alpha=0.6, s=20, + edgecolors="white", linewidths=0.3) + + # Legend + from matplotlib.lines import Line2D + handles = [Line2D([0], [0], marker="o", color="w", + markerfacecolor=chain_colors[c], markersize=8, + label=c) + for c in unique_chains if c in chain_colors] + ax2.legend(handles=handles, fontsize=8, loc="best") + ax2.set_xlabel("β_tvl (TVL elasticity)") + ax2.set_ylabel("β_vol (volatility sensitivity)") + ax2.set_title("Pool-specific coefficients by chain") + ax2.axhline(0, color="gray", linewidth=0.5, linestyle="--") + ax2.axvline(0, color="gray", linewidth=0.5, linestyle="--") + plt.tight_layout() + path2 = os.path.join(output_dir, "beta_tvl_vs_beta_vol.png") + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + return path, path2 + + +def generate_noise_params_json( + pool_params: list, + posteriors: dict, + encoding: dict, + output_path: str, + inference_method: str = "svi", +): + """Write per-pool noise params to JSON. + + Parameters + ---------- + pool_params : list of dict + From compute_noise_params. + posteriors : dict + From extract_posteriors. + encoding : dict + From _encode_covariates. + output_path : str + Output JSON path. + inference_method : str + "svi" or "nuts". + """ + output = { + "model": "bayesian_hierarchical_loglinear", + "inference_method": inference_method, + "Phi": posteriors["Phi_mean"].tolist(), + "covariate_names": encoding["covariate_names"], + "sigma_theta": posteriors["sigma_theta_mean"].tolist(), + "correlation_matrix": posteriors["correlation_matrix"].tolist(), + "beta_weekend": posteriors["beta_weekend_mean"], + "sigma_eps": posteriors["sigma_eps_mean"], + "pools": pool_params, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params → {output_path}") + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Bayesian hierarchical noise volume model for Balancer pools" + ) + parser.add_argument( + "--fetch", action="store_true", + help="Fetch pool data from Balancer API (cached to local_data/)", + ) + parser.add_argument( + "--fit", action="store_true", + help="Fit Bayesian hierarchical model (requires fetched data)", + ) + parser.add_argument( + "--nuts", action="store_true", + help="Use NUTS MCMC instead of SVI (slower, exact posteriors)", + ) + parser.add_argument( + "--plot", action="store_true", + help="Generate diagnostic plots", + ) + parser.add_argument( + "--output", default=None, + help="Output JSON path for per-pool noise params", + ) + parser.add_argument( + "--output-dir", default="results", + help="Directory for diagnostic plots (default: results)", + ) + parser.add_argument( + "--predict", action="store_true", + help="Predict noise params for a new pool", + ) + parser.add_argument( + "--chain", default=None, + help="Chain for --predict (e.g. BASE, MAINNET)", + ) + parser.add_argument( + "--tokens", nargs="+", default=None, + help="Token symbols for --predict (e.g. ETH BTC)", + ) + parser.add_argument( + "--fee", type=float, default=0.003, + help="Swap fee for --predict", + ) + parser.add_argument( + "--min-tvl", type=float, default=10000.0, + help="Minimum TVL filter for pool enumeration", + ) + parser.add_argument( + "--cache-dir", default=None, + help="Cache directory (default: local_data/noise_calibration/)", + ) + args = parser.parse_args() + + cache_dir = args.cache_dir or CACHE_DIR + + if not any([args.fetch, args.fit, args.predict]): + parser.error("At least one of --fetch, --fit, --predict is required") + + # --- Fetch --- + pools_cache = os.path.join(cache_dir, "pools.parquet") + snaps_cache = os.path.join(cache_dir, "pool_snapshots.parquet") + prices_cache = os.path.join(cache_dir, "token_prices") + panel_cache = os.path.join(cache_dir, "panel.parquet") + + if args.fetch: + print("Phase 1: Fetching data from Balancer API") + print("=" * 60) + + # Step 1: Enumerate pools + print("\n1. Enumerating pools...") + pools_df = enumerate_balancer_pools(min_tvl=args.min_tvl) + os.makedirs(cache_dir, exist_ok=True) + pools_df.to_parquet(pools_cache, index=False) + print(f" Saved {len(pools_df)} pools → {pools_cache}") + + # Step 2: Fetch snapshots + print("\n2. Fetching daily snapshots...") + snapshots_df = fetch_all_snapshots(pools_df, cache_path=snaps_cache) + + # Step 3: Fetch token prices + print("\n3. Fetching token prices...") + token_addr_by_chain = {} + for _, pool in pools_df.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + # Step 4: Assemble panel + print("\n4. Assembling panel...") + panel = assemble_panel(pools_df, snapshots_df, token_prices) + panel.to_parquet(panel_cache, index=False) + print(f" Saved panel → {panel_cache}") + + print(f"\nFetch complete. Panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # --- Fit --- + if args.fit: + inference_method = "nuts" if args.nuts else "svi" + print(f"\nPhase 2: Fitting Bayesian hierarchical model ({inference_method})") + print("=" * 60) + + # Load panel + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_cache) + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + # Filter: need at least 10 days per pool for stable estimates + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel_filtered = panel[panel["pool_id"].isin(valid_pools)] + print(f" After filtering (≥10 days): {len(panel_filtered)} obs, " + f"{panel_filtered['pool_id'].nunique()} pools") + + samples, encoding = fit_bayesian_model( + panel_filtered, use_nuts=args.nuts, + ) + posteriors = extract_posteriors(samples, encoding) + pool_params = compute_noise_params(posteriors, panel_filtered) + + # Print key diagnostics + Phi_mean = posteriors["Phi_mean"] + col_names = encoding["covariate_names"] + intercept_idx = col_names.index("intercept") + log_fee_idx = col_names.index("log_fee") + + print(f"\n Key results:") + print(f" Population intercept (Φ[intercept]):") + print(f" α: {Phi_mean[intercept_idx, 0]:.4f}") + print(f" β_tvl: {Phi_mean[intercept_idx, 1]:.4f}") + print(f" β_vol: {Phi_mean[intercept_idx, 2]:.4f}") + print(f" Fee effect (Φ[log_fee]):") + print(f" α: {Phi_mean[log_fee_idx, 0]:.4f}") + print(f" β_tvl: {Phi_mean[log_fee_idx, 1]:.4f}") + print(f" β_vol: {Phi_mean[log_fee_idx, 2]:.4f}") + print(f" β_weekend: {posteriors['beta_weekend_mean']:.4f}") + print(f" σ_θ: {posteriors['sigma_theta_mean']}") + print(f" σ_ε: {posteriors['sigma_eps_mean']:.4f}") + + # Verify pool-specific variation + b_sigmas = [p["noise_params"]["b_sigma"] for p in pool_params] + b_cs = [p["noise_params"]["b_c"] for p in pool_params] + print(f" b_sigma range: [{min(b_sigmas):.4f}, {max(b_sigmas):.4f}]") + print(f" b_c range: [{min(b_cs):.4f}, {max(b_cs):.4f}]") + + # Cache posteriors + encoding for --predict and --plot + posteriors_cache = os.path.join(cache_dir, "posteriors.json") + cache_data = { + "Phi_mean": posteriors["Phi_mean"].tolist(), + "sigma_theta_mean": posteriors["sigma_theta_mean"].tolist(), + "correlation_matrix": posteriors["correlation_matrix"].tolist(), + "beta_weekend_mean": posteriors["beta_weekend_mean"], + "sigma_eps_mean": posteriors["sigma_eps_mean"], + "pool_effects": posteriors["pool_effects"], + "covariate_names": encoding["covariate_names"], + "ref_chain": encoding["ref_chain"], + "ref_tier_a": encoding["ref_tier_a"], + "ref_tier_b": encoding["ref_tier_b"], + "chains": encoding["chains"], + "inference_method": inference_method, + } + with open(posteriors_cache, "w") as f: + json.dump(cache_data, f, indent=2, default=str) + print(f" Cached posteriors → {posteriors_cache}") + + if args.output: + generate_noise_params_json( + pool_params, posteriors, encoding, + args.output, inference_method=inference_method, + ) + + if args.plot: + print("\nPhase 3: Generating diagnostics") + print("=" * 60) + plot_hierarchical_diagnostics( + panel_filtered, posteriors, encoding, + output_dir=args.output_dir, + ) + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + parser.error("--predict requires --chain and --tokens") + + print(f"\nPredicting noise params for new pool:") + print(f" Chain: {args.chain}") + print(f" Tokens: {args.tokens}") + print(f" Fee: {args.fee}") + + # Load cached posteriors + encoding metadata + posteriors_cache = os.path.join(cache_dir, "posteriors.json") + if not os.path.exists(posteriors_cache): + print(f"ERROR: Posteriors cache not found at {posteriors_cache}", + file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + with open(posteriors_cache) as f: + cache_data = json.load(f) + + posteriors = { + "Phi_mean": np.array(cache_data["Phi_mean"]), + "sigma_theta_mean": np.array(cache_data["sigma_theta_mean"]), + "correlation_matrix": np.array(cache_data["correlation_matrix"]), + "beta_weekend_mean": cache_data["beta_weekend_mean"], + "sigma_eps_mean": cache_data["sigma_eps_mean"], + "pool_effects": cache_data["pool_effects"], + } + encoding = { + "covariate_names": cache_data["covariate_names"], + "ref_chain": cache_data["ref_chain"], + "ref_tier_a": cache_data["ref_tier_a"], + "ref_tier_b": cache_data["ref_tier_b"], + "chains": cache_data["chains"], + } + + params = predict_new_pool( + posteriors, encoding, args.chain, args.tokens, args.fee, + ) + print(f"\n Predicted noise_params:") + print(json.dumps(params, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_unified.py b/scripts/calibrate_noise_unified.py new file mode 100644 index 00000000..0e355271 --- /dev/null +++ b/scripts/calibrate_noise_unified.py @@ -0,0 +1,6 @@ +"""Thin wrapper — all logic lives in quantammsim.noise_calibration.""" +from quantammsim.noise_calibration import * # noqa: F401, F403 +from quantammsim.noise_calibration.cli import main + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_reclamm_noise.py b/scripts/calibrate_reclamm_noise.py new file mode 100644 index 00000000..8d095841 --- /dev/null +++ b/scripts/calibrate_reclamm_noise.py @@ -0,0 +1,842 @@ +"""OLS calibration of the Tsoukalas noise volume model for reClAMM pools. + +Fits the structural volume equation: + V_daily/1e6 = a_0 + a_sigma*sigma + a_c*sqrt(c_eff/1e6) + +where c_eff = (Ra+Va)*pA + (Rb+Vb)*pB is the effective TVL (real + virtual). + +From daily pool snapshots (volume, TVL, volatility). Outputs a noise_params dict +compatible with run_fingerprint["reclamm_noise_params"]. + +Usage: + # From a pre-assembled CSV + python scripts/calibrate_reclamm_noise.py --csv daily_data.csv --base-fee 0.003 + + # End-to-end from API + DB + parquets + python scripts/calibrate_reclamm_noise.py --pool cbBTC_WETH +""" + +import argparse +import json +import os +import sqlite3 +import sys +import urllib.request +from datetime import datetime, timezone + +import numpy as np +import pandas as pd + + +# --------------------------------------------------------------------------- +# Balancer V3 API +# --------------------------------------------------------------------------- + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAIN = { + "base": "BASE", + "ethereum": "MAINNET", + "gnosis": "GNOSIS", + "avalanche": "AVALANCHE", + "arbitrum": "ARBITRUM", + "polygon": "POLYGON", + "optimism": "OPTIMISM", + "sonic": "SONIC", +} + + +def fetch_balancer_snapshots(chain, pool_address, start_ts, end_ts, + base_url=BALANCER_API_URL): + """Fetch daily pool snapshots from Balancer V3 GraphQL API. + + Parameters + ---------- + chain : str + Chain name (e.g. 'base', 'ethereum'). + pool_address : str + Pool contract address (hex, no 0x prefix). + start_ts : int + Start unix timestamp (seconds). + end_ts : int + End unix timestamp (seconds). + base_url : str + Balancer API base URL. + + Returns + ------- + pd.DataFrame + Columns: date, volume_usd, total_liquidity_usd. Indexed by date string. + """ + api_chain = BALANCER_API_CHAIN.get(chain) + if api_chain is None: + raise ValueError(f"Unknown chain for Balancer API: {chain!r}") + + pool_id = f"0x{pool_address}" if not pool_address.startswith("0x") else pool_address + + # Paginate: API may limit results. Fetch in 90-day windows. + all_snapshots = [] + window = 90 * 86400 + cursor = start_ts + + while cursor < end_ts: + window_end = min(cursor + window, end_ts) + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + } + } + """, + "variables": { + "poolId": pool_id, + "chain": api_chain, + "range": "ALL_TIME", + }, + } + + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + + with urllib.request.urlopen(req, timeout=30) as resp: + body = json.loads(resp.read().decode("utf-8")) + + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + if not snapshots: + break + + for snap in snapshots: + ts = int(snap["timestamp"]) + if start_ts <= ts <= end_ts: + all_snapshots.append({ + "timestamp": ts, + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + }) + + # The API returns ALL_TIME, so no need to paginate further + break + + if not all_snapshots: + raise ValueError( + f"No Balancer snapshots for {pool_id} on {chain} " + f"between {start_ts} and {end_ts}" + ) + + df = pd.DataFrame(all_snapshots) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + # Deduplicate by date (keep last snapshot per day) + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df.set_index("date") + + +# --------------------------------------------------------------------------- +# DB-based daily pool state +# --------------------------------------------------------------------------- + +def load_daily_pool_state(pool, db_path, data_root): + """Load daily pool state from pools_history.db, compute effective TVL. + + Parameters + ---------- + pool : PoolConfig + Pool configuration (from pool_registry). + db_path : str + Path to pools_history.db. + data_root : str + Directory containing {TICKER}_USD.parquet files. + + Returns + ------- + pd.DataFrame + Indexed by date, columns: effective_tvl_usd, real_tvl_usd. + """ + conn = sqlite3.connect(db_path) + cur = conn.cursor() + cur.execute( + f"""SELECT timestamp, balance_0, balance_1, virtual_0, virtual_1 + FROM {pool.db_label} + ORDER BY timestamp""" + ) + rows = cur.fetchall() + conn.close() + + if not rows: + raise ValueError(f"No DB data for {pool.db_label}") + + df = pd.DataFrame(rows, columns=["timestamp", "bal_0", "bal_1", "virt_0", "virt_1"]) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + + # Keep last snapshot per day + daily = df.sort_values("timestamp").drop_duplicates("date", keep="last").set_index("date") + + # Load USD prices for each token + if pool.reverse: + tickers_in_db_order = [pool.tokens[1], pool.tokens[0]] + else: + tickers_in_db_order = [pool.tokens[0], pool.tokens[1]] + + price_dfs = {} + for ticker in tickers_in_db_order: + if ticker == "USDC": + price_dfs[ticker] = None # constant $1 + else: + path = os.path.join(data_root, f"{ticker}_USD.parquet") + pdf = pd.read_parquet(path) + pdf["date"] = pd.to_datetime(pdf["unix"], unit="ms").dt.date + # Daily close: last price per day + price_dfs[ticker] = ( + pdf.sort_values("unix") + .drop_duplicates("date", keep="last") + .set_index("date")["close"] + ) + + # Compute USD prices at each daily snapshot + records = [] + for date, row in daily.iterrows(): + b0, b1, v0, v1 = row["bal_0"], row["bal_1"], row["virt_0"], row["virt_1"] + + p0 = 1.0 if tickers_in_db_order[0] == "USDC" else price_dfs[tickers_in_db_order[0]].get(date, np.nan) + p1 = 1.0 if tickers_in_db_order[1] == "USDC" else price_dfs[tickers_in_db_order[1]].get(date, np.nan) + + if np.isnan(p0) or np.isnan(p1): + continue + + real_tvl = b0 * p0 + b1 * p1 + effective_tvl = (b0 + v0) * p0 + (b1 + v1) * p1 + + records.append({ + "date": date, + "real_tvl_usd": real_tvl, + "effective_tvl_usd": effective_tvl, + }) + + result = pd.DataFrame(records).set_index("date") + return result + + +# --------------------------------------------------------------------------- +# Daily volatility from price parquets +# --------------------------------------------------------------------------- + +def compute_daily_volatility(tokens, data_root, start_ts, end_ts): + """Compute daily annualised volatility of the price ratio. + + Uses 5-minute subsampled log returns within each day, then + annualises with sqrt(365). + + Parameters + ---------- + tokens : list + Token tickers in quantammsim sorted order (e.g. ['BTC', 'ETH']). + data_root : str + Directory containing {TICKER}_USD.parquet files. + start_ts : int + Start unix timestamp (seconds). + end_ts : int + End unix timestamp (seconds). + + Returns + ------- + pd.Series + Indexed by date, values are annualised daily volatility. + """ + # Load minute-level prices for both tokens + prices = {} + for ticker in tokens: + if ticker == "USDC": + prices[ticker] = None + else: + path = os.path.join(data_root, f"{ticker}_USD.parquet") + df = pd.read_parquet(path) + df = df[(df["unix"] >= start_ts * 1000) & (df["unix"] <= end_ts * 1000)] + df["datetime"] = pd.to_datetime(df["unix"], unit="ms") + df = df.set_index("datetime")["close"] + prices[ticker] = df + + # Compute price ratio (token[0] / token[1]) + t0, t1 = tokens[0], tokens[1] + if prices[t0] is not None and prices[t1] is not None: + # Align on common timestamps + combined = pd.DataFrame({"p0": prices[t0], "p1": prices[t1]}).dropna() + ratio = combined["p0"] / combined["p1"] + elif prices[t0] is not None: + ratio = prices[t0] # t1 is USDC ($1) + elif prices[t1] is not None: + ratio = 1.0 / prices[t1] # t0 is USDC + else: + raise ValueError("Both tokens are USDC — cannot compute ratio") + + # Subsample to 5-min intervals + ratio_5m = ratio.resample("5min").last().dropna() + log_returns = np.log(ratio_5m / ratio_5m.shift(1)).dropna() + + # Group by date, compute daily vol + log_returns_df = log_returns.to_frame("lr") + log_returns_df["date"] = log_returns_df.index.date + + daily_vol = log_returns_df.groupby("date")["lr"].std() + # Annualise: each day has ~288 5-min periods, scale by sqrt(288 * 365) + daily_vol_ann = daily_vol * np.sqrt(288 * 365) + + return daily_vol_ann + + +# --------------------------------------------------------------------------- +# Calibration DataFrame assembly +# --------------------------------------------------------------------------- + +def build_calibration_df(pool, data_root=None): + """Build daily calibration DataFrame from Balancer API + price parquets. + + All pool state (volume, effective TVL) comes from the Balancer V3 API. + The API's ``totalLiquidity`` is the effective TVL: for a reClAMM pool + on Balancer V3, the router sees real + virtual reserves, and + ``totalLiquidity`` reflects that full depth. Only the volatility + computation requires price parquets. + + Parameters + ---------- + pool : PoolConfig + Pool configuration (must have pool_address field). + data_root : str, optional + Directory containing {TICKER}_USD.parquet price files. + + Returns + ------- + pd.DataFrame + Columns: volume_usd, effective_tvl_usd, volatility. Indexed by date. + """ + from experiments.pool_registry import ( + get_data_end_date, + _date_to_unix, + ) + + if data_root is None: + data_root = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "quantammsim", "data", + ) + + start_ts = _date_to_unix(pool.plausible_start) + end_str = get_data_end_date(pool.tokens, data_root) + end_ts = _date_to_unix(end_str) + + print(f" Fetching Balancer snapshots for {pool.label} " + f"({pool.chain}, {pool.pool_address})...") + api_df = fetch_balancer_snapshots( + pool.chain, pool.pool_address, start_ts, end_ts, + ) + print(f" Got {len(api_df)} daily snapshots from API") + print(f" TVL range: ${api_df['total_liquidity_usd'].min():,.0f} — " + f"${api_df['total_liquidity_usd'].max():,.0f}") + + print(f" Computing daily volatility from price parquets...") + vol_series = compute_daily_volatility(pool.tokens, data_root, start_ts, end_ts) + print(f" Got {len(vol_series)} daily volatility values") + + # Assemble: volume + TVL from API, volatility from parquets + combined = api_df[["volume_usd", "total_liquidity_usd"]].copy() + combined = combined.rename(columns={"total_liquidity_usd": "effective_tvl_usd"}) + combined["volatility"] = vol_series + combined = combined.dropna() + + print(f" Combined: {len(combined)} days after join") + return combined + + +# --------------------------------------------------------------------------- +# OLS calibration +# --------------------------------------------------------------------------- + +def run_ols_calibration(daily_df, base_fee, model="sqrt"): + """OLS regression for Tsoukalas model params. + + Parameters + ---------- + daily_df : pd.DataFrame + Must contain columns: volume_usd, volatility, effective_tvl_usd. + base_fee : float + Static swap fee (e.g. 0.003). + model : str + 'sqrt' or 'log' — TVL regressor transformation. + + Returns + ------- + noise_params : dict + Coefficients for run_fingerprint["reclamm_noise_params"]. + diagnostics : dict + Standard errors, R², residual summary. + """ + if model == "loglinear": + # Multiplicative model: log(V) = b_0 + b_sigma·σ + b_c·log(TVL) + # Implies: V = exp(b_0) · TVL^b_c · exp(b_sigma·σ) + mask = daily_df["volume_usd"].values > 0 + n_dropped = int((~mask).sum()) + df_fit = daily_df[mask] + + y_log = np.log(df_fit["volume_usd"].values) + X = np.column_stack([ + np.ones(len(df_fit)), + df_fit["volatility"].values, + np.log(df_fit["effective_tvl_usd"].values), + ]) + + beta, _, _, _ = np.linalg.lstsq(X, y_log, rcond=None) + b_0, b_sigma, b_c = beta + + residuals = y_log - X @ beta + n, k = X.shape + bread = np.linalg.inv(X.T @ X) + hc1_scale = n / max(n - k, 1) + meat = X.T @ np.diag(residuals**2 * hc1_scale) @ X + robust_cov = bread @ meat @ bread + se = np.sqrt(np.diag(robust_cov)) + + ss_res = np.sum(residuals**2) + ss_tot = np.sum((y_log - y_log.mean())**2) + r_squared = 1.0 - ss_res / max(ss_tot, 1e-30) + + # Pseudo-R² in levels (median predictor) + y_pred_level = np.exp(X @ beta) + y_actual_level = df_fit["volume_usd"].values + res_level = y_actual_level - y_pred_level + r_sq_level = 1.0 - np.sum(res_level**2) / max( + np.sum((y_actual_level - y_actual_level.mean())**2), 1e-30) + + noise_params = { + "b_0": float(b_0), "b_sigma": float(b_sigma), + "b_c": float(b_c), "base_fee": float(base_fee), + } + diagnostics = { + "se": {"b_0": float(se[0]), "b_sigma": float(se[1]), + "b_c": float(se[2])}, + "r_squared": float(r_squared), + "r_squared_level": float(r_sq_level), + "n_obs": int(n), + "n_dropped_zero": n_dropped, + "residual_mean": float(np.mean(residuals)), + "residual_std": float(np.std(residuals)), + "smearing_factor": float(np.exp(np.var(residuals, ddof=1) / 2)), + "model": "loglinear", + } + return noise_params, diagnostics + + # --- Linear models (sqrt / log) --- + y = daily_df["volume_usd"].values / 1e6 + + if model == "sqrt": + tvl_eff = np.sqrt(daily_df["effective_tvl_usd"].values / 1e6) + elif model == "log": + tvl_eff = np.log(np.maximum(daily_df["effective_tvl_usd"].values / 1e6, 1e-30)) + else: + raise ValueError(f"Unknown model: {model!r}. Use 'sqrt', 'log', or 'loglinear'.") + + X = np.column_stack([ + np.ones(len(daily_df)), # a_0 + daily_df["volatility"].values, # a_sigma + tvl_eff, # a_c + ]) + + beta, residuals_ss, rank, sv = np.linalg.lstsq(X, y, rcond=None) + a_0, a_sigma, a_c = beta + + # Heteroskedasticity-robust standard errors (HC1) + residuals = y - X @ beta + n, k = X.shape + bread = np.linalg.inv(X.T @ X) + hc1_scale = n / max(n - k, 1) + meat = X.T @ np.diag(residuals**2 * hc1_scale) @ X + robust_cov = bread @ meat @ bread + se = np.sqrt(np.diag(robust_cov)) + + # R-squared + ss_res = np.sum(residuals**2) + ss_tot = np.sum((y - np.mean(y))**2) + r_squared = 1.0 - ss_res / max(ss_tot, 1e-30) + + noise_params = { + "a_0_base": float(a_0), + "a_f": 0.0, # not identified with static fees + "a_sigma": float(a_sigma), + "a_c": float(a_c), + "base_fee": float(base_fee), + } + + diagnostics = { + "se": dict(zip( + ["a_0", "a_sigma", "a_c"], + se.tolist(), + )), + "r_squared": float(r_squared), + "n_obs": int(n), + "residual_mean": float(np.mean(residuals)), + "residual_std": float(np.std(residuals)), + "model": model, + } + + return noise_params, diagnostics + + +# --------------------------------------------------------------------------- +# Plotting +# --------------------------------------------------------------------------- + +def plot_calibration_diagnostics(daily_df, noise_params, diagnostics, + pool_label="", model="sqrt", + output_dir="results"): + """Generate diagnostic plots for the noise volume calibration. + + Produces a 2×2 figure: + Top-left: Time series — real vs predicted daily volume + effective TVL + Top-right: Scatter — predicted vs actual with 45° line + Bot-left: Residuals vs time + residuals vs fitted + Bot-right: Component decomposition (stacked contributions) + + Parameters + ---------- + daily_df : pd.DataFrame + Calibration DataFrame (indexed by date). + noise_params : dict + Fitted coefficients from run_ols_calibration. + diagnostics : dict + Diagnostics dict from run_ols_calibration. + pool_label : str + Pool name for titles. + model : str + 'sqrt' or 'log'. + output_dir : str + Directory for output PNGs. + """ + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + from matplotlib.dates import DateFormatter + import matplotlib.dates as mdates + + os.makedirs(output_dir, exist_ok=True) + + # Extract data + dates = pd.to_datetime(daily_df.index) + y_actual = daily_df["volume_usd"].values / 1e6 # in $M + eff_tvl = daily_df["effective_tvl_usd"].values + vol = daily_df["volatility"].values + + r2 = diagnostics["r_squared"] + n = diagnostics["n_obs"] + + if model == "loglinear": + b_0 = noise_params["b_0"] + b_sigma = noise_params["b_sigma"] + b_c = noise_params["b_c"] + log_tvl = np.log(np.maximum(eff_tvl, 1.0)) + y_pred_log = b_0 + b_sigma * vol + b_c * log_tvl + y_pred = np.exp(y_pred_log) / 1e6 # median prediction in $M + + # Log-space residuals + mask_pos = daily_df["volume_usd"].values > 0 + residuals = np.full(len(dates), np.nan) + residuals[mask_pos] = ( + np.log(daily_df["volume_usd"].values[mask_pos]) + - y_pred_log[mask_pos] + ) + resid_unit = "log scale" + r2_level = diagnostics.get("r_squared_level") + else: + a_0 = noise_params["a_0_base"] + a_sigma = noise_params["a_sigma"] + a_c = noise_params["a_c"] + + if model == "sqrt": + tvl_term = a_c * np.sqrt(eff_tvl / 1e6) + else: + tvl_term = a_c * np.log(np.maximum(eff_tvl / 1e6, 1e-30)) + + y_pred = a_0 + a_sigma * vol + tvl_term + residuals = y_actual - y_pred + resid_unit = "$M" + r2_level = None + + # --- Figure 1: Main diagnostics (2×2) --- + fig, axes = plt.subplots(2, 2, figsize=(16, 12)) + + # (0,0) Time series: real vs predicted + TVL on secondary axis + ax = axes[0, 0] + ax.plot(dates, y_actual, color="steelblue", alpha=0.7, linewidth=1, + label="Actual volume") + ax.plot(dates, y_pred, color="crimson", linewidth=1.5, + label="Predicted volume") + ax.set_ylabel("Daily volume ($M)", color="steelblue") + ax.tick_params(axis="y", labelcolor="steelblue") + ax.legend(loc="upper left", fontsize=8) + r2_str = (f"R²(log)={r2:.3f}, R²(level)={r2_level:.3f}" + if r2_level is not None else f"R²={r2:.3f}") + ax.set_title(f"Daily volume: actual vs predicted ({r2_str}, n={n})") + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + + ax2 = ax.twinx() + ax2.fill_between(dates, eff_tvl / 1e6, alpha=0.15, color="green", + label="Effective TVL ($M)") + ax2.set_ylabel("Effective TVL ($M)", color="green") + ax2.tick_params(axis="y", labelcolor="green") + ax2.legend(loc="upper right", fontsize=8) + + # (0,1) Scatter: predicted vs actual + ax = axes[0, 1] + ax.scatter(y_pred, y_actual, alpha=0.5, s=15, color="steelblue", + edgecolors="none") + lims = [min(y_pred.min(), y_actual.min()), max(y_pred.max(), y_actual.max())] + margin = (lims[1] - lims[0]) * 0.05 + lims = [lims[0] - margin, lims[1] + margin] + ax.plot(lims, lims, "k--", linewidth=0.8, alpha=0.5, label="45° line") + ax.set_xlabel("Predicted ($M)") + ax.set_ylabel("Actual ($M)") + ax.set_title("Predicted vs actual") + ax.legend(fontsize=8) + ax.set_aspect("equal", adjustable="box") + + # (1,0) Residuals: vs time (top) and vs fitted (bottom) + ax = axes[1, 0] + ax.scatter(dates, residuals, alpha=0.5, s=12, color="steelblue", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + # 7-day rolling mean of residuals + res_series = pd.Series(residuals, index=dates) + rolling_mean = res_series.rolling(7, min_periods=1).mean() + ax.plot(dates, rolling_mean, color="crimson", linewidth=1.5, + label="7-day rolling mean") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs time") + ax.legend(fontsize=8) + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + + # (1,1) Component decomposition + ax = axes[1, 1] + if model == "loglinear": + # Log-space additive decomposition as line plots + log_tvl_plot = np.log(np.maximum(eff_tvl, 1.0)) + comp_base = np.full(len(dates), b_0) + comp_tvl = b_c * log_tvl_plot + comp_vol = b_sigma * vol + total_log = comp_base + comp_tvl + comp_vol + vol_usd = daily_df["volume_usd"].values.copy() + vol_usd[vol_usd <= 0] = np.nan + actual_log = np.log(vol_usd) + ax.plot(dates, comp_base, color="grey", linestyle="--", linewidth=1, + label=f"b_0 = {b_0:.2f}") + ax.plot(dates, comp_base + comp_tvl, color="green", linewidth=1.5, + label=f"b_0 + b_c·log(TVL) (b_c={b_c:.4f})") + ax.plot(dates, total_log, color="crimson", linewidth=1.5, + label="Full prediction") + ax.scatter(dates, actual_log, color="steelblue", s=10, alpha=0.5, + label="Actual log(V)", zorder=5) + ax.set_ylabel("log(Volume, USD)") + ax.set_title("Component decomposition (log space)") + + fig.suptitle( + f"{pool_label} — noise calibration ({model})\n" + f"log(V) = {b_0:.2f} + {b_sigma:.4f}·σ + {b_c:.4f}·log(TVL)", + fontsize=11, + ) + else: + intercept_contrib = np.full(len(dates), a_0) + vol_contrib = a_sigma * vol + tvl_contrib = tvl_term + + ax.fill_between(dates, 0, intercept_contrib, alpha=0.3, color="grey", + label=f"a_0 = {a_0:.4f}") + ax.fill_between(dates, intercept_contrib, intercept_contrib + vol_contrib, + alpha=0.3, color="orange", + label=f"a_σ·σ (a_σ={a_sigma:.4f})") + ax.fill_between(dates, intercept_contrib + vol_contrib, + intercept_contrib + vol_contrib + tvl_contrib, + alpha=0.3, color="green", + label=f"a_c·{model}(TVL) (a_c={a_c:.4f})") + ax.plot(dates, y_actual, color="steelblue", linewidth=1, alpha=0.7, + label="Actual") + ax.set_ylabel("Volume ($M)") + ax.set_title("Component decomposition") + + fig.suptitle( + f"{pool_label} — Tsoukalas noise calibration ({model})\n" + f"V/1e6 = {a_0:.4f} + {a_sigma:.4f}·σ + {a_c:.4f}·{model}(TVL_eff/1e6)", + fontsize=11, + ) + ax.legend(fontsize=7, loc="upper left") + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + plt.tight_layout() + + fname = f"noise_calibration_{pool_label}_{model}.png" + path = os.path.join(output_dir, fname) + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- Figure 2: Residuals vs each regressor --- + fig2, axes2 = plt.subplots(1, 3, figsize=(16, 5)) + + # Residuals vs volatility + ax = axes2[0] + ax.scatter(vol, residuals, alpha=0.5, s=12, color="orange", edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Volatility (annualised)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs volatility") + + # Residuals vs effective TVL + ax = axes2[1] + ax.scatter(eff_tvl / 1e6, residuals, alpha=0.5, s=12, color="green", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Effective TVL ($M)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs effective TVL") + + # Residuals vs fitted + ax = axes2[2] + ax.scatter(y_pred, residuals, alpha=0.5, s=12, color="steelblue", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Fitted ($M)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs fitted") + + fig2.suptitle(f"{pool_label} — Residual diagnostics ({model})", fontsize=11) + plt.tight_layout() + + fname2 = f"noise_residuals_{pool_label}_{model}.png" + path2 = os.path.join(output_dir, fname2) + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + return path, path2 + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Calibrate Tsoukalas noise volume model for reClAMM" + ) + parser.add_argument( + "--csv", default=None, + help="Path to CSV with columns: volume_usd, volatility, effective_tvl_usd", + ) + parser.add_argument( + "--pool", default=None, + help="Pool label from pool_registry (e.g. cbBTC_WETH) for end-to-end calibration", + ) + parser.add_argument("--base-fee", type=float, default=None, + help="Override base fee (default: use pool's swap_fee)") + parser.add_argument("--model", choices=["sqrt", "log", "loglinear"], + default="sqrt") + parser.add_argument( + "--output", default=None, + help="Output JSON file path. Defaults to stdout.", + ) + parser.add_argument( + "--plot", action="store_true", + help="Generate diagnostic plots (saved to --output-dir)", + ) + parser.add_argument( + "--output-dir", default="results", + help="Directory for diagnostic plots (default: results)", + ) + args = parser.parse_args() + + if args.csv is None and args.pool is None: + parser.error("One of --csv or --pool is required") + + if args.pool is not None: + # End-to-end mode: fetch data, assemble, calibrate + from experiments.pool_registry import POOL_REGISTRY + + if args.pool not in POOL_REGISTRY: + print(f"Unknown pool: {args.pool}", file=sys.stderr) + print(f"Available: {list(POOL_REGISTRY.keys())}", file=sys.stderr) + sys.exit(1) + + pool = POOL_REGISTRY[args.pool] + base_fee = args.base_fee if args.base_fee is not None else pool.swap_fee + + print(f"Calibrating noise model for {pool.label} ({pool.chain})") + print(f" Swap fee: {base_fee}") + print(f" Model: {args.model}") + + df = build_calibration_df(pool) + else: + # CSV mode + df = pd.read_csv(args.csv) + required_cols = {"volume_usd", "volatility", "effective_tvl_usd"} + missing = required_cols - set(df.columns) + if missing: + print(f"Error: missing columns: {missing}", file=sys.stderr) + sys.exit(1) + base_fee = args.base_fee if args.base_fee is not None else 0.003 + + noise_params, diagnostics = run_ols_calibration(df, base_fee, args.model) + + # Print diagnostics + print(f"\n OLS Results ({args.model} model):") + print(f" R² = {diagnostics['r_squared']:.4f}") + if "r_squared_level" in diagnostics: + print(f" R²(level) = {diagnostics['r_squared_level']:.4f}") + if "n_dropped_zero" in diagnostics and diagnostics["n_dropped_zero"] > 0: + print(f" Dropped {diagnostics['n_dropped_zero']} zero-volume days") + if "smearing_factor" in diagnostics: + print(f" Smearing factor = {diagnostics['smearing_factor']:.4f} " + f"(E[V]/median[V])") + print(f" n = {diagnostics['n_obs']}") + print(f" Coefficients:") + if args.model == "loglinear": + coef_keys = ["b_0", "b_sigma", "b_c"] + else: + coef_keys = ["a_0", "a_sigma", "a_c"] + for key in coef_keys: + param_key = "a_0_base" if key == "a_0" else key + val = noise_params[param_key] + se = diagnostics["se"][key] + t_stat = val / se if se > 0 else float("inf") + print(f" {key:>8} = {val:>10.4f} (SE={se:.4f}, t={t_stat:.2f})") + print(f" Residual: mean={diagnostics['residual_mean']:.6f}, " + f"std={diagnostics['residual_std']:.4f}") + + # Plot diagnostics + if args.plot: + label = args.pool if args.pool else "custom" + plot_calibration_diagnostics( + df, noise_params, diagnostics, + pool_label=label, model=args.model, + output_dir=args.output_dir, + ) + + result = { + "noise_params": noise_params, + "diagnostics": diagnostics, + } + + output_str = json.dumps(result, indent=2) + if args.output: + with open(args.output, "w") as f: + f.write(output_str + "\n") + print(f"\nWrote calibration to {args.output}", file=sys.stderr) + else: + print(f"\n{output_str}") + + +if __name__ == "__main__": + main() diff --git a/scripts/compare_reclamm_thermostats.py b/scripts/compare_reclamm_thermostats.py new file mode 100644 index 00000000..8a2c374c --- /dev/null +++ b/scripts/compare_reclamm_thermostats.py @@ -0,0 +1,379 @@ +"""Compare geometric vs constant-arc-length thermostats on historic data. + +Runs AAVE/ETH reClAMM pool simulations with both interpolation methods. +Plots: pool value, cumulative LVR, price path, empirical weights, +value difference, LVR ratio, and per-step LVR distribution (∝ Δs²). + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/compare_reclamm_thermostats.py +""" + +import jax.numpy as jnp +import numpy as np +import matplotlib.pyplot as plt +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(daily_price_shift_exponent): + """Convert shift rate to daily price shift base (matches Solidity).""" + return 1.0 - daily_price_shift_exponent / 124649.0 + + +# Pool configurations to compare +CONFIGS = [ + { + "name": "AAVE/ETH on-chain (25bps, narrow range)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 1.5, + "centeredness_margin": 0.5, + "daily_price_shift_exponent": 0.1, + }, + { + "name": "AAVE/ETH wide range (25bps)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 4.0, + "centeredness_margin": 0.2, + "daily_price_shift_exponent": 1.0, + }, + { + "name": "AAVE/ETH zero fees (narrow)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0, + "price_ratio": 1.5, + "centeredness_margin": 0.5, + "daily_price_shift_exponent": 0.1, + }, +] + + +def make_fingerprint(cfg, interpolation_method, centeredness_scaling=False): + """Build run fingerprint for a given config and interpolation method.""" + return { + "tokens": cfg["tokens"], + "rule": "reclamm", + "startDateString": cfg["start"], + "endDateString": cfg["end"], + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": cfg["fees"], + "gas_cost": 0.0, + "arb_fees": 0.0, + "reclamm_interpolation_method": interpolation_method, + "reclamm_arc_length_speed": None, # auto-calibrate + "reclamm_centeredness_scaling": centeredness_scaling, + } + + +def make_params(cfg): + """Build pool params from config.""" + return { + "price_ratio": jnp.array(cfg["price_ratio"]), + "centeredness_margin": jnp.array(cfg["centeredness_margin"]), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(cfg["daily_price_shift_exponent"]) + ), + } + + +def run_comparison(cfg): + """Run all thermostat variants, return results dict.""" + params = make_params(cfg) + + results = {} + for method in ["geometric", "constant_arc_length"]: + fp = make_fingerprint(cfg, method) + results[method] = do_run_on_historic_data( + run_fingerprint=fp, params=params + ) + + # Geometric + centeredness-proportional scaling (scales decay duration) + fp_geo_scaled = make_fingerprint(cfg, "geometric", centeredness_scaling=True) + results["geometric_scaled"] = do_run_on_historic_data( + run_fingerprint=fp_geo_scaled, params=params + ) + + # Arc-length + centeredness-proportional scaling (scales speed) + fp_cal_scaled = make_fingerprint(cfg, "constant_arc_length", centeredness_scaling=True) + results["cal_scaled"] = do_run_on_historic_data( + run_fingerprint=fp_cal_scaled, params=params + ) + + return results + + +def print_comparison(cfg, results): + """Print text summary table.""" + methods = [ + ("Geometric", results["geometric"]), + ("Geo+Scaled", results["geometric_scaled"]), + ("Const Arc", results["constant_arc_length"]), + ("Arc+Scaled", results["cal_scaled"]), + ] + + hodl_value = float((methods[0][1]["reserves"][0] * methods[0][1]["prices"][-1]).sum()) + + print("=" * 105) + print(f" {cfg['name']}") + print(f" price_ratio={cfg['price_ratio']}, " + f"margin={cfg['centeredness_margin']}, " + f"shift_exp={cfg['daily_price_shift_exponent']}, " + f"fees={cfg['fees']}") + print("-" * 105) + header = " {:20s}".format("") + for name, _ in methods: + header += f" {name:>14s}" + print(header) + + row = " {:20s}".format("Final value") + for _, r in methods: + row += f" ${float(r['final_value']):>13,.0f}" + print(row) + + print(f" {'HODL value':20s} ${hodl_value:>13,.0f}") + + row = " {:20s}".format("LVR (HODL - final)") + for _, r in methods: + lvr = hodl_value - float(r["final_value"]) + row += f" ${lvr:>13,.0f}" + print(row) + + row = " {:20s}".format("Return") + for _, r in methods: + ret = (float(r["final_value"]) / float(r["value"][0]) - 1) * 100 + row += f" {ret:>13.2f}%" + print(row) + + row = " {:20s}".format("vs HODL") + for _, r in methods: + vs = (float(r["final_value"]) / hodl_value - 1) * 100 + row += f" {vs:>13.2f}%" + print(row) + print("=" * 105) + + +def plot_comparison(cfg, results, fig_idx): + """Plot 4-panel comparison for one config.""" + # Method name → (result dict, color, linestyle) + variants = { + "Geometric": (results["geometric"], "C0", "-"), + "Geo+Scaled": (results["geometric_scaled"], "C1", "-"), + "Const arc-len": (results["constant_arc_length"], "C2", "--"), + "Arc+Scaled": (results["cal_scaled"], "C3", "--"), + } + + geo = results["geometric"] + geo_prices = np.array(geo["prices"]) + geo_reserves = np.array(geo["reserves"]) + n_steps = len(np.array(geo["value"])) + t_days = np.arange(n_steps) / (60 * 24) + + hodl_traj = (geo_reserves[0] * geo_prices[:n_steps]).sum(axis=-1) + price_ratio_traj = geo_prices[:n_steps, 0] / geo_prices[:n_steps, 1] + + fig, axes = plt.subplots(2, 2, figsize=(14, 10)) + fig.suptitle(cfg["name"], fontsize=13, fontweight="bold") + + # (0,0) Pool value over time + ax = axes[0, 0] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + ax.plot(t_days, vals / 1e6, color=color, ls=ls, label=name, alpha=0.9) + ax.plot(t_days, np.array(hodl_traj) / 1e6, color="gray", ls=":", + alpha=0.5, label="HODL") + ax.set_xlabel("Days") + ax.set_ylabel("Pool value ($M)") + ax.set_title("Pool value") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (0,1) Cumulative LVR + ax = axes[0, 1] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + lvr = np.array(hodl_traj) - vals + ax.plot(t_days, lvr / 1e3, color=color, ls=ls, label=name, alpha=0.9) + ax.set_xlabel("Days") + ax.set_ylabel("Cumulative LVR ($K)") + ax.set_title("Cumulative LVR (HODL - pool value)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (1,0) Price ratio + ax = axes[1, 0] + ax.plot(t_days, price_ratio_traj, color="C4", alpha=0.7) + ax.set_xlabel("Days") + ax.set_ylabel(f"{cfg['tokens'][0]}/{cfg['tokens'][1]} price ratio") + ax.set_title("Price path") + ax.grid(True, alpha=0.3) + + # (1,1) Empirical weights + ax = axes[1, 1] + for name, (r, color, ls) in variants.items(): + w = np.array(r["weights"]) + n_w = min(len(w), n_steps) + t_w = np.arange(n_w) / (60 * 24) + ax.plot(t_w, w[:n_w, 0], color=color, ls=ls, label=name, alpha=0.9) + ax.set_xlabel("Days") + ax.set_ylabel(f"Weight ({cfg['tokens'][0]})") + ax.set_title("Empirical weight (token 0)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + fname = f"reclamm_thermostat_comparison_{fig_idx}.png" + plt.savefig(fname, dpi=150) + print(f"Saved {fname}") + plt.close(fig) + + # Second figure: diagnostics + geo_values = np.array(geo["value"]) + geo_lvr = np.array(hodl_traj) - geo_values + + fig2, axes2 = plt.subplots(1, 3, figsize=(18, 5)) + fig2.suptitle(f"{cfg['name']} — diagnostics", fontsize=13, fontweight="bold") + + # (left) Value difference vs geometric + ax = axes2[0] + for name, (r, color, ls) in variants.items(): + if name == "Geometric": + continue + vals = np.array(r["value"]) + ax.plot(t_days, (vals - geo_values) / 1e3, color=color, ls=ls, + label=name, alpha=0.9) + ax.axhline(0, color="gray", ls="--", alpha=0.5) + ax.set_xlabel("Days") + ax.set_ylabel("Value difference ($K)") + ax.set_title("Minus Geometric") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (middle) LVR ratio over time + ax = axes2[1] + mask = np.abs(geo_lvr) > 100 + if mask.any(): + for name, (r, color, ls) in variants.items(): + if name == "Geometric": + continue + vals = np.array(r["value"]) + method_lvr = np.array(hodl_traj) - vals + ratio = np.full_like(geo_lvr, np.nan) + ratio[mask] = method_lvr[mask] / geo_lvr[mask] + ax.plot(t_days, ratio, color=color, ls=ls, alpha=0.7, label=name) + ax.axhline(1.0, color="gray", ls="--", alpha=0.5) + ax.set_ylabel("LVR ratio (method / geometric)") + ax.legend(fontsize=8) + else: + ax.text(0.5, 0.5, "LVR too small to compare", + transform=ax.transAxes, ha="center", va="center") + ax.set_xlabel("Days") + ax.set_title("Relative LVR") + ax.grid(True, alpha=0.3) + + # (right) Per-step LVR histogram + ax = axes2[2] + all_pos = [] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + method_lvr = np.array(hodl_traj) - vals + step_lvr = np.diff(method_lvr) + pos = step_lvr[step_lvr > 0] + all_pos.append((name, pos, color)) + has_data = [len(p) > 10 for _, p, _ in all_pos] + if any(has_data): + max_val = max(np.percentile(p, 99) for _, p, _ in all_pos if len(p) > 10) + bins = np.linspace(0, max_val, 50) + for name, pos, color in all_pos: + if len(pos) > 10: + ax.hist(pos, bins=bins, color=color, alpha=0.3, label=name, + density=True) + ax.set_xlabel("Per-step LVR ($)") + ax.set_ylabel("Density") + ax.legend(fontsize=8) + else: + ax.text(0.5, 0.5, "Too few thermostat steps", + transform=ax.transAxes, ha="center", va="center") + ax.set_title("Per-step LVR distribution") + ax.grid(True, alpha=0.3) + + plt.tight_layout() + fname2 = f"reclamm_thermostat_diff_{fig_idx}.png" + plt.savefig(fname2, dpi=150) + print(f"Saved {fname2}") + plt.close(fig2) + + +if __name__ == "__main__": + all_results = [] + for i, cfg in enumerate(CONFIGS): + print(f"\n>>> Running {cfg['name']}...") + try: + results = run_comparison(cfg) + print_comparison(cfg, results) + plot_comparison(cfg, results, i) + all_results.append((cfg, results)) + except Exception as e: + print(f" FAILED: {e}") + import traceback + traceback.print_exc() + + # Summary overlay: all configs on one figure (pool value normalised) + if len(all_results) > 1: + fig, axes = plt.subplots(1, 2, figsize=(16, 5)) + fig.suptitle("Cross-config comparison (normalised)", fontsize=13, + fontweight="bold") + + method_keys = [ + ("geometric", "geo", "-"), + ("geometric_scaled", "geo+s", "-."), + ("constant_arc_length", "arc", "--"), + ("cal_scaled", "arc+s", ":"), + ] + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + + for j, (key, suffix, ls) in enumerate(method_keys): + v = np.array(results[key]["value"]) + color_idx = i * len(method_keys) + j + + # (left) Normalised pool value + axes[0].plot(t, v / v[0], ls=ls, alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}") + + # (right) Value difference vs geometric (skip geo itself) + if key != "geometric": + pct_diff = (v - geo_v) / geo_v * 100 + axes[1].plot(t, pct_diff, ls=ls, alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}") + + axes[0].set_xlabel("Days") + axes[0].set_ylabel("Normalised pool value") + axes[0].set_title("Pool value (V/V0)") + axes[0].legend(fontsize=6, ncol=2) + axes[0].grid(True, alpha=0.3) + + axes[1].set_xlabel("Days") + axes[1].set_ylabel("(Method - Geo) / Geo (%)") + axes[1].set_title("Relative value difference vs Geometric") + axes[1].axhline(0, color="gray", ls="--", alpha=0.5) + axes[1].legend(fontsize=6, ncol=2) + axes[1].grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_thermostat_summary.png", dpi=150) + print("\nSaved reclamm_thermostat_summary.png") + plt.close(fig) diff --git a/scripts/demo_run_chunks_from_chain_data.py b/scripts/demo_run_chunks_from_chain_data.py index 09c19d14..fbd9cf9c 100644 --- a/scripts/demo_run_chunks_from_chain_data.py +++ b/scripts/demo_run_chunks_from_chain_data.py @@ -28,7 +28,6 @@ import numpy as np import pandas as pd import matplotlib as mpl -from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames import matplotlib.pyplot as plt import jax.numpy as jnp @@ -475,12 +474,10 @@ def _df_meta_and_head(df, name, n=3): run_fingerprint=fingerprint, coarse_weights=cw_window, params=params, - dynamic_input_frames=DynamicInputFrames( - fees=scraped["fees_df"], - gas_cost=scraped["gas_cost_df"], - lp_supply=scraped["lp_supply_df"], - arb_fees=scraped["arb_fees_df"], - ), + fees_df=scraped["fees_df"], + gas_cost_df=scraped["gas_cost_df"], + lp_supply_df=scraped["lp_supply_df"], + arb_fees_df=scraped["arb_fees_df"], ) # ---------------- Correct, window-aligned plotting block (time-aware + plain y) ---------------- diff --git a/scripts/demo_run_from_chain_data.py b/scripts/demo_run_from_chain_data.py index b1f41398..78ecbc4d 100644 --- a/scripts/demo_run_from_chain_data.py +++ b/scripts/demo_run_from_chain_data.py @@ -1,5 +1,4 @@ import jax.numpy as jnp -from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.core_simulator.param_utils import ( memory_days_to_logit_lamb, ) @@ -998,12 +997,10 @@ def generate_daily_variations(start_date_str, end_date_str): run_fingerprint=config["fingerprint"], coarse_weights=config["coarse_weights"], params=config["params"], - dynamic_input_frames=DynamicInputFrames( - fees=config["fees_df"], - gas_cost=config["gas_cost_df"], - lp_supply=config["lp_supply_df"], - arb_fees=config["arb_fees_df"], - ), + fees_df=config["fees_df"], + gas_cost_df=config["gas_cost_df"], + lp_supply_df=config["lp_supply_df"], + arb_fees_df=config["arb_fees_df"], ) print("-" * 80) print(f"Pool Type: {config['fingerprint']['rule']}") @@ -1194,4 +1191,4 @@ def generate_daily_variations(start_date_str, end_date_str): # actual_reserves_np=local_reserves, # actual_unix_values=datetime_array, # ) - # raise Exception("Stop here") + # raise Exception("Stop here") \ No newline at end of file diff --git a/scripts/demo_run_reclamm.py b/scripts/demo_run_reclamm.py new file mode 100644 index 00000000..3ea21ec6 --- /dev/null +++ b/scripts/demo_run_reclamm.py @@ -0,0 +1,207 @@ +"""Demo runs for reClAMM pools vs Balancer 50/50 baseline. + +Runs reClAMM pool simulations with parameters pulled from on-chain pools +(AAVE/ETH) and hypothetical configurations, each paired with a Balancer +50/50 constant-weight pool at the same fee level for comparison. + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/demo_run_reclamm.py +""" + +import jax.numpy as jnp +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(daily_price_shift_exponent): + """Convert shift rate to daily price shift base (matches Solidity).""" + return 1.0 - daily_price_shift_exponent / 124649.0 + + +def balancer_fingerprint(tokens, start, end, fees): + """Build a Balancer 50/50 fingerprint matching the given reclamm config.""" + return { + "tokens": tokens, + "rule": "balancer", + "startDateString": start, + "endDateString": end, + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": fees, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + } + + +SCENARIOS = [ + { + "name": "AAVE/ETH on-chain (25bps)", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0025, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.1) + ), + }, + }, + }, + { + "name": "AAVE/ETH zero fees", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.1) + ), + }, + }, + }, + { + "name": "AAVE/ETH wide range (25bps)", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0025, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(1.0) + ), + }, + }, + }, + { + "name": "BTC/ETH (10bps)", + "reclamm": { + "fingerprint": { + "tokens": ["BTC", "ETH"], + "rule": "reclamm", + "startDateString": "2024-01-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.001, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(2.0), + "centeredness_margin": jnp.array(0.3), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.5) + ), + }, + }, + }, +] + + +def run_scenario(scenario): + """Run a reClAMM config and its Balancer 50/50 baseline, print comparison.""" + rc = scenario["reclamm"] + fp = rc["fingerprint"] + + # Run reClAMM + reclamm_result = do_run_on_historic_data( + run_fingerprint=fp, params=rc["params"] + ) + + # Run Balancer 50/50 with same tokens, dates, fees + bal_fp = balancer_fingerprint( + fp["tokens"], fp["startDateString"], fp["endDateString"], fp["fees"] + ) + bal_params = { + "initial_weights_logits": jnp.zeros(len(fp["tokens"])), + } + balancer_result = do_run_on_historic_data( + run_fingerprint=bal_fp, params=bal_params + ) + + # HODL value (from reClAMM initial reserves at final prices) + hodl_value = float( + (reclamm_result["reserves"][0] * reclamm_result["prices"][-1]).sum() + ) + + rc_final = float(reclamm_result["final_value"]) + bal_final = float(balancer_result["final_value"]) + rc_init = float(reclamm_result["value"][0]) + bal_init = float(balancer_result["value"][0]) + + print("=" * 80) + print(f" {scenario['name']}") + print(f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']}") + print("-" * 80) + print(f" {'':30s} {'reClAMM':>14s} {'Balancer 50/50':>14s}") + print(f" {'Initial value':30s} ${rc_init:>13,.0f} ${bal_init:>13,.0f}") + print(f" {'Final value':30s} ${rc_final:>13,.0f} ${bal_final:>13,.0f}") + print( + f" {'Return':30s} " + f"{(rc_final / rc_init - 1) * 100:>13.2f}% " + f"{(bal_final / bal_init - 1) * 100:>13.2f}%" + ) + print( + f" {'vs HODL':30s} " + f"{(rc_final / hodl_value - 1) * 100:>13.2f}% " + f"{(bal_final / hodl_value - 1) * 100:>13.2f}%" + ) + print( + f" {'reClAMM vs Balancer':30s} " + f"{(rc_final / bal_final - 1) * 100:>13.2f}%" + ) + print("=" * 80) + + +if __name__ == "__main__": + for scenario in SCENARIOS: + print(f"\n>>> {scenario['name']}...") + try: + run_scenario(scenario) + except Exception as e: + print(f" FAILED: {e}") + import traceback + + traceback.print_exc() diff --git a/scripts/plot_predicted_vs_real_volume.py b/scripts/plot_predicted_vs_real_volume.py new file mode 100644 index 00000000..83232ab1 --- /dev/null +++ b/scripts/plot_predicted_vs_real_volume.py @@ -0,0 +1,151 @@ +"""Plot predicted vs real daily volume for pool registry pools. + +Uses the fitted noise model (from calibrate_noise_unified.py) to compute +predicted daily log-volume for each pool in the registry, and overlays +the actual observed volume from the Balancer API panel data. +""" + +import json +import os +import sys + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "experiments")) +from pool_registry import POOL_REGISTRY, BALANCER_API_CHAIN + + +def main(): + fitted_path = "results/unified_full_90d.json" + panel_path = "local_data/noise_calibration/panel.parquet" + output_dir = "results/unified_full_90d" + os.makedirs(output_dir, exist_ok=True) + + with open(fitted_path) as f: + fitted = json.load(f) + + panel = pd.read_parquet(panel_path) + + # Deduplicate registry: multiple entries can share the same pool address + # (e.g. cbBTC_WETH and cbBTC_WETH_post_oct). Group by address. + unique_pools = {} + for label, pool in POOL_REGISTRY.items(): + addr = pool.pool_address.lower() + if addr not in unique_pools: + unique_pools[addr] = (label, pool) + + # Match to panel + matched = [] + for addr, (label, pool) in unique_pools.items(): + pid_matches = [ + pid for pid in fitted["pools"] + if addr in pid.lower() + ] + if pid_matches: + pid = pid_matches[0] + matched.append((label, pool, pid)) + else: + print(f" {label}: not in fitted model (skipping)") + + if not matched: + print("No registry pools found in the fitted model.") + return + + print(f"Plotting {len(matched)} pools: {[m[0] for m in matched]}") + + # Determine grid layout + n = len(matched) + ncols = min(n, 2) + nrows = (n + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(7 * ncols, 5 * nrows), + squeeze=False) + + for idx, (label, pool, pid) in enumerate(matched): + ax = axes[idx // ncols][idx % ncols] + pool_data = fitted["pools"][pid] + theta = np.array(pool_data["theta_median"]) + # theta = [intercept, b_tvl, b_sigma, b_weekend] + + # Get panel data for this pool + pool_panel = panel[panel["pool_id"] == pid].copy() + pool_panel = pool_panel.sort_values("date") + + if len(pool_panel) == 0: + ax.set_title(f"{label}: no panel data") + continue + + # Filter to last 90 days (matching training window) + max_date = panel["date"].max() + if hasattr(max_date, "date"): + max_date = max_date + from datetime import date, timedelta + if isinstance(max_date, date): + cutoff = max_date - timedelta(days=90) + else: + cutoff = pd.Timestamp(max_date) - pd.Timedelta(days=90) + pool_panel = pool_panel[ + pool_panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if len(pool_panel) < 5: + ax.set_title(f"{label}: <5 obs in 90d window") + continue + + # Build x_obs: [1, log_tvl_lag1, volatility, weekend] + x_obs = np.column_stack([ + np.ones(len(pool_panel)), + pool_panel["log_tvl_lag1"].values, + pool_panel["volatility"].values, + pool_panel["weekend"].values, + ]) + + predicted_log_vol = x_obs @ theta + actual_log_vol = pool_panel["log_volume"].values + + # Convert to USD volume for interpretability + predicted_vol = np.exp(predicted_log_vol) + actual_vol = np.exp(actual_log_vol) + + dates = pd.to_datetime(pool_panel["date"].values) + + # Plot + ax.plot(dates, actual_vol, "o-", color="steelblue", markersize=3, + linewidth=1, alpha=0.7, label="Actual") + ax.plot(dates, predicted_vol, "s--", color="orangered", markersize=3, + linewidth=1, alpha=0.7, label="Predicted") + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)") + ax.set_title(f"{label} ({pool_data['chain']})\n" + f"b_c={theta[1]:.2f} b_σ={theta[2]:.2f} " + f"b_wknd={theta[3]:.2f}") + ax.legend(fontsize=8) + ax.tick_params(axis="x", rotation=30) + + # Annotate R² for this pool + ss_res = np.sum((actual_log_vol - predicted_log_vol) ** 2) + ss_tot = np.sum((actual_log_vol - actual_log_vol.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + ax.text(0.02, 0.95, f"R²={r2:.3f}\nn={len(pool_panel)}", + transform=ax.transAxes, fontsize=8, va="top", + bbox=dict(boxstyle="round,pad=0.3", fc="white", alpha=0.8)) + + # Hide unused axes + for idx in range(n, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle("Noise model: predicted vs actual daily volume\n" + "(registry pools, 90-day training window)", fontsize=13) + fig.tight_layout() + out_path = os.path.join(output_dir, "registry_predicted_vs_real.png") + fig.savefig(out_path, dpi=150, bbox_inches="tight") + print(f"Saved: {out_path}") + plt.close(fig) + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_reclamm_optuna_result.py b/scripts/plot_reclamm_optuna_result.py new file mode 100644 index 00000000..35242dac --- /dev/null +++ b/scripts/plot_reclamm_optuna_result.py @@ -0,0 +1,451 @@ +#!/usr/bin/env python3 +"""Plot reClAMM pool performance from Optuna tuning results. + +Reads the SGD-compatible JSON output of tune_reclamm_params.py (or any Optuna +run), extracts the best trial's pool params, re-runs a forward pass over the +full train+test window, and produces a value-over-time plot with on-chain +baselines and cumulative fee revenue. + +Usage: + python scripts/plot_reclamm_optuna_result.py results/run_.json + python scripts/plot_reclamm_optuna_result.py results/run_.json --output my_plot.png + python scripts/plot_reclamm_optuna_result.py results/run_.json --top-k 3 +""" + +import argparse +import json +import sys + +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from datetime import datetime + +from quantammsim.runners.jax_runners import do_run_on_historic_data + +# ── On-chain baselines ──────────────────────────────────────────────────── +ONCHAIN_LAUNCH_PARAMS = { + "price_ratio": 1.5, "centeredness_margin": 0.5, "shift_exponent": 0.1, +} +ONCHAIN_CURRENT_PARAMS = { + "price_ratio": 4.0, "centeredness_margin": 0.1, "shift_exponent": 0.001, +} + +BG = "#162536" +TEXT_COLOR = "#E6CE97" +COLORS = [ + "#3498db", "#2ecc71", "#e74c3c", # top-k + "#f39c12", # on-chain launch + "#9b59b6", # on-chain current +] + + +def _plot_order(configs): + """Yield (name, meta, color_idx) with baselines first, optimized trials last.""" + optimized = [] + baselines = [] + for i, (name, meta) in enumerate(configs.items()): + if "On-Chain" in name: + baselines.append((name, meta, i)) + else: + optimized.append((name, meta, i)) + return baselines + optimized + + +def parse_args(): + p = argparse.ArgumentParser( + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + p.add_argument("results_json", help="Path to run_.json from Optuna") + p.add_argument("--top-k", type=int, default=1, + help="Plot top K trials by objective (default 1)") + p.add_argument("--output", default=None, + help="Output PNG path (default: auto-generated)") + p.add_argument("--no-onchain", action="store_true", + help="Skip on-chain baseline runs") + p.add_argument("--end-test-date", default=None, + help="Override endTestDateString (e.g. '2026-02-15 00:00:00')") + p.add_argument("--noise-trader-ratio", type=float, default=None, + help="Override noise_trader_ratio from results config") + return p.parse_args() + + +def load_results(path): + """Load the double-encoded JSONL from Optuna results.""" + with open(path) as f: + raw = f.read() + data = json.loads(raw) + if isinstance(data, str): + data = json.loads(data) + if not isinstance(data, list) or len(data) < 2: + print(f"ERROR: Expected [config, trial1, trial2, ...], got {type(data)}") + sys.exit(1) + config = data[0] + trials = data[1:] + return config, trials + + +def extract_pool_params(trial, config): + """Extract reClAMM pool params from a trial entry.""" + param_keys = ["price_ratio", "centeredness_margin", "shift_exponent", + "arc_length_speed", "fees"] + params = {} + for k in param_keys: + if k in trial: + params[k] = trial[k] + return params + + +def run_full_period(params, config, fees_override=None): + """Run forward pass over the full train+test window.""" + fees = fees_override if fees_override is not None else config["fees"] + fp = { + "rule": "reclamm", + "tokens": config["tokens"], + "startDateString": config["startDateString"], + "endDateString": config["endTestDateString"], # full period + "initial_pool_value": config["initial_pool_value"], + "do_arb": config["do_arb"], + "fees": fees, + "gas_cost": config.get("gas_cost", 1.0), + "arb_fees": config.get("arb_fees", 0.0), + "protocol_fee_split": config.get("protocol_fee_split", 0.0), + "noise_trader_ratio": config.get("noise_trader_ratio", 0.0), + "reclamm_use_shift_exponent": config.get("reclamm_use_shift_exponent", True), + "reclamm_interpolation_method": config.get("reclamm_interpolation_method", "geometric"), + "reclamm_centeredness_scaling": config.get("reclamm_centeredness_scaling", False), + "reclamm_learn_arc_length_speed": config.get("reclamm_learn_arc_length_speed", False), + } + jax_params = {k: jnp.array(v) for k, v in params.items()} + return do_run_on_historic_data(run_fingerprint=fp, params=jax_params) + + +def plot_results(configs, time_series, hodl_values, config, args): + """Two-panel plot: value-over-time + cumulative fee revenue.""" + train_end_str = config["endDateString"] + train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range( + start=datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S"), + periods=n_minutes, freq="1min", + ) + step = 1440 + dates_daily = dates[::step] + + has_fee_revenue = any( + "fee_revenue" in time_series[n] and time_series[n]["fee_revenue"] is not None + for n in time_series + ) + n_panels = 2 if has_fee_revenue else 1 + fig, axes = plt.subplots( + n_panels, 1, figsize=(14, 5 * n_panels), + sharex=True, gridspec_kw={"height_ratios": [3, 1] if n_panels == 2 else [1]}, + ) + if n_panels == 1: + axes = [axes] + ax_val = axes[0] + + # ── Panel 1: Value over time ────────────────────────────────────── + for name, meta, ci in _plot_order(configs): + out = time_series[name] + vals = np.array(out["value"][::step]) / 1e6 + label = f"{name}" + if "test_objective" in meta: + obj_name = config.get("return_val", "objective") + label += f" (OOS {obj_name}={meta['test_objective']:.4f})" + is_optimized = "On-Chain" not in name + ax_val.plot(dates_daily[:len(vals)], vals, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=label, + zorder=3 if is_optimized else 2) + + hodl_daily = hodl_values[::step] / 1e6 + ax_val.plot(dates_daily[:len(hodl_daily)], hodl_daily, linewidth=2, + color="white", alpha=0.7, linestyle="--", label="HODL") + + ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax_val.get_ylim() + ax_val.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax_val.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") + + _style_axis(ax_val) + ax_val.set_ylabel("Pool Value ($M USD)", color=TEXT_COLOR, fontsize=12) + tokens_str = "/".join(config["tokens"]) + obj_name = config.get("return_val", "objective") + ntr = config.get("noise_trader_ratio", 0.0) + ax_val.set_title( + f"reClAMM Optuna-Optimized ({obj_name}, noise={ntr}) — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15, + ) + ax_val.legend(loc="upper left", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + # ── Panel 2: Cumulative fee revenue ─────────────────────────────── + if has_fee_revenue: + ax_fee = axes[1] + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + fr = out.get("fee_revenue") + if fr is None: + continue + fr = np.array(fr) + cumfee = np.cumsum(fr)[::step] / 1e3 + is_optimized = "On-Chain" not in name + ax_fee.plot(dates_daily[:len(cumfee)], cumfee, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=name, + zorder=3 if is_optimized else 2) + + ax_fee.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + _style_axis(ax_fee) + ax_fee.set_ylabel("Cumulative Fee Revenue ($K)", color=TEXT_COLOR, fontsize=12) + ax_fee.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax_fee.legend(loc="upper left", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + else: + ax_val.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + + output = args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png" + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"\nSaved plot to {output}") + plt.close() + + +def plot_test_only(configs, time_series, hodl_values, config, args): + """Test-period plot with all curves normalised to start at 1.0.""" + train_end_str = config["endDateString"] + train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") + start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") + + # Find the index of the train/test boundary + train_minutes = int((train_end_dt - start_dt).total_seconds() / 60) + test_start_idx = min(train_minutes, n_minutes - 1) + + step = 1440 + test_dates = dates[test_start_idx::step] + + fig, ax = plt.subplots(1, 1, figsize=(14, 6)) + + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + vals = np.array(out["value"]) + test_vals = vals[test_start_idx::step] + if len(test_vals) == 0: + continue + normalised = test_vals / test_vals[0] + is_optimized = "On-Chain" not in name + ax.plot(test_dates[:len(normalised)], normalised, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=name, + zorder=3 if is_optimized else 2) + + hodl_test = hodl_values[test_start_idx::step] + if len(hodl_test) > 0: + hodl_norm = hodl_test / hodl_test[0] + ax.plot(test_dates[:len(hodl_norm)], hodl_norm, linewidth=2, + color="white", alpha=0.7, linestyle="--", label="HODL") + + ax.axhline(1.0, color="white", linestyle=":", alpha=0.3, linewidth=1) + _style_axis(ax) + tokens_str = "/".join(config["tokens"]) + obj_name = config.get("return_val", "objective") + ntr = config.get("noise_trader_ratio", 0.0) + ax.set_title(f"Test Period Only (normalised) — {obj_name}, noise={ntr} — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15) + ax.set_ylabel("Normalised Value", color=TEXT_COLOR, fontsize=12) + ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax.legend(loc="best", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") + output = base.replace(".png", "_test_only.png") + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"Saved plot to {output}") + plt.close() + + +def plot_weights(configs, time_series, config, args): + """Effective weight (value fraction) of token 0 over time.""" + start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") + train_end_dt = datetime.strptime(config["endDateString"], "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") + step = 1440 + dates_daily = dates[::step] + + token_name = config["tokens"][0] + + fig, ax = plt.subplots(1, 1, figsize=(14, 5)) + + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + weights = np.array(out["weights"]) # (T, 2) + w0 = weights[::step, 0] + is_optimized = "On-Chain" not in name + ax.plot(dates_daily[:len(w0)], w0, + linewidth=2.0 if is_optimized else 1.5, + color=COLORS[ci % len(COLORS)], label=name, + alpha=0.9 if is_optimized else 0.7, + zorder=3 if is_optimized else 2) + + ax.axhline(0.5, color="white", linestyle="--", alpha=0.3, linewidth=1) + ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax.get_ylim() + ax.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") + + _style_axis(ax) + tokens_str = "/".join(config["tokens"]) + ax.set_title(f"Effective {token_name} Weight — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15) + ax.set_ylabel(f"{token_name} weight (value fraction)", color=TEXT_COLOR, fontsize=12) + ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax.legend(loc="best", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") + output = base.replace(".png", "_weights.png") + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"Saved plot to {output}") + plt.close() + + +def _style_axis(ax): + ax.set_facecolor(BG) + ax.tick_params(colors=TEXT_COLOR) + for spine in ax.spines.values(): + spine.set_color(TEXT_COLOR) + spine.set_alpha(0.3) + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.grid(True, alpha=0.15, color=TEXT_COLOR) + + +def main(): + args = parse_args() + config, trials = load_results(args.results_json) + if args.end_test_date: + config["endTestDateString"] = args.end_test_date + if args.noise_trader_ratio is not None: + config["noise_trader_ratio"] = args.noise_trader_ratio + tokens = config["tokens"] + obj_name = config.get("return_val", "objective") + + # Sort trials by penalised objective + trials_sorted = sorted(trials, key=lambda t: t.get("objective", 0), reverse=True) + top_trials = trials_sorted[:args.top_k] + + print("=" * 80) + print(f"reClAMM Optuna Result Plotter — objective: {obj_name}") + print("=" * 80) + print(f" Results: {args.results_json}") + print(f" Tokens: {'/'.join(tokens)}") + print(f" Train: {config['startDateString']} → {config['endDateString']}") + print(f" Test: {config['endDateString']} → {config['endTestDateString']}") + print(f" Fees: {config['fees']}, Gas: {config.get('gas_cost', 1.0)}") + print(f" Trials: {len(trials)} total, plotting top {len(top_trials)}") + + configs = {} + for i, trial in enumerate(top_trials): + params = extract_pool_params(trial, config) + name = f"#{trial.get('optuna_trial_number', i)} (rank {i+1})" + configs[name] = { + "params": params, + "objective": trial.get("objective", 0), + "train_objective": trial.get("train_objective", 0), + "test_objective": trial.get("test_objective", 0), + "train_sharpe": trial.get("train_sharpe", 0), + "validation_sharpe": trial.get("validation_sharpe", 0), + } + print(f"\n {name}:") + print(f" {obj_name}: train={trial.get('train_objective', 0):.4f} " + f"test={trial.get('test_objective', 0):.4f} " + f"penalised={trial.get('objective', 0):.4f}") + print(f" sharpe: train={trial.get('train_sharpe', 0):+.4f} " + f"val={trial.get('validation_sharpe', 0):+.4f}") + for k, v in params.items(): + print(f" {k}: {v:.6g}") + + if not args.no_onchain: + configs["On-Chain (launch)"] = {"params": dict(ONCHAIN_LAUNCH_PARAMS)} + configs["On-Chain (current)"] = {"params": dict(ONCHAIN_CURRENT_PARAMS)} + + # ── Full-period runs ────────────────────────────────────────────── + print(f"\n--- Running full-period simulations ({config['startDateString']} → " + f"{config['endTestDateString']}) ---") + time_series = {} + for name, cfg in configs.items(): + print(f" {name}...", end=" ", flush=True) + out = run_full_period(cfg["params"], config) + time_series[name] = out + fv = float(out["final_value"]) + fr = out.get("fee_revenue") + fr_total = float(np.array(fr).sum()) if fr is not None else 0 + hodl = float((out["reserves"][0] * out["prices"][-1]).sum()) + print(f"final=${fv:,.0f} hodl=${hodl:,.0f} RoH={fv/hodl - 1:+.2%} " + f"fee_rev=${fr_total:,.0f}") + + first_out = next(iter(time_series.values())) + hodl_reserves = first_out["reserves"][0] + hodl_values = np.sum( + np.array(hodl_reserves) * np.array(first_out["prices"]), axis=1, + ) + + # ── Plots ───────────────────────────────────────────────────────── + plot_results(configs, time_series, hodl_values, config, args) + plot_test_only(configs, time_series, hodl_values, config, args) + plot_weights(configs, time_series, config, args) + + # ── Summary table ───────────────────────────────────────────────── + print(f"\n{'=' * 120}") + print(f"SUMMARY — {'/'.join(tokens)} — {obj_name}") + print(f"{'=' * 120}") + hdr = (f"{'Config':<28s} {'Train '+obj_name:>20s} {'Test '+obj_name:>20s} " + f"{'Train SR':>10s} {'Val SR':>10s} " + f"{'PR':>7s} {'Margin':>7s} {'ShiftExp':>10s} {'Full RoH':>10s}") + print(hdr) + print("-" * 120) + + for name, cfg in configs.items(): + cp = cfg["params"] + fv = float(time_series[name]["final_value"]) + full_roh = fv / float(hodl_values[-1]) - 1 + print( + f"{name:<28s} " + f"{cfg.get('train_objective', float('nan')):>20.4f} " + f"{cfg.get('test_objective', float('nan')):>20.4f} " + f"{cfg.get('train_sharpe', float('nan')):>+10.4f} " + f"{cfg.get('validation_sharpe', float('nan')):>+10.4f} " + f"{cp.get('price_ratio', float('nan')):>7.3f} " + f"{cp.get('centeredness_margin', float('nan')):>7.4f} " + f"{cp.get('shift_exponent', float('nan')):>10.4g} " + f"{full_roh * 100:>+9.2f}%" + ) + print("=" * 120) + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_top50_predicted_vs_real.py b/scripts/plot_top50_predicted_vs_real.py new file mode 100644 index 00000000..0951a209 --- /dev/null +++ b/scripts/plot_top50_predicted_vs_real.py @@ -0,0 +1,521 @@ +"""Plot predicted vs real volume for top 50 pools by TVL on Feb 1st 2026. + +Enumerates WEIGHTED (min_tvl=1000) and RECLAMM (min_tvl=0) pools, +fetches their snapshots, filters to those with TVL >= $10k on Feb 1st 2026, +takes the top 50 by TVL, and plots predicted vs actual daily volume using +the inference artifact from calibrate_noise_unified.py. + +For pools that were in the model's training set, uses their per-pool theta. +For pools not in the training set, uses population-level prediction from B. +""" + +import ast +import json +import os +import sys +import time +from datetime import date, timedelta + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# Reuse functions from the noise calibration package +from quantammsim.noise_calibration import ( + BALANCER_API_CHAINS, + _graphql_request, + assemble_panel, + classify_token_tier, + encode_covariates, + fetch_pool_snapshots, + fetch_token_prices, +) + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_top50" +) +TVL_DATE = date(2026, 2, 1) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "top50_feb1" +) +# Inference artifact from the main unified model run +FITTED_JSON = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "unified_full_90d.json" +) + + +def enumerate_all_pools(): + """Enumerate WEIGHTED (min_tvl=1000) and RECLAMM (min_tvl=0) pools.""" + all_pools = [] + + for chain in BALANCER_API_CHAINS: + for pool_type, min_tvl in [("WEIGHTED", 1000), ("RECLAMM", 0)]: + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { chainIn: [$chain], poolTypeIn: $types, minTvl: $minTvl } + ) { + id chain type protocolVersion + poolTokens { symbol weight address } + dynamicData { totalLiquidity swapFee } + } + } + """, + "variables": { + "chain": chain, + "types": [pool_type], + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f" FAILED {chain} {pool_type}: {e}") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "tokens": tokens, + "token_addresses": addresses, + "swap_fee": fee, + "current_tvl": tvl, + }) + + if pools: + print(f" {chain:>10} {pool_type:>10}: {len(pools)}") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools") + return df + + +def fetch_all_snapshots_cached(pools_df, cache_dir): + """Fetch snapshots for all pools, caching per-pool.""" + snap_dir = os.path.join(cache_dir, "snapshots") + os.makedirs(snap_dir, exist_ok=True) + + all_snaps = [] + n = len(pools_df) + + for i, (_, pool) in enumerate(pools_df.iterrows()): + pid = pool["pool_id"] + chain = pool["chain"] + cache_file = os.path.join(snap_dir, f"{pid}.parquet") + + if os.path.exists(cache_file): + df = pd.read_parquet(cache_file) + else: + if (i + 1) % 20 == 0 or i == 0: + print(f" Fetching snapshots {i+1}/{n}...", flush=True) + try: + df = fetch_pool_snapshots(pid, chain) + if len(df) > 0: + df.to_parquet(cache_file, index=False) + time.sleep(0.3) + except Exception as e: + print(f" FAILED {pid[:20]}: {e}") + continue + + if len(df) > 0: + df["pool_id"] = pid + df["chain"] = chain + all_snaps.append(df) + + if all_snaps: + return pd.concat(all_snaps, ignore_index=True) + return pd.DataFrame() + + +def get_tvl_on_date(snapshots_df, target_date, window_days=3): + """Get TVL for each pool on/near target_date.""" + results = [] + for pid in snapshots_df["pool_id"].unique(): + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pid] + + best_row = None + best_dist = float("inf") + for _, row in pool_snaps.iterrows(): + d = row["date"] + if isinstance(d, date): + dist = abs((d - target_date).days) + else: + dist = abs((pd.Timestamp(d).date() - target_date).days) + if dist < best_dist: + best_dist = dist + best_row = row + + if best_row is not None and best_dist <= window_days: + results.append({ + "pool_id": pid, + "tvl_feb1": float(best_row["total_liquidity_usd"]), + "date_used": best_row["date"], + }) + + return pd.DataFrame(results) + + +def _get_theta_for_pool(pid, fitted, panel_90d, pop_B, pop_cov_names): + """Get theta for a pool: from fitted artifact if available, else population. + + Returns (theta, source) where source is 'fitted' or 'population'. + For IBP models, population fallback adds marginal feature effect (pi @ W). + """ + if pid in fitted["pools"]: + return np.array(fitted["pools"][pid]["theta_median"]), "fitted" + + # Population-level prediction: theta = B @ z_pool + # Build z_pool from the pool's covariates + pp = panel_90d[panel_90d["pool_id"] == pid] + if len(pp) == 0: + return None, "no_data" + + chain = pp["chain"].iloc[0] + tokens = pp["tokens"].iloc[0] + if isinstance(tokens, str): + tokens = tokens.split(",") + fee = pp["swap_fee"].iloc[0] if "swap_fee" in pp.columns else 0.003 + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + + # Build covariate vector matching the model's encoding + z = np.zeros(len(pop_cov_names)) + for i, name in enumerate(pop_cov_names): + if name == "intercept": + z[i] = 1.0 + elif name == f"chain_{chain}": + z[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z[i] = 1.0 + elif name == "log_fee": + z[i] = np.log(max(fee, 1e-6)) + + # B is (K_coeff, K_cov), theta = B @ z + B = np.array(pop_B) # (K_coeff, K_cov) + theta = B @ z + + # IBP: add marginal feature effect (pi @ W) + pop = fitted["population_effects"] + if "W" in pop and "feature_prevalences" in pop: + W = np.array(pop["W"]) # (K_features, K_coeff) + pi = np.array(pop["feature_prevalences"]) # (K_features,) + theta = theta + pi @ W + + return theta, "population" + + +def plot_pages(plot_pools, pool_idx_map_fitted, fitted, panel_90d, + pools_df, tvl_lookup, pop_B, pop_cov_names, + output_dir=OUTPUT_DIR): + """Generate paginated plots, 10 pools per page.""" + n_pools = len(plot_pools) + per_page = 10 + n_pages = (n_pools + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, n_pools) + page_pools = plot_pools[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(14, 4 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + + for idx, (pid, feb_tvl) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + + theta, source = _get_theta_for_pool( + pid, fitted, panel_90d, pop_B, pop_cov_names + ) + if theta is None: + ax.set_visible(False) + continue + + pp = panel_90d[panel_90d["pool_id"] == pid].sort_values("date") + if len(pp) < 5: + ax.set_visible(False) + continue + + x_obs = np.column_stack([ + np.ones(len(pp)), + pp["log_tvl_lag1"].values, + pp["volatility"].values, + pp["weekend"].values, + ]) + + pred_log = x_obs @ theta + actual_log = pp["log_volume"].values + pred_vol = np.exp(pred_log) + actual_vol = np.exp(actual_log) + dates = pd.to_datetime(pp["date"].values) + + ax.plot(dates, actual_vol, "o-", color="steelblue", markersize=2.5, + linewidth=0.9, alpha=0.7, label="Actual") + ax.plot(dates, pred_vol, "s--", color="orangered", markersize=2.5, + linewidth=0.9, alpha=0.7, label="Predicted") + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + ss_res = np.sum((actual_log - pred_log) ** 2) + ss_tot = np.sum((actual_log - actual_log.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + meta = pools_df[pools_df["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tokens = m["tokens"] + tok_str = "/".join(str(t)[:8] for t in tokens[:2]) + chain = str(m["chain"]) + ptype = str(m["pool_type"]) + else: + tok_str = pid[:16] + chain = "?" + ptype = "?" + + type_tag = "R" if ptype == "RECLAMM" else "W" + src_tag = "*" if source == "population" else "" + ax.set_title( + "{} ({}, {}){}\n" + "TVL ${:,.0f} on Feb 1 | " + "R\u00b2={:.3f} b_c={:.2f} b_\u03c3={:.2f} " + "b_wknd={:.2f} n={}".format( + tok_str, chain, type_tag, src_tag, feb_tvl, + r2, theta[1], theta[2], theta[3], len(pp)), + fontsize=8) + ax.legend(fontsize=7) + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + "Predicted vs actual daily volume \u2014 page {}/{} " + "(sorted by TVL on {}) [* = population prediction]".format( + page + 1, n_pages, TVL_DATE), + fontsize=11) + fig.tight_layout() + out = os.path.join(output_dir, "pred_vs_real_page{}.png".format(page + 1)) + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(" Saved: {}".format(out)) + + +def main(): + import argparse + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--artifact", default=FITTED_JSON, + help="Path to inference artifact JSON") + parser.add_argument("--output-dir", default=None, + help="Output directory (default: auto from model name)") + args = parser.parse_args() + + artifact_path = args.artifact + os.makedirs(CACHE_DIR, exist_ok=True) + + # ---- Load inference artifact ---- + print(f"Loading inference artifact: {artifact_path}") + with open(artifact_path) as f: + fitted = json.load(f) + n_fitted = len(fitted["pools"]) + model_name = fitted.get("model", "unknown") + print(f" Model: {model_name}") + print(f" {n_fitted} pools with fitted theta") + + # Output dir: use CLI override or auto from model name + output_dir = args.output_dir or os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", f"top50_feb1_{model_name}", + ) + os.makedirs(output_dir, exist_ok=True) + + # Extract population-level B matrix for pools not in the model + pop_cov_names = fitted["model_spec"]["covariate_names"] + pop_B = np.array(fitted["population_effects"]["B"]) # (K_coeff, K_cov) + print(f" Population B: {pop_B.shape}, covariates: {pop_cov_names}") + + # ---- Step 1: Enumerate pools ---- + pools_cache = os.path.join(CACHE_DIR, "pools.parquet") + if os.path.exists(pools_cache): + pools_df = pd.read_parquet(pools_cache) + if isinstance(pools_df["tokens"].iloc[0], str): + pools_df["tokens"] = pools_df["tokens"].apply(ast.literal_eval) + pools_df["token_addresses"] = pools_df["token_addresses"].apply( + ast.literal_eval + ) + print(f"\nLoaded {len(pools_df)} pools from cache") + else: + print("\n1. Enumerating pools...") + pools_df = enumerate_all_pools() + pools_df.to_parquet(pools_cache, index=False) + + # ---- Step 2: Fetch snapshots ---- + print("\n2. Fetching snapshots...") + snapshots_df = fetch_all_snapshots_cached(pools_df, CACHE_DIR) + print(f" {len(snapshots_df)} pool-days") + + # ---- Step 3: TVL on Feb 1st ---- + print(f"\n3. Finding TVL on {TVL_DATE}...") + tvl_df = get_tvl_on_date(snapshots_df, TVL_DATE) + tvl_df = tvl_df[tvl_df["tvl_feb1"] >= 10_000].copy() + tvl_df = tvl_df.sort_values("tvl_feb1", ascending=False).head(50) + top50_ids = set(tvl_df["pool_id"]) + tvl_lookup = dict(zip(tvl_df["pool_id"], tvl_df["tvl_feb1"])) + print(f" {len(tvl_df)} pools with TVL >= $10k") + + # How many are in the fitted model? + n_in_model = sum(1 for pid in top50_ids if pid in fitted["pools"]) + print(f" {n_in_model} in fitted model, " + f"{len(top50_ids) - n_in_model} will use population prediction") + + # ---- Step 4: Fetch token prices & assemble panel ---- + panel_cache = os.path.join(CACHE_DIR, "panel.parquet") + if os.path.exists(panel_cache): + panel = pd.read_parquet(panel_cache) + print(f"\n4. Loaded panel from cache: {len(panel)} obs") + else: + top50_pools = pools_df[pools_df["pool_id"].isin(top50_ids)].copy() + top50_snaps = snapshots_df[snapshots_df["pool_id"].isin(top50_ids)].copy() + + print("\n4. Fetching token prices...") + prices_cache = os.path.join(CACHE_DIR, "token_prices") + token_addr_by_chain = {} + for _, pool in top50_pools.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + print("\n Assembling panel...") + panel = assemble_panel(top50_pools, top50_snaps, token_prices) + panel.to_parquet(panel_cache, index=False) + + # ---- Step 5: Filter to 90 days ---- + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=90) + panel_90d = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if "log_tvl_lag1" not in panel_90d.columns: + panel_90d = panel_90d.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel_90d["log_tvl_lag1"] = panel_90d.groupby("pool_id")["log_tvl"].shift(1) + panel_90d = panel_90d.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel_90d.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel_90d = panel_90d[panel_90d["pool_id"].isin(valid_pools)].copy() + print(f"\n5. 90-day panel: {len(panel_90d)} obs, " + f"{panel_90d['pool_id'].nunique()} pools") + + # ---- Step 6: Plot ---- + # Sort by Feb 1 TVL, only include pools with panel data + plot_pools = [] + for pid in panel_90d["pool_id"].unique(): + if pid in tvl_lookup: + plot_pools.append((pid, tvl_lookup[pid])) + plot_pools.sort(key=lambda x: -x[1]) + print(f"\n6. Plotting {len(plot_pools)} pools...") + + plot_pages(plot_pools, fitted, fitted, panel_90d, pools_df, + tvl_lookup, pop_B, pop_cov_names, output_dir=output_dir) + + # ---- Summary table ---- + summary = [] + for pid, feb_tvl in plot_pools: + theta, source = _get_theta_for_pool( + pid, fitted, panel_90d, pop_B, pop_cov_names + ) + if theta is None: + continue + pp = panel_90d[panel_90d["pool_id"] == pid] + x = np.column_stack([ + np.ones(len(pp)), + pp["log_tvl_lag1"].values, + pp["volatility"].values, + pp["weekend"].values, + ]) + pred = x @ theta + actual = pp["log_volume"].values + ss_res = np.sum((actual - pred) ** 2) + ss_tot = np.sum((actual - actual.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + meta = pools_df[pools_df["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tok_str = "/".join(str(t) for t in m["tokens"][:2]) + chain = str(m["chain"]) + ptype = str(m["pool_type"]) + else: + tok_str = pid[:16] + chain = "?" + ptype = "?" + + summary.append({ + "pool_id": pid[:20], + "tokens": tok_str, + "chain": chain, + "type": ptype, + "tvl_feb1": feb_tvl, + "n_obs": len(pp), + "R2": r2, + "b_c": theta[1], + "b_sigma": theta[2], + "b_weekend": theta[3], + "source": source, + }) + + summary_df = pd.DataFrame(summary) + summary_path = os.path.join(output_dir, "top50_summary.csv") + summary_df.to_csv(summary_path, index=False) + print(f"\n Saved: {summary_path}") + + n_pools = len(summary_df) + n_fitted_used = (summary_df["source"] == "fitted").sum() + n_pop = (summary_df["source"] == "population").sum() + n_reclamm = (summary_df["type"] == "RECLAMM").sum() + print(f"\n{'='*70}") + print(f"Summary: {n_pools} pools ({n_fitted_used} fitted, {n_pop} population)") + print(f" RECLAMM: {n_reclamm} WEIGHTED: {n_pools - n_reclamm}") + print(f" Median R\u00b2: {summary_df['R2'].median():.3f}") + print(f" Mean b_c: {summary_df['b_c'].mean():.3f}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_structural_top50.py b/scripts/run_structural_top50.py new file mode 100644 index 00000000..0459bd5e --- /dev/null +++ b/scripts/run_structural_top50.py @@ -0,0 +1,456 @@ +"""Fit the structural mixture model and plot predicted vs actual for top 50 pools. + +Uses the cached panel (last 90 days), fits with vanilla SVI, then generates +paginated plots showing V_arb + V_noise decomposition and predicted vs actual. +""" + +import json +import os +import sys +from datetime import date, timedelta + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "structural_top50_nogas", +) +OUTPUT_JSON = os.path.join(OUTPUT_DIR, "structural_fit.json") +TRAIN_DAYS = 90 +SVI_STEPS = 20_000 +SVI_LR = 1e-3 +NUM_SAMPLES = 1000 +SEED = 42 +TOP_N = 50 + + +def load_and_filter_panel(): + """Load cached panel, filter to 90 days, keep pools with >= 10 obs.""" + panel = pd.read_parquet(PANEL_CACHE) + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=TRAIN_DAYS) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if "log_tvl_lag1" not in panel.columns: + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel.groupby("pool_id").size() + valid = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid)].copy() + + print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " + f"{cutoff} to {max_date}") + return panel + + +def fit_structural(panel): + """Run SVI on the structural mixture model.""" + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.postprocessing import ( + check_convergence, extract_structural_params, + ) + from quantammsim.noise_calibration.output import generate_output_json + + import numpyro + numpyro.enable_x64() + + # Load gas costs for mainnet from CSV if available + gas_csv = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "formula_vs_real", "mainnet_gas_cost_daily.csv", + ) + gas_arr = None + if os.path.exists(gas_csv): + gas_df = pd.read_csv(gas_csv) + # CSV has columns: unix (ms timestamp), USD (gas cost) + gas_df["date"] = pd.to_datetime(gas_df["unix"], unit="ms").dt.date + gas_lookup = dict(zip(gas_df["date"], gas_df["USD"])) + + # Build per-observation gas array + gas_vals = [] + for _, row in panel.iterrows(): + d = row["date"] + if not isinstance(d, date): + d = pd.Timestamp(d).date() + chain = row["chain"] + if chain == "MAINNET" and d in gas_lookup: + gas_vals.append(gas_lookup[d]) + elif chain == "MAINNET": + gas_vals.append(1.0) # median fallback + else: + # L2 chains: ~$0.005 + from quantammsim.noise_calibration.constants import GAS_COSTS + gas_vals.append(GAS_COSTS.get(chain, 0.005)) + gas_arr = np.array(gas_vals, dtype=np.float64) + print(f"Gas costs: loaded ({len(gas_lookup)} mainnet days from CSV)") + else: + print("Gas costs: using defaults (no mainnet CSV)") + + # No-gas variant: rely on cadence alone to modulate V_arb + # Gas threshold kills V_arb=0 for TVL<$300k (most pools), making cadence + # unlearnable. Without gas, V_arb>0 for all pools and cadence has gradient. + gas_arr = np.zeros(len(panel), dtype=np.float64) + print("Gas costs: DISABLED (cadence-only mode)") + + data = encode_covariates_structural(panel, gas=gas_arr) + + print(f"\nFitting structural model: {SVI_STEPS} SVI steps, lr={SVI_LR}") + samples, elbo_losses = run_svi( + data, + num_steps=SVI_STEPS, + lr=SVI_LR, + seed=SEED, + num_samples=NUM_SAMPLES, + model_fn=structural_noise_model, + ) + convergence = check_convergence(elbo_losses, method="svi") + + pool_params = extract_structural_params(samples, data) + + # Save output JSON + os.makedirs(OUTPUT_DIR, exist_ok=True) + inference_config = { + "method": "svi", "svi_steps": SVI_STEPS, + "svi_lr": SVI_LR, "num_samples": NUM_SAMPLES, + } + generate_output_json( + pool_params, samples, data, convergence, + OUTPUT_JSON, inference_config, + ) + + return samples, data, pool_params, elbo_losses + + +def compute_predictions(samples, data, panel): + """Compute per-observation predicted V_arb and V_noise.""" + from quantammsim.noise_calibration.formula_arb import ( + formula_arb_volume_daily_jax, + ) + from quantammsim.noise_calibration.model import _pad_with_ref + from quantammsim.noise_calibration.constants import OBS_COEFF_NAMES + import jax.numpy as jnp + + sample_dict = samples + agg_fn = np.median + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # MoE parameters + W_gate = agg_fn(np.array(sample_dict["W_gate"]), axis=0) + beta = agg_fn(np.array(sample_dict["beta"]), axis=0) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + chain_idx = np.array(data["chain_idx"]) + tier_idx = np.array(data["tier_idx"]) + sigma_daily = np.array(data["sigma_daily"]) + lag_log_tvl = np.array(data["lag_log_tvl"]) + fee = np.array(data["fee"]) + gas = np.array(data["gas"]) + + # Per-pool cadence + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + N_pools = data["N_pools"] + pool_log_cadence = np.zeros(N_pools) + for p in range(N_pools): + pool_log_cadence[p] = ( + alpha_0 + + padded_chain[chain_idx[p]] + + padded_tier[tier_idx[p]] + + alpha_tvl * np.median(lag_log_tvl[pool_idx == p]) + ) + + # Per-obs V_arb + log_cad_obs = pool_log_cadence[pool_idx] + cadence_obs = np.exp(np.clip(log_cad_obs, -2.0, 6.0)) + tvl_obs = np.exp(lag_log_tvl) + + V_arb = np.array(formula_arb_volume_daily_jax( + jnp.array(sigma_daily), jnp.array(tvl_obs), + jnp.array(fee), jnp.array(gas), jnp.array(cadence_obs), + )) + + # Per-pool noise coefficients via MoE + logits = X_pool @ W_gate + w = np.exp(logits - logits.max(axis=1, keepdims=True)) + w = w / w.sum(axis=1, keepdims=True) + beta_pool = w @ beta # (N_pools, K_obs_coeff) + + # Per-obs V_noise + log_V_noise = np.sum(beta_pool[pool_idx] * x_obs, axis=1) + V_noise = np.exp(log_V_noise) + + # Predicted total + V_total_pred = V_arb + V_noise + log_V_pred = np.log(np.maximum(V_total_pred, 1e-6)) + + return V_arb, V_noise, V_total_pred, log_V_pred, cadence_obs + + +def plot_top50(panel, data, pool_params, V_arb, V_noise, log_V_pred): + """Plot top 50 pools by median TVL.""" + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + pool_idx = np.array(data["pool_idx"]) + y_obs = np.array(data["y_obs"]) + + # Rank pools by median TVL + pool_tvl = {} + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + pool_tvl[pid] = np.median(np.exp(np.array(data["lag_log_tvl"])[mask])) + + ranked = sorted(pool_tvl.items(), key=lambda x: -x[1])[:TOP_N] + + # Build param lookup + param_lookup = {p["pool_id"]: p for p in pool_params} + + per_page = 10 + n_pages = (len(ranked) + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, len(ranked)) + page_pools = ranked[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(16, 4.5 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + + for idx, (pid, median_tvl) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + p_idx = pool_ids.index(pid) + mask = pool_idx == p_idx + + pp = panel[panel["pool_id"] == pid].sort_values("date") + dates = pd.to_datetime(pp["date"].values) + actual_vol = np.exp(y_obs[mask]) + pred_arb = V_arb[mask] + pred_noise = V_noise[mask] + pred_total = pred_arb + pred_noise + + # R2 + actual_log = y_obs[mask] + pred_log = log_V_pred[mask] + ss_res = np.sum((actual_log - pred_log) ** 2) + ss_tot = np.sum((actual_log - actual_log.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + # Arb fraction + arb_frac = np.median(pred_arb / np.maximum(pred_total, 1.0)) + + # Plot + ax.fill_between(dates, 0, pred_arb, alpha=0.3, color="orangered", + label="V_arb (LVR)") + ax.fill_between(dates, pred_arb, pred_total, alpha=0.3, + color="steelblue", label="V_noise (MoE)") + ax.plot(dates, actual_vol, "k-", linewidth=0.8, alpha=0.7, + label="Actual") + ax.plot(dates, pred_total, "--", color="purple", linewidth=0.8, + alpha=0.7, label="Predicted total") + + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + meta = pool_meta[pool_meta["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tokens = m["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + tok_str = "/".join(str(t)[:8] for t in tokens[:2]) + chain = str(m["chain"]) + else: + tok_str = pid[:16] + chain = "?" + + params = param_lookup.get(pid, {}) + arb_freq = params.get("arb_frequency", "?") + + ax.set_title( + f"{tok_str} ({chain})\n" + f"TVL ${median_tvl:,.0f} | R\u00b2={r2:.3f} " + f"arb_freq={arb_freq}min arb_frac={arb_frac:.1%} " + f"n={mask.sum()}", + fontsize=8, + ) + ax.legend(fontsize=6, loc="upper right") + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + f"Structural mixture model: V_arb + V_noise decomposition " + f"— page {page + 1}/{n_pages} " + f"(top {TOP_N} by median TVL, 90d window)", + fontsize=11, + ) + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, f"structural_top50_page{page + 1}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_elbo(elbo_losses): + """Plot ELBO convergence.""" + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + ax = axes[0] + ax.plot(elbo_losses, alpha=0.3, color="steelblue", linewidth=0.5) + window = min(100, len(elbo_losses) // 10) + if window > 1: + smoothed = pd.Series(elbo_losses).rolling(window).mean().values + ax.plot(smoothed, color="red", linewidth=1.5, label=f"Rolling {window}") + ax.legend() + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence") + + ax = axes[1] + start = len(elbo_losses) * 4 // 5 + ax.plot(range(start, len(elbo_losses)), elbo_losses[start:], + color="steelblue", linewidth=0.8) + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence (last 20%)") + + plt.tight_layout() + out = os.path.join(OUTPUT_DIR, "elbo_convergence.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {out}") + + +def plot_summary(data, pool_params, V_arb, V_noise, log_V_pred): + """Summary plots: arb frequency distribution, arb fraction, R2.""" + pool_idx = np.array(data["pool_idx"]) + y_obs = np.array(data["y_obs"]) + pool_ids = data["pool_ids"] + + fig, axes = plt.subplots(1, 3, figsize=(16, 5)) + + # 1. Arb frequency histogram + ax = axes[0] + freqs = [p["arb_frequency"] for p in pool_params] + ax.hist(freqs, bins=range(0, 62, 2), color="orangered", alpha=0.7, + edgecolor="white") + ax.set_xlabel("Arb frequency (minutes)") + ax.set_ylabel("Count") + ax.set_title(f"Arb frequency distribution (n={len(freqs)})") + ax.axvline(np.median(freqs), color="black", linestyle="--", + label=f"Median={np.median(freqs):.0f}min") + ax.legend() + + # 2. Arb fraction per pool + ax = axes[1] + arb_fracs = [] + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + total = V_arb[mask] + V_noise[mask] + arb_fracs.append(np.median(V_arb[mask] / np.maximum(total, 1.0))) + ax.hist(arb_fracs, bins=30, color="steelblue", alpha=0.7, edgecolor="white") + ax.set_xlabel("Median arb fraction") + ax.set_ylabel("Count") + ax.set_title("Arb fraction distribution") + ax.axvline(np.median(arb_fracs), color="black", linestyle="--", + label=f"Median={np.median(arb_fracs):.2f}") + ax.legend() + + # 3. Per-pool R2 + ax = axes[2] + r2_vals = [] + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + actual = y_obs[mask] + pred = log_V_pred[mask] + ss_res = np.sum((actual - pred) ** 2) + ss_tot = np.sum((actual - actual.mean()) ** 2) + r2_vals.append(1 - ss_res / ss_tot if ss_tot > 0 else float("nan")) + r2_vals = np.array(r2_vals) + ax.hist(r2_vals[np.isfinite(r2_vals)], bins=30, color="green", alpha=0.7, + edgecolor="white") + ax.set_xlabel("R²") + ax.set_ylabel("Count") + ax.set_title("Per-pool R² distribution") + ax.axvline(np.nanmedian(r2_vals), color="black", linestyle="--", + label=f"Median={np.nanmedian(r2_vals):.3f}") + ax.legend() + + plt.tight_layout() + out = os.path.join(OUTPUT_DIR, "structural_summary.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {out}") + + +def main(): + print("=" * 70) + print("Structural Mixture Model: Fit + Top 50 Plots") + print("=" * 70) + + panel = load_and_filter_panel() + samples, data, pool_params, elbo_losses = fit_structural(panel) + + print("\nComputing predictions...") + V_arb, V_noise, V_total, log_V_pred, cadence = compute_predictions( + samples, data, panel, + ) + print(f" V_arb median: ${np.median(V_arb):,.0f}") + print(f" V_noise median: ${np.median(V_noise):,.0f}") + print(f" Arb fraction (median pool): {np.median(V_arb / np.maximum(V_total, 1)):.2%}") + + print("\nGenerating plots...") + os.makedirs(OUTPUT_DIR, exist_ok=True) + plot_elbo(elbo_losses) + plot_summary(data, pool_params, V_arb, V_noise, log_V_pred) + plot_top50(panel, data, pool_params, V_arb, V_noise, log_V_pred) + + # Summary stats + print(f"\n{'=' * 70}") + print(f"Done. Output in: {OUTPUT_DIR}") + arb_freqs = [p["arb_frequency"] for p in pool_params] + print(f" Arb frequency: median={np.median(arb_freqs):.0f}min, " + f"range=[{np.min(arb_freqs)}, {np.max(arb_freqs)}]") + print(f" JSON: {OUTPUT_JSON}") + + +if __name__ == "__main__": + main() diff --git a/scripts/sim_vs_world_comparison.py b/scripts/sim_vs_world_comparison.py new file mode 100644 index 00000000..ca3046f5 --- /dev/null +++ b/scripts/sim_vs_world_comparison.py @@ -0,0 +1,972 @@ +#!/usr/bin/env python3 +"""Compare quantammsim reClAMM / Balancer vs reclamm-simulations repo + on-chain. + +Runs: + 1. Zero-fee Balancer pool (quantammsim) — the normalization baseline + 2. reClAMM pool with on-chain params (quantammsim) + 3. Loads reclamm-simulations results + world values from CSV + 4. Gas-experiment runs: time-varying gas from on-chain percentiles, + 50% protocol fee take, on-chain fees + +All comparisons align quantammsim's minute-level output to the world state +CSV's actual Unix timestamps, eliminating timing drift from block-time +variability. + +4-panel plot matching the reclamm-simulations format: + Top-left: Price (WETH/AAVE) — both repos overlaid + Top-right: (legend) + Bottom-left: Absolute value in WETH + Bottom-right: Value relative to feeless weighted (Balancer = 1.0) + +Usage: + python scripts/sim_vs_world_comparison.py + python scripts/sim_vs_world_comparison.py --csv /path/to/csv + python scripts/sim_vs_world_comparison.py --gas-experiment +""" + +import argparse +import numpy as np +import pandas as pd +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import jax.numpy as jnp +from pathlib import Path +from datetime import datetime, timezone + +from quantammsim.runners.jax_runners import do_run_on_historic_data + +# ── On-chain reClAMM params ─────────────────────────────────────────────────── +ONCHAIN_FEES = 0.0025 + +ONCHAIN_LAUNCH_PARAMS = { # deployment through 2025-12-18 + "price_ratio": 1.5014, + "centeredness_margin": 0.5, + "shift_exponent": 0.1, +} +ONCHAIN_CURRENT_PARAMS = { # post 2025-12-18 governance + "price_ratio": 4.0, + "centeredness_margin": 0.1, + "shift_exponent": 0.001, +} +GOVERNANCE_DATE = "2025-12-18" + +# CSV starts at ~17.2 WETH ≈ $50k at $2900/ETH. +INITIAL_POOL_VALUE = 50_000.0 + +# Gas cost = arb profit threshold in USD. +# reclamm-simulations uses profit_threshold = 3e-4 WETH (in token1 units). +# quantammsim's arb_thresh is in USD: 3 * 3e-4 WETH × ~$3000/ETH ≈ $2.70. +ARB_GAS_COST = 2.7 + +DEFAULT_CSV = ( + "/Users/matthew/Projects/reclamm-simulations" + "/data/sim_vs_world_values_AAVE_WETH.csv" +) +ZEROFEE_CSV = ( + "/Users/matthew/Projects/reclamm-simulations" + "/data/sim_vs_world_zerofee_centered_AAVE_WETH.csv" +) +ZEROFEE_MINUTE_CSV = ( + "/Users/matthew/Projects/reclamm-simulations" + "/data/sim_vs_world_zerofee_centered_minute_AAVE_WETH.csv" +) +WORLD_STATE_CSV = ( + "/Users/matthew/Projects/reclamm-simulations" + "/data/sim_vs_world_world_AAVE_WETH.csv" +) +DEFAULT_START = "2025-08-16 00:00:00" +DEFAULT_END = "2026-01-04 00:00:00" +DEFAULT_TOKENS = ["AAVE", "ETH"] +HALF_DAY = 720 # minutes + +# Gas experiment +GAS_CSV_DIR = Path(__file__).resolve().parent.parent / "gas_csvs" +GAS_PERCENTILES = ["50p", "75p", "90p", "95p"] +GAS_SCALE_FACTORS = [0.25, 0.5, 0.75, 1.0] +FLAT_GAS_USD = [0.0, 0.25, 0.50, 1.0, 2.0, 3.0, 5.0] +PROTOCOL_FEE_SPLIT = 0.5 + + +def parse_args(): + p = argparse.ArgumentParser( + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + p.add_argument("--csv", default=DEFAULT_CSV) + p.add_argument("--start", default=DEFAULT_START) + p.add_argument("--end", default=DEFAULT_END) + p.add_argument("--tokens", nargs="+", default=DEFAULT_TOKENS) + p.add_argument("--output", default="sim_vs_world_comparison.png") + p.add_argument( + "--gas-experiment", action="store_true", + help="Run gas-experiment sweep (time-varying gas, 50%% protocol fee)", + ) + p.add_argument( + "--launch-params", action="store_true", + help="Use launch params instead of current params in gas experiment", + ) + p.add_argument( + "--gas-scale-sweep", action="store_true", + help="Sweep gas cost scale factors, rebase to world, truncate at governance", + ) + p.add_argument( + "--best-gas", action="store_true", + help="Run the 3 best gas configs vs world (clean plot)", + ) + return p.parse_args() + + +def load_onchain_initial_state(): + """Load the on-chain pool state at t=0 from the world state CSV. + + Returns (state_dict, start_time_str) where state_dict has + Ra, Rb, Va, Vb (token units) and start_time_str is rounded + to the nearest minute for alignment with minute-level price data. + """ + df = pd.read_csv(WORLD_STATE_CSV) + r = df.iloc[0] + state = { + "Ra": float(r.balance_0), + "Rb": float(r.balance_1), + "Va": float(r.virtual_0), + "Vb": float(r.virtual_1), + } + # Round to nearest minute for price data alignment + ts_sec = int(r.timestamp) + ts_minute = (ts_sec // 60) * 60 + start_str = datetime.utcfromtimestamp(ts_minute).strftime("%Y-%m-%d %H:%M:%S") + return state, start_str + + +def load_world_timestamps(): + """Load Unix timestamps (seconds) from the world state CSV.""" + df = pd.read_csv(WORLD_STATE_CSV) + return df["timestamp"].values + + +def load_world_normalized_balances(): + """Load BPT-normalized on-chain balances and timestamps. + + Normalizes balances to initial BPT supply so that value tracks a + fixed LP position (accounts for joins/exits changing BPT supply). + + Returns (norm_bal_0, norm_bal_1, timestamps_sec). + """ + df = pd.read_csv(WORLD_STATE_CSV) + bpt_0 = df["bpt_supply"].iloc[0] + norm = bpt_0 / df["bpt_supply"].values + return ( + df["balance_0"].values * norm, + df["balance_1"].values * norm, + df["timestamp"].values, + ) + + +def sample_at_timestamps(minute_vals, start_unix_sec, timestamps_sec): + """Sample a minute-level array at specific Unix timestamps. + + For each target timestamp, finds the nearest minute index in the + sim output and returns the corresponding value. + + Parameters + ---------- + minute_vals : array, shape (N,) + Minute-level sim output. + start_unix_sec : float + Unix timestamp (seconds) of minute_vals[0]. + timestamps_sec : array + Unix timestamps (seconds) to sample at. + + Returns + ------- + array : values at the nearest minute to each target timestamp. + """ + indices = np.round((timestamps_sec - start_unix_sec) / 60).astype(int) + indices = np.clip(indices, 0, len(minute_vals) - 1) + return minute_vals[indices] + + +def run_pool(tokens, start, end, rule, fees, params, gas_cost=0.0, + protocol_fee_split=0.0, gas_cost_df=None, + onchain_initial_state=None): + """Run a quantammsim pool and return minute-level results. + + Returns (val_eth, price_ratio, start_unix_sec) where val_eth and + price_ratio are minute-level arrays and start_unix_sec is the Unix + timestamp (seconds) of the first element. + """ + fp = { + "tokens": tokens, + "rule": rule, + "startDateString": start, + "endDateString": end, + "initial_pool_value": INITIAL_POOL_VALUE, + "fees": fees, + "gas_cost": gas_cost, + "arb_fees": 0.0, + "do_arb": True, + "arb_frequency": 1, + "chunk_period": 1440, + "weight_interpolation_period": 1440, + } + if rule == "reclamm": + fp["reclamm_use_shift_exponent"] = True + fp["reclamm_interpolation_method"] = "geometric" + fp["reclamm_centeredness_scaling"] = False + if protocol_fee_split != 0.0: + fp["protocol_fee_split"] = protocol_fee_split + if onchain_initial_state is not None: + fp["reclamm_initial_state"] = onchain_initial_state + + result = do_run_on_historic_data( + run_fingerprint=fp, params=params, gas_cost_df=gas_cost_df, + ) + + # Prices: sorted tokens → [AAVE, ETH] in USD + prices = np.array(result["prices"]) + eth_usd = prices[:, 1] + price_ratio = prices[:, 0] / prices[:, 1] # WETH/AAVE + + # Pool value in ETH + val_eth = np.array(result["value"]) / eth_usd + + # Compute start timestamp from startDateString + start_unix_sec = datetime.strptime( + start, "%Y-%m-%d %H:%M:%S" + ).replace(tzinfo=timezone.utc).timestamp() + + return val_eth, price_ratio, start_unix_sec + + +def load_gas_csv(percentile): + """Load a gas CSV and return a DataFrame with columns [unix, trade_gas_cost_usd]. + + Gas CSV timestamps are offset by ~59s from exact minutes. Round down + to the nearest minute so they align with the simulator's minute-level index. + """ + path = GAS_CSV_DIR / f"Gas_{percentile}.csv" + df = pd.read_csv(path) + df = df.rename(columns={"USD": "trade_gas_cost_usd"}) + df["unix"] = (df["unix"] // 60000) * 60000 # floor to minute boundary + return df + + +def run_gas_experiment(args): + """Run gas-experiment sweep and produce comparison plot.""" + tokens = args.tokens + start, end = args.start, args.end + + # ── Select params ───────────────────────────────────────────────── + if args.launch_params: + param_source = ONCHAIN_LAUNCH_PARAMS + param_label = "launch" + else: + param_source = ONCHAIN_CURRENT_PARAMS + param_label = "current" + pool_params = {k: jnp.array(v) for k, v in param_source.items()} + + # ── Baselines ────────────────────────────────────────────────────── + print("Running Balancer (zero-fee 50/50)...") + bal_params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + bal_eth_min, qsim_price_min, start_sec = run_pool( + tokens, start, end, "balancer", 0.0, bal_params, + ) + + print(f"Running reClAMM ({param_label} params, flat gas, no protocol fee)...") + reclamm_flat_min, _, _ = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + gas_cost=ARB_GAS_COST, + ) + + # ── Load world values from CSV ───────────────────────────────────── + print("Loading reclamm-simulations CSV...") + df = pd.read_csv(args.csv) + + # ── Gas percentile runs ──────────────────────────────────────────── + gas_results_min = {} + for pct in GAS_PERCENTILES: + print(f"Running reClAMM ({param_label} params, gas={pct}, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + gas_df = load_gas_csv(pct) + val_eth_min, _, _ = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df, + ) + gas_results_min[pct] = val_eth_min + + # ── Sample at world timestamps ──────────────────────────────────── + world_ts = load_world_timestamps() + n = min(len(df), len(world_ts)) + world_ts = world_ts[:n] + + bal_eth = sample_at_timestamps(bal_eth_min, start_sec, world_ts) + reclamm_flat_eth = sample_at_timestamps(reclamm_flat_min, start_sec, world_ts) + qsim_price = sample_at_timestamps(qsim_price_min, start_sec, world_ts) + gas_results = { + pct: sample_at_timestamps(v, start_sec, world_ts) + for pct, v in gas_results_min.items() + } + + csv_world = df["world"].values[:n] + csv_feeless = df["feeless weighted"].values[:n] + print(f" Aligned: {n} world-timestamp points") + t = np.arange(n) + + # Governance half-day index + gov_unix = datetime.strptime( + GOVERNANCE_DATE, "%Y-%m-%d" + ).replace(tzinfo=timezone.utc).timestamp() + gov_idx = np.searchsorted(world_ts, gov_unix) + + # ── Plot: relative to feeless weighted ───────────────────────────── + fig, (ax_price, ax_rel) = plt.subplots(2, 1, figsize=(14, 9), + gridspec_kw={"height_ratios": [1, 2]}) + + # Top: price + ax_price.plot(t, qsim_price, color="gray", alpha=0.6, linewidth=1) + ax_price.set_ylabel("AAVE/ETH") + ax_price.set_title("Price") + ax_price.set_ylim(bottom=0) + if gov_idx < n: + ax_price.axvline(x=gov_idx, color="gray", linestyle=":", alpha=0.6) + + # Bottom: relative values + ax_rel.axhline(y=1.0, color="blue", linewidth=2, label="feeless weighted") + + # Flat-gas baseline (no protocol fee) + flat_rel = reclamm_flat_eth / bal_eth + ax_rel.plot(t, flat_rel, linewidth=2, color="gray", linestyle="--", + label=f"flat gas ${ARB_GAS_COST}, no protocol fee") + + # Gas percentile runs + colors = {"50p": "#2ca02c", "75p": "#ff7f0e", "90p": "#d62728", "95p": "#9467bd"} + for pct in GAS_PERCENTILES: + vals = gas_results[pct] + rel = vals / bal_eth + ax_rel.plot(t, rel, linewidth=1.5, color=colors[pct], + label=f"gas {pct}, {int(PROTOCOL_FEE_SPLIT*100)}% protocol fee") + + # World values + world_rel = csv_world / csv_feeless + ax_rel.plot(t, world_rel, linewidth=1.5, marker=".", markersize=2, + color="brown", label="world (on-chain)") + + ax_rel.set_xlabel("half days") + ax_rel.set_ylabel("value / feeless weighted") + ax_rel.set_title("LP value relative to feeless weighted (Balancer 50/50)") + ax_rel.legend(fontsize=8, loc="lower left") + ax_rel.grid(True, alpha=0.2) + if gov_idx < n: + ax_rel.axvline(x=gov_idx, color="gray", linestyle=":", alpha=0.6) + ax_rel.text(gov_idx + 1, ax_rel.get_ylim()[1] * 0.98, + "governance", fontsize=7, color="gray", va="top") + + tokens_str = "/".join(tokens) + fig.suptitle( + f"reClAMM gas experiment ({param_label} params) — {tokens_str}\n" + f"params: {list(param_source.values())}, " + f"fees: {ONCHAIN_FEES}, protocol fee: {PROTOCOL_FEE_SPLIT}", + fontsize=10, + ) + plt.tight_layout() + out = args.output.replace(".png", f"_gas_experiment_{param_label}.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + print(f"\nSaved: {out}") + plt.close() + + # ── Summary table ────────────────────────────────────────────────── + print(f"\n{'Scenario':<45} {'Final rel':>10} {'vs world':>10}") + print("-" * 65) + world_final_rel = world_rel[-1] if len(world_rel) > 0 else float("nan") + print(f"{'Flat gas, no protocol fee':<45} {flat_rel[-1]:>10.4f} " + f"{flat_rel[-1] - world_final_rel:>+10.4f}") + for pct in GAS_PERCENTILES: + rel = gas_results[pct] / bal_eth + print(f"{'Gas ' + pct + f', {int(PROTOCOL_FEE_SPLIT*100)}% protocol fee':<45} " + f"{rel[-1]:>10.4f} {rel[-1] - world_final_rel:>+10.4f}") + print(f"{'World (on-chain)':<45} {world_final_rel:>10.4f}") + + +def run_gas_scale_experiment(args): + """Sweep gas cost scale factors, rebase to world, truncate at governance.""" + tokens = args.tokens + end = args.end + + if args.launch_params: + param_source = ONCHAIN_LAUNCH_PARAMS + param_label = "launch" + else: + param_source = ONCHAIN_CURRENT_PARAMS + param_label = "current" + pool_params = {k: jnp.array(v) for k, v in param_source.items()} + + # Load on-chain initial state and derive start time + onchain_state, onchain_start = load_onchain_initial_state() + start = onchain_start + print(f"On-chain initial state: Ra={onchain_state['Ra']:.2f}, " + f"Rb={onchain_state['Rb']:.2f}, Va={onchain_state['Va']:.2f}, " + f"Vb={onchain_state['Vb']:.2f}") + print(f"Sim start time (from on-chain): {start}") + + # Load world + reclamm-simulations values + print("Loading reclamm-simulations CSV...") + df = pd.read_csv(args.csv) + + # Load world timestamps and find governance cutoff + world_ts = load_world_timestamps() + gov_unix = datetime.strptime( + GOVERNANCE_DATE, "%Y-%m-%d" + ).replace(tzinfo=timezone.utc).timestamp() + gov_idx = np.searchsorted(world_ts, gov_unix) + + # Run all (percentile, scale) combinations + results_min = {} + price_ratio_min = None + for pct in GAS_PERCENTILES: + gas_df_raw = load_gas_csv(pct) + for scale in GAS_SCALE_FACTORS: + label = f"{pct} × {scale}" + print(f"Running reClAMM ({param_label}, gas={label}, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + gas_df = gas_df_raw.copy() + gas_df["trade_gas_cost_usd"] = gas_df_raw["trade_gas_cost_usd"] * scale + val_eth_min, pr_min, start_sec = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df, + onchain_initial_state=onchain_state, + ) + results_min[(pct, scale)] = (val_eth_min, start_sec) + if price_ratio_min is None: + price_ratio_min = pr_min + + # Flat gas cost runs + flat_results_min = {} + for gas_usd in FLAT_GAS_USD: + print(f"Running reClAMM ({param_label}, flat gas=${gas_usd}, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + val_eth_min, _, start_sec = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + gas_cost=gas_usd, protocol_fee_split=PROTOCOL_FEE_SPLIT, + onchain_initial_state=onchain_state, + ) + flat_results_min[gas_usd] = (val_eth_min, start_sec) + + # ── World values: on-chain balances × quantammsim prices ────────── + world_bal_0, world_bal_1, world_ts = load_world_normalized_balances() + + # reclamm-sim comparison uses its own CSV (self-consistent pricing) + csv_world = df["world"].values + csv_sim = df["simulation"].values + + n = min(gov_idx, len(world_bal_0), len(csv_world), len(csv_sim), len(world_ts)) + print(f" Truncated at governance: {n} world-timestamp points") + t = np.arange(n) + world_ts_trunc = world_ts[:n] + + # Repriced world for quantammsim comparison + price_at_world = sample_at_timestamps( + price_ratio_min, start_sec, world_ts_trunc, + ) + world_val = world_bal_0[:n] * price_at_world + world_bal_1[:n] + world_growth = world_val / world_val[0] + + # CSV-based world for reclamm-sim comparison (self-consistent pricing) + csv_world = csv_world[:n] + csv_sim = csv_sim[:n] + world_growth_csv = csv_world / csv_world[0] + recsim_growth = csv_sim / csv_sim[0] + + # Sample all sim runs at world timestamps + start_sec = flat_results_min[FLAT_GAS_USD[0]][1] + + results = {} + for key, (val_min, _) in results_min.items(): + results[key] = sample_at_timestamps(val_min, start_sec, world_ts_trunc) + + flat_results = {} + for gas_usd, (val_min, _) in flat_results_min.items(): + flat_results[gas_usd] = sample_at_timestamps(val_min, start_sec, world_ts_trunc) + + # Compute growth ratios + flat_growths = {} + for gas_usd in FLAT_GAS_USD: + vals = flat_results[gas_usd] + flat_growths[gas_usd] = vals / vals[0] + + # ── Plot (% deviation from world: positive = sim below world) ─── + fig, (ax_ts, ax_pct, ax_flat) = plt.subplots( + 1, 3, figsize=(20, 7), gridspec_kw={"width_ratios": [3, 1, 1]}, + ) + + # Left: time series of % deviation from world + ax_ts.axhline(y=0.0, color="brown", linewidth=2, label="world (on-chain)") + + # reclamm-simulations (uses CSV-based world for self-consistent pricing) + recsim_dev = (1 - recsim_growth / world_growth_csv) * 100 + ax_ts.plot(t, recsim_dev, color="red", linewidth=2, + linestyle="--", label="reclamm-sim") + + # Gas scale sweep (percentile-based) + colors = {"50p": "#2ca02c", "75p": "#ff7f0e", "90p": "#d62728", "95p": "#9467bd"} + for pct in GAS_PERCENTILES: + for scale in GAS_SCALE_FACTORS: + vals = results[(pct, scale)] + sim_growth = vals / vals[0] + dev = (1 - sim_growth / world_growth) * 100 + alpha = 0.3 + 0.7 * scale + lw = 0.8 + 1.2 * scale + if scale == 1.0: + label = f"{pct} × {scale}" + elif pct == "50p": + label = f"50p × {scale}" + else: + label = None + ax_ts.plot(t, dev, color=colors[pct], alpha=alpha, + linewidth=lw, label=label) + + # Flat gas runs + flat_cmap = plt.cm.copper + for i, gas_usd in enumerate(FLAT_GAS_USD): + c = flat_cmap(i / max(len(FLAT_GAS_USD) - 1, 1)) + dev = (1 - flat_growths[gas_usd] / world_growth) * 100 + ax_ts.plot(t, dev, color=c, linewidth=1.5, linestyle="-.", + label=f"flat ${gas_usd}") + + ax_ts.set_xlabel("half days") + ax_ts.set_ylabel("% deviation from world") + ax_ts.set_title("LP value vs world (pre-governance)") + ax_ts.legend(fontsize=6, loc="best", ncol=2) + ax_ts.grid(True, alpha=0.2) + + # Reference lines for both summary panels (as % deviation) + recsim_final_dev = (1 - recsim_growth[-1] / world_growth_csv[-1]) * 100 + + # Middle: final % deviation vs percentile scale factor + ax_pct.axhline(y=0.0, color="brown", linewidth=2, label="world") + ax_pct.axhline(y=recsim_final_dev, color="red", linewidth=1.5, + linestyle="--", label=f"reclamm-sim ({recsim_final_dev:+.2f}%)") + + for pct in GAS_PERCENTILES: + finals = [] + for scale in GAS_SCALE_FACTORS: + vals = results[(pct, scale)] + sim_growth = vals / vals[0] + finals.append((1 - sim_growth[-1] / world_growth[-1]) * 100) + ax_pct.plot(GAS_SCALE_FACTORS, finals, marker="o", + color=colors[pct], linewidth=2, label=pct) + + ax_pct.set_xlabel("gas scale factor\n(1.0 = 450k gas)") + ax_pct.set_ylabel("% deviation from world") + ax_pct.set_title("Percentile gas") + ax_pct.legend(fontsize=6) + ax_pct.grid(True, alpha=0.2) + + # Right: final % deviation vs flat gas cost + ax_flat.axhline(y=0.0, color="brown", linewidth=2, label="world") + ax_flat.axhline(y=recsim_final_dev, color="red", linewidth=1.5, + linestyle="--", label=f"reclamm-sim ({recsim_final_dev:+.2f}%)") + + flat_finals = [] + for gas_usd in FLAT_GAS_USD: + flat_finals.append( + (1 - flat_growths[gas_usd][-1] / world_growth[-1]) * 100 + ) + ax_flat.plot(FLAT_GAS_USD, flat_finals, marker="s", color="black", + linewidth=2, label="flat gas") + + ax_flat.set_xlabel("flat gas cost (USD)") + ax_flat.set_ylabel("% deviation from world") + ax_flat.set_title("Flat gas") + ax_flat.legend(fontsize=6) + ax_flat.grid(True, alpha=0.2) + + tokens_str = "/".join(tokens) + fig.suptitle( + f"reClAMM gas sweep ({param_label} params) — {tokens_str}\n" + f"params: {list(param_source.values())}, " + f"fees: {ONCHAIN_FEES}, protocol fee: {PROTOCOL_FEE_SPLIT}", + fontsize=10, + ) + plt.tight_layout() + out = args.output.replace(".png", f"_gas_scale_{param_label}.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + print(f"\nSaved: {out}") + plt.close() + + # ── Summary table (% deviation from world) ───────────────────────── + print(f"\n{'Scenario':<35} {'% dev from world':>16}") + print("-" * 52) + print(f"{'reclamm-sim':<35} {recsim_final_dev:>+16.2f}%") + print() + for gas_usd in FLAT_GAS_USD: + dev = (1 - flat_growths[gas_usd][-1] / world_growth[-1]) * 100 + print(f"{'Flat $' + f'{gas_usd}':<35} {dev:>+16.2f}%") + print() + for pct in GAS_PERCENTILES: + for scale in GAS_SCALE_FACTORS: + vals = results[(pct, scale)] + sim_growth = vals / vals[0] + dev = (1 - sim_growth[-1] / world_growth[-1]) * 100 + print(f"{'Gas ' + pct + f' × {scale}':<35} {dev:>+16.2f}%") + + +def run_best_gas_experiment(args): + """Run the 3 best gas configs vs world on a clean single-panel plot.""" + tokens = args.tokens + end = args.end + + if args.launch_params: + param_source = ONCHAIN_LAUNCH_PARAMS + param_label = "launch" + else: + param_source = ONCHAIN_CURRENT_PARAMS + param_label = "current" + pool_params = {k: jnp.array(v) for k, v in param_source.items()} + + # Load on-chain initial state and derive start time + onchain_state, onchain_start = load_onchain_initial_state() + start = onchain_start + print(f"On-chain initial state: Ra={onchain_state['Ra']:.2f}, " + f"Rb={onchain_state['Rb']:.2f}, Va={onchain_state['Va']:.2f}, " + f"Vb={onchain_state['Vb']:.2f}") + print(f"Sim start time (from on-chain): {start}") + + # Find governance cutoff from world timestamps + world_ts_all = load_world_timestamps() + gov_unix = datetime.strptime( + GOVERNANCE_DATE, "%Y-%m-%d" + ).replace(tzinfo=timezone.utc).timestamp() + gov_idx = np.searchsorted(world_ts_all, gov_unix) + + # ── The best configs ───────────────────────────────────────────── + configs = [ + ("Flat $1.00", "black", "-"), + ("50p × 1.0", "#2ca02c", "-"), + ("75p × 0.75", "#ff7f0e", "-"), + ("90p × 0.25", "#d62728", "-"), + ] + + # 1) Flat $1.00 + print(f"Running reClAMM ({param_label}, flat gas=$1.00, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + flat1_min, price_ratio_min, start_sec = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + gas_cost=1.0, protocol_fee_split=PROTOCOL_FEE_SPLIT, + onchain_initial_state=onchain_state, + ) + + # 2) 50p × 1.0 + print(f"Running reClAMM ({param_label}, gas=50p × 1.0, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + gas_df_50p = load_gas_csv("50p") + g50_min, _, _ = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_50p, + onchain_initial_state=onchain_state, + ) + + # 3) 75p × 0.75 + print(f"Running reClAMM ({param_label}, gas=75p × 0.75, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + gas_df_75p = load_gas_csv("75p") + gas_df_75p_scaled = gas_df_75p.copy() + gas_df_75p_scaled["trade_gas_cost_usd"] *= 0.75 + g75_min, _, _ = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_75p_scaled, + onchain_initial_state=onchain_state, + ) + + # 4) 90p × 0.25 + print(f"Running reClAMM ({param_label}, gas=90p × 0.25, " + f"protocol_fee={PROTOCOL_FEE_SPLIT})...") + gas_df_90p = load_gas_csv("90p") + gas_df_90p_scaled = gas_df_90p.copy() + gas_df_90p_scaled["trade_gas_cost_usd"] *= 0.25 + g90_min, _, _ = run_pool( + tokens, start, end, "reclamm", ONCHAIN_FEES, pool_params, + protocol_fee_split=PROTOCOL_FEE_SPLIT, gas_cost_df=gas_df_90p_scaled, + onchain_initial_state=onchain_state, + ) + + # ── World values: on-chain balances × quantammsim prices ────────── + # Both sim and world valued at the same price at each point, + # so price fluctuations cancel in the growth ratio comparison. + world_bal_0, world_bal_1, world_ts = load_world_normalized_balances() + n = min(gov_idx, len(world_bal_0), len(world_ts)) + print(f" Truncated at governance: {n} world-timestamp points") + t = np.arange(n) + world_ts_trunc = world_ts[:n] + + # Sample quantammsim price ratio at world timestamps + price_at_world = sample_at_timestamps( + price_ratio_min, start_sec, world_ts_trunc, + ) + # World value in ETH = norm_AAVE * (AAVE/ETH) + norm_ETH + world_val = world_bal_0[:n] * price_at_world + world_bal_1[:n] + world_growth = world_val / world_val[0] + + run_vals = [ + sample_at_timestamps(flat1_min, start_sec, world_ts_trunc), + sample_at_timestamps(g50_min, start_sec, world_ts_trunc), + sample_at_timestamps(g75_min, start_sec, world_ts_trunc), + sample_at_timestamps(g90_min, start_sec, world_ts_trunc), + ] + + growths = [v / v[0] for v in run_vals] + + # ── Plot ────────────────────────────────────────────────────────── + fig, ax = plt.subplots(figsize=(14, 6)) + + ax.axhline(y=0.0, color="brown", linewidth=2, label="world (on-chain)") + + # Best 3 + for (label, color, ls), g in zip(configs, growths): + dev = (1 - g / world_growth) * 100 + final_dev = dev[-1] + ax.plot(t, dev, color=color, linewidth=2, linestyle=ls, + label=f"{label} (final {final_dev:+.2f}%)") + + ax.set_xlabel("half days") + ax.set_ylabel("% deviation from world") + ax.set_title( + f"Best gas configs vs world ({param_label} params) — " + f"{'/'.join(tokens)}\n" + f"params: {list(param_source.values())}, " + f"fees: {ONCHAIN_FEES}, protocol fee: {PROTOCOL_FEE_SPLIT}", + ) + ax.legend(fontsize=9, loc="best") + ax.grid(True, alpha=0.3) + + plt.tight_layout() + out = args.output.replace(".png", f"_best_gas_{param_label}.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + print(f"\nSaved: {out}") + plt.close() + + # Summary + labels = [c[0] for c in configs] + print(f"\n{'Scenario':<25} {'% dev from world':>16}") + print("-" * 42) + for label, g in zip(labels, growths): + dev = (1 - g[-1] / world_growth[-1]) * 100 + print(f"{label:<25} {dev:>+16.2f}%") + + +def main(): + args = parse_args() + + if args.best_gas: + run_best_gas_experiment(args) + return + + if args.gas_scale_sweep: + run_gas_scale_experiment(args) + return + + if args.gas_experiment: + run_gas_experiment(args) + return + + # ── Load CSVs ───────────────────────────────────────────────────── + print("Loading reclamm-simulations CSV...") + df = pd.read_csv(args.csv) + n_csv = len(df) + print(f" {n_csv} half-day points") + + print("Loading zero-fee minute-level CSV...") + df_zf_min = pd.read_csv(ZEROFEE_MINUTE_CSV) + print(f" {len(df_zf_min)} minute points") + + # Load world timestamps for alignment + world_ts = load_world_timestamps() + + # ── Run quantammsim pools (minute-level) ────────────────────────── + print("Running Balancer (zero-fee 50/50)...") + bal_params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + bal_eth_min, qsim_price_min, start_sec = run_pool( + args.tokens, args.start, args.end, "balancer", 0.0, bal_params, + ) + + print("Running reClAMM (launch, zero-fee, zero-gas)...") + launch_params = {k: jnp.array(v) for k, v in ONCHAIN_LAUNCH_PARAMS.items()} + reclamm_zerofee_min, _, _ = run_pool( + args.tokens, args.start, args.end, "reclamm", 0.0, launch_params, + gas_cost=0.0, + ) + + print(f"Running reClAMM (launch params, gas=${ARB_GAS_COST})...") + reclamm_launch_min, _, _ = run_pool( + args.tokens, args.start, args.end, "reclamm", ONCHAIN_FEES, launch_params, + gas_cost=ARB_GAS_COST, + ) + + print(f"Running reClAMM (current params, gas=${ARB_GAS_COST})...") + current_params = {k: jnp.array(v) for k, v in ONCHAIN_CURRENT_PARAMS.items()} + reclamm_current_min, _, _ = run_pool( + args.tokens, args.start, args.end, "reclamm", ONCHAIN_FEES, current_params, + gas_cost=ARB_GAS_COST, + ) + + # ── Sample at world timestamps ──────────────────────────────────── + n = min(n_csv, len(world_ts)) + world_ts_trunc = world_ts[:n] + + bal_eth = sample_at_timestamps(bal_eth_min, start_sec, world_ts_trunc) + reclamm_zerofee_eth = sample_at_timestamps(reclamm_zerofee_min, start_sec, world_ts_trunc) + reclamm_launch_eth = sample_at_timestamps(reclamm_launch_min, start_sec, world_ts_trunc) + reclamm_current_eth = sample_at_timestamps(reclamm_current_min, start_sec, world_ts_trunc) + qsim_price = sample_at_timestamps(qsim_price_min, start_sec, world_ts_trunc) + + print(f" Aligned: {n} world-timestamp points " + f"(qsim minutes={len(bal_eth_min)}, csv={n_csv})") + t = np.arange(n) + + csv_price = df["price"].values[:n] + csv_feeless = df["feeless weighted"].values[:n] + csv_sim = df["simulation"].values[:n] + csv_hold = df["hold"].values[:n] + csv_world = df["world"].values[:n] + + # Governance change index + gov_unix = datetime.strptime( + GOVERNANCE_DATE, "%Y-%m-%d" + ).replace(tzinfo=timezone.utc).timestamp() + gov_idx = np.searchsorted(world_ts_trunc, gov_unix) + + # Normalize quantammsim to same starting value as CSV + v0 = csv_feeless[0] + bal_norm = bal_eth * (v0 / bal_eth[0]) + zerofee_norm = reclamm_zerofee_eth * (v0 / reclamm_zerofee_eth[0]) + launch_norm = reclamm_launch_eth * (v0 / reclamm_launch_eth[0]) + current_norm = reclamm_current_eth * (v0 / reclamm_current_eth[0]) + + # Relative values (÷ respective feeless weighted baseline) + zerofee_rel = reclamm_zerofee_eth / bal_eth + launch_rel = reclamm_launch_eth / bal_eth + current_rel = reclamm_current_eth / bal_eth + csv_sim_rel = csv_sim / csv_feeless + csv_hold_rel = csv_hold / csv_feeless + csv_world_rel = csv_world / csv_feeless + + # ── Plot ────────────────────────────────────────────────────────── + fig, axs = plt.subplots(2, 2, figsize=(13, 8)) + + # Top-left: price + axs[0][0].plot(t, csv_price, label="reclamm-sim", alpha=0.8) + axs[0][0].plot(t, qsim_price, label="quantammsim", alpha=0.8, linestyle="--") + axs[0][0].set_ylabel("WETH/AAVE") + axs[0][0].set_title("Price") + axs[0][0].set_ylim(bottom=0) + axs[0][0].legend(fontsize=8) + if gov_idx < n: + axs[0][0].axvline(x=gov_idx, color="gray", linestyle=":", alpha=0.6) + + # Top-right: remove (legend is on other panels) + axs[0][1].remove() + + # Bottom-left: absolute values in WETH + axs[1][0].plot(t, bal_norm, label="qsim feeless weighted", linewidth=2, color="blue") + axs[1][0].plot(t, launch_norm, label="qsim reClAMM (launch)", linewidth=2, color="orange") + axs[1][0].plot(t, current_norm, label="qsim reClAMM (current)", linewidth=2, + color="purple", linestyle="-.") + axs[1][0].plot(t, csv_sim, label="reclamm-sim simulation", linewidth=1.5, + linestyle="--", color="red") + axs[1][0].plot(t, csv_hold, label="hold", linewidth=1.5, color="green") + axs[1][0].plot(t, csv_world, label="world values", linewidth=1.5, + marker=".", markersize=2, color="brown") + axs[1][0].set_title("Value histories") + axs[1][0].set_xlabel("half days") + axs[1][0].set_ylabel("Value in WETH") + axs[1][0].set_ylim(bottom=0) + axs[1][0].legend(fontsize=7, loc="upper right") + if gov_idx < n: + axs[1][0].axvline(x=gov_idx, color="gray", linestyle=":", alpha=0.6) + axs[1][0].text(gov_idx + 1, axs[1][0].get_ylim()[1] * 0.95, + "governance", fontsize=7, color="gray", va="top") + + # Bottom-right: relative to feeless weighted + axs[1][1].axhline(y=1.0, color="blue", linewidth=2, label="feeless weighted") + axs[1][1].plot(t, launch_rel, label="qsim reClAMM (launch)", linewidth=2, color="orange") + axs[1][1].plot(t, current_rel, label="qsim reClAMM (current)", linewidth=2, + color="purple", linestyle="-.") + axs[1][1].plot(t, csv_sim_rel, label="reclamm-sim simulation", linewidth=1.5, + linestyle="--", color="red") + axs[1][1].plot(t, csv_hold_rel, label="hold", linewidth=1.5, color="green") + axs[1][1].plot(t, csv_world_rel, label="world values", linewidth=1.5, + marker=".", markersize=2, color="brown") + axs[1][1].set_title("Value relative to feeless weighted") + axs[1][1].set_xlabel("half days") + axs[1][1].set_ylabel("relative value") + axs[1][1].legend(fontsize=7, loc="lower left") + if gov_idx < n: + axs[1][1].axvline(x=gov_idx, color="gray", linestyle=":", alpha=0.6) + + tokens_str = "/".join(args.tokens) + fig.suptitle( + f"quantammsim vs reclamm-simulations — {tokens_str}\n" + f"Launch: {list(ONCHAIN_LAUNCH_PARAMS.values())}, " + f"Current: {list(ONCHAIN_CURRENT_PARAMS.values())}, " + f"fees: {ONCHAIN_FEES}", + fontsize=10, + ) + plt.tight_layout() + plt.savefig(args.output, dpi=150, bbox_inches="tight") + print(f"\nSaved: {args.output}") + + # ── Zero-fee comparison plot (minute-level) ─────────────────────── + # Revalue reclamm-sim balances at quantammsim's price so both sides + # use the same price and the comparison is purely about balances. + # Skip row 0 of the CSV (initial state before first arb) to align + # with quantammsim's reserves[0] which is post-first-step. + ext_bal_0 = df_zf_min["balance_0"].values[1:] + ext_bal_1 = df_zf_min["balance_1"].values[1:] + n_zf = min(len(reclamm_zerofee_min), len(ext_bal_0), len(qsim_price_min)) + ext_val_repriced = ( + ext_bal_0[:n_zf] * qsim_price_min[:n_zf] + ext_bal_1[:n_zf] + ) + qsim_growth = reclamm_zerofee_min[:n_zf] / reclamm_zerofee_min[0] + ext_growth = ext_val_repriced[:n_zf] / ext_val_repriced[0] + pct_dev = (qsim_growth / ext_growth - 1) * 100 + days = np.arange(n_zf) / 1440 + + zerofee_title = ( + f"Zero-fee zero-gas reClAMM: quantammsim / reclamm-sim (minute-level) — {tokens_str}\n" + f"params: {list(ONCHAIN_LAUNCH_PARAMS.values())}" + ) + daily_smooth = pd.Series(pct_dev).rolling(1440, center=True, min_periods=720).mean() + + # Plot 1: with daily smoothing overlay + fig2, ax2 = plt.subplots(figsize=(12, 5)) + ax2.plot(days, pct_dev, linewidth=0.5, color="teal", alpha=0.6) + ax2.plot(days, daily_smooth, linewidth=2, color="darkblue", label="daily smoothed") + ax2.axhline(y=0.0, color="gray", linestyle="--", alpha=0.6) + ax2.set_xlabel("days") + ax2.set_ylabel("deviation (%)") + ax2.set_title(zerofee_title, fontsize=11) + ax2.legend(fontsize=9) + ax2.grid(True, alpha=0.3) + plt.tight_layout() + zerofee_path = args.output.replace(".png", "_zerofee_ratio.png") + plt.savefig(zerofee_path, dpi=150, bbox_inches="tight") + print(f"Saved: {zerofee_path}") + plt.close() + + # Plot 2: raw minute-level only (no smoothing) + fig3, ax3 = plt.subplots(figsize=(12, 5)) + ax3.plot(days, pct_dev, linewidth=0.5, color="teal", alpha=0.8) + ax3.axhline(y=0.0, color="gray", linestyle="--", alpha=0.6) + ax3.set_xlabel("days") + ax3.set_ylabel("deviation (%)") + ax3.set_title(zerofee_title, fontsize=11) + ax3.grid(True, alpha=0.3) + plt.tight_layout() + zerofee_raw_path = args.output.replace(".png", "_zerofee_ratio_raw.png") + plt.savefig(zerofee_raw_path, dpi=150, bbox_inches="tight") + print(f"Saved: {zerofee_raw_path}") + plt.close() + + +if __name__ == "__main__": + main() From 427044a17e51f2bcc6495ffce5d25b17b27272d5 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:51:35 +0000 Subject: [PATCH 013/115] add calibration to pyproject --- pyproject.toml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/pyproject.toml b/pyproject.toml index 413faf9e..8859b551 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,6 +45,10 @@ docs = [ "sphinx-automodapi", "sphinx-rtd-theme", ] +calibration = [ + "numpyro>=0.15.0", + "arviz>=0.15.0", +] [tool.hatch.build.targets.wheel] packages = [ From fa39d72fa28ed3d1f3f285164489907f504b8e3d Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:51:47 +0000 Subject: [PATCH 014/115] core noise model in base pool updates --- quantammsim/pools/base_pool.py | 106 ++++++++++++++++++- quantammsim/pools/noise_trades.py | 164 ++++++++++++++++++++++++++++++ 2 files changed, 266 insertions(+), 4 deletions(-) diff --git a/quantammsim/pools/base_pool.py b/quantammsim/pools/base_pool.py index d8c6d759..01845dbb 100644 --- a/quantammsim/pools/base_pool.py +++ b/quantammsim/pools/base_pool.py @@ -1,11 +1,12 @@ from abc import ABC, abstractmethod -from typing import Dict, Any, Optional +from typing import Dict, Any, Optional, Tuple +from functools import partial import numpy as np import jax.numpy as jnp from jax.nn import softmax -from jax.lax import stop_gradient -from jax import tree_util +from jax.lax import stop_gradient, dynamic_slice +from jax import tree_util, jit, vmap from quantammsim.core_simulator.param_utils import make_vmap_in_axes_dict @@ -93,7 +94,11 @@ def calculate_reserves_with_dynamic_inputs( run_fingerprint: Dict[str, Any], prices: jnp.ndarray, start_index: jnp.ndarray, - dynamic_inputs: Any, + fees_array: jnp.ndarray, + arb_thresh_array: jnp.ndarray, + arb_fees_array: jnp.ndarray, + trade_array: jnp.ndarray, + lp_supply_array: jnp.ndarray = None, additional_oracle_input: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: pass @@ -278,6 +283,99 @@ def add_noise( params[key] = jnp.array(params[key]) return params + @partial(jit, static_argnums=(2, 3)) + def calculate_volatility_array(self, prices, run_fingerprint, subsample_freq=5): + """Annualised daily realised volatility broadcast to minute-level array. + + Pure-JAX implementation (vmap + dynamic_slice) — JIT-compatible and + callable from within traced contexts (e.g. forward_pass). + + Parameters + ---------- + prices : jnp.ndarray, shape (T, 2) + Minute-level prices for two tokens. + run_fingerprint : dict + Must contain ``tokens`` and ``numeraire`` for ordering. + subsample_freq : int + Subsample within each day to reduce microstructure noise. + + Returns + ------- + jnp.ndarray, shape (T,) + Annualised volatility, constant within each day. + """ + ordered_prices, needs_swap = self._handle_numeraire_ordering( + prices, run_fingerprint, + ) + asset_prices = ordered_prices[:, 0] / ordered_prices[:, 1] + n_minutes = len(asset_prices) + + # Guard: need at least one full day for vmap + dynamic_slice + if n_minutes < 1440: + return jnp.full(n_minutes, 0.1) * jnp.sqrt(365.0) + + n_days = n_minutes // 1440 + + def calculate_daily_volatility(day_idx): + start_idx = day_idx * 1440 + window_prices = dynamic_slice(asset_prices, [start_idx], [1440]) + subsampled_prices = window_prices[::subsample_freq] + log_prices = jnp.log(jnp.maximum(subsampled_prices, 1e-8)) + returns = jnp.diff(log_prices) + num_nonzero_returns = jnp.sum(returns != 0) + total_returns = len(returns) + adjusted_variance = ( + num_nonzero_returns * jnp.var(returns) / total_returns + ) + dt = subsample_freq / 1440 + vol = jnp.sqrt(adjusted_variance) / jnp.sqrt(dt) + return vol + + daily_volatilities = vmap(calculate_daily_volatility)(jnp.arange(n_days)) + volatility_array = jnp.repeat(daily_volatilities, 1440) + + remaining_minutes = n_minutes - len(volatility_array) + if remaining_minutes > 0: + last_vol = ( + daily_volatilities[-1] if len(daily_volatilities) > 0 else 0.1 + ) + volatility_array = jnp.concatenate( + [volatility_array, jnp.full(remaining_minutes, last_vol)] + ) + + return volatility_array * jnp.sqrt(365.0) + + @partial(jit, static_argnums=(2,)) + def _handle_numeraire_ordering( + self, + prices: jnp.ndarray, + run_fingerprint: Dict[str, Any], + ) -> Tuple[jnp.ndarray, bool]: + """Reorder prices so numeraire token is in second position. + + Parameters + ---------- + prices : jnp.ndarray, shape (..., 2) + Price array with two tokens. + run_fingerprint : dict + Must contain ``tokens`` (sorted) and ``numeraire``. + + Returns + ------- + (ordered_prices, needs_swap) : (jnp.ndarray, bool) + """ + tokens = sorted(run_fingerprint["tokens"]) + numeraire = run_fingerprint["numeraire"] + if numeraire is None or numeraire not in tokens: + numeraire = tokens[-1] + needs_swap = tokens.index(numeraire) == 0 + + if needs_swap: + ordered_prices = prices[..., ::-1] + else: + ordered_prices = prices + return ordered_prices, needs_swap + def _tree_flatten(self): children = () aux_data = dict() # static values diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index f78b75c1..21a574ce 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -110,3 +110,167 @@ def calculate_reserves_after_noise_trade( ) reserves = current_reserves * ratio_of_value_of_trade_to_reserves return reserves + + +@jit +def reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """reClAMM Tsoukalas sqrt model: effective TVL regressor. + + Predicts per-minute noise trader volume using: + V_daily = (a_0 - a_f*fee + a_sigma*sigma + + a_c*sqrt(c_eff/1e6)) * 1e6 + V_noise = max(0, V_daily/1440 - arb_volume_this_period) + + where c_eff = (Ra+Va)*pA + (Rb+Vb)*pB is the effective TVL (real + + virtual reserves valued in USD). For a concentrated liquidity pool, + effective reserves determine execution quality and routing decisions, + so they are the natural driver of noise volume. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Regression coefficients. Keys: a_0_base, a_f, a_sigma, + a_c, base_fee. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + a_0_base = noise_params.get("a_0_base", 0.5) + a_f = noise_params.get("a_f", 0.0) + a_sigma = noise_params.get("a_sigma", 2.0) + a_c = noise_params.get("a_c", 1.0) + base_fee = noise_params.get("base_fee", 0.003) + + fee = 1.0 - gamma + a_0 = a_0_base + base_fee * a_f + daily_vol = ( + a_0 - a_f * fee + + a_sigma * volatility + + a_c * jnp.sqrt(effective_value_usd / 1e6) + ) * 1e6 + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + +@jit +def reclamm_tsoukalas_log_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """reClAMM Tsoukalas log model: log(c_eff/1e6) instead of sqrt. + + Same specification as the sqrt variant but uses log regressor, + which may fit better for pools spanning a wide TVL range. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Regression coefficients (same keys as sqrt variant). + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + a_0_base = noise_params.get("a_0_base", 0.5) + a_f = noise_params.get("a_f", 0.0) + a_sigma = noise_params.get("a_sigma", 2.0) + a_c = noise_params.get("a_c", 1.0) + base_fee = noise_params.get("base_fee", 0.003) + + fee = 1.0 - gamma + a_0 = a_0_base + base_fee * a_f + daily_vol = ( + a_0 - a_f * fee + + a_sigma * volatility + + a_c * jnp.log(jnp.maximum(effective_value_usd / 1e6, 1e-30)) + ) * 1e6 + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + +@jit +def reclamm_loglinear_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """Loglinear noise volume from hierarchical cross-pool calibration. + + Predicts per-minute noise volume using: + log(V_daily) = b_0 + b_sigma * volatility + b_c * log(TVL) + V_noise = max(0, exp(log_daily_vol) / 1440 - arb_volume) + + where b_0 is a pool-specific intercept (BLUP from the hierarchical + model, absorbing chain, token tier, and fee effects), and b_sigma, + b_c are shared fixed effects estimated from cross-pool variation. + + Note: ``gamma`` is accepted for interface compatibility with the + other noise volume functions but is not used; fee effects are + absorbed into ``b_0`` via the hierarchical model's BLUP. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). Unused — kept for uniform + calling convention across noise models. + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Hierarchical model coefficients. Keys: b_0, b_sigma, b_c. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + b_0 = noise_params.get("b_0", -6.7) + b_sigma = noise_params.get("b_sigma", -0.0007) + b_c = noise_params.get("b_c", 1.04) + + log_daily_vol = ( + b_0 + + b_sigma * volatility + + b_c * jnp.log(jnp.maximum(effective_value_usd, 1.0)) + ) + daily_vol = jnp.exp(log_daily_vol) + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + From fde87af188ddf061506898ca2361b3ab5e664cee Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:56:19 +0000 Subject: [PATCH 015/115] add noise tests --- tests/noise/__init__.py | 0 tests/noise/conftest.py | 274 ++++++++++++ tests/noise/test_covariate_encoding.py | 326 ++++++++++++++ tests/noise/test_formula_arb.py | 142 ++++++ tests/noise/test_model_and_inference.py | 328 ++++++++++++++ tests/noise/test_model_dp_sigma.py | 305 +++++++++++++ tests/noise/test_model_structural.py | 215 +++++++++ tests/noise/test_output.py | 303 +++++++++++++ tests/noise/test_panel_assembly.py | 442 +++++++++++++++++++ tests/noise/test_postprocessing.py | 533 +++++++++++++++++++++++ tests/noise/test_token_classification.py | 66 +++ 11 files changed, 2934 insertions(+) create mode 100644 tests/noise/__init__.py create mode 100644 tests/noise/conftest.py create mode 100644 tests/noise/test_covariate_encoding.py create mode 100644 tests/noise/test_formula_arb.py create mode 100644 tests/noise/test_model_and_inference.py create mode 100644 tests/noise/test_model_dp_sigma.py create mode 100644 tests/noise/test_model_structural.py create mode 100644 tests/noise/test_output.py create mode 100644 tests/noise/test_panel_assembly.py create mode 100644 tests/noise/test_postprocessing.py create mode 100644 tests/noise/test_token_classification.py diff --git a/tests/noise/__init__.py b/tests/noise/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/noise/conftest.py b/tests/noise/conftest.py new file mode 100644 index 00000000..5bfeadc6 --- /dev/null +++ b/tests/noise/conftest.py @@ -0,0 +1,274 @@ +"""Shared fixtures for noise calibration tests.""" + +from datetime import date, timedelta + +import numpy as np +import pandas as pd +import pytest + + +# --------------------------------------------------------------------------- +# synthetic_panel: 3 pools × 10 days = 30 obs (before lag drop → 27 obs) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_panel() -> pd.DataFrame: + """3 pools × 10 days with known structure. + + Pool A: MAINNET, WETH/USDC, tier_A=0, tier_B=0, fee=0.003 + Pool B: ARBITRUM, BAL/WETH, tier_A=0, tier_B=1, fee=0.01 + Pool C: BASE, RATS/WETH, tier_A=0, tier_B=2, fee=0.005 + + Dates: 2026-01-01 to 2026-01-10 + Weekend flags: Sat 2026-01-03 and Sun 2026-01-04 are weekends. + (2026-01-01 is Thursday, ..., 01-03 Sat, 01-04 Sun, 01-05 Mon, ...) + """ + np.random.seed(42) + + pools = [ + ("pool_A", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("pool_B", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("pool_C", "BASE", "RATS,WETH", 0.005, 0, 2), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + + records = [] + for pool_id, chain, tokens, fee, tier_a, tier_b in pools: + log_tvl_base = 14.0 + np.random.randn() * 0.5 + for d in dates: + log_tvl = log_tvl_base + np.random.randn() * 0.1 + log_vol = log_tvl - 2.0 + np.random.randn() * 0.3 + vol = 0.3 + np.random.rand() * 0.2 + is_weekend = 1.0 if d.weekday() >= 5 else 0.0 + + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": d, + "log_volume": log_vol, + "log_tvl": log_tvl, + "volatility": vol, + "weekend": is_weekend, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": tokens, + }) + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + # Structural model covariates + panel["log_sigma"] = np.log(np.maximum(panel["volatility"].values, 1e-6)) + dow = panel["date"].apply( + lambda d: d.weekday() if hasattr(d, "weekday") else pd.Timestamp(d).weekday() + ) + panel["dow_sin"] = np.sin(2.0 * np.pi * dow / 7.0) + panel["dow_cos"] = np.cos(2.0 * np.pi * dow / 7.0) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + return panel + + +# --------------------------------------------------------------------------- +# synthetic_encoded_data: output of encode_covariates(synthetic_panel) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_encoded_data(synthetic_panel): + from quantammsim.noise_calibration import encode_covariates + return encode_covariates(synthetic_panel) + + +# --------------------------------------------------------------------------- +# synthetic_samples: deterministic posterior-like dict +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_samples(synthetic_encoded_data): + """Deterministic posterior samples with eta=0, L_Omega=I. + + With this structure: theta = X_pool @ B^T exactly. + """ + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_coeff = 4 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + sigma_theta = np.ones((S, K_coeff)) + L_Omega = np.tile(np.eye(K_coeff), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_coeff)) + df = np.full((S,), 5.0) + sigma_eps = np.tile([0.5, 0.8, 0.6], (S, 1)) + + return { + "B": B, + "sigma_theta": sigma_theta, + "L_Omega": L_Omega, + "eta": eta, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_pools_df: matches enumerate_balancer_pools output schema +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_pools_df() -> pd.DataFrame: + return pd.DataFrame([ + { + "pool_id": "pool_A", + "chain": "MAINNET", + "pool_type": "WEIGHTED", + "tokens": ["WETH", "USDC"], + "token_addresses": ["0xweth", "0xusdc"], + "swap_fee": 0.003, + "current_tvl": 1_000_000, + }, + { + "pool_id": "pool_B", + "chain": "ARBITRUM", + "pool_type": "WEIGHTED", + "tokens": ["BAL", "WETH"], + "token_addresses": ["0xbal", "0xweth"], + "swap_fee": 0.01, + "current_tvl": 500_000, + }, + { + "pool_id": "pool_C", + "chain": "BASE", + "pool_type": "WEIGHTED", + "tokens": ["RATS", "WETH"], + "token_addresses": ["0xrats", "0xweth"], + "swap_fee": 0.005, + "current_tvl": 100_000, + }, + ]) + + +# --------------------------------------------------------------------------- +# synthetic_ibp_samples: IBP posterior-like dict (no eta, L_Omega, sigma_theta) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_ibp_samples(synthetic_encoded_data): + """Deterministic IBP posterior samples (marginalized model). + + Contains B, W, v_ibp, alpha_ibp — no z_logit, eta, L_Omega, sigma_theta. + Z is analytically marginalized; MAP assignments are computed from data. + """ + data = synthetic_encoded_data + K_cov = data["K_cov"] + K_coeff = 4 + K_features = 6 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + W = np.random.randn(S, K_features, K_coeff) * 0.3 + v_ibp = np.random.beta(2, 1, size=(S, K_features)) + alpha_ibp = np.full((S,), 2.0) + sigma_w = np.full((S,), 1.0) + df = np.full((S,), 5.0) + sigma_eps = np.full((S,), 0.5) + + return { + "B": B, + "W": W, + "v_ibp": v_ibp, + "alpha_ibp": alpha_ibp, + "sigma_w": sigma_w, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_ibp_dp_samples: hybrid IBP+DP posterior-like dict +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_ibp_dp_samples(synthetic_encoded_data): + """Deterministic hybrid IBP+DP posterior samples. + + Contains both IBP keys (B, W, v_ibp, alpha_ibp, sigma_w) and + DP keys (v, alpha_dp, sigma_eps as vector). No z_logit, eta, L_Omega, + sigma_theta. + """ + data = synthetic_encoded_data + K_cov = data["K_cov"] + K_coeff = 4 + K_features = 6 + K_clusters = 6 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + W = np.random.randn(S, K_features, K_coeff) * 0.3 + v_ibp = np.random.beta(2, 1, size=(S, K_features)) + alpha_ibp = np.full((S,), 2.0) + sigma_w = np.full((S,), 1.0) + v = np.random.beta(1, 2, size=(S, K_clusters - 1)) + alpha_dp = np.full((S,), 1.0) + df = np.full((S,), 5.0) + sigma_eps = np.abs(np.random.randn(S, K_clusters)) + 0.1 + + return { + "B": B, + "W": W, + "v_ibp": v_ibp, + "alpha_ibp": alpha_ibp, + "sigma_w": sigma_w, + "v": v, + "alpha_dp": alpha_dp, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_snapshots_df: matches fetch_all_snapshots output schema +# --------------------------------------------------------------------------- + +# --------------------------------------------------------------------------- +# synthetic_structural_data: output of encode_covariates_structural() +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_structural_data(synthetic_panel): + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + return encode_covariates_structural(synthetic_panel) + + +# --------------------------------------------------------------------------- +# synthetic_snapshots_df: matches fetch_all_snapshots output schema +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_snapshots_df() -> pd.DataFrame: + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + records = [] + np.random.seed(42) + for pool_id, chain in [("pool_A", "MAINNET"), ("pool_B", "ARBITRUM"), + ("pool_C", "BASE")]: + for d in dates: + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": d, + "volume_usd": np.exp(10.0 + np.random.randn() * 0.5), + "total_liquidity_usd": np.exp(14.0 + np.random.randn() * 0.3), + }) + return pd.DataFrame(records) diff --git a/tests/noise/test_covariate_encoding.py b/tests/noise/test_covariate_encoding.py new file mode 100644 index 00000000..dececb5b --- /dev/null +++ b/tests/noise/test_covariate_encoding.py @@ -0,0 +1,326 @@ +"""Tests for encode_covariates and encode_covariates_structural.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import encode_covariates +from quantammsim.noise_calibration.constants import K_OBS_COEFF, OBS_COEFF_NAMES + + +class TestEncodeCovariates: + def test_x_pool_shape(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + N_pools = data["N_pools"] + K_cov = data["K_cov"] + assert data["X_pool"].shape == (N_pools, K_cov) + + def test_x_obs_shape(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + N_obs = len(synthetic_panel) + assert data["x_obs"].shape == (N_obs, 4) + + def test_x_obs_column_0_is_intercept(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal(data["x_obs"][:, 0], 1.0) + + def test_x_obs_column_1_is_lagged_tvl(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 1], + synthetic_panel["log_tvl_lag1"].values, + ) + + def test_x_obs_column_2_is_volatility(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 2], + synthetic_panel["volatility"].values, + ) + + def test_x_obs_column_3_is_weekend(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 3], + synthetic_panel["weekend"].values, + ) + + def test_intercept_column_all_ones(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal(data["X_pool"][:, 0], 1.0) + + def test_chain_dummies_one_hot(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + chain_cols = [i for i, n in enumerate(col_names) if n.startswith("chain_")] + X = data["X_pool"] + + for row in range(X.shape[0]): + chain_vals = X[row, chain_cols] + assert chain_vals.sum() <= 1.0 + + def test_reference_chain_is_alphabetically_first(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + chains = sorted(synthetic_panel["chain"].unique()) + ref_chain = chains[0] # ARBITRUM + + pool_meta = data["pool_meta"] + ref_idx = pool_meta[pool_meta["chain"] == ref_chain].index[0] + col_names = data["covariate_names"] + chain_cols = [i for i, n in enumerate(col_names) if n.startswith("chain_")] + X = data["X_pool"] + assert all(X[ref_idx, c] == 0.0 for c in chain_cols) + + def test_tier_a_dummies_match(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + tier_a_cols = [ + (i, n) for i, n in enumerate(col_names) if n.startswith("tier_A_") + ] + pool_meta = data["pool_meta"] + X = data["X_pool"] + + for idx, row in pool_meta.iterrows(): + tier_a_str = str(row["tier_A"]) + for col_idx, col_name in tier_a_cols: + expected_tier = col_name.split("_")[-1] + expected = 1.0 if tier_a_str == expected_tier else 0.0 + assert X[idx, col_idx] == expected, ( + f"Pool {idx} tier_A={tier_a_str}, col {col_name}: " + f"expected {expected}, got {X[idx, col_idx]}" + ) + + def test_tier_b_dummies_match(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + tier_b_cols = [ + (i, n) for i, n in enumerate(col_names) if n.startswith("tier_B_") + ] + pool_meta = data["pool_meta"] + X = data["X_pool"] + + for idx, row in pool_meta.iterrows(): + tier_b_str = str(row["tier_B"]) + for col_idx, col_name in tier_b_cols: + expected_tier = col_name.split("_")[-1] + expected = 1.0 if tier_b_str == expected_tier else 0.0 + assert X[idx, col_idx] == expected + + def test_log_fee_column(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + fee_idx = col_names.index("log_fee") + pool_meta = data["pool_meta"] + + for idx, row in pool_meta.iterrows(): + expected = np.log(max(row["swap_fee"], 1e-6)) + np.testing.assert_allclose( + data["X_pool"][idx, fee_idx], expected, rtol=1e-10, + ) + + def test_pool_idx_maps_observations(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + pool_ids = data["pool_ids"] + pool_idx = data["pool_idx"] + + assert pool_idx.min() >= 0 + assert pool_idx.max() < len(pool_ids) + + for i, pid in enumerate(pool_ids): + mask = synthetic_panel["pool_id"] == pid + obs_indices = np.where(mask.values)[0] + assert (pool_idx[obs_indices] == i).all() + + def test_y_obs_matches_panel(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["y_obs"], + synthetic_panel["log_volume"].values, + ) + + def test_covariate_names_length(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + assert len(data["covariate_names"]) == data["K_cov"] + + def test_covariate_column_ordering(self, synthetic_panel): + """X_pool columns must follow: intercept, chain dummies, tier_A dummies, + tier_B dummies, log_fee. This ordering is load-bearing because B is + indexed by column position.""" + data = encode_covariates(synthetic_panel) + names = data["covariate_names"] + + assert names[0] == "intercept" + assert names[-1] == "log_fee" + + # Find boundaries + chain_start = None + tier_a_start = None + tier_b_start = None + fee_idx = len(names) - 1 + + for i, n in enumerate(names): + if n.startswith("chain_") and chain_start is None: + chain_start = i + if n.startswith("tier_A_") and tier_a_start is None: + tier_a_start = i + if n.startswith("tier_B_") and tier_b_start is None: + tier_b_start = i + + # Verify ordering: intercept < chains < tier_A < tier_B < log_fee + if chain_start is not None: + assert chain_start > 0 # after intercept + if tier_a_start is not None and chain_start is not None: + assert tier_a_start > chain_start + if tier_b_start is not None and tier_a_start is not None: + assert tier_b_start > tier_a_start + if tier_b_start is not None: + assert fee_idx > tier_b_start + + # All chain dummies are contiguous + chain_names = [n for n in names if n.startswith("chain_")] + if chain_names: + chain_indices = [names.index(n) for n in chain_names] + assert chain_indices == list(range(min(chain_indices), + max(chain_indices) + 1)) + + def test_output_dict_has_all_required_keys(self, synthetic_panel): + """encode_covariates must return all keys consumed by downstream + functions (predict_new_pool, generate_output_json, _save_sample_cache).""" + data = encode_covariates(synthetic_panel) + required_keys = { + "pool_idx", "X_pool", "x_obs", "y_obs", "pool_ids", "pool_meta", + "covariate_names", "tier_A_per_pool", "N_pools", "K_cov", + "ref_chain", "ref_tier_a", "ref_tier_b", "chains", + } + assert required_keys.issubset(data.keys()), ( + f"Missing keys: {required_keys - data.keys()}" + ) + + +class TestEncodeCovariatesNoTiers: + """Tests for encode_covariates(include_tiers=False).""" + + def test_no_tier_columns_in_covariate_names(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + for name in data["covariate_names"]: + assert not name.startswith("tier_A_"), ( + f"Found tier_A column {name} with include_tiers=False" + ) + assert not name.startswith("tier_B_"), ( + f"Found tier_B column {name} with include_tiers=False" + ) + + def test_still_has_intercept_and_log_fee(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + assert "intercept" in data["covariate_names"] + assert "log_fee" in data["covariate_names"] + + def test_still_has_chain_dummies(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + chain_cols = [n for n in data["covariate_names"] + if n.startswith("chain_")] + assert len(chain_cols) > 0 + + def test_k_cov_smaller_than_with_tiers(self, synthetic_panel): + data_tiers = encode_covariates(synthetic_panel, include_tiers=True) + data_no_tiers = encode_covariates(synthetic_panel, include_tiers=False) + assert data_no_tiers["K_cov"] < data_tiers["K_cov"] + + def test_default_is_include_tiers_true(self, synthetic_panel): + data_default = encode_covariates(synthetic_panel) + data_explicit = encode_covariates(synthetic_panel, include_tiers=True) + assert data_default["K_cov"] == data_explicit["K_cov"] + + def test_tier_A_per_pool_still_present(self, synthetic_panel): + """tier_A_per_pool is still returned (for other downstream uses).""" + data = encode_covariates(synthetic_panel, include_tiers=False) + assert "tier_A_per_pool" in data + + def test_x_pool_shape_matches_k_cov(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + assert data["X_pool"].shape == (data["N_pools"], data["K_cov"]) + + +class TestEncodeStructuralCovariates: + """Tests for encode_covariates_structural().""" + + @pytest.fixture() + def struct_data(self, synthetic_panel): + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + return encode_covariates_structural(synthetic_panel) + + def test_encode_structural_x_obs_shape(self, struct_data, synthetic_panel): + N_obs = len(synthetic_panel) + assert struct_data["x_obs"].shape == (N_obs, K_OBS_COEFF) + + def test_encode_structural_x_obs_columns(self, struct_data, synthetic_panel): + """Columns must match OBS_COEFF_NAMES ordering: + [1, lag_log_tvl, log_sigma, tvl_x_sigma, tvl_x_fee, sigma_x_fee, + dow_sin, dow_cos].""" + x = struct_data["x_obs"] + np.testing.assert_array_equal(x[:, 0], 1.0) # intercept + np.testing.assert_array_equal( + x[:, 1], synthetic_panel["log_tvl_lag1"].values, + ) + np.testing.assert_array_equal( + x[:, 2], synthetic_panel["log_sigma"].values, + ) + np.testing.assert_allclose( + x[:, 3], synthetic_panel["tvl_x_sigma"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 4], synthetic_panel["tvl_x_fee"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 5], synthetic_panel["sigma_x_fee"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 6], synthetic_panel["dow_sin"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 7], synthetic_panel["dow_cos"].values, rtol=1e-12, + ) + + def test_encode_structural_has_sigma_daily(self, struct_data): + """sigma_daily = volatility / sqrt(365), de-annualised.""" + assert "sigma_daily" in struct_data + assert len(struct_data["sigma_daily"]) > 0 + + def test_encode_structural_has_gas(self, struct_data): + assert "gas" in struct_data + assert len(struct_data["gas"]) > 0 + assert (struct_data["gas"] >= 0).all() + + def test_encode_structural_has_chain_idx_tier_idx(self, struct_data): + assert "chain_idx" in struct_data + assert "tier_idx" in struct_data + assert struct_data["chain_idx"].dtype in (np.int32, np.int64) + assert struct_data["tier_idx"].dtype in (np.int32, np.int64) + + def test_encode_structural_has_fee(self, struct_data): + """fee array is raw (not log), for the formula.""" + assert "fee" in struct_data + assert (struct_data["fee"] > 0).all() + assert (struct_data["fee"] < 1).all() # fees are fractions + + def test_encode_structural_tier_idx_is_pair(self, struct_data): + """tier_idx encodes the (tier_A, tier_B) PAIR, not individual tokens. + (0,0)->0, (0,1)->1, (0,2)->2, (1,1)->3, (1,2)->4, (2,2)->5.""" + tier_idx = struct_data["tier_idx"] + pool_meta = struct_data["pool_meta"] + + for i, row in pool_meta.iterrows(): + a, b = int(row["tier_A"]), int(row["tier_B"]) + expected = a * (5 - a) // 2 + b - a + assert tier_idx[i] == expected, ( + f"Pool {i}: tier ({a},{b}) expected idx {expected}, " + f"got {tier_idx[i]}" + ) + + def test_encode_structural_n_chains_n_tiers(self, struct_data): + """n_chains and n_tiers computed from data.""" + assert "n_chains" in struct_data + assert "n_tiers" in struct_data + assert struct_data["n_chains"] >= 1 + assert struct_data["n_tiers"] >= 1 diff --git a/tests/noise/test_formula_arb.py b/tests/noise/test_formula_arb.py new file mode 100644 index 00000000..f862d516 --- /dev/null +++ b/tests/noise/test_formula_arb.py @@ -0,0 +1,142 @@ +"""Tests for JAX-differentiable LVR formula.""" + +import numpy as np +import pytest + + +class TestFormulaArbJax: + @pytest.fixture(autouse=True) + def _import(self): + from quantammsim.noise_calibration.formula_arb import ( + formula_arb_volume_daily_jax, + ) + self.formula = formula_arb_volume_daily_jax + + def test_formula_arb_zero_vol_returns_zero(self): + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.0), + tvl=jnp.float64(1e6), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-10) + + def test_formula_arb_zero_tvl_returns_zero(self): + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.03), + tvl=jnp.float64(0.0), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-10) + + def test_formula_arb_quadratic_in_sigma(self): + """V_arb(2σ) / V_arb(σ) ≈ 4 for small gas (correction ≈ 1).""" + import jax.numpy as jnp + sigma = jnp.float64(0.01) + tvl = jnp.float64(1e8) # large TVL so gas is negligible + fee = jnp.float64(0.003) + gas = jnp.float64(0.001) # tiny gas + cadence = jnp.float64(0.01) # very fast arb + + v1 = float(self.formula(sigma, tvl, fee, gas, cadence)) + v2 = float(self.formula(2.0 * sigma, tvl, fee, gas, cadence)) + assert v1 > 0 + ratio = v2 / v1 + assert ratio == pytest.approx(4.0, rel=0.1) + + def test_formula_arb_linear_in_tvl(self): + """V_arb(2V) / V_arb(V) ≈ 2 for small gas.""" + import jax.numpy as jnp + sigma = jnp.float64(0.02) + tvl = jnp.float64(1e8) + fee = jnp.float64(0.003) + gas = jnp.float64(0.0001) + cadence = jnp.float64(0.01) + + v1 = float(self.formula(sigma, tvl, fee, gas, cadence)) + v2 = float(self.formula(sigma, 2.0 * tvl, fee, gas, cadence)) + assert v1 > 0 + ratio = v2 / v1 + assert ratio == pytest.approx(2.0, rel=0.1) + + def test_formula_arb_gas_kills_small_pools(self): + """High gas, small TVL → V_arb ≈ 0.""" + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.02), + tvl=jnp.float64(1000.0), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1000.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-6) + + def test_formula_arb_matches_numpy_reference(self): + """Compare to the numpy formula in plot_formula_arb_vs_real.py.""" + import jax.numpy as jnp + + # Reference implementation (from plot_formula_arb_vs_real.py:58) + def ref(sigma_daily, tvl, fee, block_time_s, gas_usd): + if tvl <= 0 or fee <= 0 or sigma_daily <= 0: + return 0.0 + gamma = fee + delta = 2.0 * np.sqrt(2.0 * gas_usd / tvl) if gas_usd > 0 else 0.0 + bLVR = sigma_daily**2 * tvl / 8.0 + sqrt_s2_2l = sigma_daily * np.sqrt(block_time_s / (2.0 * 86400.0)) + bFEE = bLVR * max( + 1.0 - delta / (2.0 * gamma) - sqrt_s2_2l / (gamma + delta / 2.0), + 0.0, + ) + return bFEE / gamma + + test_cases = [ + (0.03, 1e6, 0.003, 1.0, 1.0), + (0.05, 5e5, 0.01, 0.5, 0.005), + (0.01, 1e7, 0.005, 2.0, 0.01), + (0.1, 1e4, 0.03, 10.0, 5.0), + (0.02, 1e5, 0.001, 1.0, 0.001), + ] + + for sigma, tvl, fee, cadence_min, gas in test_cases: + block_time_s = cadence_min * 60.0 + expected = ref(sigma, tvl, fee, block_time_s, gas) + actual = float(self.formula( + jnp.float64(sigma), jnp.float64(tvl), + jnp.float64(fee), jnp.float64(gas), + jnp.float64(cadence_min), + )) + np.testing.assert_allclose( + actual, expected, rtol=1e-10, + err_msg=f"Mismatch for sigma={sigma}, tvl={tvl}, " + f"fee={fee}, cadence={cadence_min}, gas={gas}", + ) + + def test_formula_arb_is_jax_differentiable(self): + """jax.grad w.r.t. sigma should run without error.""" + import jax + import jax.numpy as jnp + + grad_fn = jax.grad(self.formula, argnums=0) + result = grad_fn( + jnp.float64(0.03), jnp.float64(1e6), + jnp.float64(0.003), jnp.float64(1.0), + jnp.float64(1.0), + ) + assert np.isfinite(float(result)) + + def test_formula_arb_cadence_reduces_volume(self): + """Higher cadence → less frequent arb → lower volume.""" + import jax.numpy as jnp + sigma = jnp.float64(0.03) + tvl = jnp.float64(1e6) + fee = jnp.float64(0.003) + gas = jnp.float64(1.0) + + v_fast = float(self.formula(sigma, tvl, fee, gas, jnp.float64(1.0))) + v_slow = float(self.formula(sigma, tvl, fee, gas, jnp.float64(10.0))) + assert v_fast > v_slow diff --git a/tests/noise/test_model_and_inference.py b/tests/noise/test_model_and_inference.py new file mode 100644 index 00000000..ee1fda5c --- /dev/null +++ b/tests/noise/test_model_and_inference.py @@ -0,0 +1,328 @@ +"""Tests for noise_model, _get_theta_samples, _build_model_kwargs, and SVI smoke.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + noise_model, + _get_theta_samples, + _build_model_kwargs, + K_COEFF, +) + + +# =========================================================================== +# TestNoiseModelDefinition +# =========================================================================== + + +class TestNoiseModelDefinition: + def test_model_traces_without_error(self, synthetic_encoded_data): + import jax + import jax.numpy as jnp + import numpyro + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + assert trace is not None + + def test_required_sites_present(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + required = {"B", "sigma_theta", "L_Omega", "df", "sigma_eps", "eta", "y"} + assert required.issubset(trace.keys()) + + def test_site_shapes(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + N_pools = data["N_pools"] + K_cov = data["K_cov"] + + assert trace["B"]["value"].shape == (K_COEFF, K_cov) + assert trace["sigma_theta"]["value"].shape == (K_COEFF,) + assert trace["L_Omega"]["value"].shape == (K_COEFF, K_COEFF) + assert trace["df"]["value"].shape == () + assert trace["sigma_eps"]["value"].shape == (3,) + assert trace["eta"]["value"].shape == (N_pools, K_COEFF) + + def test_theta_deterministic_site(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + assert "theta" in trace + assert trace["theta"]["value"].shape == (data["N_pools"], K_COEFF) + + def test_prior_predictive_produces_y(self, synthetic_encoded_data): + import jax + from numpyro.infer import Predictive + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + kwargs["y_obs"] = None + + predictive = Predictive(noise_model, num_samples=5) + rng_key = jax.random.PRNGKey(42) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 5 + + +# =========================================================================== +# TestGetThetaSamples +# =========================================================================== + + +class TestGetThetaSamples: + def test_returns_theta_directly_when_present(self, synthetic_encoded_data): + data = synthetic_encoded_data + theta_direct = np.random.randn(10, data["N_pools"], K_COEFF) + sample_dict = {"theta": theta_direct} + result = _get_theta_samples(sample_dict, data["X_pool"]) + np.testing.assert_array_equal(result, theta_direct) + + def test_reconstructs_from_non_centered( + self, synthetic_encoded_data, synthetic_samples + ): + data = synthetic_encoded_data + result = _get_theta_samples(synthetic_samples, data["X_pool"]) + assert result.shape == (10, data["N_pools"], K_COEFF) + + def test_eta_zero_identity_gives_mu( + self, synthetic_encoded_data, synthetic_samples + ): + """With eta=0 and L_Omega=I, theta = X_pool @ B^T.""" + data = synthetic_encoded_data + X_pool = data["X_pool"] + B = synthetic_samples["B"] + + result = _get_theta_samples(synthetic_samples, X_pool) + + # Expected: mu[s,p,j] = sum_d X_pool[p,d] * B[s,j,d] + expected = np.einsum("pd,sjd->spj", X_pool, B) + np.testing.assert_allclose(result, expected, atol=1e-12) + + def test_output_shape(self, synthetic_encoded_data, synthetic_samples): + data = synthetic_encoded_data + result = _get_theta_samples(synthetic_samples, data["X_pool"]) + S = synthetic_samples["B"].shape[0] + assert result.shape == (S, data["N_pools"], K_COEFF) + + def test_reconstructed_matches_direct(self, synthetic_encoded_data): + """When both theta and raw params are present, reconstruction matches.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + S = 8 + X_pool = data["X_pool"] + + np.random.seed(123) + B = np.random.randn(S, K_COEFF, K_cov) * 0.3 + sigma_theta = np.abs(np.random.randn(S, K_COEFF)) + 0.1 + # Random lower-triangular L + L_raw = np.zeros((S, K_COEFF, K_COEFF)) + for s in range(S): + A = np.random.randn(K_COEFF, K_COEFF) + L_raw[s] = np.linalg.cholesky(A @ A.T + np.eye(K_COEFF)) + eta = np.random.randn(S, N_pools, K_COEFF) + + sample_dict = { + "B": B, "sigma_theta": sigma_theta, + "L_Omega": L_raw, "eta": eta, + } + + theta_recon = _get_theta_samples(sample_dict, X_pool) + + # Compute expected directly + mu = np.einsum("pd,sjd->spj", X_pool, B) + L_Sigma = sigma_theta[:, :, None] * L_raw + offset = np.einsum("spi,sji->spj", eta, L_Sigma) + expected = mu + offset + + np.testing.assert_allclose(theta_recon, expected, atol=1e-10) + + +# =========================================================================== +# TestBuildModelKwargs +# =========================================================================== + + +class TestBuildModelKwargs: + def test_all_outputs_are_jnp(self, synthetic_encoded_data): + import jax.numpy as jnp + + kwargs = _build_model_kwargs(synthetic_encoded_data) + for key in ["pool_idx", "X_pool", "x_obs", "y_obs", "tier_A_per_pool"]: + assert isinstance(kwargs[key], jnp.ndarray), ( + f"{key} should be jnp array" + ) + + def test_shapes_preserved(self, synthetic_encoded_data): + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + assert kwargs["pool_idx"].shape == data["pool_idx"].shape + assert kwargs["X_pool"].shape == data["X_pool"].shape + assert kwargs["x_obs"].shape == data["x_obs"].shape + assert kwargs["y_obs"].shape == data["y_obs"].shape + assert kwargs["N_pools"] == data["N_pools"] + assert kwargs["K_cov"] == data["K_cov"] + + def test_dp_model_kwargs_exclude_tier_A(self, synthetic_encoded_data): + """When model_fn is noise_model_dp_sigma, tier_A_per_pool is excluded + and K_clusters is included.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + data = dict(synthetic_encoded_data) + data["K_clusters"] = 6 + kwargs = _build_model_kwargs(data, model_fn=noise_model_dp_sigma) + assert "tier_A_per_pool" not in kwargs + assert kwargs["K_clusters"] == 6 + + def test_tier_model_kwargs_include_tier_A(self, synthetic_encoded_data): + """When model_fn is noise_model (default), tier_A_per_pool is included + and K_clusters is not.""" + kwargs = _build_model_kwargs(synthetic_encoded_data) + assert "tier_A_per_pool" in kwargs + assert "K_clusters" not in kwargs + + def test_default_model_fn_is_noise_model(self, synthetic_encoded_data): + """Calling without model_fn should behave identically to model_fn=noise_model.""" + kwargs_default = _build_model_kwargs(synthetic_encoded_data) + kwargs_explicit = _build_model_kwargs( + synthetic_encoded_data, model_fn=noise_model + ) + assert set(kwargs_default.keys()) == set(kwargs_explicit.keys()) + + +# =========================================================================== +# TestSVISmoke +# =========================================================================== + + +class TestSVISmoke: + @pytest.mark.slow + def test_svi_converges_small_data(self): + """SVI on tiny synthetic data: ELBO should decrease.""" + import jax + import numpyro + + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + + # Build minimal panel: 5 pools × 20 days + np.random.seed(42) + from datetime import date, timedelta + + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("p2", "BASE", "RATS,WETH", 0.005, 0, 2), + ("p3", "MAINNET", "LINK,WETH", 0.005, 0, 1), + ("p4", "ARBITRUM", "AAVE,USDC", 0.003, 0, 1), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(21)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + + import pandas as pd + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + data = encode_covariates(panel) + samples, losses = run_svi(data, num_steps=2000, lr=1e-3, seed=0, + num_samples=50) + + # ELBO should decrease: mean of last 100 < mean of first 100 + assert np.mean(losses[-100:]) < np.mean(losses[:100]) + + @pytest.mark.slow + def test_svi_samples_have_required_keys(self): + """SVI output dict has all expected latent variable keys.""" + import jax + import numpyro + + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + + np.random.seed(42) + from datetime import date, timedelta + import pandas as pd + + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(15)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + data = encode_covariates(panel) + samples, _ = run_svi(data, num_steps=500, lr=1e-3, seed=0, + num_samples=10) + + required = {"B", "sigma_theta", "L_Omega", "eta", "df", "sigma_eps"} + assert required.issubset(samples.keys()) diff --git a/tests/noise/test_model_dp_sigma.py b/tests/noise/test_model_dp_sigma.py new file mode 100644 index 00000000..3d2be214 --- /dev/null +++ b/tests/noise/test_model_dp_sigma.py @@ -0,0 +1,305 @@ +"""Tests for stick_breaking_weights and noise_model_dp_sigma.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import K_COEFF +from quantammsim.noise_calibration.constants import K_CLUSTERS_DEFAULT + + +# =========================================================================== +# TestStickBreakingWeights +# =========================================================================== + + +class TestStickBreakingWeights: + def test_sums_to_one(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.5, 0.3, 0.4, 0.6, 0.2]) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(jnp.sum(w)), 1.0, atol=1e-6) + + def test_correct_length(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + K = 7 + v = jnp.ones(K - 1) * 0.3 + w = stick_breaking_weights(v) + assert w.shape == (K,) + + def test_non_negative(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.1, 0.9, 0.5, 0.7, 0.3]) + w = stick_breaking_weights(v) + assert jnp.all(w >= 0.0) + + def test_first_weight_equals_first_v(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.7, 0.4, 0.2]) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(w[0]), 0.7, atol=1e-6) + + def test_jit_compatible(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax + import jax.numpy as jnp + + v = jnp.array([0.5, 0.3, 0.4]) + w_eager = stick_breaking_weights(v) + w_jit = jax.jit(stick_breaking_weights)(v) + np.testing.assert_allclose( + np.array(w_eager), np.array(w_jit), atol=1e-6 + ) + + def test_all_v_one_concentrates_on_first(self): + """If v = [1, 1, ...], all mass goes to first component.""" + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.ones(5) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(w[0]), 1.0, atol=1e-6) + np.testing.assert_allclose(float(jnp.sum(w[1:])), 0.0, atol=1e-6) + + +# =========================================================================== +# TestDPModelDefinition +# =========================================================================== + + +class TestDPModelDefinition: + def _get_dp_model_kwargs(self, data, K_clusters=6): + """Build kwargs for noise_model_dp_sigma from encoded data.""" + import jax.numpy as jnp + + return dict( + pool_idx=jnp.array(data["pool_idx"]), + X_pool=jnp.array(data["X_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K_coeff=K_COEFF, + K_cov=data["K_cov"], + K_clusters=K_clusters, + ) + + def test_model_traces_without_error(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + assert trace is not None + + def test_has_dp_sites(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + dp_sites = {"alpha_dp", "v", "sigma_eps", "log_lik"} + assert dp_sites.issubset(trace.keys()), ( + f"Missing DP sites: {dp_sites - trace.keys()}" + ) + + def test_no_y_site_when_obs_provided(self, synthetic_encoded_data): + """With y_obs provided, the marginalized model uses factor, not obs.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + assert "y" not in trace, ( + "DP model should use numpyro.factor, not obs=y_obs" + ) + + def test_shared_mean_structure_sites(self, synthetic_encoded_data): + """The DP model must share the same mean structure as the tier model.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + shared_sites = {"B", "sigma_theta", "L_Omega", "df", "eta", "theta"} + assert shared_sites.issubset(trace.keys()), ( + f"Missing shared sites: {shared_sites - trace.keys()}" + ) + + def test_correct_shapes(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + K_clusters = 6 + kwargs = self._get_dp_model_kwargs( + synthetic_encoded_data, K_clusters=K_clusters + ) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + N_pools = synthetic_encoded_data["N_pools"] + K_cov = synthetic_encoded_data["K_cov"] + + assert trace["B"]["value"].shape == (K_COEFF, K_cov) + assert trace["sigma_theta"]["value"].shape == (K_COEFF,) + assert trace["L_Omega"]["value"].shape == (K_COEFF, K_COEFF) + assert trace["df"]["value"].shape == () + assert trace["sigma_eps"]["value"].shape == (K_clusters,) + assert trace["eta"]["value"].shape == (N_pools, K_COEFF) + assert trace["v"]["value"].shape == (K_clusters - 1,) + assert trace["alpha_dp"]["value"].shape == () + + def test_no_tier_A_per_pool_in_signature(self): + """The DP model should not accept tier_A_per_pool.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import inspect + + sig = inspect.signature(noise_model_dp_sigma) + assert "tier_A_per_pool" not in sig.parameters + + def test_prior_predictive_produces_y(self, synthetic_encoded_data): + """With y_obs=None, the model should sample y explicitly.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + from numpyro.infer import Predictive + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + kwargs["y_obs"] = None + + predictive = Predictive(noise_model_dp_sigma, num_samples=5) + rng_key = jax.random.PRNGKey(42) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 5 + + def test_k_clusters_configurable(self, synthetic_encoded_data): + """K_clusters=4 should produce different sigma_eps shape.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + K_clusters = 4 + kwargs = self._get_dp_model_kwargs( + synthetic_encoded_data, K_clusters=K_clusters + ) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + assert trace["sigma_eps"]["value"].shape == (K_clusters,) + assert trace["v"]["value"].shape == (K_clusters - 1,) + + +# =========================================================================== +# TestDPModelSVISmoke +# =========================================================================== + + +class TestDPModelSVISmoke: + def _build_dp_panel(self): + """Build a minimal panel for DP SVI testing.""" + from datetime import date, timedelta + import pandas as pd + + np.random.seed(42) + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("p2", "BASE", "RATS,WETH", 0.005, 0, 2), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(15)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + return panel + + def test_svi_converges_dp_model(self): + """SVI on DP model with tiny data: ELBO should decrease.""" + import numpyro + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + panel = self._build_dp_panel() + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = 4 + + samples, losses = run_svi( + data, num_steps=2000, lr=1e-3, seed=0, + num_samples=50, model_fn=noise_model_dp_sigma, + ) + assert np.mean(losses[-100:]) < np.mean(losses[:100]) + + def test_svi_samples_have_dp_keys(self): + """SVI output for DP model has v, alpha_dp, sigma_eps.""" + import numpyro + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + panel = self._build_dp_panel() + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = 4 + + samples, _ = run_svi( + data, num_steps=500, lr=1e-3, seed=0, + num_samples=10, model_fn=noise_model_dp_sigma, + ) + required = {"B", "sigma_theta", "L_Omega", "eta", "df", + "sigma_eps", "v", "alpha_dp"} + assert required.issubset(samples.keys()), ( + f"Missing keys: {required - samples.keys()}" + ) diff --git a/tests/noise/test_model_structural.py b/tests/noise/test_model_structural.py new file mode 100644 index 00000000..9243ba79 --- /dev/null +++ b/tests/noise/test_model_structural.py @@ -0,0 +1,215 @@ +"""Tests for structural_noise_model definition and SVI integration.""" + +import numpy as np +import pytest + + +class TestStructuralModelDefinition: + """Tests for the structural_noise_model numpyro model.""" + + def test_model_traces_without_error(self, synthetic_structural_data): + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + rng_key = jax.random.PRNGKey(0) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + + def test_required_sites_present(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + required = { + "alpha_0", "alpha_chain", "alpha_tier", "alpha_tvl", + "W_gate", "beta", "df", "sigma_eps", "y", + } + assert required.issubset(samples.keys()), ( + f"Missing sites: {required - samples.keys()}" + ) + + def test_no_eta_sigma_theta_L_Omega(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + old_sites = {"eta", "sigma_theta", "L_Omega", "theta", "B"} + for site in old_sites: + assert site not in samples, f"Old site '{site}' should not be present" + + def test_W_gate_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + K_pool_cov = synthetic_structural_data["K_cov"] + K_archetypes = 3 # default + assert samples["W_gate"].shape == (3, K_pool_cov, K_archetypes) + + def test_beta_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + from quantammsim.noise_calibration.constants import K_OBS_COEFF + K_archetypes = 3 + assert samples["beta"].shape == (3, K_archetypes, K_OBS_COEFF) + + def test_alpha_chain_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + n_chains = synthetic_structural_data["n_chains"] + assert samples["alpha_chain"].shape == (3, n_chains - 1) + + def test_prior_predictive_produces_y(self, synthetic_structural_data): + """y_obs=None path produces y samples.""" + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + kwargs["y_obs"] = None + predictive = Predictive(structural_noise_model, num_samples=10) + samples = predictive(jax.random.PRNGKey(42), **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 10 + + def test_prior_predictive_range_reasonable(self, synthetic_structural_data): + """Prior y median should be in a plausible range for log(volume). + + The tails can be wide (V_noise = exp(beta @ x_obs) with large + covariates), but the central mass should be reasonable. + """ + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + kwargs["y_obs"] = None + predictive = Predictive(structural_noise_model, num_samples=200) + samples = predictive(jax.random.PRNGKey(42), **kwargs) + y = np.array(samples["y"]) + median = np.median(y) + p25 = np.percentile(y, 25) + p75 = np.percentile(y, 75) + # Median should be in a plausible log-volume range + assert median > -20, f"Prior y median too low: {median}" + assert median < 50, f"Prior y median too high: {median}" + # IQR should not be enormous + iqr = p75 - p25 + assert iqr < 500, f"Prior y IQR too wide: {iqr}" + + def test_K_archetypes_configurable(self, synthetic_structural_data): + """K=2 and K=4 both work.""" + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + for K in [2, 4]: + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + kwargs["K_archetypes"] = K + predictive = Predictive(structural_noise_model, num_samples=2) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + assert samples["W_gate"].shape[-1] == K + assert samples["beta"].shape[1] == K + + +class TestSVIStructural: + """SVI convergence tests for structural model.""" + + @pytest.mark.slow + def test_svi_structural_converges(self, synthetic_structural_data): + """2000 SVI steps, ELBO should decrease.""" + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, losses = run_svi( + synthetic_structural_data, + num_steps=2000, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + assert losses[-100:].mean() < losses[:100].mean() + + @pytest.mark.slow + def test_svi_structural_samples_have_required_keys( + self, synthetic_structural_data, + ): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + required = {"W_gate", "beta", "alpha_0", "alpha_chain", + "alpha_tier", "alpha_tvl", "df", "sigma_eps"} + assert required.issubset(samples.keys()), ( + f"Missing keys: {required - samples.keys()}" + ) + + @pytest.mark.slow + def test_svi_structural_no_eta_keys(self, synthetic_structural_data): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + old_keys = {"eta", "sigma_theta", "L_Omega", "B"} + for key in old_keys: + assert key not in samples, f"Old key '{key}' should not be in samples" diff --git a/tests/noise/test_output.py b/tests/noise/test_output.py new file mode 100644 index 00000000..7661c9d6 --- /dev/null +++ b/tests/noise/test_output.py @@ -0,0 +1,303 @@ +"""Tests for generate_output_json and _save_sample_cache.""" + +import json +import os + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + extract_noise_params, + generate_output_json, + _save_sample_cache, + K_COEFF, +) + + +# =========================================================================== +# TestGenerateOutputJSON +# =========================================================================== + + +class TestGenerateOutputJSON: + @pytest.fixture() + def _output_setup(self, tmp_path, synthetic_samples, synthetic_encoded_data): + """Produce the JSON file and return (path, data dict).""" + pool_params = extract_noise_params( + synthetic_samples, synthetic_encoded_data + ) + output_path = str(tmp_path / "test_output.json") + convergence = {"method": "svi", "final_elbo": 1234.0} + inference_config = {"method": "svi", "svi_steps": 1000} + + generate_output_json( + pool_params, synthetic_samples, synthetic_encoded_data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + data = json.load(f) + return output_path, data + + def test_writes_valid_json(self, _output_setup): + path, data = _output_setup + assert isinstance(data, dict) + + def test_top_level_keys(self, _output_setup): + _, data = _output_setup + expected = { + "model", "model_spec", "inference", "population_effects", + "convergence", "n_pools", "n_obs", "pools", + } + assert expected.issubset(data.keys()) + + def test_model_spec_fields(self, _output_setup): + _, data = _output_setup + spec = data["model_spec"] + assert "K_coeff" in spec + assert "K_cov" in spec + assert "coeff_names" in spec + assert "covariate_names" in spec + assert "likelihood" in spec + assert spec["likelihood"] == "StudentT" + assert "tvl_lag" in spec + assert spec["tvl_lag"] == "log_tvl_lag1" + + def test_population_effects_fields(self, _output_setup): + _, data = _output_setup + pe = data["population_effects"] + assert "B" in pe + assert "sigma_theta" in pe + assert "sigma_eps" in pe + assert "df" in pe + assert "correlation_matrix" in pe + + def test_pool_entries(self, _output_setup, synthetic_encoded_data): + _, data = _output_setup + pools = data["pools"] + for pid in synthetic_encoded_data["pool_ids"]: + assert pid in pools + entry = pools[pid] + assert "chain" in entry + assert "tokens" in entry + assert "theta_median" in entry + assert "noise_params" in entry + + def test_correlation_matrix_symmetric_unit_diagonal(self, _output_setup): + _, data = _output_setup + Omega = np.array(data["population_effects"]["correlation_matrix"]) + np.testing.assert_allclose(Omega, Omega.T, atol=1e-10) + np.testing.assert_allclose(np.diag(Omega), 1.0, atol=1e-10) + + def test_n_pools_and_n_obs(self, _output_setup, synthetic_encoded_data): + _, data = _output_setup + assert data["n_pools"] == synthetic_encoded_data["N_pools"] + assert data["n_obs"] == len(synthetic_encoded_data["y_obs"]) + + def test_model_name(self, _output_setup): + _, data = _output_setup + assert data["model"] == "unified_hierarchical_student_t" + + +# =========================================================================== +# TestSaveSampleCache +# =========================================================================== + + +class TestSaveSampleCache: + def test_creates_npz_and_json( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + cache_dir = str(tmp_path / "cache") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + assert os.path.exists(os.path.join(cache_dir, "unified_samples.npz")) + assert os.path.exists(os.path.join(cache_dir, "unified_data.json")) + + def test_npz_excludes_y_and_theta( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """Both 'y' and 'theta' must be excluded from the npz cache.""" + samples_with_extras = dict(synthetic_samples) + samples_with_extras["y"] = np.random.randn(10, 27) + samples_with_extras["theta"] = np.random.randn(10, 3, 4) + + cache_dir = str(tmp_path / "cache2") + _save_sample_cache(samples_with_extras, synthetic_encoded_data, cache_dir) + + npz_path = os.path.join(cache_dir, "unified_samples.npz") + loaded = np.load(npz_path) + assert "y" not in loaded.files + assert "theta" not in loaded.files + + def test_npz_contains_required_keys( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """The npz must contain B, sigma_theta, L_Omega, eta, df, sigma_eps.""" + cache_dir = str(tmp_path / "cache3") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + npz_path = os.path.join(cache_dir, "unified_samples.npz") + loaded = np.load(npz_path) + required = {"B", "sigma_theta", "L_Omega", "eta", "df", "sigma_eps"} + assert required.issubset(set(loaded.files)) + + def test_json_contains_metadata_keys( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """The JSON cache must contain all keys needed by --predict.""" + cache_dir = str(tmp_path / "cache4") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + json_path = os.path.join(cache_dir, "unified_data.json") + with open(json_path) as f: + meta = json.load(f) + required = { + "pool_ids", "covariate_names", "K_cov", "N_pools", + "ref_chain", "ref_tier_a", "ref_tier_b", "chains", + } + assert required.issubset(meta.keys()) + + +# =========================================================================== +# TestGenerateOutputJSONDP +# =========================================================================== + + +class TestGenerateOutputJSONDP: + @pytest.fixture() + def _dp_output_setup(self, tmp_path, synthetic_encoded_data): + """Produce DP model JSON output and return (path, data dict).""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_clusters = 4 + S = 10 + + np.random.seed(99) + dp_samples = { + "B": np.random.randn(S, K_COEFF, K_cov) * 0.5, + "sigma_theta": np.ones((S, K_COEFF)), + "L_Omega": np.tile(np.eye(K_COEFF), (S, 1, 1)), + "eta": np.zeros((S, N_pools, K_COEFF)), + "df": np.full((S,), 5.0), + "sigma_eps": np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)), + "v": np.tile([0.6, 0.3, 0.05], (S, 1)), + "alpha_dp": np.full((S,), 1.5), + } + + pool_params = extract_noise_params(dp_samples, data) + output_path = str(tmp_path / "dp_output.json") + convergence = {"method": "svi", "final_elbo": 1234.0} + inference_config = {"method": "svi", "svi_steps": 1000} + + generate_output_json( + pool_params, dp_samples, data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + result = json.load(f) + return output_path, result + + def test_sigma_eps_structure_is_dp_mixture(self, _dp_output_setup): + _, data = _dp_output_setup + assert data["model_spec"]["sigma_eps_structure"] == "dp_mixture" + + def test_model_name_includes_dp(self, _dp_output_setup): + _, data = _dp_output_setup + assert "dp_sigma" in data["model"] + + def test_has_cluster_weights(self, _dp_output_setup): + _, data = _dp_output_setup + assert "cluster_weights" in data["population_effects"] + + def test_sigma_eps_length_equals_k_clusters(self, _dp_output_setup): + _, data = _dp_output_setup + sigma_eps = data["population_effects"]["sigma_eps"] + assert len(sigma_eps) == 4 # K_clusters = 4 + + def test_cluster_weights_sum_to_one(self, _dp_output_setup): + _, data = _dp_output_setup + w = data["population_effects"]["cluster_weights"] + np.testing.assert_allclose(sum(w), 1.0, atol=1e-4) + + def test_still_has_standard_fields(self, _dp_output_setup): + _, data = _dp_output_setup + expected = { + "model", "model_spec", "inference", "population_effects", + "convergence", "n_pools", "n_obs", "pools", + } + assert expected.issubset(data.keys()) + + +# =========================================================================== +# TestGenerateOutputJSONStructural +# =========================================================================== + + +class TestGenerateOutputJSONStructural: + @pytest.fixture() + def _structural_output_setup(self, tmp_path, synthetic_structural_data): + """Produce structural model JSON output and return (path, data dict).""" + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + from quantammsim.noise_calibration.constants import K_OBS_COEFF + + data = synthetic_structural_data + K_cov = data["K_cov"] + K_archetypes = 3 + n_chains = data["n_chains"] + n_tiers = data["n_tiers"] + S = 10 + + np.random.seed(77) + structural_samples = { + "alpha_0": np.random.randn(S) * 0.1 + 2.0, + "alpha_chain": np.random.randn(S, n_chains - 1) * 0.1, + "alpha_tier": np.random.randn(S, n_tiers - 1) * 0.1, + "alpha_tvl": np.random.randn(S) * 0.01, + "W_gate": np.random.randn(S, K_cov, K_archetypes) * 0.3, + "beta": np.random.randn(S, K_archetypes, K_OBS_COEFF) * 0.5, + "df": np.full((S,), 5.0), + "sigma_eps": np.full((S,), 0.5), + } + + pool_params = extract_structural_params(structural_samples, data) + output_path = str(tmp_path / "structural_output.json") + convergence = {"method": "svi", "final_elbo": 999.0} + inference_config = {"method": "svi", "svi_steps": 2000} + + generate_output_json( + pool_params, structural_samples, data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + result = json.load(f) + return output_path, result + + def test_output_model_name(self, _structural_output_setup): + _, data = _structural_output_setup + assert data["model"] == "structural_mixture" + + def test_output_has_arb_params(self, _structural_output_setup): + _, data = _structural_output_setup + pe = data["population_effects"] + assert "alpha_0" in pe + assert "alpha_chain" in pe + assert "alpha_tier" in pe + assert "alpha_tvl" in pe + + def test_output_has_archetype_info(self, _structural_output_setup): + _, data = _structural_output_setup + pe = data["population_effects"] + assert "W_gate" in pe + assert "beta" in pe + assert "K_archetypes" in pe + + def test_output_pools_have_arb_frequency(self, _structural_output_setup): + _, data = _structural_output_setup + pools = data["pools"] + for pid, entry in pools.items(): + assert "arb_frequency" in entry + assert isinstance(entry["arb_frequency"], int) + assert 1 <= entry["arb_frequency"] <= 60 diff --git a/tests/noise/test_panel_assembly.py b/tests/noise/test_panel_assembly.py new file mode 100644 index 00000000..dd7cd315 --- /dev/null +++ b/tests/noise/test_panel_assembly.py @@ -0,0 +1,442 @@ +"""Tests for compute_pair_volatility, assemble_panel, validate_panel.""" + +from datetime import date, timedelta + +import numpy as np +import pandas as pd +import pytest + +from quantammsim.noise_calibration import ( + compute_pair_volatility, + assemble_panel, + validate_panel, +) + + +# =========================================================================== +# TestComputePairVolatility +# =========================================================================== + + +class TestComputePairVolatility: + @pytest.fixture() + def _snap_dates(self): + """10 unique dates for snapshot stub.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + return pd.DataFrame({"date": dates}) + + def test_stablecoin_pair_returns_001(self, _snap_dates): + pool_row = pd.Series({ + "tokens": ["USDC", "DAI"], + "chain": "MAINNET", + }) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert (vol == 0.01).all() + + def test_both_missing_non_stable(self, _snap_dates): + pool_row = pd.Series({ + "tokens": ["FOO", "BAR"], + "chain": "MAINNET", + }) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert (vol == 0.5).all() + + def test_one_missing_non_stable(self, _snap_dates): + np.random.seed(77) + pool_row = pd.Series({ + "tokens": ["WETH", "BAR"], + "chain": "MAINNET", + }) + prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": [1735689600 + i * 3600 for i in range(48)], + "price": [3000.0 + np.random.randn() * 10 for _ in range(48)], + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, prices) + assert (vol == 0.5).all() + + def test_synthetic_hourly_prices_positive_finite(self, _snap_dates): + np.random.seed(42) + n_hours = 240 # 10 days x 24 hours + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + prices_a = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 8) + prices_b = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 7) + + pool_row = pd.Series({ + "tokens": ["WETH", "LINK"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": prices_a, + }), + ("MAINNET", "LINK"): pd.DataFrame({ + "timestamp": timestamps, "price": prices_b, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + assert len(vol) > 0 + assert (vol > 0).all() + assert np.all(np.isfinite(vol)) + + def test_annualisation_uses_sqrt_24x365(self, _snap_dates): + """Verify the annualisation factor is sqrt(24*365).""" + np.random.seed(7) + n_hours = 240 + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + raw_prices = np.exp(np.cumsum(np.random.randn(n_hours) * 0.005) + 8) + + pool_row = pd.Series({ + "tokens": ["WETH", "USDC"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": raw_prices, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + + # Reconstruct manually + df = pd.DataFrame({"timestamp": timestamps, "price": raw_prices}) + df["datetime"] = pd.to_datetime(df["timestamp"], unit="s") + df["date"] = df["datetime"].dt.date + df["ratio"] = df["price"] # WETH vs stable => ratio = price + df["log_return"] = np.log(df["ratio"] / df["ratio"].shift(1)) + df = df.dropna(subset=["log_return"]) + daily_std = df.groupby("date")["log_return"].std() + expected = daily_std * np.sqrt(24 * 365) + + common = vol.index.intersection(expected.index) + assert len(common) > 0 + np.testing.assert_allclose( + vol.loc[common].values, expected.loc[common].values, rtol=1e-10, + ) + + def test_single_token_pool_returns_empty(self, _snap_dates): + pool_row = pd.Series({"tokens": ["WETH"], "chain": "MAINNET"}) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert len(vol) == 0 + + def test_no_overlapping_price_dates(self, _snap_dates): + """Two tokens with non-overlapping timestamps -> fallback 0.5.""" + pool_row = pd.Series({ + "tokens": ["WETH", "LINK"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": [1000000 + i for i in range(10)], + "price": [3000.0] * 10, + }), + ("MAINNET", "LINK"): pd.DataFrame({ + "timestamp": [9000000 + i for i in range(10)], + "price": [15.0] * 10, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + assert (vol == 0.5).all() + + def test_cross_chain_price_fallback(self, _snap_dates): + """Prices keyed to a different chain SHOULD be used as fallback. + + Token prices are chain-agnostic (WETH is WETH regardless of chain), + so an ARBITRUM pool should use MAINNET WETH prices if ARBITRUM + prices aren't available. + """ + np.random.seed(55) + n_hours = 240 + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + prices = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 8) + + pool_row = pd.Series({ + "tokens": ["WETH", "USDC"], + "chain": "ARBITRUM", + }) + token_prices = { + # Only MAINNET prices, but ARBITRUM pool should still use them + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": prices.tolist(), + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + # Should get real volatility, NOT the 0.5 fallback + assert len(vol) > 0 + assert not (vol == 0.5).all(), ( + "Cross-chain price fallback should have produced real volatility" + ) + + +# =========================================================================== +# TestAssemblePanel +# =========================================================================== + + +class TestAssemblePanel: + def test_lagged_tvl_drops_first_obs_per_pool( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + counts = panel.groupby("pool_id").size() + assert (counts == 9).all() + + def test_lagged_tvl_exact_values( + self, synthetic_pools_df, synthetic_snapshots_df + ): + """log_tvl_lag1[t] must equal log_tvl[t-1] for same pool (shift(1)).""" + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + for pid in panel["pool_id"].unique(): + pool = panel[panel["pool_id"] == pid].sort_values("date") + tvl_vals = pool["log_tvl"].values + lag_vals = pool["log_tvl_lag1"].values + # After dropping the first obs, lag[i] = tvl[i-1] in the original + # pre-drop series. Since the panel is sorted by date, each + # lag value should equal the log_tvl of the chronologically + # preceding observation. We verify consecutive pairs: for rows + # i and i+1, lag[i+1] == tvl[i]. + for i in range(len(tvl_vals) - 1): + np.testing.assert_allclose( + lag_vals[i + 1], tvl_vals[i], rtol=1e-14, + err_msg=f"Pool {pid} row {i+1}: lag should equal previous tvl", + ) + + def test_weekend_flag(self, synthetic_pools_df, synthetic_snapshots_df): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + for _, row in panel.iterrows(): + d = row["date"] + if not isinstance(d, date): + d = pd.Timestamp(d).date() + expected = 1.0 if d.weekday() >= 5 else 0.0 + assert row["weekend"] == expected, f"Wrong weekend flag for {d}" + + def test_log_volume_is_natural_log( + self, synthetic_pools_df, synthetic_snapshots_df + ): + """log_volume must equal ln(volume_usd), not log10 or log2.""" + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + snaps = synthetic_snapshots_df.copy() + # Join snapshots to panel by pool_id + date to verify exact log values + for _, row in panel.iterrows(): + pid = row["pool_id"] + d = row["date"] + snap_match = snaps[ + (snaps["pool_id"] == pid) & (snaps["date"] == d) + ] + assert len(snap_match) == 1, f"No snapshot for {pid} on {d}" + expected = np.log(snap_match.iloc[0]["volume_usd"]) + np.testing.assert_allclose( + row["log_volume"], expected, rtol=1e-14, + err_msg=f"log_volume for {pid} on {d} should be ln(volume_usd)", + ) + + def test_tier_assignment_min_first( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + # Pool A: WETH(0), USDC(0) -> tier_A=0, tier_B=0 + pool_a = panel[panel["pool_id"] == "pool_A"].iloc[0] + assert pool_a["tier_A"] == 0 + assert pool_a["tier_B"] == 0 + + # Pool B: BAL(1), WETH(0) -> tier_A=0, tier_B=1 + pool_b = panel[panel["pool_id"] == "pool_B"].iloc[0] + assert pool_b["tier_A"] == 0 + assert pool_b["tier_B"] == 1 + + # Pool C: RATS(2), WETH(0) -> tier_A=0, tier_B=2 + pool_c = panel[panel["pool_id"] == "pool_C"].iloc[0] + assert pool_c["tier_A"] == 0 + assert pool_c["tier_B"] == 2 + + def test_log_fee(self, synthetic_pools_df, synthetic_snapshots_df): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + pool_a = panel[panel["pool_id"] == "pool_A"].iloc[0] + assert np.isclose(pool_a["log_fee"], np.log(0.003)) + + def test_all_expected_columns( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + expected_cols = { + "pool_id", "chain", "date", "log_volume", "log_tvl", + "log_tvl_lag1", "volatility", "weekend", "log_fee", + "swap_fee", "tier_A", "tier_B", "tokens", + } + assert expected_cols.issubset(set(panel.columns)) + + def test_zero_volume_rows_dropped(self, synthetic_pools_df): + """Rows with volume_usd <= 0 must be excluded from the panel.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(5)] + records = [] + for d in dates: + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", "date": d, + "volume_usd": 1000.0, + "total_liquidity_usd": 100000.0, + }) + # Add a zero-volume row + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", + "date": date(2026, 1, 6), + "volume_usd": 0.0, + "total_liquidity_usd": 100000.0, + }) + snaps = pd.DataFrame(records) + panel = assemble_panel(synthetic_pools_df, snaps, {}) + pool_a = panel[panel["pool_id"] == "pool_A"] + # 5 valid rows, minus 1 for lag = 4 (the zero-volume row is skipped) + assert len(pool_a) == 4 + + def test_zero_tvl_rows_dropped(self, synthetic_pools_df): + """Rows with total_liquidity_usd <= 0 must be excluded.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(5)] + records = [] + for d in dates: + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", "date": d, + "volume_usd": 1000.0, + "total_liquidity_usd": 100000.0, + }) + # Add a zero-TVL row + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", + "date": date(2026, 1, 6), + "volume_usd": 1000.0, + "total_liquidity_usd": 0.0, + }) + snaps = pd.DataFrame(records) + panel = assemble_panel(synthetic_pools_df, snaps, {}) + pool_a = panel[panel["pool_id"] == "pool_A"] + assert len(pool_a) == 4 + + +# =========================================================================== +# TestSyntheticPanelColumns — structural model covariates in the fixture +# =========================================================================== + + +class TestSyntheticPanelColumns: + """Tests that the synthetic_panel fixture has new structural columns.""" + + def test_panel_has_log_sigma(self, synthetic_panel): + assert "log_sigma" in synthetic_panel.columns + expected = np.log(np.maximum(synthetic_panel["volatility"].values, 1e-6)) + np.testing.assert_allclose( + synthetic_panel["log_sigma"].values, expected, rtol=1e-12, + ) + + def test_panel_has_dow_harmonics(self, synthetic_panel): + assert "dow_sin" in synthetic_panel.columns + assert "dow_cos" in synthetic_panel.columns + assert (synthetic_panel["dow_sin"] >= -1.0).all() + assert (synthetic_panel["dow_sin"] <= 1.0).all() + assert (synthetic_panel["dow_cos"] >= -1.0).all() + assert (synthetic_panel["dow_cos"] <= 1.0).all() + + def test_panel_has_interactions(self, synthetic_panel): + assert "tvl_x_sigma" in synthetic_panel.columns + assert "tvl_x_fee" in synthetic_panel.columns + assert "sigma_x_fee" in synthetic_panel.columns + + def test_dow_harmonics_correct_for_known_date(self, synthetic_panel): + """2026-01-03 is Saturday (weekday=5), so dow=5.""" + sat_rows = synthetic_panel[ + synthetic_panel["date"] == date(2026, 1, 3) + ] + if len(sat_rows) == 0: + pytest.skip("No Saturday rows in fixture") + expected_sin = np.sin(2 * np.pi * 5 / 7) + expected_cos = np.cos(2 * np.pi * 5 / 7) + np.testing.assert_allclose( + sat_rows["dow_sin"].values[0], expected_sin, atol=1e-12, + ) + np.testing.assert_allclose( + sat_rows["dow_cos"].values[0], expected_cos, atol=1e-12, + ) + + def test_interactions_use_lagged_tvl(self, synthetic_panel): + """tvl_x_sigma must use log_tvl_lag1, not log_tvl.""" + expected = ( + synthetic_panel["log_tvl_lag1"].values + * synthetic_panel["log_sigma"].values + ) + np.testing.assert_allclose( + synthetic_panel["tvl_x_sigma"].values, expected, rtol=1e-12, + ) + + +# =========================================================================== +# TestAssemblePanelStructuralColumns — new columns from real pipeline +# =========================================================================== + + +class TestAssemblePanelStructuralColumns: + """Tests that assemble_panel() produces new structural columns.""" + + def test_assemble_panel_has_log_sigma( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "log_sigma" in panel.columns + expected = np.log(np.maximum(panel["volatility"].values, 1e-6)) + np.testing.assert_allclose( + panel["log_sigma"].values, expected, rtol=1e-12, + ) + + def test_assemble_panel_has_dow_harmonics( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "dow_sin" in panel.columns + assert "dow_cos" in panel.columns + + def test_assemble_panel_has_interactions( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "tvl_x_sigma" in panel.columns + assert "tvl_x_fee" in panel.columns + assert "sigma_x_fee" in panel.columns + # Interactions use lagged TVL + expected = panel["log_tvl_lag1"].values * panel["log_sigma"].values + np.testing.assert_allclose( + panel["tvl_x_sigma"].values, expected, rtol=1e-12, + ) + + +# =========================================================================== +# TestValidatePanel +# =========================================================================== + + +class TestValidatePanel: + def test_flags_constant_volume(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + mask = panel["pool_id"] == "pool_A" + panel.loc[mask, "log_volume"] = 10.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "near-constant" in captured.out or "constant" in captured.out.lower() + + def test_flags_tvl_jumps(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + idx = panel[panel["pool_id"] == "pool_A"].index[1] + panel.loc[idx, "log_tvl"] = panel.loc[idx, "log_tvl"] + 5.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "TVL jumps" in captured.out + + def test_flags_volume_exceeds_tvl(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + panel["log_volume"] = panel["log_tvl"] + 1.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "volume > TVL" in captured.out + + def test_returns_dataframe_unchanged(self, synthetic_panel): + result = validate_panel(synthetic_panel) + pd.testing.assert_frame_equal(result, synthetic_panel) diff --git a/tests/noise/test_postprocessing.py b/tests/noise/test_postprocessing.py new file mode 100644 index 00000000..3d7d2e41 --- /dev/null +++ b/tests/noise/test_postprocessing.py @@ -0,0 +1,533 @@ +"""Tests for extract_noise_params, predict_new_pool, check_convergence, +assign_dp_clusters, and structural model post-processing.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + extract_noise_params, + predict_new_pool, + check_convergence, + classify_token_tier, + _get_theta_samples, + K_COEFF, + COEFF_NAMES, +) +from quantammsim.noise_calibration.constants import K_OBS_COEFF, OBS_COEFF_NAMES + + +# =========================================================================== +# TestExtractNoiseParams +# =========================================================================== + + +class TestExtractNoiseParams: + def test_output_length(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + assert len(result) == synthetic_encoded_data["N_pools"] + + def test_weekend_absorption(self, synthetic_samples, synthetic_encoded_data): + """b_0_eff = b_0_raw + b_weekend * (2/7).""" + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + + theta = _get_theta_samples( + synthetic_samples, synthetic_encoded_data["X_pool"] + ) + theta_med = np.median(theta, axis=0) + + for i, p in enumerate(result): + b_0_raw = theta_med[i, 0] + b_weekend = theta_med[i, 3] + expected_b_0 = b_0_raw + b_weekend * (2.0 / 7.0) + np.testing.assert_allclose( + p["noise_params"]["b_0"], expected_b_0, atol=1e-10, + ) + + def test_noise_params_keys(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + expected_keys = {"b_0", "b_sigma", "b_c", "b_weekend", "base_fee"} + for p in result: + assert set(p["noise_params"].keys()) == expected_keys + + def test_theta_median_length(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + assert len(p["theta_median"]) == K_COEFF + + def test_b_c_equals_theta_1(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + np.testing.assert_allclose( + p["noise_params"]["b_c"], p["theta_median"][1], atol=1e-10, + ) + + def test_b_sigma_equals_theta_2( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + np.testing.assert_allclose( + p["noise_params"]["b_sigma"], p["theta_median"][2], atol=1e-10, + ) + + def test_base_fee_matches_pool( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + pool_meta = synthetic_encoded_data["pool_meta"] + for i, p in enumerate(result): + expected_fee = pool_meta.iloc[i]["swap_fee"] + np.testing.assert_allclose( + p["noise_params"]["base_fee"], expected_fee, atol=1e-10, + ) + + def test_pool_id_and_chain_preserved( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + pool_ids = synthetic_encoded_data["pool_ids"] + pool_meta = synthetic_encoded_data["pool_meta"] + for i, p in enumerate(result): + assert p["pool_id"] == pool_ids[i] + assert p["chain"] == str(pool_meta.iloc[i]["chain"]) + + def test_use_median_false_uses_mean( + self, synthetic_samples, synthetic_encoded_data + ): + """use_median=False must produce different values than use_median=True.""" + result_med = extract_noise_params( + synthetic_samples, synthetic_encoded_data, use_median=True + ) + result_mean = extract_noise_params( + synthetic_samples, synthetic_encoded_data, use_median=False + ) + assert len(result_mean) == len(result_med) + + # With random B samples (S=10, seed 99), median != mean for at least + # one pool. Check that at least one theta_median value differs. + any_differ = False + for pm, pn in zip(result_med, result_mean): + for tm, tn in zip(pm["theta_median"], pn["theta_median"]): + if not np.isclose(tm, tn, atol=1e-14): + any_differ = True + break + if any_differ: + break + assert any_differ, "median and mean paths produced identical values" + + +# =========================================================================== +# TestPredictNewPool +# =========================================================================== + + +class TestPredictNewPool: + def _build_z_new(self, data, chain, tokens, fee): + """Reconstruct z_new the same way predict_new_pool does internally.""" + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b}": + z_new[i] = 1.0 + return z_new + + def test_known_chain_sets_dummy( + self, synthetic_samples, synthetic_encoded_data + ): + """ARBITRUM dummy must be 1 and must affect the prediction vs MAINNET.""" + data = synthetic_encoded_data + + result_arb = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "USDC"], fee=0.003, + ) + result_main = predict_new_pool( + synthetic_samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], fee=0.003, + ) + # MAINNET is the reference chain (alphabetically: ARBITRUM < BASE < MAINNET). + # Wait — ARBITRUM is alphabetically first, so ARBITRUM is the reference. + # MAINNET has a chain_MAINNET dummy. Both should produce different mu. + # If chain dummies are ignored, these would be identical. + arb_b0 = result_arb["noise_params"]["b_0"] + main_b0 = result_main["noise_params"]["b_0"] + assert not np.isclose(arb_b0, main_b0, atol=1e-10), ( + "Different chains should produce different predictions" + ) + + def test_tier_assignment_affects_prediction( + self, synthetic_samples, synthetic_encoded_data + ): + """WETH/RATS (tier 0,2) vs WETH/USDC (tier 0,0) must differ.""" + data = synthetic_encoded_data + + result_rats = predict_new_pool( + synthetic_samples, data, + chain="BASE", tokens=["WETH", "RATS"], fee=0.005, + ) + result_usdc = predict_new_pool( + synthetic_samples, data, + chain="BASE", tokens=["WETH", "USDC"], fee=0.005, + ) + # Different tier_B dummies should give different mu + rats_b0 = result_rats["noise_params"]["b_0"] + usdc_b0 = result_usdc["noise_params"]["b_0"] + assert not np.isclose(rats_b0, usdc_b0, atol=1e-10), ( + "Different tier assignments should produce different predictions" + ) + + def test_weekend_absorption_arithmetic( + self, synthetic_samples, synthetic_encoded_data + ): + """b_0 in noise_params must equal mu_median[0] + mu_median[3] * (2/7).""" + data = synthetic_encoded_data + result = predict_new_pool( + synthetic_samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], fee=0.003, + ) + # Reconstruct mu_median independently + z_new = self._build_z_new(data, "MAINNET", ["WETH", "USDC"], 0.003) + B = synthetic_samples["B"] + mu_samples = np.einsum("skd,d->sk", B, z_new) + mu_median = np.median(mu_samples, axis=0) + + b_0_raw = mu_median[0] + b_weekend = mu_median[3] + expected_b_0 = b_0_raw + b_weekend * (2.0 / 7.0) + np.testing.assert_allclose( + result["noise_params"]["b_0"], expected_b_0, atol=1e-10, + ) + + def test_mu_equals_b_at_z_new( + self, synthetic_samples, synthetic_encoded_data + ): + """With known B samples, verify credible interval medians = median(B @ z_new).""" + data = synthetic_encoded_data + z_new = self._build_z_new(data, "ARBITRUM", ["WETH", "RATS"], 0.005) + + B = synthetic_samples["B"] + mu_expected = np.einsum("skd,d->sk", B, z_new) + mu_median_expected = np.median(mu_expected, axis=0) + + result = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "RATS"], fee=0.005, + ) + for k, name in enumerate(COEFF_NAMES): + np.testing.assert_allclose( + result["credible_intervals_90"][name]["median"], + mu_median_expected[k], + atol=1e-10, + ) + + def test_unseen_chain_uses_reference( + self, synthetic_samples, synthetic_encoded_data + ): + """A chain not in training data should get reference-chain prediction + (all chain dummies = 0), not raise an error.""" + data = synthetic_encoded_data + result = predict_new_pool( + synthetic_samples, data, + chain="SONIC", tokens=["WETH", "USDC"], fee=0.003, + ) + # Should be same as ARBITRUM (the reference chain, all dummies 0) + result_ref = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "USDC"], fee=0.003, + ) + np.testing.assert_allclose( + result["noise_params"]["b_0"], + result_ref["noise_params"]["b_0"], + atol=1e-10, + ) + + +# =========================================================================== +# TestCheckConvergence +# =========================================================================== + + +class TestCheckConvergence: + def test_svi_returns_expected_keys(self): + losses = np.random.randn(1000).cumsum() + 5000 + result = check_convergence(losses, method="svi") + assert "final_elbo" in result + assert "elbo_last_100_std" in result + assert "elbo_last_100_mean" in result + + def test_svi_method_key(self): + losses = np.linspace(5000, 1000, 500) + result = check_convergence(losses, method="svi") + assert result["method"] == "svi" + + def test_svi_elbo_last_100_std_correct(self): + np.random.seed(42) + losses = np.random.randn(500) * 10 + 1000 + result = check_convergence(losses, method="svi") + expected_std = float(np.std(losses[-100:])) + np.testing.assert_allclose( + result["elbo_last_100_std"], expected_std, atol=1e-10, + ) + + def test_svi_final_elbo_is_last_loss(self): + losses = np.array([100.0, 50.0, 25.0, 12.5]) + result = check_convergence(losses, method="svi") + assert result["final_elbo"] == 12.5 + + +# =========================================================================== +# TestAssignDPClusters +# =========================================================================== + + +class TestAssignDPClusters: + @pytest.fixture() + def dp_samples_and_data(self, synthetic_encoded_data): + """Synthetic DP posterior samples with known cluster structure.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_clusters = 4 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_COEFF, K_cov) * 0.5 + sigma_theta = np.ones((S, K_COEFF)) + L_Omega = np.tile(np.eye(K_COEFF), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_COEFF)) + df = np.full((S,), 5.0) + + # Well-separated sigma_eps clusters + sigma_eps = np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)) + # Stick-breaking weights: mostly on cluster 0 + v = np.tile([0.6, 0.3, 0.05], (S, 1)) + + samples = { + "B": B, "sigma_theta": sigma_theta, "L_Omega": L_Omega, + "eta": eta, "df": df, "sigma_eps": sigma_eps, "v": v, + } + data_with_k = dict(data) + data_with_k["K_clusters"] = K_clusters + return samples, data_with_k + + def test_returns_correct_length(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + assert len(assignments) == data["N_pools"] + + def test_valid_cluster_indices(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + K = data["K_clusters"] + assert all(0 <= a < K for a in assignments) + + def test_returns_integer_array(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + assert assignments.dtype in (np.int32, np.int64, int) + + +# =========================================================================== +# TestExtractNoiseParamsDP +# =========================================================================== + + +class TestExtractNoiseParamsDP: + @pytest.fixture() + def dp_samples_and_data(self, synthetic_encoded_data): + """Same as above for extract_noise_params testing.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_COEFF, K_cov) * 0.5 + sigma_theta = np.ones((S, K_COEFF)) + L_Omega = np.tile(np.eye(K_COEFF), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_COEFF)) + df = np.full((S,), 5.0) + sigma_eps = np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)) + v = np.tile([0.6, 0.3, 0.05], (S, 1)) + + samples = { + "B": B, "sigma_theta": sigma_theta, "L_Omega": L_Omega, + "eta": eta, "df": df, "sigma_eps": sigma_eps, "v": v, + } + return samples, data + + def test_output_length(self, dp_samples_and_data): + samples, data = dp_samples_and_data + result = extract_noise_params(samples, data) + assert len(result) == data["N_pools"] + + def test_noise_params_keys_present(self, dp_samples_and_data): + samples, data = dp_samples_and_data + result = extract_noise_params(samples, data) + expected_keys = {"b_0", "b_sigma", "b_c", "b_weekend", "base_fee"} + for p in result: + assert set(p["noise_params"].keys()) == expected_keys + + +# =========================================================================== +# TestExtractStructuralParams +# =========================================================================== + + +class TestExtractStructuralParams: + """Tests for extract_structural_params().""" + + @pytest.fixture() + def structural_samples_and_data(self, synthetic_structural_data): + """Run a quick SVI fit to get structural samples.""" + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + return samples, synthetic_structural_data + + def test_extract_returns_arb_params(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + samples, data = structural_samples_and_data + result = extract_structural_params(samples, data) + assert len(result) == data["N_pools"] + for p in result: + assert "arb_frequency" in p + assert isinstance(p["arb_frequency"], int) + assert 1 <= p["arb_frequency"] <= 60 + + def test_extract_returns_noise_params(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + samples, data = structural_samples_and_data + result = extract_structural_params(samples, data) + for p in result: + assert "noise_params" in p + coeffs = p["noise_params"] + assert len(coeffs) == K_OBS_COEFF + for name in OBS_COEFF_NAMES: + assert name in coeffs, f"Missing coefficient: {name}" + + +# =========================================================================== +# TestPredictStructural +# =========================================================================== + + +class TestPredictStructural: + @pytest.fixture() + def structural_samples_and_data(self, synthetic_structural_data): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + return samples, synthetic_structural_data + + def test_predict_returns_cadence(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + result = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + assert "arb_frequency" in result + assert isinstance(result["arb_frequency"], int) + assert 1 <= result["arb_frequency"] <= 60 + + def test_predict_returns_noise_coefficients( + self, structural_samples_and_data, + ): + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + result = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + assert "noise_params" in result + assert len(result["noise_params"]) == K_OBS_COEFF + + def test_predict_uses_W_gate(self, structural_samples_and_data): + """Different (chain, tier) → different predictions.""" + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + r1 = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + r2 = predict_new_pool_structural( + samples, data, + chain="ARBITRUM", tokens=["BAL", "WETH"], + fee=0.01, tvl_est=5e5, + ) + # Different pool characteristics should produce different noise params + assert r1["noise_params"] != r2["noise_params"] + + def test_predict_cadence_higher_for_longtail( + self, structural_samples_and_data, + ): + """Long-tail pools should have higher cadence (less efficient arb).""" + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + # This tests the structural relationship — it may not hold with + # random SVI samples on synthetic data, so we just check it doesn't + # crash and returns valid values + r1 = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + r2 = predict_new_pool_structural( + samples, data, + chain="BASE", tokens=["RATS", "WETH"], + fee=0.005, tvl_est=1e5, + ) + # Both should be valid + assert 1 <= r1["arb_frequency"] <= 60 + assert 1 <= r2["arb_frequency"] <= 60 diff --git a/tests/noise/test_token_classification.py b/tests/noise/test_token_classification.py new file mode 100644 index 00000000..dd540f84 --- /dev/null +++ b/tests/noise/test_token_classification.py @@ -0,0 +1,66 @@ +"""Tests for _normalise_symbol and classify_token_tier.""" + +import pytest +from quantammsim.noise_calibration import _normalise_symbol, classify_token_tier + + +# =========================================================================== +# TestNormaliseSymbol +# =========================================================================== + + +class TestNormaliseSymbol: + def test_passthrough_unknown(self): + assert _normalise_symbol("FOO") == "FOO" + + def test_known_mapping_preserved(self): + assert _normalise_symbol("WETH") == "WETH" + assert _normalise_symbol("WBTC") == "WBTC" + assert _normalise_symbol("cbBTC") == "cbBTC" + + def test_whitespace_stripped(self): + assert _normalise_symbol(" ETH ") == "ETH" + assert _normalise_symbol(" WETH ") == "WETH" + + def test_case_sensitivity_preserved(self): + # Lowercase is NOT normalised to uppercase + assert _normalise_symbol("weth") == "weth" + + +# =========================================================================== +# TestClassifyTokenTier +# =========================================================================== + + +class TestClassifyTokenTier: + def test_tier0_native_tokens(self): + for sym in ["ETH", "BTC"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_wrapped(self): + for sym in ["WETH", "WBTC", "cbBTC"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_stablecoins(self): + for sym in ["USDC", "USDT", "DAI"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_chain_natives(self): + for sym in ["MATIC", "AVAX", "GNO", "S", "wS"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier1_defi_bluechips(self): + for sym in ["AAVE", "BAL", "COW", "LINK", "ARB"]: + assert classify_token_tier(sym) == 1, f"{sym} should be tier 1" + + def test_tier2_unknown_tokens(self): + for sym in ["RATS", "PEPE"]: + assert classify_token_tier(sym) == 2, f"{sym} should be tier 2" + + def test_tier2_empty_string(self): + assert classify_token_tier("") == 2 + + def test_wrapped_variant_normalisation(self): + # "wS" is in the mapping table AND in _TIER_0 + assert classify_token_tier("wS") == 0 + assert classify_token_tier(" wS ") == 0 From 9812099cf50fca3c97c2bb5130857ba4edf753c8 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 6 Mar 2026 16:56:45 +0000 Subject: [PATCH 016/115] port over and combine with dynamic obj noise modelling approach --- quantammsim/pools/reCLAMM/reclamm.py | 190 +++++++++- quantammsim/pools/reCLAMM/reclamm_reserves.py | 329 +++++++++++++++--- 2 files changed, 460 insertions(+), 59 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index fbfecab4..aa57d3ed 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -10,7 +10,7 @@ config.update("jax_enable_x64", True) import jax.numpy as jnp -from jax import jit, tree_util +from jax import jit, tree_util, vmap from jax.lax import dynamic_slice from functools import partial from typing import Dict, Any, Optional, NamedTuple @@ -34,6 +34,73 @@ SHIFT_EXPONENT_DIVISOR = 124649.0 +def _prepare_dynamic_array(arr, start_index, bout_length, arb_frequency, max_len): + """Slice and decimate a dynamic input array to match arb_prices shape.""" + arr = jnp.asarray(arr) + if arr.ndim == 0: + return jnp.full((max_len,), arr, dtype=arr.dtype) + if arr.shape[0] <= 1: + return jnp.broadcast_to(arr, (max_len,) + arr.shape[1:]) + + start = (start_index[0],) + (0,) * (arr.ndim - 1) + slice_sizes = (bout_length - 1,) + arr.shape[1:] + sliced = dynamic_slice(arr, start, slice_sizes) + if arb_frequency != 1: + sliced = sliced[::arb_frequency] + return sliced + + +def _align_prices_to_numeraire(prices, run_fingerprint): + """Ensure the numeraire token is in column 1 for ratio-volatility calc.""" + tokens = run_fingerprint.get("tokens") + numeraire = run_fingerprint.get("numeraire") + if tokens is None or numeraire is None or len(tokens) != 2: + return prices + + token_labels = [str(token).lower() for token in tokens] + numeraire_label = str(numeraire).lower() + if token_labels[0] == numeraire_label: + return prices[:, ::-1] + return prices + + +def _calculate_annualized_ratio_volatility( + prices, run_fingerprint, subsample_freq=5, +): + """Annualized daily realized volatility broadcast to minute-level array.""" + ordered_prices = _align_prices_to_numeraire(prices, run_fingerprint) + asset_prices = ordered_prices[:, 0] / ordered_prices[:, 1] + n_minutes = asset_prices.shape[0] + + if n_minutes < 1440: + return jnp.full((n_minutes,), 0.1 * jnp.sqrt(365.0), dtype=prices.dtype) + + n_days = n_minutes // 1440 + + def calculate_daily_volatility(day_idx): + start_idx = day_idx * 1440 + window_prices = dynamic_slice(asset_prices, (start_idx,), (1440,)) + subsampled_prices = window_prices[::subsample_freq] + log_prices = jnp.log(jnp.maximum(subsampled_prices, 1e-8)) + returns = jnp.diff(log_prices) + num_nonzero_returns = jnp.sum(returns != 0) + total_returns = jnp.maximum(returns.shape[0], 1) + adjusted_variance = num_nonzero_returns * jnp.var(returns) / total_returns + dt = subsample_freq / 1440 + return jnp.sqrt(adjusted_variance) / jnp.sqrt(dt) + + daily_volatilities = vmap(calculate_daily_volatility)(jnp.arange(n_days)) + volatility_array = jnp.repeat(daily_volatilities, 1440) + + remaining_minutes = n_minutes - volatility_array.shape[0] + if remaining_minutes > 0: + volatility_array = jnp.concatenate( + [volatility_array, jnp.full((remaining_minutes,), daily_volatilities[-1])] + ) + + return volatility_array * jnp.sqrt(365.0) + + class _PoolState(NamedTuple): """Intermediate state produced by _init_pool_state. @@ -203,6 +270,48 @@ def _resolve_ste_temperature(run_fingerprint): """Resolve STE gate temperature for differentiable reCLAMM transitions.""" return run_fingerprint.get("ste_temperature") + def _resolve_noise_inputs( + self, + run_fingerprint: Dict[str, Any], + prices: jnp.ndarray, + start_index: jnp.ndarray, + arb_len: int, + lp_supply_array: Optional[jnp.ndarray] = None, + ): + """Prepare optional lp-supply and noise-model inputs for reserve scans.""" + bout_length = run_fingerprint["bout_length"] + arb_freq = run_fingerprint["arb_frequency"] + + lp_prepared = None + if lp_supply_array is not None: + lp_prepared = _prepare_dynamic_array( + lp_supply_array, + start_index=start_index, + bout_length=bout_length, + arb_frequency=arb_freq, + max_len=arb_len, + ) + + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + arb_vol = None + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility_array = _calculate_annualized_ratio_volatility( + prices, run_fingerprint + ) + arb_vol = _prepare_dynamic_array( + volatility_array, + start_index=start_index, + bout_length=bout_length, + arb_frequency=arb_freq, + max_len=arb_len, + ) + + return lp_prepared, noise_model, noise_params, arb_vol + @partial(jit, static_argnums=(2,)) def calculate_reserves_with_fees( self, @@ -211,9 +320,17 @@ def calculate_reserves_with_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) ste_temperature = self._resolve_ste_temperature(run_fingerprint) + lp_prepared, noise_model, noise_params, arb_vol = self._resolve_noise_inputs( + run_fingerprint, + prices, + start_index, + s.arb_prices.shape[0], + lp_supply_array=lp_supply_array, + ) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_with_fees( @@ -232,6 +349,11 @@ def calculate_reserves_with_fees( centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), ste_temperature=ste_temperature, + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=lp_prepared, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -243,6 +365,7 @@ def calculate_reserves_and_fee_revenue_with_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ): """Calculate reserves and LP fee revenue with fees. @@ -254,6 +377,13 @@ def calculate_reserves_and_fee_revenue_with_fees( """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) ste_temperature = self._resolve_ste_temperature(run_fingerprint) + lp_prepared, noise_model, noise_params, arb_vol = self._resolve_noise_inputs( + run_fingerprint, + prices, + start_index, + s.arb_prices.shape[0], + lp_supply_array=lp_supply_array, + ) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -272,6 +402,11 @@ def calculate_reserves_and_fee_revenue_with_fees( centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), ste_temperature=ste_temperature, + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=lp_prepared, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), @@ -298,10 +433,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) ste_temperature = self._resolve_ste_temperature(run_fingerprint) - bout_length = run_fingerprint["bout_length"] - max_len = bout_length - 1 - if run_fingerprint["arb_frequency"] != 1: - max_len = max_len // run_fingerprint["arb_frequency"] + max_len = s.arb_prices.shape[0] materialized_inputs = materialize_dynamic_inputs( dynamic_inputs, run_fingerprint.get("dynamic_input_flags"), @@ -310,6 +442,13 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( do_trades=False, dtype=s.arb_prices.dtype, ) + _, noise_model, noise_params, arb_vol = self._resolve_noise_inputs( + run_fingerprint, + prices, + start_index, + s.arb_prices.shape[0], + lp_supply_array=None, + ) return _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, @@ -321,6 +460,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( arb_thresh=materialized_inputs.gas_cost, arb_fees=materialized_inputs.arb_fees, price_ratio_updates=materialized_inputs.reclamm_price_ratio_updates, + lp_supply_array=materialized_inputs.lp_supply, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), @@ -328,6 +468,10 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), ste_temperature=ste_temperature, + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) @partial(jit, static_argnums=(2,)) @@ -338,10 +482,20 @@ def _calculate_reserves_zero_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: """Protected zero-fee implementation for hooks and weight calculation.""" s = self._init_pool_state(params, run_fingerprint, prices, start_index) ste_temperature = self._resolve_ste_temperature(run_fingerprint) + lp_prepared = None + if lp_supply_array is not None: + lp_prepared = _prepare_dynamic_array( + lp_supply_array, + start_index=start_index, + bout_length=run_fingerprint["bout_length"], + arb_frequency=run_fingerprint["arb_frequency"], + max_len=s.arb_prices.shape[0], + ) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_zero_fees( @@ -353,6 +507,7 @@ def _calculate_reserves_zero_fees( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, ste_temperature=ste_temperature, + lp_supply_array=lp_prepared, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -363,9 +518,15 @@ def calculate_reserves_zero_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: return self._calculate_reserves_zero_fees( - params, run_fingerprint, prices, start_index, additional_oracle_input + params, + run_fingerprint, + prices, + start_index, + additional_oracle_input, + lp_supply_array, ) @partial(jit, static_argnums=(2,)) @@ -380,10 +541,7 @@ def calculate_reserves_with_dynamic_inputs( ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) ste_temperature = self._resolve_ste_temperature(run_fingerprint) - bout_length = run_fingerprint["bout_length"] - max_len = bout_length - 1 - if run_fingerprint["arb_frequency"] != 1: - max_len = max_len // run_fingerprint["arb_frequency"] + max_len = s.arb_prices.shape[0] materialized_inputs = materialize_dynamic_inputs( dynamic_inputs, run_fingerprint.get("dynamic_input_flags"), @@ -392,6 +550,13 @@ def calculate_reserves_with_dynamic_inputs( do_trades=False, dtype=s.arb_prices.dtype, ) + _, noise_model, noise_params, arb_vol = self._resolve_noise_inputs( + run_fingerprint, + prices, + start_index, + s.arb_prices.shape[0], + lp_supply_array=None, + ) return _jax_calc_reclamm_reserves_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, @@ -403,6 +568,7 @@ def calculate_reserves_with_dynamic_inputs( arb_thresh=materialized_inputs.gas_cost, arb_fees=materialized_inputs.arb_fees, price_ratio_updates=materialized_inputs.reclamm_price_ratio_updates, + lp_supply_array=materialized_inputs.lp_supply, all_sig_variations=jnp.array( run_fingerprint["all_sig_variations"] ), @@ -410,6 +576,10 @@ def calculate_reserves_with_dynamic_inputs( centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), ste_temperature=ste_temperature, + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) def init_base_parameters( diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 7e0bab8c..3a0e89ad 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -31,6 +31,12 @@ from quantammsim.pools.G3M.G3M_trades import ( _jax_calc_G3M_trade_from_exact_in_given_out, ) +from quantammsim.pools.noise_trades import ( + calculate_reserves_after_noise_trade, + reclamm_tsoukalas_sqrt_noise_volume, + reclamm_tsoukalas_log_noise_volume, + reclamm_loglinear_noise_volume, +) # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) _INITIALIZATION_MAX_BALANCE_A = 1e6 @@ -622,7 +628,7 @@ def apply_target_price_ratio_to_virtual_balances(Ra, Rb, Va, Vb, target_price_ra def _reclamm_scan_step_zero_fees( carry_list, - prices, + input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, @@ -636,11 +642,23 @@ def _reclamm_scan_step_zero_fees( 1. Update virtual balances (path-dependent) 2. Compute analytical constant-product arb (no fee friction) - Carry: [real_reserves (2,), Va (0-d), Vb (0-d)] + Carry: [real_reserves (2,), Va (0-d), Vb (0-d), prev_lp_supply (0-d)] + Input: [prices (2,), lp_supply (0-d)] """ prev_reserves = carry_list[0] Va = carry_list[1] Vb = carry_list[2] + prev_lp_supply = carry_list[3] + + prices = input_list[0] + lp_supply = input_list[1] + + # Scale both real and virtual reserves by LP supply ratio. + scale = lp_supply / prev_lp_supply + lp_supply_change = lp_supply != prev_lp_supply + prev_reserves = jnp.where(lp_supply_change, prev_reserves * scale, prev_reserves) + Va = jnp.where(lp_supply_change, Va * scale, Va) + Vb = jnp.where(lp_supply_change, Vb * scale, Vb) Ra = prev_reserves[0] Rb = prev_reserves[1] @@ -717,7 +735,7 @@ def _reclamm_scan_step_zero_fees( Rb_new = jnp.where(clamp_a, Rb + edge_a[1], jnp.where(clamp_b, Rb + edge_b[1], Rb_new)) new_reserves = jnp.array([Ra_new, Rb_new]) - return [new_reserves, Va, Vb], new_reserves + return [new_reserves, Va, Vb, lp_supply], new_reserves # --------------------------------------------------------------------------- @@ -729,7 +747,7 @@ def _reclamm_scan_step_zero_fees( def _reclamm_scan_step_zero_fees_full_state( carry_list, - prices, + input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, @@ -739,7 +757,7 @@ def _reclamm_scan_step_zero_fees_full_state( ): """TEST-ONLY: scan step that outputs (reserves, Va, Vb).""" new_carry, new_reserves = _reclamm_scan_step_zero_fees( - carry_list, prices, centeredness_margin, daily_price_shift_base, seconds_per_step, + carry_list, input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, ste_temperature=ste_temperature, @@ -761,15 +779,20 @@ def _reclamm_scan_step_with_fees_and_revenue( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + noise_model="ratio", + noise_params=None, ): """Single scan step for reClAMM pool with fees, returning LP fee revenue. Primary implementation — ``_reclamm_scan_step_with_fees`` wraps this. Carry: [real_reserves (2,), Va, Vb, step_idx, active_start_ratio, - active_target_ratio, active_start_step, active_end_step, active_enabled] + active_target_ratio, active_start_step, active_end_step, active_enabled, + prev_lp_supply] Input: [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_update] + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_update, + lp_supply] Returns ------- @@ -786,9 +809,7 @@ def _reclamm_scan_step_with_fees_and_revenue( active_start_step = carry_list[6] active_end_step = carry_list[7] active_enabled = carry_list[8] - - Ra = prev_reserves[0] - Rb = prev_reserves[1] + prev_lp_supply = carry_list[9] prices = input_list[0] active_initial_weights = input_list[1] @@ -798,6 +819,18 @@ def _reclamm_scan_step_with_fees_and_revenue( arb_thresh = input_list[5] arb_fees = input_list[6] price_ratio_update = input_list[7] + lp_supply = input_list[8] + + # Scale both real and virtual reserves by LP supply ratio so liquidity + # add/remove events preserve proportional pool state. + scale = lp_supply / prev_lp_supply + lp_supply_change = lp_supply != prev_lp_supply + prev_reserves = jnp.where(lp_supply_change, prev_reserves * scale, prev_reserves) + Va = jnp.where(lp_supply_change, Va * scale, Va) + Vb = jnp.where(lp_supply_change, Vb * scale, Vb) + + Ra = prev_reserves[0] + Rb = prev_reserves[1] event_has = price_ratio_update[0] > 0.5 event_target_ratio = jnp.maximum( @@ -973,6 +1006,39 @@ def _skip_schedule_state(_): Ra_new = Ra + applied_trade[0] Rb_new = Rb + applied_trade[1] + # Optional noise-trader model. + if noise_model == "ratio": + noisy_reserves = calculate_reserves_after_noise_trade( + applied_trade, jnp.array([Ra_new, Rb_new]), prices, + noise_trader_ratio, gamma, + ) + Ra_new = jnp.where(noise_trader_ratio > 0, noisy_reserves[0], Ra_new) + Rb_new = jnp.where(noise_trader_ratio > 0, noisy_reserves[1], Rb_new) + elif noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility = input_list[9] + arb_volume = 0.5 * jnp.sum(jnp.abs(applied_trade) * prices) + real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) + effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + + _np = noise_params if noise_params is not None else {} + if noise_model == "tsoukalas_sqrt": + noise_vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value, gamma, volatility, arb_volume, _np + ) + elif noise_model == "tsoukalas_log": + noise_vol = reclamm_tsoukalas_log_noise_volume( + effective_value, gamma, volatility, arb_volume, _np + ) + else: + noise_vol = reclamm_loglinear_noise_volume( + effective_value, gamma, volatility, arb_volume, _np + ) + + noise_fee_income = (1.0 - gamma) * noise_vol + noise_scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) + Ra_new = Ra_new * noise_scale + Rb_new = Rb_new * noise_scale + # Clamp-to-edge: if a real reserve would go negative, apply an # exact-in-given-out edge trade that drains that token to _DUST_USD # worth of reserves (preserving the AMM invariant). @@ -1021,6 +1087,7 @@ def _skip_schedule_state(_): active_start_step, active_end_step, active_enabled, + lp_supply, ], (new_reserves, lp_fee_revenue_usd) @@ -1038,6 +1105,9 @@ def _reclamm_scan_step_with_fees( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + noise_model="ratio", + noise_params=None, ): """Single scan step for reClAMM pool with fees (reserves only). @@ -1057,6 +1127,9 @@ def _reclamm_scan_step_with_fees( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params, ) return new_carry, new_reserves @@ -1075,6 +1148,9 @@ def _reclamm_scan_step_with_fees_full_state( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + noise_model="ratio", + noise_params=None, ): """TEST-ONLY: fee scan step that also outputs virtual balances.""" new_carry, (new_reserves, _fee_rev) = _reclamm_scan_step_with_fees_and_revenue( @@ -1090,6 +1166,9 @@ def _reclamm_scan_step_with_fees_full_state( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params, ) return new_carry, (new_reserves, new_carry[1], new_carry[2]) @@ -1106,6 +1185,7 @@ def _jax_calc_reclamm_reserves_zero_fees( arc_length_speed=0.0, centeredness_scaling=False, ste_temperature=10.0, + lp_supply_array=None, ): """Calculate reClAMM reserves over time with zero fees. @@ -1127,12 +1207,22 @@ def _jax_calc_reclamm_reserves_zero_fees( If > 0, use constant-arc-length thermostat instead of geometric. centeredness_scaling : bool If True, scale speed by margin/centeredness (proportional controller). + lp_supply_array : jnp.ndarray, optional + LP token supply over time, shape (T,). Defaults to constant 1.0. Returns ------- reserves : jnp.ndarray, shape (T, 2) Real reserves over time. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + scan_fn = Partial( _reclamm_scan_step_zero_fees, centeredness_margin=centeredness_margin, @@ -1143,8 +1233,8 @@ def _jax_calc_reclamm_reserves_zero_fees( ste_temperature=ste_temperature, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] - _, reserves = scan(scan_fn, carry_init, prices) + carry_init = [initial_reserves, initial_Va, initial_Vb, lp_supply_array[0]] + _, reserves = scan(scan_fn, carry_init, [prices, lp_supply_array]) return reserves @@ -1160,6 +1250,7 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( arc_length_speed=0.0, centeredness_scaling=False, ste_temperature=10.0, + lp_supply_array=None, ): """TEST-ONLY: Like _jax_calc_reclamm_reserves_zero_fees but returns Va/Vb. @@ -1169,6 +1260,14 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( Va_history : jnp.ndarray, shape (T,) Vb_history : jnp.ndarray, shape (T,) """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + scan_fn = Partial( _reclamm_scan_step_zero_fees_full_state, centeredness_margin=centeredness_margin, @@ -1179,12 +1278,14 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( ste_temperature=ste_temperature, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] - _, (reserves, Va_history, Vb_history) = scan(scan_fn, carry_init, prices) + carry_init = [initial_reserves, initial_Va, initial_Vb, lp_supply_array[0]] + _, (reserves, Va_history, Vb_history) = scan( + scan_fn, carry_init, [prices, lp_supply_array] + ) return reserves, Va_history, Vb_history -@jit +@partial(jit, static_argnames=("noise_model",)) def _jax_calc_reclamm_reserves_with_fees( initial_reserves, initial_Va, @@ -1201,12 +1302,25 @@ def _jax_calc_reclamm_reserves_with_fees( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves over time with fees. Uses the G3M optimal arb machinery with constant weights [0.5, 0.5] applied to effective reserves (real + virtual). """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) gamma = 1.0 - fees @@ -1242,6 +1356,9 @@ def _jax_calc_reclamm_reserves_with_fees( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) carry_init = [ @@ -1254,17 +1371,27 @@ def _jax_calc_reclamm_reserves_with_fees( jnp.float64(0.0), # active_start_step jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled + lp_supply_array[0], # prev_lp_supply ] - _, reserves = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma_array, + arb_thresh_array, + arb_fees_array, + price_ratio_updates, + lp_supply_array, + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + + _, reserves = scan(scan_fn, carry_init, scan_inputs) return reserves -@partial(jit, static_argnums=(11,)) +@partial(jit, static_argnums=(11,), static_argnames=("noise_model",)) def _jax_calc_reclamm_reserves_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1284,8 +1411,21 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) @@ -1334,6 +1474,9 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) carry_init = [ @@ -1346,17 +1489,27 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( jnp.float64(0.0), # active_start_step jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled + lp_supply_array[0], # prev_lp_supply ] - _, reserves = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + price_ratio_updates, + lp_supply_array, + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + + _, reserves = scan(scan_fn, carry_init, scan_inputs) return reserves -@partial(jit, static_argnums=(11,)) +@partial(jit, static_argnums=(11,), static_argnames=("noise_model",)) def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( initial_reserves, initial_Va, @@ -1376,8 +1529,21 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """TEST-ONLY: dynamic-input reserve path returning virtual-balance history.""" + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) @@ -1425,6 +1591,9 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) carry_init = [ @@ -1437,17 +1606,27 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( jnp.float64(0.0), # active_start_step jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled + lp_supply_array[0], # prev_lp_supply ] - _, (reserves, Va_history, Vb_history) = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + price_ratio_updates, + lp_supply_array, + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + + _, (reserves, Va_history, Vb_history) = scan(scan_fn, carry_init, scan_inputs) return reserves, Va_history, Vb_history -@jit +@partial(jit, static_argnames=("noise_model",)) def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( initial_reserves, initial_Va, @@ -1464,6 +1643,11 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1473,6 +1657,14 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( fee_revenue : jnp.ndarray, shape (T,) LP fee revenue per timestep in USD. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) gamma = 1.0 - fees @@ -1507,6 +1699,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) carry_init = [ @@ -1519,17 +1714,27 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( jnp.float64(0.0), # active_start_step jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled + lp_supply_array[0], # prev_lp_supply ] - _, (reserves, fee_revenue) = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma_array, + arb_thresh_array, + arb_fees_array, + price_ratio_updates, + lp_supply_array, + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + + _, (reserves, fee_revenue) = scan(scan_fn, carry_init, scan_inputs) return reserves, fee_revenue -@partial(jit, static_argnums=(11,)) +@partial(jit, static_argnums=(11,), static_argnames=("noise_model",)) def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1549,6 +1754,11 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( centeredness_scaling=False, protocol_fee_split=0.0, ste_temperature=10.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1558,6 +1768,14 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( fee_revenue : jnp.ndarray, shape (T,) LP fee revenue per timestep in USD. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) @@ -1605,6 +1823,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, ste_temperature=ste_temperature, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) carry_init = [ @@ -1617,11 +1838,21 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( jnp.float64(0.0), # active_start_step jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled + lp_supply_array[0], # prev_lp_supply ] - _, (reserves, fee_revenue) = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], - ) + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + price_ratio_updates, + lp_supply_array, + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + + _, (reserves, fee_revenue) = scan(scan_fn, carry_init, scan_inputs) return reserves, fee_revenue From d39ddcb035a47ede5517b4fbc23fb9fab9df0e07 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Sat, 7 Mar 2026 15:59:58 +0000 Subject: [PATCH 017/115] add missing imports --- pyproject.toml | 4 +++- setup.py | 12 +++++++++--- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 8859b551..19cb8f4e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -24,6 +24,8 @@ dependencies = [ "plotly", "dask", "Historic-Crypto", + "gdown", + "binance_historical_data", "bidask", "optax", "jsonpickle", @@ -113,4 +115,4 @@ exclude_lines = [ "def __repr__", "raise NotImplementedError", "if __name__ == .__main__.:", -] \ No newline at end of file +] diff --git a/setup.py b/setup.py index 20eb403f..1aa0b94f 100644 --- a/setup.py +++ b/setup.py @@ -20,7 +20,7 @@ "pyarrow", "plotly", "bidask", - "Historic_Crypto", + "Historic-Crypto", "gdown", "binance_historical_data", "dask", @@ -30,10 +30,12 @@ ], extras_require={ "dev": [ - "pytest>=6.0", + "pytest>=7.0", + "pytest-cov>=4.0", + "pytest-xdist>=3.0", + "pytest-timeout>=2.0", "black", "flake8", - "pytest-cov", "hypothesis", ], "docs": [ @@ -41,6 +43,10 @@ "sphinx-automodapi", "sphinx-rtd-theme", ], + "calibration": [ + "numpyro>=0.15.0", + "arviz>=0.15.0", + ], }, python_requires=">=3.9", ) From 9e2485761961c0ec9a8f1a9e3cfa0411556075f3 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Sat, 7 Mar 2026 16:16:21 +0000 Subject: [PATCH 018/115] add reclamm private repo port fixes --- quantammsim/pools/reCLAMM/reclamm.py | 57 +--------------------------- 1 file changed, 2 insertions(+), 55 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index aa57d3ed..afd24967 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -10,7 +10,7 @@ config.update("jax_enable_x64", True) import jax.numpy as jnp -from jax import jit, tree_util, vmap +from jax import jit, tree_util from jax.lax import dynamic_slice from functools import partial from typing import Dict, Any, Optional, NamedTuple @@ -50,57 +50,6 @@ def _prepare_dynamic_array(arr, start_index, bout_length, arb_frequency, max_len return sliced -def _align_prices_to_numeraire(prices, run_fingerprint): - """Ensure the numeraire token is in column 1 for ratio-volatility calc.""" - tokens = run_fingerprint.get("tokens") - numeraire = run_fingerprint.get("numeraire") - if tokens is None or numeraire is None or len(tokens) != 2: - return prices - - token_labels = [str(token).lower() for token in tokens] - numeraire_label = str(numeraire).lower() - if token_labels[0] == numeraire_label: - return prices[:, ::-1] - return prices - - -def _calculate_annualized_ratio_volatility( - prices, run_fingerprint, subsample_freq=5, -): - """Annualized daily realized volatility broadcast to minute-level array.""" - ordered_prices = _align_prices_to_numeraire(prices, run_fingerprint) - asset_prices = ordered_prices[:, 0] / ordered_prices[:, 1] - n_minutes = asset_prices.shape[0] - - if n_minutes < 1440: - return jnp.full((n_minutes,), 0.1 * jnp.sqrt(365.0), dtype=prices.dtype) - - n_days = n_minutes // 1440 - - def calculate_daily_volatility(day_idx): - start_idx = day_idx * 1440 - window_prices = dynamic_slice(asset_prices, (start_idx,), (1440,)) - subsampled_prices = window_prices[::subsample_freq] - log_prices = jnp.log(jnp.maximum(subsampled_prices, 1e-8)) - returns = jnp.diff(log_prices) - num_nonzero_returns = jnp.sum(returns != 0) - total_returns = jnp.maximum(returns.shape[0], 1) - adjusted_variance = num_nonzero_returns * jnp.var(returns) / total_returns - dt = subsample_freq / 1440 - return jnp.sqrt(adjusted_variance) / jnp.sqrt(dt) - - daily_volatilities = vmap(calculate_daily_volatility)(jnp.arange(n_days)) - volatility_array = jnp.repeat(daily_volatilities, 1440) - - remaining_minutes = n_minutes - volatility_array.shape[0] - if remaining_minutes > 0: - volatility_array = jnp.concatenate( - [volatility_array, jnp.full((remaining_minutes,), daily_volatilities[-1])] - ) - - return volatility_array * jnp.sqrt(365.0) - - class _PoolState(NamedTuple): """Intermediate state produced by _init_pool_state. @@ -299,9 +248,7 @@ def _resolve_noise_inputs( arb_vol = None if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): - volatility_array = _calculate_annualized_ratio_volatility( - prices, run_fingerprint - ) + volatility_array = self.calculate_volatility_array(prices, run_fingerprint) arb_vol = _prepare_dynamic_array( volatility_array, start_index=start_index, From 6d45c270c54e567dbee9f0e774208572669f6b06 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 12:28:42 +0000 Subject: [PATCH 019/115] ci: install calibration extra and add reclamm branch trigger --- .github/workflows/tests.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 6e0239df..f311c9ca 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -2,9 +2,9 @@ name: Tests on: push: - branches: [main, master, dev] + branches: [main, master, dev, reclamm] pull_request: - branches: [main, master, dev] + branches: [main, master, dev, reclamm] jobs: test: @@ -25,7 +25,7 @@ jobs: - name: Install dependencies run: | python -m pip install --upgrade pip - pip install -e ".[dev]" + pip install -e ".[dev,calibration]" - name: Run tests with coverage run: | From 8371fc43ef9dbc9c6882b31802e5408b95d0b6a3 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 13:10:30 +0000 Subject: [PATCH 020/115] fix: use contract's centeredness-preserving formula for price ratio updates Replace the ad-hoc "keep overvalued, solve undervalued" virtual balance recalculation with the closed-form quadratic from ReClammMath.sol computeVirtualBalancesUpdatingPriceRatio. The old code silently drove centeredness to 1.0 for off-center pools. Add parametrized test mirroring the Foundry fuzz test testCalculateVirtualBalancesUpdatingPriceRatio__Fuzz, asserting that centeredness is preserved and the target price ratio is achieved. --- quantammsim/pools/reCLAMM/reclamm_reserves.py | 44 +++++++++++-------- .../test_reclamm_price_ratio_updates.py | 44 +++++++++++++++++++ 2 files changed, 69 insertions(+), 19 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 5f1ac85d..3825f110 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -557,32 +557,38 @@ def initialise_reclamm_reserves(initial_pool_value, initial_prices, price_ratio) # --------------------------------------------------------------------------- def apply_target_price_ratio_to_virtual_balances(Ra, Rb, Va, Vb, target_price_ratio): - """Retarget virtual balances to a desired price ratio while preserving orientation. + """Retarget virtual balances to a desired price ratio while preserving centeredness. - The overvalued-side virtual balance is preserved (subject to floor), and the - undervalued-side virtual balance is solved from the reCLAMM ratio constraint. + Uses the closed-form quadratic solution from ReClammMath.sol + ``computeVirtualBalancesUpdatingPriceRatio``: + + Vu = Ru * (1 + C + sqrt(1 + C*(C + 4*Q0 - 2))) / (2*(Q0 - 1)) + Vo = Vu * lastVo / lastVu + + where Q0 = sqrt(price_ratio), C = centeredness, Ru is the real balance of + the undervalued token. The overvalued virtual balance is then scaled + proportionally so that Va/Vb is preserved, which keeps centeredness constant. """ safe_ratio = jnp.maximum(target_price_ratio, 1.0 + 1e-12) - sqrt_ratio = jnp.sqrt(safe_ratio) - fourth_root_ratio = jnp.sqrt(sqrt_ratio) + Q0 = jnp.sqrt(safe_ratio) # sqrt(price_ratio) centeredness, is_above = compute_centeredness(Ra, Rb, Va, Vb) + C = centeredness - # Above center => B overvalued, so keep Vb and solve Va. - v_over_b_floor = Rb / jnp.maximum(fourth_root_ratio - 1.0, 1e-30) - Vb_kept = jnp.maximum(Vb, v_over_b_floor) - Va_from_b = Ra * (Vb_kept + Rb) / jnp.maximum( - (sqrt_ratio - 1.0) * Vb_kept - Rb, 1e-30 - ) + # Closed-form quadratic solution for the undervalued virtual balance. + discriminant = jnp.maximum(1.0 + C * (C + 4.0 * Q0 - 2.0), 0.0) + numerator_factor = 1.0 + C + jnp.sqrt(discriminant) + denominator = 2.0 * jnp.maximum(Q0 - 1.0, 1e-30) - # Below center => A overvalued, so keep Va and solve Vb. - v_over_a_floor = Ra / jnp.maximum(fourth_root_ratio - 1.0, 1e-30) - Va_kept = jnp.maximum(Va, v_over_a_floor) - Vb_from_a = Rb * (Va_kept + Ra) / jnp.maximum( - (sqrt_ratio - 1.0) * Va_kept - Ra, 1e-30 - ) + # Above center: A is undervalued (Ra abundant), B is overvalued. + Vu_above = Ra * numerator_factor / denominator # new Va + Vo_above = Vu_above * Vb / jnp.maximum(Va, 1e-30) # new Vb, scaled + + # Below center: B is undervalued (Rb abundant), A is overvalued. + Vu_below = Rb * numerator_factor / denominator # new Vb + Vo_below = Vu_below * Va / jnp.maximum(Vb, 1e-30) # new Va, scaled - Va_new = jnp.where(is_above, Va_from_b, Va_kept) - Vb_new = jnp.where(is_above, Vb_kept, Vb_from_a) + Va_new = jnp.where(is_above, Vu_above, Vo_below) + Vb_new = jnp.where(is_above, Vo_above, Vu_below) # When centeredness is degenerate (e.g. both sides zero), preserve current virtuals. invalid_centeredness = ~jnp.isfinite(centeredness) diff --git a/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py index db24aca8..955d16db 100644 --- a/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py +++ b/tests/pools/reCLAMM/test_reclamm_price_ratio_updates.py @@ -6,6 +6,8 @@ import pytest from quantammsim.pools.reCLAMM.reclamm_reserves import ( + apply_target_price_ratio_to_virtual_balances, + compute_centeredness, compute_price_ratio, initialise_reclamm_reserves, _jax_calc_reclamm_reserves_with_dynamic_inputs, @@ -290,6 +292,48 @@ def test_replacement_event_supersedes_active_event(self): ) assert ratio_after_replacement == pytest.approx(2.0, rel=1e-4, abs=1e-4) + @pytest.mark.parametrize( + "Ra, Rb, Va, Vb, target_price_ratio", + [ + # Centered pool, widen ratio + (500_000.0, 500_000.0, 100_000.0, 100_000.0, 9.0), + # Centered pool, narrow ratio + (500_000.0, 500_000.0, 100_000.0, 100_000.0, 2.0), + # Above center (Ra abundant) + (800_000.0, 200_000.0, 100_000.0, 100_000.0, 16.0), + # Below center (Rb abundant) + (200_000.0, 800_000.0, 100_000.0, 100_000.0, 16.0), + # Asymmetric virtuals + (300_000.0, 600_000.0, 50_000.0, 200_000.0, 5.0), + # Large ratio change + (500_000.0, 500_000.0, 100_000.0, 100_000.0, 100.0), + ], + ) + def test_price_ratio_update_preserves_centeredness( + self, Ra, Rb, Va, Vb, target_price_ratio + ): + """Mirrors Foundry fuzz test testCalculateVirtualBalancesUpdatingPriceRatio__Fuzz. + + Asserts that apply_target_price_ratio_to_virtual_balances preserves + centeredness and achieves the target price ratio. + """ + Ra, Rb = jnp.float64(Ra), jnp.float64(Rb) + Va, Vb = jnp.float64(Va), jnp.float64(Vb) + + old_centeredness, _ = compute_centeredness(Ra, Rb, Va, Vb) + Va_new, Vb_new = apply_target_price_ratio_to_virtual_balances( + Ra, Rb, Va, Vb, target_price_ratio + ) + new_centeredness, _ = compute_centeredness(Ra, Rb, Va_new, Vb_new) + new_price_ratio = float(compute_price_ratio(Ra, Rb, Va_new, Vb_new)) + + assert float(new_centeredness) == pytest.approx( + float(old_centeredness), rel=1e-6, abs=1e-10 + ), "Centeredness should be preserved" + assert new_price_ratio == pytest.approx( + target_price_ratio, rel=1e-4 + ), "Price ratio should match target" + def test_dynamic_fee_revenue_path_with_schedule(self): reserves, Va, Vb = _init_pool() n_steps = 10 From 1b9d1ebe807783e5bab150a46dcf93124d601514 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 13:29:14 +0000 Subject: [PATCH 021/115] fix: add ste_temperature to test fingerprints Tests that construct run_fingerprint dicts directly (bypassing recursive_default_set) need the ste_temperature key now that the STE-enabled scan steps read it from the fingerprint. --- tests/pools/reCLAMM/test_reclamm_fee_revenue.py | 3 +++ tests/pools/reCLAMM/test_reclamm_reserves.py | 5 +++++ 2 files changed, 8 insertions(+) diff --git a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py index bfbd21b1..106f8a3d 100644 --- a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py +++ b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py @@ -297,6 +297,7 @@ def test_pool_method_with_fees(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -342,6 +343,7 @@ def test_pool_method_with_dynamic_inputs(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -408,6 +410,7 @@ def test_forward_pass_returns_fee_revenue(self): "rule": "reclamm", "training_data_kind": "historic", "do_trades": False, + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index cf402174..396f50f5 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -272,6 +272,7 @@ def test_calculate_reserves_with_fees(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -314,6 +315,7 @@ def test_calculate_reserves_zero_fees(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -353,6 +355,7 @@ def test_calculate_weights(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -485,6 +488,7 @@ def test_fingerprint_dispatch(self): "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "reclamm_interpolation_method": "constant_arc_length", "reclamm_arc_length_speed": None, # auto-calibrate + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -690,6 +694,7 @@ def test_learnable_arc_length_speed_forward_pass(self): "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "reclamm_interpolation_method": "constant_arc_length", "reclamm_learn_arc_length_speed": True, + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) From 8bd2c598205310415f7828477b39166d4473ba83 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 13:55:23 +0000 Subject: [PATCH 022/115] fix: replace STE gate on fees check with hard check MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The fees_gate STE was unnecessary — gamma (1 - fees) is either a static config value or a learnable param constrained to be nonzero, so the fee/zero-fee branch selection never benefits from soft gradients. --- quantammsim/pools/reCLAMM/reclamm_reserves.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 3a0e89ad..1ea939ab 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -987,10 +987,8 @@ def _skip_schedule_state(_): 0, ) - fees_gate = _ste_greater_than( - jnp.abs(gamma - 1.0), jnp.asarray(1e-12, dtype=gamma.dtype), ste_temperature - ) - optimal_arb_trade = _ste_select(fees_gate, fee_trade, zero_fee_trade) + fees_are_being_charged = gamma != 1.0 + optimal_arb_trade = jnp.where(fees_are_being_charged, fee_trade, zero_fee_trade) # Check profitability for arb profit_to_arb = -(optimal_arb_trade * prices).sum() - arb_thresh From 75a654001d5a1cdc3a74f9f152379209780ca997 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:42:35 +0000 Subject: [PATCH 023/115] feat: add volatility calculation, relative arb invariant, and LP supply forwarding - base_pool: add calculate_volatility_array and _handle_numeraire_ordering - optimal_n_pool_arb: use relative invariant check instead of absolute slack - balancer/balancer_reserves: forward lp_supply through scan functions - balancer: use materialized_inputs.lp_supply in dynamic_inputs path - TFMM_base_pool: use materialized_inputs.lp_supply, fix trade_array ref - core_simulator/__init__: enable JAX compilation cache --- quantammsim/core_simulator/__init__.py | 1 + quantammsim/pools/G3M/balancer/balancer.py | 1 + .../pools/G3M/balancer/balancer_reserves.py | 54 +++++++--- quantammsim/pools/G3M/optimal_n_pool_arb.py | 41 ++++--- .../pools/G3M/quantamm/TFMM_base_pool.py | 4 + quantammsim/pools/base_pool.py | 100 +++++++++++++++++- 6 files changed, 160 insertions(+), 41 deletions(-) diff --git a/quantammsim/core_simulator/__init__.py b/quantammsim/core_simulator/__init__.py index 10c118ae..010490e0 100644 --- a/quantammsim/core_simulator/__init__.py +++ b/quantammsim/core_simulator/__init__.py @@ -17,6 +17,7 @@ import jax.numpy as jnp # noqa: F401 from jax import config config.update("jax_enable_x64", True) + config.update("jax_compilation_cache_dir", "/tmp/jax_cache") except ImportError as e: raise ImportError( "JAX is required for core simulator. Please install jax and jaxlib." diff --git a/quantammsim/pools/G3M/balancer/balancer.py b/quantammsim/pools/G3M/balancer/balancer.py index fcee2722..89cf46dc 100644 --- a/quantammsim/pools/G3M/balancer/balancer.py +++ b/quantammsim/pools/G3M/balancer/balancer.py @@ -331,6 +331,7 @@ def calculate_reserves_with_dynamic_inputs( materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], + materialized_inputs.lp_supply, ) return reserves diff --git a/quantammsim/pools/G3M/balancer/balancer_reserves.py b/quantammsim/pools/G3M/balancer/balancer_reserves.py index 82c66295..e36463b2 100644 --- a/quantammsim/pools/G3M/balancer/balancer_reserves.py +++ b/quantammsim/pools/G3M/balancer/balancer_reserves.py @@ -147,7 +147,7 @@ def _jax_calc_balancer_reserves_with_fees_scan_function_using_precalcs( tokens_to_drop, gamma, n, - 0, + -1e-15, ) optimal_arb_trade = jnp.where( @@ -350,6 +350,7 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using prev_reserves = carry_list[1] counter = carry_list[2] + prev_lp_supply = carry_list[3] # input_list contains weights, prices, precalcs and fee/arb amounts prices = input_list[0] @@ -359,7 +360,16 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using gamma = input_list[4] arb_thresh = input_list[5] arb_fees = input_list[6] - trade = input_list[7] if do_trades else None + trade = input_list[7] + lp_supply = input_list[8] + + # Scale reserves for LP supply changes (proportional deposits/withdrawals) + lp_supply_change = lp_supply != prev_lp_supply + prev_reserves = jnp.where( + lp_supply_change, + prev_reserves * lp_supply / prev_lp_supply, + prev_reserves, + ) fees_are_being_charged = gamma != 1.0 @@ -385,7 +395,7 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using tokens_to_drop, gamma, n, - 0, + -1e-15, ) optimal_arb_trade = jnp.where( @@ -421,6 +431,7 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using prices, reserves, counter, + lp_supply, ], reserves @@ -436,6 +447,7 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( trades=None, do_trades=False, do_arb=True, + lp_supply_array=None, ): """ Calculate AMM reserves considering fees and arbitrage opportunities using signature variations, @@ -497,6 +509,14 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( if do_trades and trades is None: raise ValueError("Trades must be provided when do_trades=True.") + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + # pre-calculate some values that are repeatedly used in optimal arb calculations _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( precalc_shared_values_for_all_signatures(all_sig_variations, n_assets) @@ -529,18 +549,22 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( initial_prices, initial_reserves, 0, + lp_supply_array[0], ] - scan_inputs = [ - prices, - active_initial_weights, - per_asset_ratios, - all_other_assets_ratios, - gamma, - arb_thresh, - arb_fees, - ] - if do_trades: - scan_inputs.append(trades) - _, reserves = scan(scan_fn, carry_list_init, scan_inputs) + _, reserves = scan( + scan_fn, + carry_list_init, + [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + trades, + lp_supply_array, + ], + ) return reserves diff --git a/quantammsim/pools/G3M/optimal_n_pool_arb.py b/quantammsim/pools/G3M/optimal_n_pool_arb.py index 822e7ce9..d9e3964b 100644 --- a/quantammsim/pools/G3M/optimal_n_pool_arb.py +++ b/quantammsim/pools/G3M/optimal_n_pool_arb.py @@ -164,21 +164,18 @@ def construct_optimal_trade_jnp( valid_post_trade_reserves = ( jnp.sum(initial_reserves + active_overall_trade > 0) == n ) - valid_post_trade_constant = ( - jnp.prod( - ( - initial_reserves - + active_overall_trade - * (fee_gamma ** (trade_to_direction_jnp(active_overall_trade))) - ) - ** initial_weights + post_trade_constant = jnp.prod( + ( + initial_reserves + + active_overall_trade + * (fee_gamma ** (trade_to_direction_jnp(active_overall_trade))) ) - - initial_constant - >= slack + ** initial_weights ) + relative_diff = (post_trade_constant - initial_constant) / initial_constant + valid_post_trade_constant = relative_diff >= slack valid_trade = jnp.logical_and(valid_post_trade_reserves, valid_post_trade_constant) return jnp.where(valid_trade, active_overall_trade, 0) - # return active_overall_trade, valid_post_trade_reserves * valid_post_trade_constant construct_optimal_trade_jnp_vmapped = vmap( @@ -336,18 +333,16 @@ def calc_optimal_trade_for_one_signature( valid_post_trade_reserves = ( jnp.sum(initial_reserves + active_overall_trade > 0) == n ) - valid_post_trade_constant = ( - jnp.prod( - ( - initial_reserves - + active_overall_trade - * (fee_gamma ** (trade_to_direction_jnp(active_overall_trade))) - ) - ** initial_weights + post_trade_constant = jnp.prod( + ( + initial_reserves + + active_overall_trade + * (fee_gamma ** (trade_to_direction_jnp(active_overall_trade))) ) - - initial_constant - >= slack + ** initial_weights ) + relative_diff = (post_trade_constant - initial_constant) / initial_constant + valid_post_trade_constant = relative_diff >= slack valid_trade = jnp.logical_and(valid_post_trade_reserves, valid_post_trade_constant) return jnp.where(valid_trade, active_overall_trade, 0) # return { @@ -411,7 +406,7 @@ def parallelised_optimal_trade_sifter( tokens_to_drop, fee_gamma, n, - 0, + slack, ) profits = -(overall_trades * local_prices).sum(-1) @@ -457,7 +452,7 @@ def wrapped_parallelised_optimal_trade_sifter( tokens_to_drop, fee_gamma, n, - slack=0, + slack=slack, ) return trade diff --git a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py index 7091ee3f..b739abef 100644 --- a/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py +++ b/quantammsim/pools/G3M/quantamm/TFMM_base_pool.py @@ -417,6 +417,10 @@ def calculate_reserves_with_dynamic_inputs( do_trades=run_fingerprint["do_trades"], dtype=arb_acted_upon_local_prices.dtype, ) + lp_supply_array_broadcast = materialized_inputs.lp_supply + # if we are doing trades, the trades array must be of the same length as the other arrays + if run_fingerprint["do_trades"]: + assert materialized_inputs.trades.shape[0] == max_len protocol_fee_split = run_fingerprint.get("protocol_fee_split", 0.0) reserves = _jax_calc_quantAMM_reserves_with_dynamic_inputs( initial_reserves, diff --git a/quantammsim/pools/base_pool.py b/quantammsim/pools/base_pool.py index 0415ef99..b695d8cd 100644 --- a/quantammsim/pools/base_pool.py +++ b/quantammsim/pools/base_pool.py @@ -1,11 +1,12 @@ from abc import ABC, abstractmethod -from typing import Dict, Any, Optional +from typing import Dict, Any, Optional, Tuple +from functools import partial import numpy as np import jax.numpy as jnp from jax.nn import softmax -from jax.lax import stop_gradient -from jax import tree_util +from jax.lax import stop_gradient, dynamic_slice +from jax import tree_util, jit, vmap from quantammsim.core_simulator.param_utils import make_vmap_in_axes_dict @@ -283,6 +284,99 @@ def add_noise( params[key] = jnp.array(params[key]) return params + @partial(jit, static_argnums=(2, 3)) + def calculate_volatility_array(self, prices, run_fingerprint, subsample_freq=5): + """Annualised daily realised volatility broadcast to minute-level array. + + Pure-JAX implementation (vmap + dynamic_slice) — JIT-compatible and + callable from within traced contexts (e.g. forward_pass). + + Parameters + ---------- + prices : jnp.ndarray, shape (T, 2) + Minute-level prices for two tokens. + run_fingerprint : dict + Must contain ``tokens`` and ``numeraire`` for ordering. + subsample_freq : int + Subsample within each day to reduce microstructure noise. + + Returns + ------- + jnp.ndarray, shape (T,) + Annualised volatility, constant within each day. + """ + ordered_prices, needs_swap = self._handle_numeraire_ordering( + prices, run_fingerprint, + ) + asset_prices = ordered_prices[:, 0] / ordered_prices[:, 1] + n_minutes = len(asset_prices) + + # Guard: need at least one full day for vmap + dynamic_slice + if n_minutes < 1440: + return jnp.full(n_minutes, 0.1) * jnp.sqrt(365.0) + + n_days = n_minutes // 1440 + + def calculate_daily_volatility(day_idx): + start_idx = day_idx * 1440 + window_prices = dynamic_slice(asset_prices, [start_idx], [1440]) + subsampled_prices = window_prices[::subsample_freq] + log_prices = jnp.log(jnp.maximum(subsampled_prices, 1e-8)) + returns = jnp.diff(log_prices) + num_nonzero_returns = jnp.sum(returns != 0) + total_returns = len(returns) + adjusted_variance = ( + num_nonzero_returns * jnp.var(returns) / total_returns + ) + dt = subsample_freq / 1440 + vol = jnp.sqrt(adjusted_variance) / jnp.sqrt(dt) + return vol + + daily_volatilities = vmap(calculate_daily_volatility)(jnp.arange(n_days)) + volatility_array = jnp.repeat(daily_volatilities, 1440) + + remaining_minutes = n_minutes - len(volatility_array) + if remaining_minutes > 0: + last_vol = ( + daily_volatilities[-1] if len(daily_volatilities) > 0 else 0.1 + ) + volatility_array = jnp.concatenate( + [volatility_array, jnp.full(remaining_minutes, last_vol)] + ) + + return volatility_array * jnp.sqrt(365.0) + + @partial(jit, static_argnums=(2,)) + def _handle_numeraire_ordering( + self, + prices: jnp.ndarray, + run_fingerprint: Dict[str, Any], + ) -> Tuple[jnp.ndarray, bool]: + """Reorder prices so numeraire token is in second position. + + Parameters + ---------- + prices : jnp.ndarray, shape (..., 2) + Price array with two tokens. + run_fingerprint : dict + Must contain ``tokens`` (sorted) and ``numeraire``. + + Returns + ------- + (ordered_prices, needs_swap) : (jnp.ndarray, bool) + """ + tokens = sorted(run_fingerprint["tokens"]) + numeraire = run_fingerprint["numeraire"] + if numeraire is None or numeraire not in tokens: + numeraire = tokens[-1] + needs_swap = tokens.index(numeraire) == 0 + + if needs_swap: + ordered_prices = prices[..., ::-1] + else: + ordered_prices = prices + return ordered_prices, needs_swap + def _tree_flatten(self): children = () aux_data = dict() # static values From d05fa6b5e9939b82238a7ecf525b3943f8484a5f Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:42:54 +0000 Subject: [PATCH 024/115] feat: add noise volume models: tsoukalas_sqrt, tsoukalas_log, loglinear Three noise trade volume models for reCLAMM pools, each predicting non-arbitrage trading volume from pool TVL, fee tier, volatility, and arb volume. Used by the noise model dispatch in scan steps. --- quantammsim/pools/noise_trades.py | 164 ++++++++++++++++++++++++++++++ 1 file changed, 164 insertions(+) diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index 82e76744..3f45792f 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -108,3 +108,167 @@ def calculate_reserves_after_noise_trade( ) reserves = current_reserves * ratio_of_value_of_trade_to_reserves return reserves + + +@jit +def reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """reClAMM Tsoukalas sqrt model: effective TVL regressor. + + Predicts per-minute noise trader volume using: + V_daily = (a_0 - a_f*fee + a_sigma*sigma + + a_c*sqrt(c_eff/1e6)) * 1e6 + V_noise = max(0, V_daily/1440 - arb_volume_this_period) + + where c_eff = (Ra+Va)*pA + (Rb+Vb)*pB is the effective TVL (real + + virtual reserves valued in USD). For a concentrated liquidity pool, + effective reserves determine execution quality and routing decisions, + so they are the natural driver of noise volume. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Regression coefficients. Keys: a_0_base, a_f, a_sigma, + a_c, base_fee. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + a_0_base = noise_params.get("a_0_base", 0.5) + a_f = noise_params.get("a_f", 0.0) + a_sigma = noise_params.get("a_sigma", 2.0) + a_c = noise_params.get("a_c", 1.0) + base_fee = noise_params.get("base_fee", 0.003) + + fee = 1.0 - gamma + a_0 = a_0_base + base_fee * a_f + daily_vol = ( + a_0 - a_f * fee + + a_sigma * volatility + + a_c * jnp.sqrt(effective_value_usd / 1e6) + ) * 1e6 + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + +@jit +def reclamm_tsoukalas_log_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """reClAMM Tsoukalas log model: log(c_eff/1e6) instead of sqrt. + + Same specification as the sqrt variant but uses log regressor, + which may fit better for pools spanning a wide TVL range. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Regression coefficients (same keys as sqrt variant). + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + a_0_base = noise_params.get("a_0_base", 0.5) + a_f = noise_params.get("a_f", 0.0) + a_sigma = noise_params.get("a_sigma", 2.0) + a_c = noise_params.get("a_c", 1.0) + base_fee = noise_params.get("base_fee", 0.003) + + fee = 1.0 - gamma + a_0 = a_0_base + base_fee * a_f + daily_vol = ( + a_0 - a_f * fee + + a_sigma * volatility + + a_c * jnp.log(jnp.maximum(effective_value_usd / 1e6, 1e-30)) + ) * 1e6 + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + +@jit +def reclamm_loglinear_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + noise_params=None, +): + """Loglinear noise volume from hierarchical cross-pool calibration. + + Predicts per-minute noise volume using: + log(V_daily) = b_0 + b_sigma * volatility + b_c * log(TVL) + V_noise = max(0, exp(log_daily_vol) / 1440 - arb_volume) + + where b_0 is a pool-specific intercept (BLUP from the hierarchical + model, absorbing chain, token tier, and fee effects), and b_sigma, + b_c are shared fixed effects estimated from cross-pool variation. + + Note: ``gamma`` is accepted for interface compatibility with the + other noise volume functions but is not used; fee effects are + absorbed into ``b_0`` via the hierarchical model's BLUP. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). Unused — kept for uniform + calling convention across noise models. + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + noise_params : dict, optional + Hierarchical model coefficients. Keys: b_0, b_sigma, b_c. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + b_0 = noise_params.get("b_0", -6.7) + b_sigma = noise_params.get("b_sigma", -0.0007) + b_c = noise_params.get("b_c", 1.04) + + log_daily_vol = ( + b_0 + + b_sigma * volatility + + b_c * jnp.log(jnp.maximum(effective_value_usd, 1.0)) + ) + daily_vol = jnp.exp(log_daily_vol) + return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) + + From 0245049c1001417a258d69c6856af4f53ce15f35 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:43:18 +0000 Subject: [PATCH 025/115] feat: add LP supply tracking and noise model dispatch to reCLAMM scan steps Scan step carry now includes prev_lp_supply (pos 3) and price-ratio schedule state (pos 4-9). Scan inputs include price_ratio_update (7), lp_supply (8), and optional volatility (9). - reclamm_reserves: LP supply scaling, noise model dispatch (ratio, tsoukalas_sqrt, tsoukalas_log, loglinear, arb_only) in scan steps - reclamm: _prepare_dynamic_array for volatility, noise/LP kwargs threading through all pool method variants - tests: LP supply and noise volume coverage --- quantammsim/pools/reCLAMM/reclamm.py | 122 +++ quantammsim/pools/reCLAMM/reclamm_reserves.py | 274 ++++- .../reCLAMM/test_reclamm_noise_volume.py | 991 ++++++++++++++++++ tests/pools/reCLAMM/test_reclamm_reserves.py | 593 ++++++++++- 4 files changed, 1930 insertions(+), 50 deletions(-) create mode 100644 tests/pools/reCLAMM/test_reclamm_noise_volume.py diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 762301c8..5b913ff2 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -18,6 +18,22 @@ from quantammsim.core_simulator.dynamic_inputs import materialize_dynamic_inputs from quantammsim.pools.base_pool import AbstractPool + + +def _prepare_dynamic_array(arr, start_index, bout_length, arb_frequency, max_len): + """Slice and decimate a dynamic input array to match arb_prices shape. + + Scalar (1,) arrays are broadcast to (max_len,). + Full-length arrays are sliced to the bout window then decimated. + """ + if arr.shape[0] <= 1: + return jnp.broadcast_to(arr, (max_len,) + arr.shape[1:]) + sliced = dynamic_slice(arr, (start_index[0],), (bout_length - 1,)) + if arb_frequency != 1: + sliced = sliced[::arb_frequency] + return sliced + + from quantammsim.pools.reCLAMM.reclamm_reserves import ( initialise_reclamm_reserves, calibrate_arc_length_speed, @@ -206,9 +222,36 @@ def calculate_reserves_with_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ) -> jnp.ndarray: s = self._init_pool_state(params, run_fingerprint, prices, start_index) + bout_length = run_fingerprint["bout_length"] + arb_freq = run_fingerprint["arb_frequency"] + lp_prepared = ( + _prepare_dynamic_array( + lp_supply_array, start_index, bout_length, + arb_freq, s.arb_prices.shape[0], + ) + if lp_supply_array is not None else None + ) + + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + arb_vol = _prepare_dynamic_array( + volatility_array, start_index, bout_length, + arb_freq, s.arb_prices.shape[0], + ) + else: + arb_vol = None + if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_with_fees( s.initial_reserves, s.Va, s.Vb, @@ -225,6 +268,11 @@ def calculate_reserves_with_fees( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=lp_prepared, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -236,6 +284,7 @@ def calculate_reserves_and_fee_revenue_with_fees( prices: jnp.ndarray, start_index: jnp.ndarray, additional_oracle_input: Optional[jnp.ndarray] = None, + lp_supply_array: Optional[jnp.ndarray] = None, ): """Calculate reserves and LP fee revenue with fees. @@ -247,6 +296,32 @@ def calculate_reserves_and_fee_revenue_with_fees( """ s = self._init_pool_state(params, run_fingerprint, prices, start_index) + bout_length = run_fingerprint["bout_length"] + arb_freq = run_fingerprint["arb_frequency"] + lp_prepared = ( + _prepare_dynamic_array( + lp_supply_array, start_index, bout_length, + arb_freq, s.arb_prices.shape[0], + ) + if lp_supply_array is not None else None + ) + + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + arb_vol = _prepare_dynamic_array( + volatility_array, start_index, bout_length, + arb_freq, s.arb_prices.shape[0], + ) + else: + arb_vol = None + if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( s.initial_reserves, s.Va, s.Vb, @@ -263,6 +338,11 @@ def calculate_reserves_and_fee_revenue_with_fees( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=lp_prepared, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), @@ -301,6 +381,22 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( dtype=s.arb_prices.dtype, ) + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + arb_vol = _prepare_dynamic_array( + volatility_array, start_index, bout_length, + run_fingerprint["arb_frequency"], max_len, + ) + else: + arb_vol = None + return _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, s.arb_prices, @@ -317,6 +413,11 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=materialized_inputs.lp_supply, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) @partial(jit, static_argnums=(2,)) @@ -379,6 +480,22 @@ def calculate_reserves_with_dynamic_inputs( dtype=s.arb_prices.dtype, ) + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + arb_vol = _prepare_dynamic_array( + volatility_array, start_index, bout_length, + run_fingerprint["arb_frequency"], max_len, + ) + else: + arb_vol = None + return _jax_calc_reclamm_reserves_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, s.arb_prices, @@ -395,6 +512,11 @@ def calculate_reserves_with_dynamic_inputs( arc_length_speed=s.arc_length_speed, centeredness_scaling=s.centeredness_scaling, protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + lp_supply_array=materialized_inputs.lp_supply, + noise_model=noise_model, + noise_params=noise_params, + volatility_array=arb_vol, ) def init_base_parameters( diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 3825f110..2a46ad3d 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -30,6 +30,12 @@ from quantammsim.pools.G3M.G3M_trades import ( _jax_calc_G3M_trade_from_exact_in_given_out, ) +from quantammsim.pools.noise_trades import ( + calculate_reserves_after_noise_trade, + reclamm_tsoukalas_sqrt_noise_volume, + reclamm_tsoukalas_log_noise_volume, + reclamm_loglinear_noise_volume, +) # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) _INITIALIZATION_MAX_BALANCE_A = 1e6 @@ -598,7 +604,7 @@ def apply_target_price_ratio_to_virtual_balances(Ra, Rb, Va, Vb, target_price_ra def _reclamm_scan_step_zero_fees( carry_list, - prices, + input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, @@ -611,11 +617,23 @@ def _reclamm_scan_step_zero_fees( 1. Update virtual balances (path-dependent) 2. Compute analytical constant-product arb (no fee friction) - Carry: [real_reserves (2,), Va (0-d), Vb (0-d)] + Carry: [real_reserves (2,), Va (0-d), Vb (0-d), prev_lp_supply (0-d)] + Input: [prices (2,), lp_supply (0-d)] """ prev_reserves = carry_list[0] Va = carry_list[1] Vb = carry_list[2] + prev_lp_supply = carry_list[3] + + prices = input_list[0] + lp_supply = input_list[1] + + # Scale both real and virtual reserves by LP supply ratio. + scale = lp_supply / prev_lp_supply + lp_supply_change = lp_supply != prev_lp_supply + prev_reserves = jnp.where(lp_supply_change, prev_reserves * scale, prev_reserves) + Va = jnp.where(lp_supply_change, Va * scale, Va) + Vb = jnp.where(lp_supply_change, Vb * scale, Vb) Ra = prev_reserves[0] Rb = prev_reserves[1] @@ -690,7 +708,7 @@ def _reclamm_scan_step_zero_fees( Rb_new = jnp.where(clamp_a, Rb + edge_a[1], jnp.where(clamp_b, Rb + edge_b[1], Rb_new)) new_reserves = jnp.array([Ra_new, Rb_new]) - return [new_reserves, Va, Vb], new_reserves + return [new_reserves, Va, Vb, lp_supply], new_reserves # --------------------------------------------------------------------------- @@ -702,7 +720,7 @@ def _reclamm_scan_step_zero_fees( def _reclamm_scan_step_zero_fees_full_state( carry_list, - prices, + input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, @@ -711,7 +729,7 @@ def _reclamm_scan_step_zero_fees_full_state( ): """TEST-ONLY: scan step that outputs (reserves, Va, Vb).""" new_carry, new_reserves = _reclamm_scan_step_zero_fees( - carry_list, prices, centeredness_margin, daily_price_shift_base, seconds_per_step, + carry_list, input_list, centeredness_margin, daily_price_shift_base, seconds_per_step, arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, ) @@ -731,15 +749,19 @@ def _reclamm_scan_step_with_fees_and_revenue( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + noise_model="ratio", + noise_params=None, ): """Single scan step for reClAMM pool with fees, returning LP fee revenue. Primary implementation — ``_reclamm_scan_step_with_fees`` wraps this. - Carry: [real_reserves (2,), Va, Vb, step_idx, active_start_ratio, + Carry: [real_reserves (2,), Va, Vb, prev_lp_supply, step_idx, active_start_ratio, active_target_ratio, active_start_step, active_end_step, active_enabled] Input: [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_update] + all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_update, + lp_supply, (optional) volatility] Returns ------- @@ -750,15 +772,13 @@ def _reclamm_scan_step_with_fees_and_revenue( prev_reserves = carry_list[0] Va = carry_list[1] Vb = carry_list[2] - step_idx = carry_list[3] - active_start_ratio = carry_list[4] - active_target_ratio = carry_list[5] - active_start_step = carry_list[6] - active_end_step = carry_list[7] - active_enabled = carry_list[8] - - Ra = prev_reserves[0] - Rb = prev_reserves[1] + prev_lp_supply = carry_list[3] + step_idx = carry_list[4] + active_start_ratio = carry_list[5] + active_target_ratio = carry_list[6] + active_start_step = carry_list[7] + active_end_step = carry_list[8] + active_enabled = carry_list[9] prices = input_list[0] active_initial_weights = input_list[1] @@ -768,7 +788,21 @@ def _reclamm_scan_step_with_fees_and_revenue( arb_thresh = input_list[5] arb_fees = input_list[6] price_ratio_update = input_list[7] + lp_supply = input_list[8] + + # Scale both real and virtual reserves by LP supply ratio. + # Matches ReClammPool.sol onBeforeAddLiquidity / onBeforeRemoveLiquidity: + # all balances (real + virtual) scale proportionally with BPT supply. + scale = lp_supply / prev_lp_supply + lp_supply_change = lp_supply != prev_lp_supply + prev_reserves = jnp.where(lp_supply_change, prev_reserves * scale, prev_reserves) + Va = jnp.where(lp_supply_change, Va * scale, Va) + Vb = jnp.where(lp_supply_change, Vb * scale, Vb) + + Ra = prev_reserves[0] + Rb = prev_reserves[1] + # Price-ratio schedule: apply target price ratio changes over time. event_has = price_ratio_update[0] > 0.5 event_target_ratio = jnp.maximum( jnp.where(jnp.isfinite(price_ratio_update[1]), price_ratio_update[1], 1.0), @@ -936,6 +970,45 @@ def _skip_schedule_state(_): Ra_new = Ra + applied_trade[0] Rb_new = Rb + applied_trade[1] + # --- Noise model dispatch --- + # noise_model is a concrete Python string (passed via Partial as static + # aux_data), so if/elif branches resolve at trace time. + if noise_model == "ratio": + noisy_reserves = calculate_reserves_after_noise_trade( + applied_trade, jnp.array([Ra_new, Rb_new]), prices, + noise_trader_ratio, gamma, + ) + Ra_new = jnp.where(noise_trader_ratio > 0, noisy_reserves[0], Ra_new) + Rb_new = jnp.where(noise_trader_ratio > 0, noisy_reserves[1], Rb_new) + elif noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + volatility = input_list[9] + arb_volume = 0.5 * jnp.sum(jnp.abs(applied_trade) * prices) + real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) + effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + + _np = noise_params if noise_params is not None else {} + if noise_model == "tsoukalas_sqrt": + noise_vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value, gamma, volatility, + arb_volume, _np, + ) + elif noise_model == "tsoukalas_log": + noise_vol = reclamm_tsoukalas_log_noise_volume( + effective_value, gamma, volatility, + arb_volume, _np, + ) + else: # loglinear + noise_vol = reclamm_loglinear_noise_volume( + effective_value, gamma, volatility, + arb_volume, _np, + ) + + noise_fee_income = (1.0 - gamma) * noise_vol + scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) + Ra_new = Ra_new * scale + Rb_new = Rb_new * scale + # else: "arb_only" — no noise trades + # Clamp-to-edge: if a real reserve would go negative, apply an # exact-in-given-out edge trade that drains that token to _DUST_USD # worth of reserves (preserving the AMM invariant). @@ -978,6 +1051,7 @@ def _skip_schedule_state(_): new_reserves, Va, Vb, + lp_supply, step_idx + 1.0, active_start_ratio, active_target_ratio, @@ -1000,6 +1074,9 @@ def _reclamm_scan_step_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + noise_model="ratio", + noise_params=None, ): """Single scan step for reClAMM pool with fees (reserves only). @@ -1018,6 +1095,9 @@ def _reclamm_scan_step_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params, ) return new_carry, new_reserves @@ -1064,6 +1144,7 @@ def _jax_calc_reclamm_reserves_zero_fees( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + lp_supply_array=None, ): """Calculate reClAMM reserves over time with zero fees. @@ -1085,12 +1166,22 @@ def _jax_calc_reclamm_reserves_zero_fees( If > 0, use constant-arc-length thermostat instead of geometric. centeredness_scaling : bool If True, scale speed by margin/centeredness (proportional controller). + lp_supply_array : jnp.ndarray, optional + LP token supply over time, shape (T,). Defaults to constant 1.0. Returns ------- reserves : jnp.ndarray, shape (T, 2) Real reserves over time. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + scan_fn = Partial( _reclamm_scan_step_zero_fees, centeredness_margin=centeredness_margin, @@ -1100,8 +1191,8 @@ def _jax_calc_reclamm_reserves_zero_fees( centeredness_scaling=centeredness_scaling, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] - _, reserves = scan(scan_fn, carry_init, prices) + carry_init = [initial_reserves, initial_Va, initial_Vb, lp_supply_array[0]] + _, reserves = scan(scan_fn, carry_init, [prices, lp_supply_array]) return reserves @@ -1116,6 +1207,7 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( seconds_per_step, arc_length_speed=0.0, centeredness_scaling=False, + lp_supply_array=None, ): """TEST-ONLY: Like _jax_calc_reclamm_reserves_zero_fees but returns Va/Vb. @@ -1125,6 +1217,14 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( Va_history : jnp.ndarray, shape (T,) Vb_history : jnp.ndarray, shape (T,) """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + scan_fn = Partial( _reclamm_scan_step_zero_fees_full_state, centeredness_margin=centeredness_margin, @@ -1134,12 +1234,12 @@ def _jax_calc_reclamm_reserves_zero_fees_full_state( centeredness_scaling=centeredness_scaling, ) - carry_init = [initial_reserves, initial_Va, initial_Vb] - _, (reserves, Va_history, Vb_history) = scan(scan_fn, carry_init, prices) + carry_init = [initial_reserves, initial_Va, initial_Vb, lp_supply_array[0]] + _, (reserves, Va_history, Vb_history) = scan(scan_fn, carry_init, [prices, lp_supply_array]) return reserves, Va_history, Vb_history -@jit +@partial(jit, static_argnames=('noise_model',)) def _jax_calc_reclamm_reserves_with_fees( initial_reserves, initial_Va, @@ -1155,12 +1255,25 @@ def _jax_calc_reclamm_reserves_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves over time with fees. Uses the G3M optimal arb machinery with constant weights [0.5, 0.5] applied to effective reserves (real + virtual). """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) gamma = 1.0 - fees @@ -1195,12 +1308,22 @@ def _jax_calc_reclamm_reserves_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) + scan_inputs = [prices, active_initial_weights, per_asset_ratios, + all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, + price_ratio_updates, lp_supply_array] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + carry_init = [ initial_reserves, initial_Va, initial_Vb, + lp_supply_array[0], jnp.float64(0.0), # step_idx jnp.float64(0.0), # active_start_ratio jnp.float64(0.0), # active_target_ratio @@ -1208,16 +1331,11 @@ def _jax_calc_reclamm_reserves_with_fees( jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled ] - _, reserves = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], - ) + _, reserves = scan(scan_fn, carry_init, scan_inputs) return reserves -@partial(jit, static_argnums=(11,)) +@partial(jit, static_argnums=(11,), static_argnames=('noise_model',)) def _jax_calc_reclamm_reserves_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1236,8 +1354,21 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) @@ -1285,12 +1416,22 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) + scan_inputs = [prices, active_initial_weights, per_asset_ratios, + all_other_assets_ratios, gamma, arb_thresh, arb_fees, + price_ratio_updates, lp_supply_array] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + carry_init = [ initial_reserves, initial_Va, initial_Vb, + lp_supply_array[0], jnp.float64(0.0), # step_idx jnp.float64(0.0), # active_start_ratio jnp.float64(0.0), # active_target_ratio @@ -1298,12 +1439,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled ] - _, reserves = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], - ) + _, reserves = scan(scan_fn, carry_init, scan_inputs) return reserves @@ -1351,6 +1487,8 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( price_ratio_updates, (prices.shape[0], price_ratio_updates.shape[1]) ) + lp_supply_array = jnp.ones(prices.shape[0], dtype=prices.dtype) + _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( precalc_shared_values_for_all_signatures(all_sig_variations, n_assets) ) @@ -1380,6 +1518,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( initial_reserves, initial_Va, initial_Vb, + lp_supply_array[0], jnp.float64(0.0), # step_idx jnp.float64(0.0), # active_start_ratio jnp.float64(0.0), # active_target_ratio @@ -1391,12 +1530,13 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs_full_state( scan_fn, carry_init, [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], + all_other_assets_ratios, gamma, arb_thresh, arb_fees, + price_ratio_updates, lp_supply_array], ) return reserves, Va_history, Vb_history -@jit +@partial(jit, static_argnames=('noise_model',)) def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( initial_reserves, initial_Va, @@ -1412,6 +1552,11 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1421,6 +1566,14 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( fee_revenue : jnp.ndarray, shape (T,) LP fee revenue per timestep in USD. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) gamma = 1.0 - fees @@ -1454,12 +1607,22 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) + scan_inputs = [prices, active_initial_weights, per_asset_ratios, + all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, + price_ratio_updates, lp_supply_array] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + carry_init = [ initial_reserves, initial_Va, initial_Vb, + lp_supply_array[0], jnp.float64(0.0), # step_idx jnp.float64(0.0), # active_start_ratio jnp.float64(0.0), # active_target_ratio @@ -1467,16 +1630,11 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled ] - _, (reserves, fee_revenue) = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma_array, arb_thresh_array, arb_fees_array, price_ratio_updates], - ) + _, (reserves, fee_revenue) = scan(scan_fn, carry_init, scan_inputs) return reserves, fee_revenue -@partial(jit, static_argnums=(11,)) +@partial(jit, static_argnums=(11,), static_argnames=('noise_model',)) def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( initial_reserves, initial_Va, @@ -1495,6 +1653,11 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=0.0, centeredness_scaling=False, protocol_fee_split=0.0, + noise_trader_ratio=0.0, + lp_supply_array=None, + noise_model="ratio", + noise_params=None, + volatility_array=None, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1504,6 +1667,14 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( fee_revenue : jnp.ndarray, shape (T,) LP fee revenue per timestep in USD. """ + if lp_supply_array is None: + lp_supply_array = jnp.array(1.0) + lp_supply_array = jnp.where( + lp_supply_array.size == 1, + jnp.full(prices.shape[0], lp_supply_array), + lp_supply_array, + ) + n_assets = 2 weights = jnp.array([0.5, 0.5]) @@ -1550,12 +1721,22 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( arc_length_speed=arc_length_speed, centeredness_scaling=centeredness_scaling, protocol_fee_split=protocol_fee_split, + noise_trader_ratio=noise_trader_ratio, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, ) + scan_inputs = [prices, active_initial_weights, per_asset_ratios, + all_other_assets_ratios, gamma, arb_thresh, arb_fees, + price_ratio_updates, lp_supply_array] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + carry_init = [ initial_reserves, initial_Va, initial_Vb, + lp_supply_array[0], jnp.float64(0.0), # step_idx jnp.float64(0.0), # active_start_ratio jnp.float64(0.0), # active_target_ratio @@ -1563,10 +1744,5 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( jnp.float64(0.0), # active_end_step jnp.array(False), # active_enabled ] - _, (reserves, fee_revenue) = scan( - scan_fn, - carry_init, - [prices, active_initial_weights, per_asset_ratios, - all_other_assets_ratios, gamma, arb_thresh, arb_fees, price_ratio_updates], - ) + _, (reserves, fee_revenue) = scan(scan_fn, carry_init, scan_inputs) return reserves, fee_revenue diff --git a/tests/pools/reCLAMM/test_reclamm_noise_volume.py b/tests/pools/reCLAMM/test_reclamm_noise_volume.py new file mode 100644 index 00000000..f413a9e3 --- /dev/null +++ b/tests/pools/reCLAMM/test_reclamm_noise_volume.py @@ -0,0 +1,991 @@ +"""Tests for reClAMM Tsoukalas noise volume model. + +Tests noise volume functions (sqrt and log variants), volatility computation, +scan step integration, pool class plumbing, and OLS calibration. +""" + +import pytest +import jax.numpy as jnp +import numpy as np +import numpy.testing as npt + +from quantammsim.pools.noise_trades import ( + reclamm_tsoukalas_sqrt_noise_volume, + reclamm_tsoukalas_log_noise_volume, + reclamm_loglinear_noise_volume, +) + +# Typical noise_params for a mid-cap pool. +# a_c=1.5 is roughly equivalent to the old a_c_real=1.0 + a_c_virt=0.5 +# for pools with comparable real and virtual TVL: 1.5/sqrt(2) ≈ 1.06 +# per-component, but we use 1.5 to ensure noise income dominates the +# arb-suppression side-effect in integration tests. +DEFAULT_NOISE_PARAMS = { + "a_0_base": 0.5, + "a_f": 0.0, + "a_sigma": 2.0, + "a_c": 1.5, + "base_fee": 0.003, +} + + +# --------------------------------------------------------------------------- +# Tests 1-7: Unit tests for noise volume functions +# --------------------------------------------------------------------------- + + +class TestPositiveOutputReasonableInputs: + """Test 1: Volume > 0 for typical inputs (both sqrt and log variants).""" + + def test_sqrt_positive_output(self): + vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=15_000_000.0, + gamma=0.997, + volatility=0.5, + arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + assert float(vol) > 0, f"Expected positive noise volume, got {float(vol)}" + + def test_log_positive_output(self): + vol = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=15_000_000.0, + gamma=0.997, + volatility=0.5, + arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + assert float(vol) > 0, f"Expected positive noise volume, got {float(vol)}" + + +class TestZeroWhenArbExceedsPredicted: + """Test 2: noise = max(0, daily/1440 - arb), so returns 0 when arb dominates.""" + + def test_sqrt_zero_when_arb_large(self): + vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=3_000_000.0, + gamma=0.997, + volatility=0.3, + arb_volume_this_period=1e12, # Absurdly large arb + noise_params=DEFAULT_NOISE_PARAMS, + ) + assert float(vol) == 0.0, f"Expected zero noise volume, got {float(vol)}" + + def test_log_zero_when_arb_large(self): + vol = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=3_000_000.0, + gamma=0.997, + volatility=0.3, + arb_volume_this_period=1e12, + noise_params=DEFAULT_NOISE_PARAMS, + ) + assert float(vol) == 0.0, f"Expected zero noise volume, got {float(vol)}" + + +class TestMonotonicInEffectiveTVL: + """Test 3: Higher effective TVL -> more predicted volume.""" + + def test_sqrt_monotonic_effective_tvl(self): + kwargs = dict( + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + vol_low = reclamm_tsoukalas_sqrt_noise_volume(effective_value_usd=3_000_000.0, **kwargs) + vol_high = reclamm_tsoukalas_sqrt_noise_volume(effective_value_usd=20_000_000.0, **kwargs) + assert float(vol_high) > float(vol_low) + + def test_log_monotonic_effective_tvl(self): + kwargs = dict( + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + vol_low = reclamm_tsoukalas_log_noise_volume(effective_value_usd=3_000_000.0, **kwargs) + vol_high = reclamm_tsoukalas_log_noise_volume(effective_value_usd=20_000_000.0, **kwargs) + assert float(vol_high) > float(vol_low) + + +class TestMonotonicInVolatility: + """Test 4: Higher volatility -> more predicted volume.""" + + def test_sqrt_monotonic_volatility(self): + kwargs = dict( + effective_value_usd=10_000_000.0, + gamma=0.997, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + vol_low = reclamm_tsoukalas_sqrt_noise_volume(volatility=0.2, **kwargs) + vol_high = reclamm_tsoukalas_sqrt_noise_volume(volatility=0.8, **kwargs) + assert float(vol_high) > float(vol_low) + + def test_log_monotonic_volatility(self): + kwargs = dict( + effective_value_usd=10_000_000.0, + gamma=0.997, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + vol_low = reclamm_tsoukalas_log_noise_volume(volatility=0.2, **kwargs) + vol_high = reclamm_tsoukalas_log_noise_volume(volatility=0.8, **kwargs) + assert float(vol_high) > float(vol_low) + + +class TestEffectiveTVLSensitivity: + """Test 5: Changing effective TVL changes output.""" + + def test_sqrt_tvl_sensitivity(self): + base = dict( + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + v1 = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=7_000_000.0, **base) + v2 = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=13_000_000.0, **base) + assert float(v1) != float(v2), "Effective TVL change should affect output" + + def test_log_tvl_sensitivity(self): + base = dict( + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=DEFAULT_NOISE_PARAMS, + ) + v1 = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=7_000_000.0, **base) + v2 = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=13_000_000.0, **base) + assert float(v1) != float(v2), "Effective TVL change should affect output" + + +class TestCustomParamsOverrideDefaults: + """Test 6: noise_params dict values are actually used.""" + + def test_sqrt_custom_params(self): + # With a_sigma=0, volatility shouldn't matter + zero_sigma_params = {**DEFAULT_NOISE_PARAMS, "a_sigma": 0.0} + v1 = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.2, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + v2 = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.8, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + npt.assert_allclose(float(v1), float(v2), rtol=1e-10, + err_msg="With a_sigma=0, volatility should not affect output") + + def test_log_custom_params(self): + zero_sigma_params = {**DEFAULT_NOISE_PARAMS, "a_sigma": 0.0} + v1 = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.2, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + v2 = reclamm_tsoukalas_log_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.8, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + npt.assert_allclose(float(v1), float(v2), rtol=1e-10, + err_msg="With a_sigma=0, volatility should not affect output") + + +class TestCalculateVolatilityArray: + """Test calculate_volatility_array on the base pool class (JIT'd, pure JAX).""" + + def _make_run_fp(self): + from quantammsim.runners.jax_runner_utils import Hashabledict + return Hashabledict({"tokens": ("ETH", "USDC"), "numeraire": "USDC"}) + + def _make_pool(self): + from quantammsim.pools.creator import create_pool + return create_pool("reclamm") + + def test_output_shape_matches_input(self): + """Volatility array length matches input price length.""" + pool = self._make_pool() + n_minutes = 1440 * 3 # 3 days + rng = np.random.default_rng(42) + log_rets = rng.normal(0, 0.001, (n_minutes, 2)) + prices = jnp.array( + np.exp(np.cumsum(log_rets, axis=0)) * np.array([2500.0, 1.0]) + ) + vol_array = pool.calculate_volatility_array(prices, self._make_run_fp()) + assert vol_array.shape == (n_minutes,), ( + f"Expected shape ({n_minutes},), got {vol_array.shape}" + ) + + def test_constant_prices_zero_vol(self): + """Constant prices should give zero volatility.""" + pool = self._make_pool() + n_minutes = 1440 * 2 + prices = jnp.tile(jnp.array([2500.0, 1.0]), (n_minutes, 1)) + vol_array = pool.calculate_volatility_array(prices, self._make_run_fp()) + npt.assert_allclose(np.array(vol_array), 0.0, atol=1e-10) + + def test_volatile_prices_positive_vol(self): + """Volatile prices should give positive volatility.""" + pool = self._make_pool() + n_minutes = 1440 * 2 + rng = np.random.default_rng(123) + log_rets = rng.normal(0, 0.01, n_minutes) + price_ratio = np.exp(np.cumsum(log_rets)) + prices = jnp.array( + np.column_stack([price_ratio * 2500.0, np.ones(n_minutes)]) + ) + vol_array = pool.calculate_volatility_array(prices, self._make_run_fp()) + assert float(jnp.mean(vol_array)) > 0, "Volatile prices should give positive vol" + + def test_partial_last_day_handled(self): + """Non-multiple-of-1440: correct shape and partial-day fill uses last day's vol.""" + pool = self._make_pool() + n_full_days = 2 + n_partial = 500 + n_minutes = 1440 * n_full_days + n_partial + + # Use volatile prices so daily vol is nonzero + rng = np.random.default_rng(77) + log_rets = rng.normal(0, 0.005, n_minutes) + price_ratio = np.exp(np.cumsum(log_rets)) + prices = jnp.array( + np.column_stack([price_ratio * 2500.0, np.ones(n_minutes)]) + ) + vol_array = pool.calculate_volatility_array(prices, self._make_run_fp()) + + assert vol_array.shape == (n_minutes,) + + # The partial-day region (last 500 minutes) should be filled + # with the last full day's volatility value + last_full_day_vol = vol_array[n_full_days * 1440 - 1] + partial_region = vol_array[n_full_days * 1440:] + npt.assert_allclose( + np.array(partial_region), + float(last_full_day_vol), + rtol=1e-10, + err_msg="Partial-day region should be filled with last full day's vol", + ) + # And the fill value should be nonzero (volatile prices) + assert float(last_full_day_vol) > 0, "Expected nonzero vol from volatile prices" + + +# --------------------------------------------------------------------------- +# Tests 8-12: Scan step integration tests +# --------------------------------------------------------------------------- + +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + initialise_reclamm_reserves, + _jax_calc_reclamm_reserves_with_fees, + _jax_calc_reclamm_reserves_and_fee_revenue_with_fees, + _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs, +) + +ALL_SIG_VARIATIONS_2 = jnp.array([[1, -1], [-1, 1]]) + +# Pool config shared by integration tests +_CM = 0.2 # centeredness_margin +_DPSB = 1.0 - 1.0 / 124000.0 # daily_price_shift_base +_SPP = 60.0 # seconds_per_step (1-min arb) +_FEES = 0.003 +_PRICE_RATIO = 4.0 +_POOL_VALUE = 1_000_000.0 + + +def _init_pool(pool_value=_POOL_VALUE, price_a=2500.0, price_b=1.0, + price_ratio=_PRICE_RATIO): + initial_prices = jnp.array([price_a, price_b]) + reserves, Va, Vb = initialise_reclamm_reserves(pool_value, initial_prices, price_ratio) + return reserves, Va, Vb + + +def _make_trending_prices(start_a, end_a, price_b, n_steps): + prices_a = jnp.linspace(start_a, end_a, n_steps) + prices_b = jnp.full(n_steps, price_b) + return jnp.stack([prices_a, prices_b], axis=1) + + +class TestRatioBackwardCompatible: + """Test 8: noise_model='ratio' matches existing noise_trader_ratio path.""" + + def test_ratio_model_matches_legacy(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + + # Legacy path: just noise_trader_ratio + res_legacy = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_trader_ratio=1.5, + ) + + # New path: noise_model="ratio" (default) + res_new = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_trader_ratio=1.5, + noise_model="ratio", + ) + npt.assert_array_equal(res_legacy, res_new) + + +class TestArbOnlyEqualsZeroRatio: + """Test 9: noise_model='arb_only' same as noise_trader_ratio=0.""" + + def test_arb_only_matches_zero_ratio(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + + res_zero = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_trader_ratio=0.0, + ) + res_arb_only = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="arb_only", + ) + npt.assert_array_equal(res_zero, res_arb_only) + + +class TestTsoukalasSqrtIncreasesReserves: + """Test 10: Tsoukalas noise income grows real TVL vs arb-only.""" + + def test_sqrt_reserves_grow(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + vol_array = jnp.full(n_steps, 0.5) # Synthetic constant volatility + + res_arb_only = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="arb_only", + ) + res_tsoukalas = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="tsoukalas_sqrt", + noise_params=DEFAULT_NOISE_PARAMS, + volatility_array=vol_array, + ) + # Noise fee income should make total real value strictly greater than arb-only + val_arb = float(jnp.sum(res_arb_only[-1] * prices[-1])) + val_tsoukalas = float(jnp.sum(res_tsoukalas[-1] * prices[-1])) + assert val_tsoukalas > val_arb, ( + f"Tsoukalas reserves ({val_tsoukalas:.2f}) should be strictly > " + f"arb-only ({val_arb:.2f})" + ) + + +class TestTsoukalasDoesNotAffectVirtualBalances: + """Test 11: Within a single scan step, noise modifies real reserves + but does NOT modify Va/Vb in the carry.""" + + def test_single_step_virtual_balances_identical(self): + """Call the scan step directly for one step with arb_only and tsoukalas_sqrt. + Assert that the carry's Va and Vb are bitwise identical.""" + from quantammsim.pools.reCLAMM.reclamm_reserves import ( + _reclamm_scan_step_with_fees_and_revenue, + ) + from quantammsim.pools.G3M.optimal_n_pool_arb import ( + precalc_shared_values_for_all_signatures, + precalc_components_of_optimal_trade_across_prices, + ) + + reserves, Va, Vb = _init_pool() + # Small price shift so arb volume is small relative to predicted noise + # (a 2500→3000 jump produces arb > predicted noise/min, zeroing noise_vol) + prices_1 = jnp.array([[2510.0, 1.0]]) + + weights = jnp.array([0.5, 0.5]) + gamma = 1.0 - _FEES + + _, active_trade_dirs, tokens_to_drop, leave_one_out_idxs = ( + precalc_shared_values_for_all_signatures(ALL_SIG_VARIATIONS_2, 2) + ) + aiw, par, aoar = precalc_components_of_optimal_trade_across_prices( + weights, prices_1, gamma, tokens_to_drop, + active_trade_dirs, leave_one_out_idxs, + ) + + carry = [ + reserves, Va, Vb, + jnp.float64(1.0), # prev_lp_supply + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] + + def run_step(noise_model, noise_params=None): + input_list = [ + prices_1[0], aiw[0], par[0], aoar[0], + jnp.float64(gamma), jnp.float64(0.0), + jnp.float64(0.0), + jnp.array([0.0, 0.0, 0.0]), # price_ratio_update (no-op) + jnp.float64(1.0), # lp_supply + ] + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + input_list.append(jnp.float64(0.5)) # volatility + + return _reclamm_scan_step_with_fees_and_revenue( + carry, input_list, + weights=weights, + tokens_to_drop=tokens_to_drop, + active_trade_directions=active_trade_dirs, + n=2, + centeredness_margin=_CM, + daily_price_shift_base=_DPSB, + seconds_per_step=_SPP, + noise_model=noise_model, + noise_params=noise_params if noise_params is not None else {}, + ) + + carry_arb, (res_arb, _) = run_step("arb_only") + carry_tsoukalas, (res_tsoukalas, _) = run_step( + "tsoukalas_sqrt", DEFAULT_NOISE_PARAMS, + ) + + # Va and Vb in carry must be bitwise identical + npt.assert_array_equal(carry_arb[1], carry_tsoukalas[1], + err_msg="Va should be unaffected by noise model") + npt.assert_array_equal(carry_arb[2], carry_tsoukalas[2], + err_msg="Vb should be unaffected by noise model") + + # But real reserves SHOULD differ (noise adds fee income). + # The noise effect is small relative to reserve magnitude, so use + # exact bitwise comparison rather than allclose (whose default + # rtol=1e-5 would mask the difference). + assert not jnp.array_equal(res_arb, res_tsoukalas), ( + "Real reserves should differ between arb_only and tsoukalas_sqrt" + ) + + # Same invariant for loglinear path + loglinear_params = {"b_0": -1.4, "b_sigma": 0.1, "b_c": 1.04} + carry_loglinear, (res_loglinear, _) = run_step( + "loglinear", loglinear_params, + ) + npt.assert_array_equal(carry_arb[1], carry_loglinear[1], + err_msg="Va should be unaffected by loglinear noise model") + npt.assert_array_equal(carry_arb[2], carry_loglinear[2], + err_msg="Vb should be unaffected by loglinear noise model") + assert not jnp.array_equal(res_arb, res_loglinear), ( + "Real reserves should differ between arb_only and loglinear" + ) + + +class TestTsoukalasWithFeeRevenue: + """Test 12: Fee revenue includes noise contribution.""" + + def test_fee_revenue_includes_noise(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + vol_array = jnp.full(n_steps, 0.5) # Synthetic constant volatility + + _, fee_rev_arb = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="arb_only", + ) + _, fee_rev_tsoukalas = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="tsoukalas_sqrt", + noise_params=DEFAULT_NOISE_PARAMS, + volatility_array=vol_array, + ) + # Tsoukalas should generate strictly more fee revenue due to noise volume + total_arb = float(fee_rev_arb.sum()) + total_tsoukalas = float(fee_rev_tsoukalas.sum()) + assert total_tsoukalas > total_arb, ( + f"Tsoukalas fee revenue ({total_tsoukalas:.4f}) should exceed " + f"arb-only ({total_arb:.4f})" + ) + + +# --------------------------------------------------------------------------- +# Tests 13-14: Pool class integration tests +# --------------------------------------------------------------------------- + + +class TestNoiseModelFromFingerprint: + """Test 13: Pool reads noise_model from fingerprint.""" + + def test_tsoukalas_sqrt_from_fingerprint(self): + from quantammsim.pools.creator import create_pool + from quantammsim.runners.jax_runner_utils import Hashabledict + + pool = create_pool("reclamm") + + params = { + "price_ratio": _PRICE_RATIO, + "centeredness_margin": _CM, + "daily_price_shift_base": _DPSB, + } + + n_steps = 50 + np.random.seed(42) + price_a = 2500.0 * np.exp(np.cumsum(np.random.normal(0, 0.01, n_steps))) + prices = jnp.stack([jnp.array(price_a), jnp.ones(n_steps)], axis=1) + + # Fingerprint with Tsoukalas noise model + run_fingerprint_tsoukalas = Hashabledict({ + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": _POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "fees": _FEES, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "noise_model": "tsoukalas_sqrt", + "reclamm_noise_params": DEFAULT_NOISE_PARAMS, + }) + + # Fingerprint without noise + run_fingerprint_arb_only = Hashabledict({ + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": _POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "fees": _FEES, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "noise_model": "arb_only", + }) + + start_index = jnp.array([0, 0]) + + res_tsoukalas, fee_rev_tsoukalas = pool.calculate_reserves_and_fee_revenue_with_fees( + params, run_fingerprint_tsoukalas, prices, start_index, + ) + res_arb, fee_rev_arb = pool.calculate_reserves_and_fee_revenue_with_fees( + params, run_fingerprint_arb_only, prices, start_index, + ) + + assert res_tsoukalas.shape == (n_steps, 2) + assert fee_rev_tsoukalas.shape == (n_steps,) + # Tsoukalas should produce more fee revenue + assert float(fee_rev_tsoukalas.sum()) > float(fee_rev_arb.sum()) + + +class TestVolatilityComputedForTsoukalas: + """Test 14: Volatility array auto-computed when noise_model is tsoukalas_*. + + Uses >= 1440 minutes so the real vmap+dynamic_slice path is exercised + (not just the <1440 fallback). Compares against arb_only to verify the + auto-computed volatility feeds through to meaningfully different fee revenue. + """ + + def test_volatility_auto_computed_affects_fee_revenue(self): + from quantammsim.pools.creator import create_pool + from quantammsim.runners.jax_runner_utils import Hashabledict + + pool = create_pool("reclamm") + + params = { + "price_ratio": _PRICE_RATIO, + "centeredness_margin": _CM, + "daily_price_shift_base": _DPSB, + } + + # Need at least 1 day of data for real volatility computation + n_steps = 1440 + 100 # Just over 1 day + np.random.seed(42) + price_a = 2500.0 * np.exp(np.cumsum(np.random.normal(0, 0.001, n_steps))) + prices = jnp.stack([jnp.array(price_a), jnp.ones(n_steps)], axis=1) + + base_fp = { + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": _POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "fees": _FEES, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + } + + fp_tsoukalas = Hashabledict({ + **base_fp, + "noise_model": "tsoukalas_sqrt", + "reclamm_noise_params": DEFAULT_NOISE_PARAMS, + }) + fp_arb_only = Hashabledict({ + **base_fp, + "noise_model": "arb_only", + }) + + start_index = jnp.array([0, 0]) + + res_tsoukalas, fee_rev_tsoukalas = pool.calculate_reserves_and_fee_revenue_with_fees( + params, fp_tsoukalas, prices, start_index, + ) + _, fee_rev_arb = pool.calculate_reserves_and_fee_revenue_with_fees( + params, fp_arb_only, prices, start_index, + ) + + assert res_tsoukalas.shape == (n_steps, 2) + assert fee_rev_tsoukalas.shape == (n_steps,) + + # The auto-computed volatility must feed through to produce + # strictly more fee revenue than arb-only + total_tsoukalas = float(fee_rev_tsoukalas.sum()) + total_arb = float(fee_rev_arb.sum()) + assert total_tsoukalas > total_arb, ( + f"Tsoukalas with auto-computed volatility ({total_tsoukalas:.4f}) " + f"should exceed arb-only ({total_arb:.4f})" + ) + + +# --------------------------------------------------------------------------- +# Tests 15-16: Calibration pipeline tests +# --------------------------------------------------------------------------- + +import pandas as pd +from scripts.calibrate_reclamm_noise import run_ols_calibration + + +class TestOLSRecoversKnownParams: + """Test 15: Synthetic data with known coefficients -> OLS recovers them.""" + + def test_ols_recovery_sqrt(self): + rng = np.random.default_rng(42) + n = 200 + + true_a_0 = 0.8 + true_a_sigma = 1.5 + true_a_c = 0.6 + + vol = rng.uniform(0.2, 1.0, n) + eff_tvl = rng.uniform(3e6, 50e6, n) + + # Construct volume from known params (in $M units) + volume_M = ( + true_a_0 + + true_a_sigma * vol + + true_a_c * np.sqrt(eff_tvl / 1e6) + ) + # Add small noise + volume_M += rng.normal(0, 0.01, n) + volume_usd = volume_M * 1e6 + + df = pd.DataFrame({ + "volume_usd": volume_usd, + "volatility": vol, + "effective_tvl_usd": eff_tvl, + }) + + noise_params, diagnostics = run_ols_calibration(df, base_fee=0.003, model="sqrt") + + npt.assert_allclose(noise_params["a_0_base"], true_a_0, atol=0.05) + npt.assert_allclose(noise_params["a_sigma"], true_a_sigma, atol=0.05) + npt.assert_allclose(noise_params["a_c"], true_a_c, atol=0.05) + assert diagnostics["r_squared"] > 0.99 + + def test_ols_recovery_log(self): + rng = np.random.default_rng(123) + n = 200 + + true_a_0 = 0.5 + true_a_sigma = 2.0 + true_a_c = 0.4 + + vol = rng.uniform(0.2, 1.0, n) + eff_tvl = rng.uniform(3e6, 50e6, n) + + volume_M = ( + true_a_0 + + true_a_sigma * vol + + true_a_c * np.log(eff_tvl / 1e6) + ) + volume_M += rng.normal(0, 0.01, n) + volume_usd = volume_M * 1e6 + + df = pd.DataFrame({ + "volume_usd": volume_usd, + "volatility": vol, + "effective_tvl_usd": eff_tvl, + }) + + noise_params, diagnostics = run_ols_calibration(df, base_fee=0.003, model="log") + + npt.assert_allclose(noise_params["a_0_base"], true_a_0, atol=0.05) + npt.assert_allclose(noise_params["a_sigma"], true_a_sigma, atol=0.05) + npt.assert_allclose(noise_params["a_c"], true_a_c, atol=0.05) + assert diagnostics["r_squared"] > 0.99 + + +class TestOutputFormatCompatible: + """Test 16: Output dict has all required keys for run_fingerprint integration.""" + + def test_output_keys(self): + rng = np.random.default_rng(99) + n = 50 + df = pd.DataFrame({ + "volume_usd": rng.uniform(1e6, 10e6, n), + "volatility": rng.uniform(0.2, 0.8, n), + "effective_tvl_usd": rng.uniform(3e6, 25e6, n), + }) + + noise_params, diagnostics = run_ols_calibration(df, base_fee=0.003) + + required_keys = {"a_0_base", "a_f", "a_sigma", "a_c", "base_fee"} + assert set(noise_params.keys()) == required_keys + + # All values are float + for k, v in noise_params.items(): + assert isinstance(v, float), f"{k} should be float, got {type(v)}" + + # a_f should be 0 for static fees + assert noise_params["a_f"] == 0.0 + + # Diagnostics should have standard errors + assert "se" in diagnostics + assert set(diagnostics["se"].keys()) == {"a_0", "a_sigma", "a_c"} + + # Can be used directly as noise_params for the noise functions + vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value_usd=15e6, + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=noise_params, + ) + assert jnp.isfinite(vol) + + +# --------------------------------------------------------------------------- +# Tests 17-24: Loglinear (hierarchical) noise volume model +# --------------------------------------------------------------------------- + +# Typical noise_params from the hierarchical model +LOGLINEAR_NOISE_PARAMS = { + "b_0": -7.1, # grand mean + BLUP + "b_sigma": -0.003, # shared volatility effect + "b_c": 1.04, # shared TVL elasticity + "base_fee": 0.003, +} + + +class TestLoglinearPositiveOutput: + """Test 17: Volume > 0 for typical inputs.""" + + def test_loglinear_positive_output(self): + vol = reclamm_loglinear_noise_volume( + effective_value_usd=15_000_000.0, + gamma=0.997, + volatility=0.5, + arb_volume_this_period=0.0, + noise_params=LOGLINEAR_NOISE_PARAMS, + ) + assert float(vol) > 0, f"Expected positive noise volume, got {float(vol)}" + + +class TestLoglinearZeroWhenArbDominates: + """Test 18: noise = max(0, ...) so returns 0 when arb dominates.""" + + def test_loglinear_zero_when_arb_large(self): + vol = reclamm_loglinear_noise_volume( + effective_value_usd=3_000_000.0, + gamma=0.997, + volatility=0.3, + arb_volume_this_period=1e12, + noise_params=LOGLINEAR_NOISE_PARAMS, + ) + assert float(vol) == 0.0, f"Expected zero, got {float(vol)}" + + +class TestLoglinearMonotonic: + """Test 19: Higher effective TVL -> more predicted volume.""" + + def test_loglinear_monotonic_tvl(self): + kwargs = dict( + gamma=0.997, volatility=0.5, arb_volume_this_period=0.0, + noise_params=LOGLINEAR_NOISE_PARAMS, + ) + vol_low = reclamm_loglinear_noise_volume( + effective_value_usd=3_000_000.0, **kwargs) + vol_high = reclamm_loglinear_noise_volume( + effective_value_usd=20_000_000.0, **kwargs) + assert float(vol_high) > float(vol_low) + + +class TestLoglinearCustomParams: + """Test 20: noise_params dict values are actually used.""" + + def test_loglinear_custom_b_sigma(self): + # With b_sigma=0, volatility shouldn't matter + zero_sigma_params = {**LOGLINEAR_NOISE_PARAMS, "b_sigma": 0.0} + v1 = reclamm_loglinear_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.2, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + v2 = reclamm_loglinear_noise_volume( + effective_value_usd=10_000_000.0, + gamma=0.997, volatility=0.8, arb_volume_this_period=0.0, + noise_params=zero_sigma_params, + ) + npt.assert_allclose(float(v1), float(v2), rtol=1e-10, + err_msg="With b_sigma=0, volatility should not affect output") + + +class TestLoglinearScanStepIntegration: + """Test 21: loglinear noise model works through the scan step.""" + + def test_loglinear_increases_reserves(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + vol_array = jnp.full(n_steps, 0.5) + + # Use a b_0 that gives reasonable volume at this TVL + # Pool TVL ~$1M → log(1e6) ≈ 13.8 → b_0 + 1.04*13.8 = b_0 + 14.4 + # Want log(V_daily) ≈ 13 (= ~$440k/day) → b_0 ≈ -1.4 + params = {"b_0": -1.4, "b_sigma": 0.1, "b_c": 1.04, "base_fee": 0.003} + + res_arb_only = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="arb_only", + ) + res_loglinear = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="loglinear", + noise_params=params, + volatility_array=vol_array, + ) + val_arb = float(jnp.sum(res_arb_only[-1] * prices[-1])) + val_loglinear = float(jnp.sum(res_loglinear[-1] * prices[-1])) + assert val_loglinear > val_arb, ( + f"Loglinear reserves ({val_loglinear:.2f}) should exceed " + f"arb-only ({val_arb:.2f})" + ) + + +class TestLoglinearFeeRevenue: + """Test 22: Fee revenue includes loglinear noise contribution.""" + + def test_loglinear_fee_revenue(self): + reserves, Va, Vb = _init_pool() + n_steps = 50 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + vol_array = jnp.full(n_steps, 0.5) + + params = {"b_0": -1.4, "b_sigma": 0.1, "b_c": 1.04, "base_fee": 0.003} + + _, fee_rev_arb = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="arb_only", + ) + _, fee_rev_loglinear = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, _CM, _DPSB, _SPP, + fees=_FEES, all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_model="loglinear", + noise_params=params, + volatility_array=vol_array, + ) + total_arb = float(fee_rev_arb.sum()) + total_loglinear = float(fee_rev_loglinear.sum()) + assert total_loglinear > total_arb, ( + f"Loglinear fee revenue ({total_loglinear:.4f}) should exceed " + f"arb-only ({total_arb:.4f})" + ) + + +class TestLoglinearPoolClassIntegration: + """Test 23: Pool reads loglinear noise_model from fingerprint.""" + + def test_loglinear_from_fingerprint(self): + from quantammsim.pools.creator import create_pool + from quantammsim.runners.jax_runner_utils import Hashabledict + + pool = create_pool("reclamm") + + params = { + "price_ratio": _PRICE_RATIO, + "centeredness_margin": _CM, + "daily_price_shift_base": _DPSB, + } + + n_steps = 50 + np.random.seed(42) + price_a = 2500.0 * np.exp(np.cumsum(np.random.normal(0, 0.01, n_steps))) + prices = jnp.stack([jnp.array(price_a), jnp.ones(n_steps)], axis=1) + + loglinear_params = { + "b_0": -1.4, "b_sigma": 0.1, "b_c": 1.04, "base_fee": 0.003, + } + + fp_loglinear = Hashabledict({ + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": _POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "fees": _FEES, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "noise_model": "loglinear", + "reclamm_noise_params": loglinear_params, + }) + + fp_arb_only = Hashabledict({ + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": _POOL_VALUE, + "arb_frequency": 1, + "do_arb": True, + "fees": _FEES, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "noise_model": "arb_only", + }) + + start_index = jnp.array([0, 0]) + + res_loglinear, fee_rev_loglinear = pool.calculate_reserves_and_fee_revenue_with_fees( + params, fp_loglinear, prices, start_index, + ) + _, fee_rev_arb = pool.calculate_reserves_and_fee_revenue_with_fees( + params, fp_arb_only, prices, start_index, + ) + + assert res_loglinear.shape == (n_steps, 2) + assert fee_rev_loglinear.shape == (n_steps,) + assert float(fee_rev_loglinear.sum()) > float(fee_rev_arb.sum()) + + +class TestLoglinearDefaultParams: + """Test 24: Function works with default params (noise_params=None).""" + + def test_loglinear_defaults(self): + vol = reclamm_loglinear_noise_volume( + effective_value_usd=15_000_000.0, + gamma=0.997, + volatility=0.5, + arb_volume_this_period=0.0, + ) + assert jnp.isfinite(vol) + assert float(vol) > 0 diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index 1db40cb7..ef70c9ec 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -13,12 +13,14 @@ from quantammsim.pools.reCLAMM.reclamm_reserves import ( compute_invariant, compute_price_ratio, + compute_centeredness, initialise_reclamm_reserves, calibrate_arc_length_speed, _jax_calc_reclamm_reserves_zero_fees, + _jax_calc_reclamm_reserves_zero_fees_full_state, _jax_calc_reclamm_reserves_with_fees, + _jax_calc_reclamm_reserves_and_fee_revenue_with_fees, ) -from tests.conftest import TEST_DATA_DIR # For n=2: sig variations with exactly one +1 and one -1 ALL_SIG_VARIATIONS_2 = jnp.array([[1, -1], [-1, 1]]) @@ -816,3 +818,592 @@ def test_train_on_historic_data_optuna(self): } result = train_on_historic_data(fp, verbose=False, root=TEST_DATA_DIR) assert result is not None + + +class TestNoiseTraderRatio: + """Noise trader fee income wiring for reClAMM pools.""" + + def _run_with_noise(self, noise_trader_ratio, n_steps=50, fees=0.003): + """Run reClAMM with fees and return reserves + fee revenue.""" + reserves, Va, Vb = _init_pool() + # Trending prices so arb trades happen and noise trade has non-zero effect + prices = _make_trending_prices(2500.0, 3000.0, 1.0, n_steps) + + return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=fees, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_trader_ratio=noise_trader_ratio, + ) + + def test_noise_trader_ratio_zero_is_default(self): + """noise_trader_ratio=0.0 should produce identical results to omitting it.""" + reserves, Va, Vb = _init_pool() + prices = _make_trending_prices(2500.0, 3000.0, 1.0, 50) + + # Explicit zero + res_zero, rev_zero = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + noise_trader_ratio=0.0, + ) + + # Default (omitted) + res_default, rev_default = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + + npt.assert_array_equal(res_zero, res_default) + npt.assert_array_equal(rev_zero, rev_default) + + def test_noise_trader_ratio_increases_reserves(self): + """Noise traders add fee income, so pool value should be higher.""" + res_no_noise, _ = self._run_with_noise(0.0) + res_noise, _ = self._run_with_noise(0.1) + + # Final pool value in USD (sum of reserves * price at last step) + final_prices = jnp.array([3000.0, 1.0]) + value_no_noise = (res_no_noise[-1] * final_prices).sum() + value_noise = (res_noise[-1] * final_prices).sum() + + assert value_noise > value_no_noise, ( + f"Noise traders should increase pool value: {value_noise} <= {value_no_noise}" + ) + + # Reserves should differ + assert not jnp.allclose(res_no_noise, res_noise), ( + "Reserves should differ with noise traders" + ) + + def test_noise_trader_ratio_through_pool_class(self): + """noise_trader_ratio flows through the pool class methods.""" + from quantammsim.pools.creator import create_pool + from quantammsim.runners.jax_runner_utils import Hashabledict + + pool = create_pool("reclamm") + + params = { + "price_ratio": DEFAULT_PRICE_RATIO, + "centeredness_margin": DEFAULT_CENTEREDNESS_MARGIN, + "daily_price_shift_base": DEFAULT_DAILY_PRICE_SHIFT_BASE, + } + + n_steps = 50 + np.random.seed(42) + price_a = 2500.0 * np.exp(np.cumsum(np.random.normal(0, 0.01, n_steps))) + prices = jnp.stack([jnp.array(price_a), jnp.ones(n_steps)], axis=1) + start_index = jnp.array([0, 0]) + + base_fp = { + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": 1_000_000.0, + "arb_frequency": 1, + "do_arb": True, + "fees": 0.003, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + } + + fp_no_noise = Hashabledict({**base_fp, "noise_trader_ratio": 0.0}) + fp_noise = Hashabledict({**base_fp, "noise_trader_ratio": 0.1}) + + res_no = pool.calculate_reserves_with_fees( + params, fp_no_noise, prices, start_index + ) + res_yes = pool.calculate_reserves_with_fees( + params, fp_noise, prices, start_index + ) + + # Should run and produce different results + assert res_no.shape == (n_steps, 2) + assert res_yes.shape == (n_steps, 2) + assert not jnp.allclose(res_no, res_yes), ( + "Pool class should produce different reserves with noise traders" + ) + + def test_noise_trade_does_not_affect_virtual_balances(self): + """Noise trade fee income only grows real reserves, not Va/Vb.""" + from quantammsim.pools.reCLAMM.reclamm_reserves import ( + _reclamm_scan_step_with_fees_and_revenue, + precalc_shared_values_for_all_signatures, + precalc_components_of_optimal_trade_across_prices, + ) + + reserves, Va, Vb = _init_pool() + # Price must differ from init (2500) to trigger an arb trade + prices_arr = jnp.array([[3000.0, 1.0]]) + prices = prices_arr[0] + + weights = jnp.array([0.5, 0.5]) + gamma = 1.0 - 0.003 + n_assets = 2 + + _, active_trade_directions, tokens_to_drop, leave_one_out_idxs = ( + precalc_shared_values_for_all_signatures(ALL_SIG_VARIATIONS_2, n_assets) + ) + + active_initial_weights, per_asset_ratios, all_other_assets_ratios = ( + precalc_components_of_optimal_trade_across_prices( + weights, prices_arr, gamma, tokens_to_drop, + active_trade_directions, leave_one_out_idxs, + ) + ) + + carry = [ + reserves, Va, Vb, + jnp.float64(1.0), # prev_lp_supply + jnp.float64(0.0), # step_idx + jnp.float64(0.0), # active_start_ratio + jnp.float64(0.0), # active_target_ratio + jnp.float64(0.0), # active_start_step + jnp.float64(0.0), # active_end_step + jnp.array(False), # active_enabled + ] + inputs = [ + prices, + active_initial_weights[0], + per_asset_ratios[0], + all_other_assets_ratios[0], + gamma, + 0.0, # arb_thresh + 0.0, # arb_fees + jnp.array([0.0, 0.0, 0.0]), # price_ratio_update (no-op) + jnp.float64(1.0), # lp_supply + ] + + # Without noise + carry_no, _ = _reclamm_scan_step_with_fees_and_revenue( + carry, inputs, + weights=weights, + tokens_to_drop=tokens_to_drop, + active_trade_directions=active_trade_directions, + n=n_assets, + centeredness_margin=DEFAULT_CENTEREDNESS_MARGIN, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + noise_trader_ratio=0.0, + ) + + # With noise + carry_yes, _ = _reclamm_scan_step_with_fees_and_revenue( + carry, inputs, + weights=weights, + tokens_to_drop=tokens_to_drop, + active_trade_directions=active_trade_directions, + n=n_assets, + centeredness_margin=DEFAULT_CENTEREDNESS_MARGIN, + daily_price_shift_base=DEFAULT_DAILY_PRICE_SHIFT_BASE, + seconds_per_step=DEFAULT_SECONDS_PER_STEP, + noise_trader_ratio=0.5, + ) + + # Virtual balances should be identical + npt.assert_array_equal(carry_no[1], carry_yes[1], err_msg="Va changed") + npt.assert_array_equal(carry_no[2], carry_yes[2], err_msg="Vb changed") + + # But real reserves should differ (noise adds fee income) + assert not jnp.allclose(carry_no[0], carry_yes[0]), ( + "Real reserves should differ with noise traders" + ) + + +class TestLpSupply: + """LP supply (BPT) scaling for reClAMM pools. + + Ported from Foundry fuzz test invariants in ReClammLiquidity.t.sol: + both real AND virtual reserves scale proportionally with BPT supply, + preserving price and centeredness. + """ + + def test_lp_supply_none_matches_default(self): + """lp_supply_array=None vs omitting param → identical results.""" + reserves, Va, Vb = _init_pool() + prices = _make_trending_prices(2500.0, 3000.0, 1.0, 20) + + result_none = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + lp_supply_array=None, + ) + result_omit = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + npt.assert_array_equal(result_none, result_omit) + + def test_lp_supply_constant_one_matches_default(self): + """lp_supply_array=jnp.ones(T) vs None → identical results.""" + reserves, Va, Vb = _init_pool() + n_steps = 20 + prices = _make_trending_prices(2500.0, 3000.0, 1.0, n_steps) + + result_ones = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + lp_supply_array=jnp.ones(n_steps), + ) + result_none = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + npt.assert_array_equal(result_ones, result_none) + + def test_lp_supply_doubling_scales_reserves(self): + """BPT supply doubling halfway → reserves ~2x at that step. + + Ported from Foundry testAddLiquidity__Fuzz: reserves scale with BPT. + """ + reserves, Va, Vb = _init_pool() + n_steps = 40 + half = n_steps // 2 + prices = _make_constant_prices(2500.0, 1.0, n_steps) + + # No LP supply change (baseline) + result_base = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + + # LP supply doubles at step `half` + lp_supply = jnp.concatenate([jnp.ones(half), 2.0 * jnp.ones(n_steps - half)]) + result_lp = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + lp_supply_array=lp_supply, + ) + + # Before doubling: identical + npt.assert_allclose(result_lp[:half], result_base[:half], rtol=1e-10) + # After doubling: reserves ~2x the baseline + npt.assert_allclose(result_lp[half], result_base[half] * 2.0, rtol=1e-6) + + def test_lp_supply_halving_scales_reserves(self): + """BPT supply halving halfway → reserves ~0.5x at that step.""" + reserves, Va, Vb = _init_pool() + n_steps = 40 + half = n_steps // 2 + prices = _make_constant_prices(2500.0, 1.0, n_steps) + + result_base = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + + lp_supply = jnp.concatenate([jnp.ones(half), 0.5 * jnp.ones(n_steps - half)]) + result_lp = _jax_calc_reclamm_reserves_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + lp_supply_array=lp_supply, + ) + + # Before halving: identical + npt.assert_allclose(result_lp[:half], result_base[:half], rtol=1e-10) + # After halving: reserves ~0.5x the baseline + npt.assert_allclose(result_lp[half], result_base[half] * 0.5, rtol=1e-6) + + def test_lp_supply_preserves_price(self): + """Price ratio (Ra+Va)/(Rb+Vb) unchanged through LP supply scaling. + + Ported from Foundry: assertEq(price_before, price_after) in + testAddLiquidity__Fuzz. + """ + reserves, Va, Vb = _init_pool() + n_steps = 40 + half = n_steps // 2 + # Trending so virtual balances shift — makes price preservation non-trivial + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + + lp_supply = jnp.concatenate([jnp.ones(half), 2.0 * jnp.ones(n_steps - half)]) + + # Use full_state variant to get Va/Vb history + result_reserves, Va_hist, Vb_hist = _jax_calc_reclamm_reserves_zero_fees_full_state( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + lp_supply_array=lp_supply, + ) + + # Price = (Rb + Vb) / (Ra + Va) — should be identical at step half-1 and half + # (the supply change happens at start of step `half`, before arb) + Ra_before = result_reserves[half - 1, 0] + Rb_before = result_reserves[half - 1, 1] + Va_before = Va_hist[half - 1] + Vb_before = Vb_hist[half - 1] + price_before = (Rb_before + Vb_before) / (Ra_before + Va_before) + + # At step `half`, the carry from step half-1 gets scaled, then arb runs. + # We need to check price right after scaling, before arb. + # The closest check: price at step `half` output should reflect + # scaled reserves with arb on top. Instead, verify that a constant-price + # run preserves exact ratio. + prices_const = _make_constant_prices(2500.0, 1.0, n_steps) + res_c, Va_c, Vb_c = _jax_calc_reclamm_reserves_zero_fees_full_state( + reserves, Va, Vb, prices_const, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + lp_supply_array=lp_supply, + ) + # With constant prices and no arb, price is just initial ratio throughout + price_at_half_minus_1 = (res_c[half - 1, 1] + Vb_c[half - 1]) / ( + res_c[half - 1, 0] + Va_c[half - 1] + ) + price_at_half = (res_c[half, 1] + Vb_c[half]) / ( + res_c[half, 0] + Va_c[half] + ) + npt.assert_allclose( + float(price_at_half_minus_1), float(price_at_half), rtol=1e-10, + err_msg="Price ratio should be preserved through LP supply change", + ) + + def test_lp_supply_preserves_centeredness(self): + """Centeredness unchanged through LP supply scaling. + + Ported from Foundry: centeredness invariance in testAddLiquidity__Fuzz. + """ + reserves, Va, Vb = _init_pool() + n_steps = 40 + half = n_steps // 2 + prices = _make_constant_prices(2500.0, 1.0, n_steps) + + lp_supply = jnp.concatenate([jnp.ones(half), 2.0 * jnp.ones(n_steps - half)]) + + res, Va_hist, Vb_hist = _jax_calc_reclamm_reserves_zero_fees_full_state( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + lp_supply_array=lp_supply, + ) + + c_before, _ = compute_centeredness( + res[half - 1, 0], res[half - 1, 1], Va_hist[half - 1], Vb_hist[half - 1] + ) + c_after, _ = compute_centeredness( + res[half, 0], res[half, 1], Va_hist[half], Vb_hist[half] + ) + npt.assert_allclose( + float(c_before), float(c_after), rtol=1e-10, + err_msg="Centeredness should be preserved through LP supply change", + ) + + def test_lp_supply_through_pool_class(self): + """Pool class passes lp_supply_array to underlying computation.""" + from quantammsim.pools.creator import create_pool + from quantammsim.runners.jax_runner_utils import Hashabledict + + pool = create_pool("reclamm") + + params = { + "price_ratio": DEFAULT_PRICE_RATIO, + "centeredness_margin": DEFAULT_CENTEREDNESS_MARGIN, + "daily_price_shift_base": DEFAULT_DAILY_PRICE_SHIFT_BASE, + } + + n_steps = 20 + prices = _make_trending_prices(2500.0, 3000.0, 1.0, n_steps) + + run_fingerprint = Hashabledict({ + "n_assets": 2, + "bout_length": n_steps + 1, + "initial_pool_value": 1_000_000.0, + "arb_frequency": 1, + "do_arb": True, + "fees": 0.003, + "gas_cost": 0.0, + "arb_fees": 0.0, + "tokens": ("ETH", "USDC"), + "numeraire": "USDC", + "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + }) + + start_index = jnp.array([0, 0]) + + # Build dynamic input arrays + fees_arr = jnp.full(n_steps, 0.003) + arb_thresh_arr = jnp.zeros(n_steps) + arb_fees_arr = jnp.zeros(n_steps) + trade_arr = jnp.zeros((n_steps, 2)) + + lp_supply = jnp.concatenate([jnp.ones(10), 2.0 * jnp.ones(10)]) + + res_with_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( + params, run_fingerprint, prices, start_index, + fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, + lp_supply_array=lp_supply, + ) + res_without_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( + params, run_fingerprint, prices, start_index, + fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, + ) + + # First 10 steps identical, then diverge + npt.assert_allclose(res_with_lp[:10], res_without_lp[:10], rtol=1e-10) + assert not jnp.allclose(res_with_lp[10:], res_without_lp[10:]), ( + "LP supply change should produce different reserves" + ) + + def test_lp_supply_with_fee_revenue(self): + """Doubling LP supply → fee revenue increases (bigger pool → bigger arb trades).""" + reserves, Va, Vb = _init_pool() + n_steps = 40 + half = n_steps // 2 + prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + + # Baseline: no supply change + _, rev_base = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + ) + + # Supply doubles halfway + lp_supply = jnp.concatenate([jnp.ones(half), 2.0 * jnp.ones(n_steps - half)]) + _, rev_lp = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( + reserves, Va, Vb, prices, + DEFAULT_CENTEREDNESS_MARGIN, + DEFAULT_DAILY_PRICE_SHIFT_BASE, + DEFAULT_SECONDS_PER_STEP, + fees=0.003, + arb_thresh=0.0, + arb_fees=0.0, + all_sig_variations=ALL_SIG_VARIATIONS_2, + lp_supply_array=lp_supply, + ) + + # After doubling, fee revenue per step should be larger + # (pool is 2x bigger → arb trades are 2x bigger → fees are 2x) + post_double_base = rev_base[half:].sum() + post_double_lp = rev_lp[half:].sum() + assert float(post_double_lp) > float(post_double_base), ( + f"Fee revenue should increase after doubling: {post_double_lp} <= {post_double_base}" + ) + + def test_lp_supply_e2e_do_run_on_historic_data(self): + """End-to-end: lp_supply_df flows through do_run_on_historic_data.""" + import pandas as pd + from quantammsim.runners.jax_runners import do_run_on_historic_data + + fp = { + "rule": "reclamm", + "tokens": ["ETH", "USDC"], + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2024-06-15 00:00:00", + "initial_pool_value": 1_000_000.0, + "do_arb": True, + "fees": 0.003, + } + params = { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array(DEFAULT_DAILY_PRICE_SHIFT_BASE), + } + + # Baseline: no LP supply change + result_base = do_run_on_historic_data( + run_fingerprint={**fp}, + params={**params}, + ) + + # LP supply doubles halfway through the period + # unix column must be in milliseconds (matches windowing_utils convention) + start_unix_ms = int(pd.Timestamp("2024-06-01").timestamp() * 1000) + mid_unix_ms = int(pd.Timestamp("2024-06-08").timestamp() * 1000) + lp_supply_df = pd.DataFrame({ + "unix": [start_unix_ms, mid_unix_ms], + "lp_supply": [1.0, 2.0], + }) + + result_lp = do_run_on_historic_data( + run_fingerprint={**fp}, + params={**params}, + lp_supply_df=lp_supply_df, + ) + + # Final values should differ — doubling LP supply changes pool dynamics + base_val = float(result_base["final_value"]) + lp_val = float(result_lp["final_value"]) + assert base_val != lp_val, ( + f"LP supply change should affect final value: base={base_val}, lp={lp_val}" + ) + # Doubled pool should have higher final value (more reserves) + assert lp_val > base_val, ( + f"Doubled LP supply should increase final value: {lp_val} <= {base_val}" + ) + From f4c4aaf090e24f3fe97abc0ab4171a0f0f87d5ca Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:43:37 +0000 Subject: [PATCH 026/115] feat: add noise calibration pipeline Hierarchical Bayesian noise model calibration using numpyro: data pipeline, covariate encoding, model definitions, inference, BLUP postprocessing, diagnostic plotting, and CLI. Adds [calibration] optional deps (numpyro, arviz) to pyproject.toml and updates CI to install them. --- .github/workflows/tests.yml | 4 +- pyproject.toml | 4 + quantammsim/noise_calibration/__init__.py | 39 ++ quantammsim/noise_calibration/cli.py | 417 +++++++++++ quantammsim/noise_calibration/constants.py | 60 ++ .../noise_calibration/covariate_encoding.py | 228 ++++++ .../noise_calibration/data_pipeline.py | 516 ++++++++++++++ .../noise_calibration/data_validation.py | 41 ++ quantammsim/noise_calibration/formula_arb.py | 36 + quantammsim/noise_calibration/inference.py | 270 +++++++ quantammsim/noise_calibration/model.py | 465 ++++++++++++ quantammsim/noise_calibration/output.py | 314 +++++++++ quantammsim/noise_calibration/plotting.py | 335 +++++++++ .../noise_calibration/postprocessing.py | 659 ++++++++++++++++++ .../noise_calibration/token_classification.py | 23 + tests/noise/__init__.py | 0 tests/noise/conftest.py | 274 ++++++++ tests/noise/test_covariate_encoding.py | 326 +++++++++ tests/noise/test_formula_arb.py | 142 ++++ tests/noise/test_model_and_inference.py | 328 +++++++++ tests/noise/test_model_dp_sigma.py | 305 ++++++++ tests/noise/test_model_structural.py | 190 +++++ tests/noise/test_output.py | 305 ++++++++ tests/noise/test_panel_assembly.py | 442 ++++++++++++ tests/noise/test_postprocessing.py | 533 ++++++++++++++ tests/noise/test_token_classification.py | 66 ++ 26 files changed, 6319 insertions(+), 3 deletions(-) create mode 100644 quantammsim/noise_calibration/__init__.py create mode 100644 quantammsim/noise_calibration/cli.py create mode 100644 quantammsim/noise_calibration/constants.py create mode 100644 quantammsim/noise_calibration/covariate_encoding.py create mode 100644 quantammsim/noise_calibration/data_pipeline.py create mode 100644 quantammsim/noise_calibration/data_validation.py create mode 100644 quantammsim/noise_calibration/formula_arb.py create mode 100644 quantammsim/noise_calibration/inference.py create mode 100644 quantammsim/noise_calibration/model.py create mode 100644 quantammsim/noise_calibration/output.py create mode 100644 quantammsim/noise_calibration/plotting.py create mode 100644 quantammsim/noise_calibration/postprocessing.py create mode 100644 quantammsim/noise_calibration/token_classification.py create mode 100644 tests/noise/__init__.py create mode 100644 tests/noise/conftest.py create mode 100644 tests/noise/test_covariate_encoding.py create mode 100644 tests/noise/test_formula_arb.py create mode 100644 tests/noise/test_model_and_inference.py create mode 100644 tests/noise/test_model_dp_sigma.py create mode 100644 tests/noise/test_model_structural.py create mode 100644 tests/noise/test_output.py create mode 100644 tests/noise/test_panel_assembly.py create mode 100644 tests/noise/test_postprocessing.py create mode 100644 tests/noise/test_token_classification.py diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 2e4b5f6b..8647a5e5 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -13,8 +13,6 @@ jobs: steps: - uses: actions/checkout@v4 - with: - lfs: true - name: Set up Python 3.9 uses: actions/setup-python@v5 @@ -25,7 +23,7 @@ jobs: - name: Install dependencies run: | python -m pip install --upgrade pip - pip install -e ".[dev]" + pip install -e ".[dev,calibration]" - name: Run tests with coverage run: | diff --git a/pyproject.toml b/pyproject.toml index 413faf9e..8859b551 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,6 +45,10 @@ docs = [ "sphinx-automodapi", "sphinx-rtd-theme", ] +calibration = [ + "numpyro>=0.15.0", + "arviz>=0.15.0", +] [tool.hatch.build.targets.wheel] packages = [ diff --git a/quantammsim/noise_calibration/__init__.py b/quantammsim/noise_calibration/__init__.py new file mode 100644 index 00000000..f1fddde2 --- /dev/null +++ b/quantammsim/noise_calibration/__init__.py @@ -0,0 +1,39 @@ +"""Noise calibration package for Balancer pool volume models. + +Public API re-exports from submodules. +""" + +# scipy.signal patch (must run before arviz import) +try: + from scipy.signal import gaussian as _ # noqa: F401 +except ImportError: + from scipy.signal.windows import gaussian as _gauss + import scipy.signal + scipy.signal.gaussian = _gauss + +from .constants import ( + K_COEFF, COEFF_NAMES, BALANCER_API_URL, BALANCER_API_CHAINS, CACHE_DIR, + K_CLUSTERS_DEFAULT, K_FEATURES_DEFAULT, +) +from .token_classification import classify_token_tier, _normalise_symbol +from .data_pipeline import ( + _graphql_request, enumerate_balancer_pools, fetch_pool_snapshots, + fetch_all_snapshots, fetch_token_prices, compute_pair_volatility, + assemble_panel, +) +from .data_validation import validate_panel +from .covariate_encoding import encode_covariates, encode_covariates_structural +from .model import noise_model, noise_model_dp_sigma, noise_model_ibp, noise_model_ibp_dp, stick_breaking_weights, structural_noise_model +from .formula_arb import formula_arb_volume_daily_jax +from .inference import ( + _get_theta_samples, _build_model_kwargs, run_svi, run_nuts, + run_svi_then_nuts, +) +from .postprocessing import ( + extract_noise_params, predict_new_pool, check_convergence, + run_prior_predictive, assign_dp_clusters, assign_ibp_dp_joint, + extract_structural_params, predict_new_pool_structural, +) +from .plotting import plot_diagnostics +from .output import generate_output_json, _save_sample_cache +from .cli import main diff --git a/quantammsim/noise_calibration/cli.py b/quantammsim/noise_calibration/cli.py new file mode 100644 index 00000000..3e169ac3 --- /dev/null +++ b/quantammsim/noise_calibration/cli.py @@ -0,0 +1,417 @@ +"""CLI entry point for noise calibration.""" + +import argparse +import json +import os +import sys +from datetime import date, timedelta + +import numpy as np +import pandas as pd + +from .constants import CACHE_DIR +from .data_pipeline import ( + enumerate_balancer_pools, fetch_all_snapshots, + fetch_token_prices, assemble_panel, +) +from .data_validation import validate_panel +from .covariate_encoding import encode_covariates +from .inference import run_svi, run_nuts, run_svi_then_nuts +from .postprocessing import ( + extract_noise_params, predict_new_pool, + check_convergence, run_prior_predictive, +) +from .plotting import plot_diagnostics +from .output import generate_output_json, _save_sample_cache + + +def _parse_args(): + parser = argparse.ArgumentParser( + description="Unified Bayesian hierarchical noise volume model " + "for Balancer pools (gold standard)" + ) + + # Actions + parser.add_argument("--fetch", action="store_true", + help="Fetch data from Balancer API") + parser.add_argument("--fit", action="store_true", + help="Run inference (SVI default)") + parser.add_argument("--nuts", action="store_true", + help="Use NUTS instead of SVI") + parser.add_argument("--svi-init-nuts", action="store_true", + help="SVI-initialized NUTS (fast warmup)") + parser.add_argument("--plot", action="store_true", + help="Generate diagnostic plots") + parser.add_argument("--prior-predictive", action="store_true", + help="Include prior predictive check") + parser.add_argument("--validate", action="store_true", + help="Run data validation pass") + parser.add_argument("--predict", action="store_true", + help="Predict for unseen pool") + + # Output + parser.add_argument("--output", default=None, + help="Output JSON path") + parser.add_argument("--output-dir", default="results", + help="Plot output directory (default: results)") + + # Predict args + parser.add_argument("--chain", default=None, + help="Chain for --predict") + parser.add_argument("--tokens", nargs="+", default=None, + help="Tokens for --predict") + parser.add_argument("--fee", type=float, default=0.003, + help="Fee for --predict") + + # NUTS hyperparameters + parser.add_argument("--num-warmup", type=int, default=1000, + help="NUTS warmup iterations (default: 1000)") + parser.add_argument("--num-samples", type=int, default=2000, + help="NUTS/SVI samples (default: 2000)") + parser.add_argument("--num-chains", type=int, default=4, + help="NUTS chains (default: 4)") + parser.add_argument("--target-accept", type=float, default=0.85, + help="NUTS target accept prob (default: 0.85)") + parser.add_argument("--max-tree-depth", type=int, default=10, + help="NUTS max tree depth (default: 10)") + parser.add_argument("--seed", type=int, default=42, + help="Random seed (default: 42)") + + # SVI hyperparameters + parser.add_argument("--svi-steps", type=int, default=20000, + help="SVI optimization steps (default: 20000)") + parser.add_argument("--svi-lr", type=float, default=1e-3, + help="SVI learning rate (default: 1e-3)") + + # Model variant + parser.add_argument("--model", choices=["tier", "dp_sigma", "ibp", "ibp_dp", + "structural"], + default="tier", + help="Noise model variant: 'tier' (per-tier sigma_eps), " + "'dp_sigma' (DP mixture on sigma_eps), " + "'ibp' (IBP latent features), " + "'ibp_dp' (IBP features + DP noise clusters), or " + "'structural' (structural mixture: arb + MoE noise)") + parser.add_argument("--k-clusters", type=int, default=6, + help="Number of DP mixture components " + "(capacity ceiling, default: 6)") + parser.add_argument("--k-features", type=int, default=6, + help="Number of IBP latent features " + "(default: 6)") + + # Data + parser.add_argument("--train-days", type=int, default=90, + help="Use only the last N days of data for fitting " + "(default: 90). Aligns with Balancer API hourly price " + "coverage window. Set to 0 to use all data.") + parser.add_argument("--min-tvl", type=float, default=10000.0, + help="Pool enumeration TVL filter") + parser.add_argument("--cache-dir", default=None, + help="Cache directory") + parser.add_argument("--device", choices=["cpu", "gpu", "auto"], + default="auto", + help="JAX device (default: auto)") + + return parser.parse_args() + + +def main(): + args = _parse_args() + + if not any([args.fetch, args.fit, args.predict, args.validate]): + print("ERROR: At least one of --fetch, --fit, --predict, --validate " + "is required", file=sys.stderr) + sys.exit(1) + + cache_dir = args.cache_dir or CACHE_DIR + + # --- JAX setup (BEFORE any JAX ops / imports) --- + if args.fit or args.predict or args.prior_predictive: + # Set device before importing JAX + if args.device == "cpu": + os.environ.setdefault("JAX_PLATFORMS", "cpu") + elif args.device == "gpu": + os.environ.setdefault("JAX_PLATFORMS", "cuda") + # auto: don't touch JAX_PLATFORMS, let JAX pick + + # Set host device count for NUTS multi-chain BEFORE JAX init + if args.nuts or args.svi_init_nuts: + import numpyro as _np_pre + _np_pre.set_host_device_count( + min(args.num_chains, os.cpu_count() or 4) + ) + + import jax + import numpyro + numpyro.enable_x64() + + # --- File paths --- + pools_cache = os.path.join(cache_dir, "pools.parquet") + snaps_cache = os.path.join(cache_dir, "pool_snapshots.parquet") + prices_cache = os.path.join(cache_dir, "token_prices") + panel_cache = os.path.join(cache_dir, "panel.parquet") + + # --- Fetch --- + if args.fetch: + print("Phase 1: Fetching data from Balancer API") + print("=" * 60) + + print("\n1. Enumerating pools...") + pools_df = enumerate_balancer_pools(min_tvl=args.min_tvl) + os.makedirs(cache_dir, exist_ok=True) + pools_df.to_parquet(pools_cache, index=False) + print(f" Saved {len(pools_df)} pools -> {pools_cache}") + + print("\n2. Fetching daily snapshots...") + snapshots_df = fetch_all_snapshots(pools_df, cache_path=snaps_cache) + + print("\n3. Fetching token prices...") + token_addr_by_chain = {} + for _, pool in pools_df.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + print("\n4. Assembling panel (with lagged TVL)...") + panel = assemble_panel(pools_df, snapshots_df, token_prices) + panel.to_parquet(panel_cache, index=False) + print(f" Saved panel -> {panel_cache}") + + print(f"\nFetch complete. Panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # --- Validate --- + if args.validate: + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + panel = pd.read_parquet(panel_cache) + validate_panel(panel) + + # --- Fit --- + if args.fit: + print("\nUnified Noise Volume Model") + print("=" * 60) + + # Load panel + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_cache) + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + # Filter to recent window for training + if args.train_days > 0: + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=args.train_days) + n_before = len(panel) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + print(f" Filtered to last {args.train_days} days " + f"(>= {cutoff}): {len(panel)} obs " + f"(dropped {n_before - len(panel)})") + + # Ensure lagged TVL exists (in case loaded from old cache) + if "log_tvl_lag1" not in panel.columns: + print(" Adding lagged TVL to cached panel...") + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + # Filter: need at least 10 days per pool + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid_pools)].copy() + print(f" After filtering (>= 10 days): {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # Select model variant + if args.model == "structural": + from .model import structural_noise_model + from .covariate_encoding import encode_covariates_structural + model_fn = structural_noise_model + data = encode_covariates_structural(panel) + print(f" Model: structural mixture (arb + MoE noise)") + elif args.model == "ibp_dp": + from .model import noise_model_ibp_dp + model_fn = noise_model_ibp_dp + data = encode_covariates(panel, include_tiers=False) + data["K_features"] = args.k_features + data["K_clusters"] = args.k_clusters + print(f" Model: IBP+DP hybrid " + f"(K_features={args.k_features}, " + f"K_clusters={args.k_clusters})") + elif args.model == "ibp": + from .model import noise_model_ibp + model_fn = noise_model_ibp + data = encode_covariates(panel, include_tiers=False) + data["K_features"] = args.k_features + print(f" Model: IBP latent features " + f"(K_features={args.k_features})") + elif args.model == "dp_sigma": + from .model import noise_model_dp_sigma + model_fn = noise_model_dp_sigma + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = args.k_clusters + print(f" Model: DP mixture on sigma_eps " + f"(K_clusters={args.k_clusters})") + else: + model_fn = None # default = noise_model + data = encode_covariates(panel) + + # Prior predictive + prior_samples = None + if args.prior_predictive: + print("\n Running prior predictive check...") + prior_samples = run_prior_predictive(data, model_fn=model_fn) + + # Inference + mcmc_obj = None + elbo_losses = None + inference_config = {"seed": args.seed} + + if args.svi_init_nuts: + inference_config["method"] = "svi_init_nuts" + inference_config["svi_steps"] = args.svi_steps + inference_config["svi_lr"] = args.svi_lr + inference_config["num_warmup"] = args.num_warmup + inference_config["num_samples"] = args.num_samples + inference_config["num_chains"] = args.num_chains + inference_config["target_accept"] = args.target_accept + inference_config["max_tree_depth"] = args.max_tree_depth + + mcmc_obj, elbo_losses = run_svi_then_nuts( + data, + svi_steps=args.svi_steps, + svi_lr=args.svi_lr, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + model_fn=model_fn, + ) + samples = mcmc_obj + convergence = check_convergence(mcmc_obj, method="nuts") + + elif args.nuts: + inference_config["method"] = "nuts" + inference_config["num_warmup"] = args.num_warmup + inference_config["num_samples"] = args.num_samples + inference_config["num_chains"] = args.num_chains + inference_config["target_accept"] = args.target_accept + inference_config["max_tree_depth"] = args.max_tree_depth + + mcmc_obj = run_nuts( + data, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + model_fn=model_fn, + ) + samples = mcmc_obj + convergence = check_convergence(mcmc_obj, method="nuts") + + else: + inference_config["method"] = "svi" + inference_config["svi_steps"] = args.svi_steps + inference_config["svi_lr"] = args.svi_lr + inference_config["num_samples"] = args.num_samples + + samples, elbo_losses = run_svi( + data, + num_steps=args.svi_steps, + lr=args.svi_lr, + seed=args.seed, + num_samples=args.num_samples, + model_fn=model_fn, + ) + convergence = check_convergence(elbo_losses, method="svi") + + if args.model == "structural": + from .postprocessing import extract_structural_params + pool_params = extract_structural_params(samples, data) + arb_freqs = [p["arb_frequency"] for p in pool_params] + print(f"\n Per-pool arb_frequency: " + f"mean={np.mean(arb_freqs):.1f}, " + f"range=[{np.min(arb_freqs)}, {np.max(arb_freqs)}]") + else: + pool_params = extract_noise_params(samples, data) + b_c_vals = [p["noise_params"]["b_c"] for p in pool_params] + b_0_vals = [p["noise_params"]["b_0"] for p in pool_params] + print(f"\n Per-pool b_c: mean={np.mean(b_c_vals):.3f}, " + f"std={np.std(b_c_vals):.3f}, " + f"range=[{np.min(b_c_vals):.3f}, {np.max(b_c_vals):.3f}]") + print(f" Per-pool b_0: mean={np.mean(b_0_vals):.3f}, " + f"std={np.std(b_0_vals):.3f}") + + if args.output: + generate_output_json( + pool_params, samples, data, convergence, + args.output, inference_config, + ) + + if args.plot: + print("\nGenerating diagnostic plots...") + plot_diagnostics( + samples, data, output_dir=args.output_dir, + elbo_losses=elbo_losses, mcmc=mcmc_obj, + prior_samples=prior_samples, + ) + + # Cache samples for --predict + _save_sample_cache(samples, data, cache_dir) + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + print("ERROR: --predict requires --chain and --tokens", + file=sys.stderr) + sys.exit(1) + + # Load cached samples + sample_cache = os.path.join(cache_dir, "unified_samples.npz") + data_cache = os.path.join(cache_dir, "unified_data.json") + + if not os.path.exists(sample_cache): + print(f"ERROR: Sample cache not found at {sample_cache}", + file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + cached = np.load(sample_cache) + sample_dict = {k: cached[k] for k in cached.files} + + with open(data_cache) as f: + data_meta = json.load(f) + + result = predict_new_pool( + sample_dict, data_meta, args.chain, args.tokens, args.fee + ) + print(json.dumps(result, indent=2)) diff --git a/quantammsim/noise_calibration/constants.py b/quantammsim/noise_calibration/constants.py new file mode 100644 index 00000000..05a217d4 --- /dev/null +++ b/quantammsim/noise_calibration/constants.py @@ -0,0 +1,60 @@ +"""Constants for noise calibration.""" + +import os + +K_COEFF = 4 +COEFF_NAMES = ["intercept", "b_tvl", "b_sigma", "b_weekend"] + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAINS = [ + "MAINNET", "POLYGON", "ARBITRUM", "GNOSIS", "BASE", "SONIC", "OPTIMISM", + "AVALANCHE", +] + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "local_data", "noise_calibration", +) + +# Tier 0: blue-chip — top by volume, wrapped native, major stables +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", +} + +K_CLUSTERS_DEFAULT = 6 +K_FEATURES_DEFAULT = 6 + +# Structural model: observation-level covariates (expanded from K_COEFF=4) +K_OBS_COEFF = 8 +OBS_COEFF_NAMES = [ + "intercept", "b_tvl", "b_sigma", + "b_tvl_sigma", "b_tvl_fee", "b_sigma_fee", + "b_dow_sin", "b_dow_cos", +] + +# Gas costs per arb transaction (USD) by chain +GAS_COSTS = { + "MAINNET": None, # time-varying, loaded from CSV + "POLYGON": 0.005, + "ARBITRUM": 0.005, + "BASE": 0.005, + "GNOSIS": 0.01, + "OPTIMISM": 0.005, + "SONIC": 0.005, + "AVALANCHE": 0.005, + "MODE": 0.005, + "FRAXTAL": 0.005, +} + +# Tier 1: mid-cap DeFi blue-chips (approx CoinGecko rank < 200) +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} diff --git a/quantammsim/noise_calibration/covariate_encoding.py b/quantammsim/noise_calibration/covariate_encoding.py new file mode 100644 index 00000000..a298a6b4 --- /dev/null +++ b/quantammsim/noise_calibration/covariate_encoding.py @@ -0,0 +1,228 @@ +"""Covariate encoding for the hierarchical noise model.""" + +import numpy as np +import pandas as pd + +from .constants import K_COEFF, K_OBS_COEFF, GAS_COSTS + + +def encode_covariates(panel: pd.DataFrame, include_tiers: bool = True) -> dict: + """Build NumPyro-ready arrays from the panel DataFrame. + + Returns dict with arrays for the model plus metadata for output/prediction. + Key difference from hierarchical script: x_obs uses log_tvl_lag1 not log_tvl. + """ + pool_meta = panel.drop_duplicates("pool_id").reset_index(drop=True) + pool_ids = pool_meta["pool_id"].values + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + N_pools = len(pool_ids) + + pool_idx = panel["pool_id"].map(pool_id_to_idx).values + + # --- Build X_pool (pool-level covariates, data-driven) --- + chains = sorted(panel["chain"].unique()) + ref_chain = chains[0] + chain_cols = [] + chain_names = [] + for c in chains[1:]: + chain_cols.append((pool_meta["chain"] == c).astype(float).values) + chain_names.append(f"chain_{c}") + + tier_a_vals = sorted(pool_meta["tier_A"].astype(str).unique()) + ref_tier_a = tier_a_vals[0] + tier_a_cols = [] + tier_a_names = [] + if include_tiers: + for t in tier_a_vals[1:]: + tier_a_cols.append( + (pool_meta["tier_A"].astype(str) == t).astype(float).values + ) + tier_a_names.append(f"tier_A_{t}") + + tier_b_vals = sorted(pool_meta["tier_B"].astype(str).unique()) + ref_tier_b = tier_b_vals[0] + tier_b_cols = [] + tier_b_names = [] + if include_tiers: + for t in tier_b_vals[1:]: + tier_b_cols.append( + (pool_meta["tier_B"].astype(str) == t).astype(float).values + ) + tier_b_names.append(f"tier_B_{t}") + + columns = [np.ones((N_pools, 1))] + col_names = ["intercept"] + + for arr, name in zip(chain_cols, chain_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_a_cols, tier_a_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_b_cols, tier_b_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + columns.append(pool_meta["log_fee"].values.reshape(-1, 1)) + col_names.append("log_fee") + + X_pool = np.hstack(columns) + K_cov = X_pool.shape[1] + + # --- Observation-level arrays (uses LAGGED TVL) --- + x_obs = np.column_stack([ + np.ones(len(panel)), + panel["log_tvl_lag1"].values, + panel["volatility"].values, + panel["weekend"].values, + ]).astype(np.float64) + + y_obs = panel["log_volume"].values.astype(np.float64) + + # --- Per-pool tier_A index for per-tier sigma_eps --- + tier_A_per_pool = pool_meta["tier_A"].values.astype(np.int32) + + print(f" Encoded: N_obs={len(y_obs)}, N_pools={N_pools}, " + f"K_coeff={K_COEFF}, K_cov={K_cov}") + print(f" Covariates: {col_names}") + print(f" Tier distribution: " + f"T0={np.sum(tier_A_per_pool == 0)}, " + f"T1={np.sum(tier_A_per_pool == 1)}, " + f"T2={np.sum(tier_A_per_pool == 2)}") + + return { + "pool_idx": pool_idx.astype(np.int32), + "X_pool": X_pool.astype(np.float64), + "x_obs": x_obs, + "y_obs": y_obs, + "pool_ids": list(pool_ids), + "pool_meta": pool_meta, + "covariate_names": col_names, + "tier_A_per_pool": tier_A_per_pool, + "N_pools": N_pools, + "K_cov": K_cov, + "ref_chain": ref_chain, + "ref_tier_a": ref_tier_a, + "ref_tier_b": ref_tier_b, + "chains": chains, + } + + +def _tier_pair_idx(a: int, b: int) -> int: + """Encode (tier_A, tier_B) pair as a single index. + + Upper triangle of 3x3 grid: + (0,0)->0, (0,1)->1, (0,2)->2, (1,1)->3, (1,2)->4, (2,2)->5. + """ + return a * (5 - a) // 2 + b - a + + +def encode_covariates_structural( + panel: pd.DataFrame, + gas: np.ndarray = None, +) -> dict: + """Build NumPyro-ready arrays for the structural mixture model. + + Extends encode_covariates with: + - x_obs: 8 columns (intercept, tvl, log_sigma, interactions, DOW harmonics) + - Additional arrays: sigma_daily, fee, gas, chain_idx, tier_idx, lag_log_tvl + - n_chains, n_tiers computed from panel + + Parameters + ---------- + panel : pd.DataFrame + Output of assemble_panel(), must have log_sigma, dow_sin, dow_cos, + tvl_x_sigma, tvl_x_fee, sigma_x_fee columns. + gas : np.ndarray, optional + Per-observation gas costs in USD. If None, uses default (0.01 for all). + """ + # Ensure structural columns exist (compute from base columns if missing) + if "log_sigma" not in panel.columns: + panel = panel.copy() + panel["log_sigma"] = np.log(np.maximum(panel["volatility"].values, 1e-6)) + dow = panel["date"].apply( + lambda d: d.weekday() if hasattr(d, "weekday") + else pd.Timestamp(d).weekday() + ) + panel["dow_sin"] = np.sin(2.0 * np.pi * dow / 7.0) + panel["dow_cos"] = np.cos(2.0 * np.pi * dow / 7.0) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + # Reuse X_pool construction from encode_covariates (with tiers for gating) + base = encode_covariates(panel, include_tiers=True) + + # --- Observation-level x_obs: 8 columns --- + x_obs = np.column_stack([ + np.ones(len(panel)), # intercept + panel["log_tvl_lag1"].values, # lagged TVL + panel["log_sigma"].values, # log(volatility) + panel["tvl_x_sigma"].values, # tvl × sigma interaction + panel["tvl_x_fee"].values, # tvl × fee interaction + panel["sigma_x_fee"].values, # sigma × fee interaction + panel["dow_sin"].values, # DOW harmonic sin + panel["dow_cos"].values, # DOW harmonic cos + ]).astype(np.float64) + + # --- Additional arrays for the structural model --- + sigma_daily = (panel["volatility"] / np.sqrt(365.0)).values.astype(np.float64) + fee_per_obs = np.exp(panel["log_fee"].values).astype(np.float64) + lag_log_tvl = panel["log_tvl_lag1"].values.astype(np.float64) + + # Gas: per-observation + if gas is not None: + gas_arr = np.asarray(gas, dtype=np.float64) + else: + gas_arr = np.full(len(panel), 0.01, dtype=np.float64) + + # Chain index: integer per pool + pool_meta = base["pool_meta"] + chains = base["chains"] + chain_to_idx = {c: i for i, c in enumerate(chains)} + chain_idx_per_pool = np.array( + [chain_to_idx[c] for c in pool_meta["chain"]], dtype=np.int32, + ) + + # Tier pair index: per pool + tier_idx_per_pool = np.array( + [_tier_pair_idx(int(row["tier_A"]), int(row["tier_B"])) + for _, row in pool_meta.iterrows()], + dtype=np.int32, + ) + + # Count unique tier pairs and chains + n_chains = len(chains) + tier_pairs = set() + for _, row in pool_meta.iterrows(): + tier_pairs.add((int(row["tier_A"]), int(row["tier_B"]))) + n_tiers = 6 # fixed: upper triangle of 3x3 + + print(f" Structural encoding: N_obs={len(panel)}, " + f"N_pools={base['N_pools']}, n_chains={n_chains}, n_tiers={n_tiers}") + + return { + # Base arrays (same as encode_covariates) + "pool_idx": base["pool_idx"], + "X_pool": base["X_pool"], + "x_obs": x_obs, + "y_obs": base["y_obs"], + "pool_ids": base["pool_ids"], + "pool_meta": pool_meta, + "covariate_names": base["covariate_names"], + "tier_A_per_pool": base["tier_A_per_pool"], + "N_pools": base["N_pools"], + "K_cov": base["K_cov"], + "ref_chain": base["ref_chain"], + "ref_tier_a": base["ref_tier_a"], + "ref_tier_b": base["ref_tier_b"], + "chains": chains, + # Structural model extras + "sigma_daily": sigma_daily, + "fee": fee_per_obs, + "gas": gas_arr, + "chain_idx": chain_idx_per_pool, + "tier_idx": tier_idx_per_pool, + "lag_log_tvl": lag_log_tvl, + "n_chains": n_chains, + "n_tiers": n_tiers, + } diff --git a/quantammsim/noise_calibration/data_pipeline.py b/quantammsim/noise_calibration/data_pipeline.py new file mode 100644 index 00000000..b824cdc4 --- /dev/null +++ b/quantammsim/noise_calibration/data_pipeline.py @@ -0,0 +1,516 @@ +"""Data pipeline: fetch pools, snapshots, prices, and assemble panel.""" + +import json +import os +import time +import urllib.request +from datetime import datetime + +import numpy as np +import pandas as pd + +from .constants import BALANCER_API_URL, BALANCER_API_CHAINS +from .token_classification import classify_token_tier + + +def _graphql_request(query: dict, base_url: str = BALANCER_API_URL, + timeout: int = 30) -> dict: + """Send a GraphQL request to the Balancer V3 API.""" + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8")) + + +def enumerate_balancer_pools( + chains: list = None, + pool_types: list = None, + min_tvl: float = 10000.0, +) -> pd.DataFrame: + """Enumerate all WEIGHTED + RECLAMM pools across chains from Balancer API.""" + if chains is None: + chains = BALANCER_API_CHAINS + if pool_types is None: + pool_types = ["WEIGHTED", "RECLAMM"] + + all_pools = [] + for chain in chains: + print(f" Querying {chain}...", end=" ", flush=True) + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { + chainIn: [$chain] + poolTypeIn: $types + minTvl: $minTvl + } + ) { + id + chain + type + createTime + protocolVersion + poolTokens { + symbol + weight + address + } + dynamicData { + totalLiquidity + swapFee + } + } + } + """, + "variables": { + "chain": chain, + "types": pool_types, + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f"FAILED ({e})") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + weights = [t.get("weight") for t in p.get("poolTokens", [])] + token_addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "protocol_version": p.get("protocolVersion", 0), + "tokens": tokens, + "token_addresses": token_addresses, + "weights": weights, + "swap_fee": fee, + "create_time": p.get("createTime", 0), + "current_tvl": tvl, + }) + + print(f"{len(pools)} pools") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools across {len(chains)} chains") + return df + + +def fetch_pool_snapshots(pool_id: str, chain: str, + base_url: str = BALANCER_API_URL) -> pd.DataFrame: + """Fetch ALL_TIME daily snapshots for a single pool.""" + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + totalShares + } + } + """, + "variables": { + "poolId": pool_id, + "chain": chain, + "range": "ALL_TIME", + }, + } + + body = _graphql_request(query) + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + + if not snapshots: + return pd.DataFrame(columns=["timestamp", "volume_usd", + "total_liquidity_usd", "total_shares"]) + + records = [] + for snap in snapshots: + records.append({ + "timestamp": int(snap["timestamp"]), + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + "total_shares": float(snap.get("totalShares", 0)), + }) + + df = pd.DataFrame(records) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df + + +def fetch_all_snapshots(pools_df: pd.DataFrame, + cache_path: str = None) -> pd.DataFrame: + """Fetch daily snapshots for all pools, with caching.""" + cached = pd.DataFrame() + cached_pool_ids = set() + if cache_path and os.path.exists(cache_path): + cached = pd.read_parquet(cache_path) + cached_pool_ids = set(cached["pool_id"].unique()) + print(f" Cache has {len(cached_pool_ids)} pools, " + f"{len(cached)} pool-days") + + if len(pools_df) == 0: + print(" No pools to fetch.") + return cached if len(cached) > 0 else pd.DataFrame( + columns=["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd", "total_shares"] + ) + to_fetch = pools_df[~pools_df["pool_id"].isin(cached_pool_ids)] + print(f" Need to fetch {len(to_fetch)} new pools") + + new_records = [] + for i, (_, pool) in enumerate(to_fetch.iterrows()): + if (i + 1) % 10 == 0 or i == 0: + print(f" Fetching {i+1}/{len(to_fetch)}: {pool['pool_id'][:10]}... " + f"({pool['chain']})", flush=True) + try: + snap_df = fetch_pool_snapshots(pool["pool_id"], pool["chain"]) + if len(snap_df) > 0: + snap_df["pool_id"] = pool["pool_id"] + snap_df["chain"] = pool["chain"] + cols = ["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + if "total_shares" in snap_df.columns: + cols.append("total_shares") + new_records.append(snap_df[cols]) + except Exception as e: + print(f" FAILED {pool['pool_id'][:10]}: {e}") + time.sleep(0.5) + + if new_records: + new_df = pd.concat(new_records, ignore_index=True) + combined = pd.concat([cached, new_df], ignore_index=True) + else: + combined = cached + + if cache_path and len(combined) > 0: + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + combined.to_parquet(cache_path, index=False) + print(f" Saved cache: {len(combined)} pool-days -> {cache_path}") + + return combined + + +def fetch_token_prices(token_addresses_by_chain: dict, + cache_dir: str = None) -> dict: + """Fetch hourly token prices from Balancer API.""" + if cache_dir: + os.makedirs(cache_dir, exist_ok=True) + + prices = {} + + for chain, tokens in token_addresses_by_chain.items(): + uncached = {} + for symbol, address in tokens.items(): + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") if cache_dir else None + + if cp and os.path.exists(cp): + prices[(chain, symbol)] = pd.read_parquet(cp) + else: + uncached[symbol] = address + + if not uncached: + continue + + addr_to_symbol = {addr: sym for sym, addr in uncached.items()} + addresses = list(uncached.values()) + + print(f" Fetching {len(addresses)} prices on {chain}...", flush=True) + + batch_size = 20 + for batch_start in range(0, len(addresses), batch_size): + batch_addrs = addresses[batch_start:batch_start + batch_size] + query = { + "query": """ + query GetPrices($chain: GqlChain!, $addresses: [String!]!, + $range: GqlTokenChartDataRange!) { + tokenGetHistoricalPrices( + addresses: $addresses, chain: $chain, range: $range + ) { + address + prices { + timestamp + price + } + } + } + """, + "variables": { + "chain": chain, + "addresses": batch_addrs, + "range": "ONE_YEAR", + }, + } + + try: + body = _graphql_request(query, timeout=60) + results = body.get("data", {}).get( + "tokenGetHistoricalPrices", []) + for result in results: + addr = result.get("address", "") + price_list = result.get("prices", []) + symbol = addr_to_symbol.get(addr) + if symbol and price_list: + pdf = pd.DataFrame(price_list) + pdf["timestamp"] = pdf["timestamp"].astype(int) + pdf["price"] = pdf["price"].astype(float) + prices[(chain, symbol)] = pdf + if cache_dir: + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") + pdf.to_parquet(cp, index=False) + except Exception as e: + print(f" FAILED batch on {chain}: {e}") + + time.sleep(0.5) + + print(f" Got prices for {len(prices)} token-chain pairs") + return prices + + +def compute_pair_volatility( + snapshots_df: pd.DataFrame, + pool_row: pd.Series, + token_prices: dict, +) -> pd.Series: + """Compute daily annualised volatility for a pool's pair ratio.""" + tokens = pool_row["tokens"] + chain = pool_row["chain"] + + if len(tokens) < 2: + return pd.Series(dtype=float) + + def _get_price_df(symbol): + key = (chain, symbol) + if key in token_prices: + return token_prices[key] + for k, v in token_prices.items(): + if k[1] == symbol: + return v + return None + + p0_df = _get_price_df(tokens[0]) + p1_df = _get_price_df(tokens[1]) + + stables = {"USDC", "USDT", "DAI", "LUSD", "GHO", "crvUSD", "sDAI", + "WXDAI", "xDAI", "USDC.e", "USDbC"} + + if tokens[0] in stables and tokens[1] in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.01, index=dates) + + if p0_df is None and tokens[0] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + if p1_df is None and tokens[1] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + + if tokens[0] in stables: + if p1_df is None or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p1_df.copy() + ratio_df["ratio"] = 1.0 / ratio_df["price"] + elif tokens[1] in stables: + if p0_df is None or len(p0_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p0_df.copy() + ratio_df["ratio"] = ratio_df["price"] + else: + if p0_df is None or p1_df is None or len(p0_df) == 0 or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + merged = pd.merge_asof( + p0_df.sort_values("timestamp"), + p1_df.sort_values("timestamp"), + on="timestamp", + suffixes=("_0", "_1"), + tolerance=7200, + ).dropna() + if len(merged) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = merged.copy() + ratio_df["ratio"] = merged["price_0"] / merged["price_1"] + + ratio_df["datetime"] = pd.to_datetime(ratio_df["timestamp"], unit="s") + ratio_df["date"] = ratio_df["datetime"].dt.date + ratio_df = ratio_df.sort_values("timestamp") + + ratio_df["log_return"] = np.log( + ratio_df["ratio"] / ratio_df["ratio"].shift(1) + ) + ratio_df = ratio_df.dropna(subset=["log_return"]) + + daily_vol = ratio_df.groupby("date")["log_return"].std() + daily_vol_ann = daily_vol * np.sqrt(24 * 365) + + return daily_vol_ann + + +def assemble_panel( + pools_df: pd.DataFrame, + snapshots_df: pd.DataFrame, + token_prices: dict, +) -> pd.DataFrame: + """Assemble the full panel DataFrame with lagged TVL. + + Adds log_tvl_lag1 = per-pool shift(1) of log_tvl to break + the TVL-volume simultaneity bias. Drops the first observation + per pool (~1 obs per pool). + """ + records = [] + pool_ids = snapshots_df["pool_id"].unique() + n_pools = len(pool_ids) + + # Track volatility fallback rate + n_obs_total = 0 + n_obs_fallback = 0 + pools_all_fallback = [] # pools where every obs hit fallback + + for i, pool_id in enumerate(pool_ids): + if (i + 1) % 20 == 0 or i == 0: + print(f" Assembling {i+1}/{n_pools}...", flush=True) + + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pool_id] + pool_meta = pools_df[pools_df["pool_id"] == pool_id] + if len(pool_meta) == 0: + continue + pool_row = pool_meta.iloc[0] + + tokens = pool_row["tokens"] + if len(tokens) < 2: + continue + + chain = pool_row["chain"] + swap_fee = pool_row["swap_fee"] + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + vol_series = compute_pair_volatility(pool_snaps, pool_row, token_prices) + + pool_obs = 0 + pool_fallback = 0 + has_shares = "total_shares" in pool_snaps.columns + for _, snap in pool_snaps.iterrows(): + date = snap["date"] + volume = snap["volume_usd"] + tvl = snap["total_liquidity_usd"] + shares = float(snap["total_shares"]) if has_shares else 0.0 + + if tvl <= 0 or volume <= 0: + continue + + used_fallback = False + if isinstance(vol_series, pd.Series) and date in vol_series.index: + vol = vol_series[date] + else: + vol = 0.5 + used_fallback = True + + if not np.isfinite(vol) or vol <= 0: + vol = 0.5 + used_fallback = True + + n_obs_total += 1 + if used_fallback: + n_obs_fallback += 1 + pool_fallback += 1 + pool_obs += 1 + + if isinstance(date, datetime): + is_weekend = date.weekday() >= 5 + else: + is_weekend = pd.Timestamp(date).weekday() >= 5 + + # DOW harmonics (deterministic from date) + if isinstance(date, datetime): + dow = date.weekday() + else: + dow = pd.Timestamp(date).weekday() + dow_sin = np.sin(2.0 * np.pi * dow / 7.0) + dow_cos = np.cos(2.0 * np.pi * dow / 7.0) + + record = { + "pool_id": pool_id, + "chain": chain, + "date": date, + "log_volume": np.log(volume), + "log_tvl": np.log(tvl), + "volatility": vol, + "log_sigma": np.log(max(vol, 1e-6)), + "weekend": 1.0 if is_weekend else 0.0, + "log_fee": np.log(max(swap_fee, 1e-6)), + "dow_sin": dow_sin, + "dow_cos": dow_cos, + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": ",".join(tokens[:2]), + "swap_fee": swap_fee, + } + if shares > 0: + record["total_shares"] = shares + records.append(record) + + if pool_obs > 0 and pool_fallback == pool_obs: + pools_all_fallback.append( + (pool_id[:16], chain, ",".join(tokens[:2])) + ) + + panel = pd.DataFrame(records) + + # Add lagged TVL to break simultaneity bias + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + n_before = len(panel) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + n_dropped = n_before - len(panel) + + # Interaction terms (use lagged TVL to break simultaneity) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + print(f"\n Panel: {len(panel)} observations, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + print(f" Dropped {n_dropped} first-day obs for lagged TVL") + + # Volatility coverage report + if n_obs_total > 0: + pct = 100 * n_obs_fallback / n_obs_total + print(f"\n Volatility coverage:") + print(f" {n_obs_fallback}/{n_obs_total} obs used fallback " + f"vol=0.5 ({pct:.1f}%)") + print(f" {len(pools_all_fallback)} pools had 100% fallback") + if pools_all_fallback: + for pid, ch, toks in pools_all_fallback[:10]: + print(f" {pid}... ({ch}) {toks}") + if len(pools_all_fallback) > 10: + print(f" ... and {len(pools_all_fallback) - 10} more") + + return panel diff --git a/quantammsim/noise_calibration/data_validation.py b/quantammsim/noise_calibration/data_validation.py new file mode 100644 index 00000000..8cb44e72 --- /dev/null +++ b/quantammsim/noise_calibration/data_validation.py @@ -0,0 +1,41 @@ +"""Data validation for noise calibration panels.""" + +import numpy as np +import pandas as pd + + +def validate_panel(panel: pd.DataFrame) -> pd.DataFrame: + """Run data validation checks. Prints warnings but does NOT drop rows.""" + print("\n Data validation:") + + # Pools with constant volume + vol_std = panel.groupby("pool_id")["log_volume"].std() + constant_vol = vol_std[vol_std < 0.01] + if len(constant_vol) > 0: + print(f" WARNING: {len(constant_vol)} pools have near-constant " + f"log(volume) (std < 0.01)") + for pid in constant_vol.index[:5]: + print(f" {pid[:16]}... std={constant_vol[pid]:.4f}") + if len(constant_vol) > 5: + print(f" ... and {len(constant_vol) - 5} more") + + # TVL jumps > 10x between consecutive days + panel_sorted = panel.sort_values(["pool_id", "date"]) + tvl_ratio = panel_sorted.groupby("pool_id")["log_tvl"].diff().abs() + big_jumps = tvl_ratio[tvl_ratio > np.log(10)] + if len(big_jumps) > 0: + affected_pools = panel_sorted.loc[big_jumps.index, "pool_id"].nunique() + print(f" WARNING: {len(big_jumps)} TVL jumps > 10x across " + f"{affected_pools} pools") + + # Days where volume > TVL + high_vol = panel[panel["log_volume"] > panel["log_tvl"]] + if len(high_vol) > 0: + affected_pools = high_vol["pool_id"].nunique() + print(f" WARNING: {len(high_vol)} days where volume > TVL across " + f"{affected_pools} pools (potential wash trading)") + + if len(constant_vol) == 0 and len(big_jumps) == 0 and len(high_vol) == 0: + print(" All checks passed.") + + return panel diff --git a/quantammsim/noise_calibration/formula_arb.py b/quantammsim/noise_calibration/formula_arb.py new file mode 100644 index 00000000..80fe3352 --- /dev/null +++ b/quantammsim/noise_calibration/formula_arb.py @@ -0,0 +1,36 @@ +"""JAX-differentiable LVR formula for arb volume. + +Based on arXiv:2305.14604v2 §6 with gas costs and discrete-time correction. +Reference: scripts/plot_formula_arb_vs_real.py:formula_arb_volume_daily (line 58). +""" + +import jax.numpy as jnp + + +def formula_arb_volume_daily_jax(sigma_daily, tvl, fee, gas_usd, cadence_minutes): + """Analytical arb volume per day for a CPMM with gas costs. + + All inputs are JAX scalars or arrays (must be broadcastable). + + Parameters + ---------- + sigma_daily : float + Daily volatility of the log price ratio (NOT annualised). + tvl : float + Pool TVL in USD. + fee : float + Swap fee as fraction (e.g. 0.003 for 30bp). + gas_usd : float + All-in gas cost per arb tx in USD. + cadence_minutes : float + Effective arb cadence in minutes (= simulator's arb_frequency). + """ + block_time_s = cadence_minutes * 60.0 + delta = 2.0 * jnp.sqrt(2.0 * jnp.maximum(gas_usd, 0.0) / jnp.maximum(tvl, 1e-6)) + bLVR = sigma_daily**2 * tvl / 8.0 + sqrt_term = sigma_daily * jnp.sqrt(block_time_s / (2.0 * 86400.0)) + correction = jnp.maximum( + 1.0 - delta / (2.0 * fee) - sqrt_term / (fee + delta / 2.0), + 0.0, + ) + return bLVR * correction / fee diff --git a/quantammsim/noise_calibration/inference.py b/quantammsim/noise_calibration/inference.py new file mode 100644 index 00000000..e315a5bd --- /dev/null +++ b/quantammsim/noise_calibration/inference.py @@ -0,0 +1,270 @@ +"""Inference runners: SVI, NUTS, SVI-initialized NUTS.""" + +import numpy as np + +from .constants import K_COEFF +from .model import noise_model + + +def _get_theta_samples(sample_dict: dict, X_pool: np.ndarray, + data: dict = None) -> np.ndarray: + """Get theta samples, reconstructing from non-centered params if needed. + + MCMC.get_samples() includes the deterministic "theta" site. + SVI's Predictive(guide, ...) does NOT — the guide only samples latent + variables. In that case, reconstruct theta manually: + mu = X_pool @ B^T + L_Sigma = diag(sigma_theta) @ L_Omega + theta = mu + eta @ L_Sigma^T + + For marginalized IBP (W present, z_logit absent): compute MAP feature + assignments from data, then theta = X_pool @ B.T + Z_MAP @ W. + Requires data dict for pool_idx, x_obs, y_obs. + + For legacy STE IBP (z_logit present): theta = X_pool @ B.T + Z_hard @ W. + """ + if "theta" in sample_dict: + return np.array(sample_dict["theta"]) + + # Marginalized IBP path: compute MAP assignments from data + if "W" in sample_dict and "z_logit" not in sample_dict: + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + W = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + + # MAP assignments: (N_pools, K_features) binary + if "v" in sample_dict: + # Hybrid IBP+DP: joint MAP over (features, clusters) + from .postprocessing import assign_ibp_dp_joint + Z_map, _ = assign_ibp_dp_joint(sample_dict, data) + else: + from .postprocessing import assign_ibp_features + Z_map = assign_ibp_features(sample_dict, data) + + mu = np.einsum("pd,sjd->spj", X_pool, B) + # Z_map doesn't vary across samples — broadcast + feature_effect = np.einsum("pk,skj->spj", Z_map.astype(float), W) + return mu + feature_effect + + # Legacy STE IBP path: theta = X_pool @ B.T + Z_hard @ W + if "z_logit" in sample_dict: + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + W = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + z_logit = np.array(sample_dict["z_logit"]) # (S, N_pools, K_features) + Z_hard = (z_logit > 0).astype(float) + + mu = np.einsum("pd,sjd->spj", X_pool, B) # (S, N_pools, K_coeff) + feature_effect = np.einsum("spk,skj->spj", Z_hard, W) + return mu + feature_effect + + B = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + sigma_theta = np.array(sample_dict["sigma_theta"]) # (S, K_coeff) + L_Omega = np.array(sample_dict["L_Omega"]) # (S, K_coeff, K_coeff) + eta = np.array(sample_dict["eta"]) # (S, N_pools, K_coeff) + + # mu[s, p, j] = sum_d X_pool[p, d] * B[s, j, d] -> (S, N_pools, K_coeff) + mu = np.einsum("pd,sjd->spj", X_pool, B) + + # L_Sigma = diag(sigma_theta) @ L_Omega -> (S, K_coeff, K_coeff) + L_Sigma = sigma_theta[:, :, None] * L_Omega + + # offset = eta @ L_Sigma^T -> (S, N_pools, K_coeff) + offset = np.einsum("spi,sji->spj", eta, L_Sigma) + + return mu + offset + + +def _build_model_kwargs(data: dict, model_fn=None) -> dict: + """Convert data dict to jnp arrays for the model. + + Uses inspect.signature on model_fn to decide which kwargs to include: + - tier_A_per_pool: only if model_fn accepts it + - K_clusters: only if model_fn accepts it and data has it + """ + import inspect + import jax.numpy as jnp + + if model_fn is None: + model_fn = noise_model + + params = set(inspect.signature(model_fn).parameters.keys()) + + kwargs = dict( + pool_idx=jnp.array(data["pool_idx"]), + X_pool=jnp.array(data["X_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K_coeff=K_COEFF, + K_cov=data["K_cov"], + ) + + if "tier_A_per_pool" in params: + kwargs["tier_A_per_pool"] = jnp.array(data["tier_A_per_pool"]) + + if "K_clusters" in params and "K_clusters" in data: + kwargs["K_clusters"] = data["K_clusters"] + + if "K_features" in params and "K_features" in data: + kwargs["K_features"] = data["K_features"] + + # Structural model parameters + if "sigma_daily" in params and "sigma_daily" in data: + kwargs["sigma_daily"] = jnp.array(data["sigma_daily"]) + if "lag_log_tvl" in params and "lag_log_tvl" in data: + kwargs["lag_log_tvl"] = jnp.array(data["lag_log_tvl"]) + if "fee" in params and "fee" in data: + kwargs["fee"] = jnp.array(data["fee"]) + if "gas" in params and "gas" in data: + kwargs["gas"] = jnp.array(data["gas"]) + if "chain_idx" in params and "chain_idx" in data: + kwargs["chain_idx"] = jnp.array(data["chain_idx"]) + if "tier_idx" in params and "tier_idx" in data: + kwargs["tier_idx"] = jnp.array(data["tier_idx"]) + if "n_chains" in params and "n_chains" in data: + kwargs["n_chains"] = data["n_chains"] + if "n_tiers" in params and "n_tiers" in data: + kwargs["n_tiers"] = data["n_tiers"] + if "K_archetypes" in params and "K_archetypes" in data: + kwargs["K_archetypes"] = data["K_archetypes"] + + return kwargs + + +def run_svi(data, num_steps=20000, lr=1e-3, seed=0, + num_samples=1000, model_fn=None) -> tuple: + """Run SVI with AutoNormal guide. + + Returns (samples_dict, elbo_losses). + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import SVI, Trace_ELBO, Predictive + from numpyro.infer.autoguide import AutoNormal + + if model_fn is None: + model_fn = noise_model + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + + print(f"\n Running SVI: {num_steps} steps, lr={lr}") + guide = AutoNormal(model_fn) + optimizer = numpyro.optim.Adam(lr) + svi = SVI(model_fn, guide, optimizer, loss=Trace_ELBO()) + + rng_key = jax.random.PRNGKey(seed) + svi_result = svi.run(rng_key, num_steps, **model_kwargs) + + elbo_losses = np.array(svi_result.losses) + print(f" SVI complete. Final ELBO: {elbo_losses[-1]:.2f}") + print(f" ELBO last 100 std: {np.std(elbo_losses[-100:]):.2f}") + + # Draw posterior samples + predictive = Predictive( + guide, params=svi_result.params, num_samples=num_samples, + ) + samples = predictive(jax.random.PRNGKey(seed + 1), **model_kwargs) + samples = {k: np.array(v) for k, v in samples.items()} + + print(f" Drew {num_samples} posterior samples.") + return samples, elbo_losses + + +def run_nuts(data, num_warmup=1000, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42, + init_values=None, model_fn=None): + """Run NUTS MCMC. + + Uses init_to_value if init_values provided (for SVI-initialized NUTS). + Returns the MCMC object. + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import MCMC, NUTS, init_to_value + + if model_fn is None: + model_fn = noise_model + + # Note: set_host_device_count must be called before JAX init. + # We handle this in main(). Here we just verify device count. + n_devices = len(jax.devices("cpu")) + if n_devices < num_chains: + print(f" WARNING: Only {n_devices} CPU devices available for " + f"{num_chains} chains. Chains will run sequentially.") + + init_strategy = None + if init_values is not None: + init_strategy = init_to_value( + values={k: jnp.array(v) for k, v in init_values.items()} + ) + + kernel = NUTS( + model_fn, + target_accept_prob=target_accept, + max_tree_depth=max_tree_depth, + init_strategy=init_strategy, + ) + mcmc = MCMC( + kernel, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + progress_bar=True, + ) + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + rng_key = jax.random.PRNGKey(seed) + + print(f"\n Running NUTS: {num_chains} chains x " + f"({num_warmup} warmup + {num_samples} samples)") + print(f" target_accept={target_accept}, max_tree_depth={max_tree_depth}") + if init_values is not None: + print(" Using SVI-initialized starting values.") + + mcmc.run(rng_key, **model_kwargs) + mcmc.print_summary(exclude_deterministic=True) + return mcmc + + +def run_svi_then_nuts(data, svi_steps=5000, svi_lr=1e-3, + num_warmup=500, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42, + model_fn=None): + """Run SVI first, then use posterior means as NUTS init. + + Returns (MCMC, elbo_losses). + """ + if model_fn is None: + model_fn = noise_model + + # Phase 1: SVI + print(" Phase 1: SVI warm-start") + samples, elbo_losses = run_svi( + data, num_steps=svi_steps, lr=svi_lr, seed=seed, num_samples=100, + model_fn=model_fn, + ) + + # Extract posterior means for init + init_values = {} + skip_keys = {"y", "theta", "w"} + for k, v in samples.items(): + if k in skip_keys: + continue + init_values[k] = np.mean(v, axis=0) + + # Phase 2: NUTS from SVI init + print("\n Phase 2: NUTS from SVI-initialized values") + mcmc = run_nuts( + data, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + target_accept=target_accept, + max_tree_depth=max_tree_depth, + seed=seed, + init_values=init_values, + model_fn=model_fn, + ) + + return mcmc, elbo_losses diff --git a/quantammsim/noise_calibration/model.py b/quantammsim/noise_calibration/model.py new file mode 100644 index 00000000..4a1ba19b --- /dev/null +++ b/quantammsim/noise_calibration/model.py @@ -0,0 +1,465 @@ +"""NumPyro noise volume models.""" + +import jax +import jax.numpy as jnp + +from .formula_arb import formula_arb_volume_daily_jax + + +def _pad_with_ref(alpha): + """Prepend a zero for the reference category.""" + return jnp.concatenate([jnp.zeros(1), alpha]) + + +def stick_breaking_weights(v): + """Convert Beta stick-breaking fractions to K-simplex weights. + + v: array of shape (K-1,) with values in (0, 1). + Returns weights of shape (K,) summing to 1. + + w_1 = v_1 + w_k = v_k * prod_{j> jnp.arange(K_features)[None, :]) & 1 + ).astype(jnp.float32) # (n_configs, K_features) + + # Log-prior for each config from IBP stick-breaking + log_pi = jnp.log(pi + 1e-30) + log_1mpi = jnp.log(1.0 - pi + 1e-30) + log_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Per-config feature effect + feature_effects = configs @ W # (n_configs, K_coeff) + + # Per-observation means + mu_pop_obs = jnp.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config + log_lik = dist.StudentT(df, mu_obs, sigma_eps).log_prob( + y_obs[:, None] + ) # (N_obs, n_configs) + + # Sum log-likelihoods within each pool + pool_log_liks = jnp.zeros((N_pools, n_configs)) + pool_log_liks = pool_log_liks.at[pool_idx].add(log_lik) + + # Marginal: logsumexp over configs per pool + log_marginal = logsumexp( + log_prior[None, :] + pool_log_liks, axis=1 + ) # (N_pools,) + numpyro.factor("log_lik", log_marginal.sum()) + else: + # === Prior predictive: sample explicit assignments === + with numpyro.plate("pools", N_pools): + z_features = numpyro.sample( + "z_features", + dist.Bernoulli(probs=pi).expand([K_features]).to_event(1), + ) + + theta = mu_pop + z_features @ W # (N_pools, K_coeff) + numpyro.deterministic("theta", theta) + + theta_obs = theta[pool_idx] + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample( + "y", dist.StudentT(df, mu_obs, sigma_eps), + ) + + +def noise_model_ibp_dp(pool_idx, X_pool, x_obs, y_obs=None, + N_pools=None, K_coeff=4, K_cov=None, + K_features=6, K_clusters=6): + """Hybrid IBP+DP noise model. + + IBP latent features for mean heterogeneity (theta = X_pool @ B.T + z @ W), + DP mixture for noise heterogeneity (per-cluster sigma_eps). Joint + marginalization over (2^K_features × K_clusters) configurations when + y_obs is provided. + """ + import numpyro + import numpyro.distributions as dist + from jax.scipy.special import logsumexp + + # --- Population effects --- + B = numpyro.sample( + "B", dist.Normal(0.0, 5.0).expand([K_coeff, K_cov]).to_event(2) + ) + + # Student-t degrees of freedom + df = numpyro.sample("df", dist.Gamma(2.0, 0.1)) + + # --- IBP prior on feature prevalences --- + alpha_ibp = numpyro.sample("alpha_ibp", dist.Gamma(2.0, 1.0)) + with numpyro.plate("features", K_features): + v_ibp = numpyro.sample("v_ibp", dist.Beta(alpha_ibp, 1.0)) + pi = jnp.cumprod(v_ibp) # decreasing prevalences + + # --- Feature effect matrix --- + sigma_w = numpyro.sample("sigma_w", dist.HalfNormal(2.0)) + W = numpyro.sample( + "W", dist.Normal(0.0, sigma_w).expand([K_features, K_coeff]).to_event(2) + ) + + # --- DP mixture on sigma_eps --- + alpha_dp = numpyro.sample("alpha_dp", dist.Gamma(1.0, 1.0)) + with numpyro.plate("sticks", K_clusters - 1): + v = numpyro.sample("v", dist.Beta(1.0, alpha_dp)) + w = numpyro.deterministic("w", stick_breaking_weights(v)) + + sigma_eps = numpyro.sample( + "sigma_eps", + dist.HalfNormal(2.0).expand([K_clusters]).to_event(1), + ) + + # --- Population mean per pool --- + mu_pop = X_pool @ B.T # (N_pools, K_coeff) + + if y_obs is not None: + # === Joint marginalization over IBP configs × DP clusters === + + # Enumerate all 2^K binary feature configurations + n_configs = 2 ** K_features + configs = ( + (jnp.arange(n_configs)[:, None] >> jnp.arange(K_features)[None, :]) & 1 + ).astype(jnp.float32) # (n_configs, K_features) + + # Log-prior for each IBP config + log_pi = jnp.log(pi + 1e-30) + log_1mpi = jnp.log(1.0 - pi + 1e-30) + log_ibp_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Per-config feature effect + feature_effects = configs @ W # (n_configs, K_coeff) + + # Per-observation means for each IBP config + mu_pop_obs = jnp.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per IBP config per DP cluster + log_lik = dist.StudentT( + df, mu_obs[:, :, None], sigma_eps[None, None, :] + ).log_prob(y_obs[:, None, None]) # (N_obs, n_configs, K_clusters) + + # Sum log-likelihoods within each pool + pool_log_liks = jnp.zeros((N_pools, n_configs, K_clusters)) + pool_log_liks = pool_log_liks.at[pool_idx].add(log_lik) + + # Joint prior: IBP config prior × DP cluster weight + log_joint_prior = log_ibp_prior[:, None] + jnp.log(w + 1e-30)[None, :] # (n_configs, K_clusters) + + # Marginal log-likelihood per pool: logsumexp over (configs, clusters) + log_marginal = logsumexp( + log_joint_prior[None, :, :] + pool_log_liks, axis=(1, 2) + ) # (N_pools,) + numpyro.factor("log_lik", log_marginal.sum()) + else: + # === Prior predictive: sample explicit assignments === + with numpyro.plate("pools", N_pools): + z_features = numpyro.sample( + "z_features", + dist.Bernoulli(probs=pi).expand([K_features]).to_event(1), + ) + z_cluster = numpyro.sample("z_cluster", dist.Categorical(probs=w)) + + theta = mu_pop + z_features @ W # (N_pools, K_coeff) + numpyro.deterministic("theta", theta) + + theta_obs = theta[pool_idx] + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) + sigma_obs = sigma_eps[z_cluster[pool_idx]] + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("y", dist.StudentT(df, mu_obs, sigma_obs)) + + +def structural_noise_model(pool_idx, X_pool, x_obs, y_obs=None, + sigma_daily=None, lag_log_tvl=None, + fee=None, gas=None, + chain_idx=None, tier_idx=None, + N_pools=None, K_obs_coeff=None, + K_cov=None, tier_A_per_pool=None, + n_chains=8, n_tiers=6, + **kwargs): + """Structural model: LVR arb + hierarchical per-pool noise. + + Decomposes observed total volume into arb (LVR formula with learnable + cadence) and noise (per-pool theta with hierarchical prior, same as + noise_model). All continuous — AutoNormal guide works directly. + """ + import numpyro + import numpyro.distributions as dist + + K_obs_coeff = K_obs_coeff or x_obs.shape[1] + K_cov = K_cov or X_pool.shape[1] + + # --- Arb cadence parameters --- + # Informative prior: exp(2.5) ≈ 12 min cadence (empirical median from + # formula-vs-real analysis on major pairs). sigma=0.5 allows range ~4-35 min. + alpha_0 = numpyro.sample("alpha_0", dist.Normal(2.5, 0.5)) + alpha_chain = numpyro.sample( + "alpha_chain", + dist.Normal(0, 0.5).expand([n_chains - 1]).to_event(1), + ) + alpha_tier = numpyro.sample( + "alpha_tier", + dist.Normal(0, 0.5).expand([n_tiers - 1]).to_event(1), + ) + alpha_tvl = numpyro.sample("alpha_tvl", dist.Normal(0, 0.3)) + + # Per-observation cadence (broadcast pool-level indices to obs) + log_cadence = ( + alpha_0 + + _pad_with_ref(alpha_chain)[chain_idx[pool_idx]] + + _pad_with_ref(alpha_tier)[tier_idx[pool_idx]] + + alpha_tvl * lag_log_tvl + ) + cadence = jnp.exp(jnp.clip(log_cadence, -2.0, 6.0)) # 0.1 to 400 min + + # V_arb per obs (deterministic given cadence + observables) + V_arb = formula_arb_volume_daily_jax( + sigma_daily, jnp.exp(lag_log_tvl), fee, gas, cadence, + ) + + # --- Hierarchical per-pool noise (same structure as noise_model) --- + B = numpyro.sample( + "B", dist.Normal(0.0, 5.0).expand([K_obs_coeff, K_cov]).to_event(2) + ) + sigma_theta = numpyro.sample( + "sigma_theta", dist.HalfNormal(2.0).expand([K_obs_coeff]).to_event(1) + ) + L_Omega = numpyro.sample( + "L_Omega", dist.LKJCholesky(K_obs_coeff, concentration=2.0) + ) + + # Non-centered pool effects + L_Sigma = jnp.diag(sigma_theta) @ L_Omega + + with numpyro.plate("pools", N_pools): + eta = numpyro.sample( + "eta", dist.Normal(0.0, 1.0).expand([K_obs_coeff]).to_event(1) + ) + + mu_pop = X_pool @ B.T # (N_pools, K_obs_coeff) + theta = mu_pop + eta @ L_Sigma.T # (N_pools, K_obs_coeff) + numpyro.deterministic("theta", theta) + + # Per-obs noise volume + theta_obs = theta[pool_idx] + log_V_noise = jnp.sum(theta_obs * x_obs, axis=1) + V_noise = jnp.exp(log_V_noise) + + # --- Observation model --- + df = numpyro.sample("df", dist.Gamma(2.0, 0.1)) + + # Per-tier sigma_eps (same as noise_model) + sigma_eps = numpyro.sample( + "sigma_eps", dist.HalfNormal(3.0).expand([3]).to_event(1) + ) + sigma_obs = sigma_eps[tier_A_per_pool[pool_idx]] + + mu = jnp.log(jnp.maximum(V_arb + V_noise, 1e-6)) + + if y_obs is not None: + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("y", dist.StudentT(df, mu, sigma_obs), obs=y_obs) + else: + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("y", dist.StudentT(df, mu, sigma_obs)) diff --git a/quantammsim/noise_calibration/output.py b/quantammsim/noise_calibration/output.py new file mode 100644 index 00000000..1d7b632b --- /dev/null +++ b/quantammsim/noise_calibration/output.py @@ -0,0 +1,314 @@ +"""JSON output and sample caching.""" + +import json +import os + +import numpy as np + +from .constants import K_COEFF, COEFF_NAMES, K_OBS_COEFF, OBS_COEFF_NAMES + + +def generate_output_json(pool_params, samples, data, convergence, + output_path, inference_config): + """Write structured JSON output. + + Dispatches format based on whether samples contain DP mixture parameters + (detected via "v" in sample_dict), or structural model parameters + (detected via "W_gate" in sample_dict). + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Structural model path: has arb cadence params + hierarchical noise + is_structural = "alpha_0" in sample_dict and "B" in sample_dict + if is_structural: + _generate_structural_output( + pool_params, sample_dict, data, convergence, + output_path, inference_config, + ) + return + + # Detection priority: check hybrid first, then pure IBP, then DP + is_ibp_dp = ("W" in sample_dict and "v" in sample_dict + and "z_logit" not in sample_dict) + is_ibp = ("W" in sample_dict and "v" not in sample_dict + and "z_logit" not in sample_dict) + is_ibp_ste = "z_logit" in sample_dict # legacy STE artifacts + is_dp = "v" in sample_dict and "W" not in sample_dict + + B_median = np.median(np.array(sample_dict["B"]), axis=0).tolist() + sigma_eps_median = np.median( + np.array(sample_dict["sigma_eps"]), axis=0 + ) + df_median = float(np.median(np.array(sample_dict["df"]))) + + if is_ibp_dp: + model_name = "hierarchical_student_t_ibp_dp" + sigma_eps_structure = "dp_mixture" + + W_median = np.median(np.array(sample_dict["W"]), axis=0).tolist() + v_ibp_median = np.median(np.array(sample_dict["v_ibp"]), axis=0) + pi = np.cumprod(v_ibp_median).tolist() + + from .model import stick_breaking_weights + import jax.numpy as jnp + v_median = np.median(np.array(sample_dict["v"]), axis=0) + cluster_weights = np.array( + stick_breaking_weights(jnp.array(v_median)) + ).tolist() + + population_effects = { + "B": B_median, + "sigma_eps": sigma_eps_median.tolist() if hasattr(sigma_eps_median, 'tolist') else sigma_eps_median, + "df": df_median, + "W": W_median, + "feature_prevalences": pi, + "alpha_ibp": float( + np.median(np.array(sample_dict["alpha_ibp"])) + ), + "cluster_weights": cluster_weights, + "alpha_dp": float( + np.median(np.array(sample_dict["alpha_dp"])) + ), + } + + # Joint MAP assignments + from .postprocessing import assign_ibp_dp_joint + feat_assignments, cluster_assignments = assign_ibp_dp_joint( + sample_dict, data + ) + + pool_entries = {} + for i, p in enumerate(pool_params): + entry = { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + "feature_assignments": feat_assignments[i].tolist(), + "cluster_assignment": int(cluster_assignments[i]), + } + pool_entries[p["pool_id"]] = entry + + elif is_ibp or is_ibp_ste: + model_name = "hierarchical_student_t_ibp" + sigma_eps_structure = "scalar" + + W_median = np.median(np.array(sample_dict["W"]), axis=0).tolist() + v_ibp_median = np.median(np.array(sample_dict["v_ibp"]), axis=0) + pi = np.cumprod(v_ibp_median).tolist() + + population_effects = { + "B": B_median, + "sigma_eps": float(sigma_eps_median), + "df": df_median, + "W": W_median, + "feature_prevalences": pi, + "alpha_ibp": float( + np.median(np.array(sample_dict["alpha_ibp"])) + ), + } + + # Per-pool feature assignments + if is_ibp_ste: + # Legacy STE path: threshold z_logit + z_logit = np.array(sample_dict["z_logit"]) + z_logit_median = np.median(z_logit, axis=0) + feature_assignments = (z_logit_median > 0).astype(int).tolist() + else: + # Marginalized path: MAP assignments from data + from .postprocessing import assign_ibp_features + feature_assignments = assign_ibp_features( + sample_dict, data + ).tolist() + + pool_entries = {} + for i, p in enumerate(pool_params): + entry = { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + "feature_assignments": feature_assignments[i], + } + pool_entries[p["pool_id"]] = entry + + elif is_dp: + model_name = "hierarchical_student_t_dp_sigma" + sigma_eps_structure = "dp_mixture" + else: + model_name = "unified_hierarchical_student_t" + sigma_eps_structure = "per_tier" + + if not is_ibp and not is_ibp_dp: + sigma_theta_median = np.median( + np.array(sample_dict["sigma_theta"]), axis=0 + ).tolist() + + # Correlation matrix + L_Omega = np.array(sample_dict["L_Omega"]) + Omega = np.einsum("sij,skj->sik", L_Omega, L_Omega) + Omega_median = np.median(Omega, axis=0).tolist() + + population_effects = { + "B": B_median, + "sigma_theta": sigma_theta_median, + "sigma_eps": sigma_eps_median.tolist() if hasattr(sigma_eps_median, 'tolist') else sigma_eps_median, + "df": df_median, + "correlation_matrix": Omega_median, + } + + if is_dp: + from .model import stick_breaking_weights + import jax.numpy as jnp + v_median = np.median(np.array(sample_dict["v"]), axis=0) + w = stick_breaking_weights(jnp.array(v_median)) + population_effects["cluster_weights"] = np.array(w).tolist() + population_effects["alpha_dp"] = float( + np.median(np.array(sample_dict["alpha_dp"])) + ) + + pool_entries = { + p["pool_id"]: { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + } + for p in pool_params + } + + output = { + "model": model_name, + "model_spec": { + "K_coeff": K_COEFF, + "K_cov": data["K_cov"], + "coeff_names": COEFF_NAMES, + "covariate_names": data["covariate_names"], + "likelihood": "StudentT", + "tvl_lag": "log_tvl_lag1", + "sigma_eps_structure": sigma_eps_structure, + }, + "inference": inference_config, + "population_effects": population_effects, + "convergence": convergence, + "n_pools": len(pool_params), + "n_obs": len(data["y_obs"]), + "pools": pool_entries, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +def _generate_structural_output(pool_params, sample_dict, data, convergence, + output_path, inference_config): + """Write structural model JSON output (LVR arb + hierarchical noise).""" + alpha_0 = float(np.median(np.array(sample_dict["alpha_0"]))) + alpha_chain = np.median(np.array(sample_dict["alpha_chain"]), axis=0).tolist() + alpha_tier = np.median(np.array(sample_dict["alpha_tier"]), axis=0).tolist() + alpha_tvl = float(np.median(np.array(sample_dict["alpha_tvl"]))) + + B_median = np.median(np.array(sample_dict["B"]), axis=0).tolist() + sigma_theta_median = np.median( + np.array(sample_dict["sigma_theta"]), axis=0 + ).tolist() + + # Correlation matrix + L_Omega = np.array(sample_dict["L_Omega"]) + Omega = np.einsum("sij,skj->sik", L_Omega, L_Omega) + Omega_median = np.median(Omega, axis=0).tolist() + + df_median = float(np.median(np.array(sample_dict["df"]))) + sigma_eps_median = np.median( + np.array(sample_dict["sigma_eps"]), axis=0 + ).tolist() + + population_effects = { + "alpha_0": alpha_0, + "alpha_chain": alpha_chain, + "alpha_tier": alpha_tier, + "alpha_tvl": alpha_tvl, + "B": B_median, + "sigma_theta": sigma_theta_median, + "correlation_matrix": Omega_median, + "df": df_median, + "sigma_eps": sigma_eps_median, + } + + pool_entries = {} + for p in pool_params: + pool_entries[p["pool_id"]] = { + "chain": p["chain"], + "tokens": p["tokens"], + "arb_frequency": p["arb_frequency"], + "noise_params": p["noise_params"], + } + + output = { + "model": "structural_mixture", + "model_spec": { + "K_obs_coeff": K_OBS_COEFF, + "obs_coeff_names": OBS_COEFF_NAMES, + "K_cov": data["K_cov"], + "covariate_names": data["covariate_names"], + "likelihood": "StudentT", + }, + "inference": inference_config, + "population_effects": population_effects, + "convergence": convergence, + "n_pools": len(pool_params), + "n_obs": len(data["y_obs"]), + "pools": pool_entries, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +def _save_sample_cache(samples, data, cache_dir): + """Cache posterior samples and data arrays for --predict reuse.""" + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + os.makedirs(cache_dir, exist_ok=True) + + # Save only the samples needed for prediction and diagnostics. + # Skip "y" (S x N_obs, can be >1GB) and "theta" (S x N_pools x K, + # reconstructible from B, eta, sigma_theta, L_Omega). + skip_keys = {"y", "theta"} + sample_cache = os.path.join(cache_dir, "unified_samples.npz") + np.savez_compressed( + sample_cache, + **{k: np.array(v) for k, v in sample_dict.items() + if k not in skip_keys}, + ) + + # Data arrays for predict + data_cache = os.path.join(cache_dir, "unified_data.json") + cache_data = { + "pool_ids": data["pool_ids"], + "covariate_names": data["covariate_names"], + "K_cov": data["K_cov"], + "N_pools": data["N_pools"], + "ref_chain": data["ref_chain"], + "ref_tier_a": data["ref_tier_a"], + "ref_tier_b": data["ref_tier_b"], + "chains": data["chains"], + } + with open(data_cache, "w") as f: + json.dump(cache_data, f, indent=2) + + print(f" Cached samples -> {sample_cache}") + print(f" Cached data metadata -> {data_cache}") diff --git a/quantammsim/noise_calibration/plotting.py b/quantammsim/noise_calibration/plotting.py new file mode 100644 index 00000000..727443bd --- /dev/null +++ b/quantammsim/noise_calibration/plotting.py @@ -0,0 +1,335 @@ +"""Diagnostic plots for noise calibration.""" + +import os + +import numpy as np +import pandas as pd + +from .constants import K_COEFF, COEFF_NAMES +from .inference import _get_theta_samples + + +def plot_diagnostics(samples, data, output_dir, elbo_losses=None, + mcmc=None, prior_samples=None): + """Generate up to 9 diagnostic plots.""" + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + os.makedirs(output_dir, exist_ok=True) + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) # (S, N_pools, K_coeff) + theta_median = np.median(theta_samples, axis=0) + + pool_idx = data["pool_idx"] + x_obs = data["x_obs"] + y_obs = data["y_obs"] + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + + # --- 1. Prior predictive check --- + if prior_samples is not None: + y_prior = prior_samples.get("y", None) + if y_prior is not None: + fig, ax = plt.subplots(figsize=(10, 5)) + # Flatten a subsample of prior draws + y_prior_flat = y_prior.flatten() + # Clip for display + clip_lo, clip_hi = np.percentile(y_prior_flat, [0.5, 99.5]) + y_prior_clipped = y_prior_flat[ + (y_prior_flat >= clip_lo) & (y_prior_flat <= clip_hi) + ] + ax.hist(y_prior_clipped, bins=100, alpha=0.5, density=True, + color="steelblue", label="Prior predictive") + ax.hist(y_obs, bins=100, alpha=0.5, density=True, + color="coral", label="Observed") + ax.set_xlabel("log(volume)") + ax.set_ylabel("Density") + ax.set_title("Prior predictive check: log-volume") + ax.legend() + plt.tight_layout() + path = os.path.join(output_dir, "prior_predictive.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 2. ELBO loss curve (SVI only) --- + if elbo_losses is not None: + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + + ax = axes[0] + ax.plot(elbo_losses, alpha=0.3, color="steelblue", linewidth=0.5) + # Smoothed + window = min(100, len(elbo_losses) // 10) + if window > 1: + smoothed = pd.Series(elbo_losses).rolling(window).mean().values + ax.plot(smoothed, color="red", linewidth=1.5, label=f"Rolling {window}") + ax.legend() + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence") + + ax = axes[1] + # Last 20% of training + start = len(elbo_losses) * 4 // 5 + ax.plot(range(start, len(elbo_losses)), elbo_losses[start:], + color="steelblue", linewidth=0.8) + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence (last 20%)") + + plt.tight_layout() + path = os.path.join(output_dir, "elbo_convergence.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 3. Trace plots (NUTS only) --- + if mcmc is not None: + try: + import arviz as az + idata = az.from_numpyro(mcmc) + var_names = ["sigma_theta", "sigma_eps", "df"] + available = [v for v in var_names if v in idata.posterior] + if available: + axes = az.plot_trace(idata, var_names=available, compact=True) + fig = axes.ravel()[0].figure + fig.set_size_inches(14, 3 * len(available)) + path = os.path.join(output_dir, "trace_plots.png") + fig.savefig(path, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {path}") + except Exception as e: + print(f" WARNING: Trace plots failed: {e}") + + # --- 4. Posterior predictive: predicted vs observed --- + # Compute y_pred and r2 here; r2 is reused in plot 9 (model summary). + theta_obs = theta_median[pool_idx] + y_pred = np.sum(theta_obs * x_obs, axis=1) + r2 = 1 - np.var(y_obs - y_pred) / np.var(y_obs) + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + ax.scatter(y_obs, y_pred, alpha=0.1, s=4, color="steelblue") + lims = [min(y_obs.min(), y_pred.min()), max(y_obs.max(), y_pred.max())] + ax.plot(lims, lims, "r--", linewidth=1) + ax.set_xlabel("Observed log(volume)") + ax.set_ylabel("Predicted log(volume)") + ax.set_title("Posterior predictive check") + ax.text(0.05, 0.95, f"R² = {r2:.3f}", transform=ax.transAxes, + fontsize=11, verticalalignment="top") + + ax = axes[1] + residuals = y_obs - y_pred + ax.hist(residuals, bins=60, color="steelblue", edgecolor="white", alpha=0.8) + ax.axvline(0, color="red", linestyle="--") + ax.set_xlabel("Residual") + + sigma_eps_samples = np.array(sample_dict.get("sigma_eps", [0])) + if sigma_eps_samples.ndim > 1: + sigma_str = ", ".join(f"{np.median(sigma_eps_samples[:, i]):.2f}" + for i in range(sigma_eps_samples.shape[1])) + else: + sigma_str = f"{np.median(sigma_eps_samples):.2f}" + ax.set_title(f"Residuals (sigma_eps ~ [{sigma_str}])") + + plt.tight_layout() + path = os.path.join(output_dir, "posterior_predictive.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 5. Per-pool b_c by chain/tier --- + b_tvl_all = theta_median[:, 1] + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + chains_present = sorted(pool_meta["chain"].unique()) + chain_data = [] + chain_labels = [] + for c in chains_present: + mask = pool_meta["chain"].values == c + if mask.sum() > 0: + chain_data.append(b_tvl_all[mask]) + chain_labels.append(f"{c}\n(n={mask.sum()})") + if chain_data: + ax.boxplot(chain_data, tick_labels=chain_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by chain") + + ax = axes[1] + tier_a_vals = pool_meta["tier_A"].values.astype(int) + tier_labels_map = {0: "Blue-chip", 1: "Mid-cap", 2: "Long-tail"} + tier_data = [] + tier_labels = [] + for t in [0, 1, 2]: + mask = tier_a_vals == t + if mask.sum() > 0: + tier_data.append(b_tvl_all[mask]) + tier_labels.append(f"{tier_labels_map[t]}\n(n={mask.sum()})") + if tier_data: + ax.boxplot(tier_data, tick_labels=tier_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by token tier (best token)") + + plt.tight_layout() + path = os.path.join(output_dir, "per_pool_b_c.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 6. Correlation matrix posterior --- + L_Omega_samples = np.array(sample_dict["L_Omega"]) # (S, K, K) + Omega_samples = np.einsum("sij,skj->sik", L_Omega_samples, L_Omega_samples) + Omega_median = np.median(Omega_samples, axis=0) + + fig, ax = plt.subplots(figsize=(7, 6)) + im = ax.imshow(Omega_median, vmin=-1, vmax=1, cmap="RdBu_r") + ax.set_xticks(range(K_COEFF)) + ax.set_yticks(range(K_COEFF)) + ax.set_xticklabels(COEFF_NAMES, rotation=45, ha="right") + ax.set_yticklabels(COEFF_NAMES) + for i in range(K_COEFF): + for j in range(K_COEFF): + ax.text(j, i, f"{Omega_median[i, j]:.2f}", ha="center", + va="center", fontsize=10, + color="white" if abs(Omega_median[i, j]) > 0.5 else "black") + plt.colorbar(im, ax=ax, shrink=0.8) + ax.set_title("Posterior median correlation matrix (Omega)") + plt.tight_layout() + path = os.path.join(output_dir, "correlation_matrix.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 7. Shrinkage plot: OLS b_c vs hierarchical b_c --- + ols_b_c = np.zeros(len(pool_ids)) + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + if mask.sum() < 5: + ols_b_c[i] = np.nan + continue + x_i = x_obs[mask] + y_i = y_obs[mask] + try: + beta, _, _, _ = np.linalg.lstsq(x_i, y_i, rcond=None) + ols_b_c[i] = beta[1] # TVL coefficient + except np.linalg.LinAlgError: + ols_b_c[i] = np.nan + + hier_b_c = theta_median[:, 1] + valid = np.isfinite(ols_b_c) + + if valid.sum() > 2: + fig, ax = plt.subplots(figsize=(8, 8)) + ax.scatter(ols_b_c[valid], hier_b_c[valid], alpha=0.6, s=20, + color="steelblue") + + pop_b_c = np.median(hier_b_c) + ax.axhline(pop_b_c, color="red", linestyle="--", linewidth=0.8, + label=f"Population median = {pop_b_c:.3f}") + + lims = [min(np.nanmin(ols_b_c[valid]), hier_b_c[valid].min()) - 0.2, + max(np.nanmax(ols_b_c[valid]), hier_b_c[valid].max()) + 0.2] + ax.plot(lims, lims, "k:", linewidth=0.8, alpha=0.5) + ax.set_xlabel("Per-pool OLS b_c (lagged TVL)") + ax.set_ylabel("Hierarchical posterior median b_c") + ax.set_title("Shrinkage: OLS vs hierarchical TVL elasticity") + ax.legend() + plt.tight_layout() + path = os.path.join(output_dir, "shrinkage_b_c.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 8. beta_tvl vs beta_vol scatter colored by chain --- + fig, ax = plt.subplots(figsize=(10, 7)) + pool_id_to_chain = dict(zip(pool_meta["pool_id"], pool_meta["chain"])) + chain_colors = {} + cmap = plt.cm.tab10 + unique_chains = sorted(pool_meta["chain"].unique()) + for i, c in enumerate(unique_chains): + chain_colors[c] = cmap(i % 10) + + beta_tvl_arr = theta_median[:, 1] + beta_vol_arr = theta_median[:, 2] + for i, pid in enumerate(pool_ids): + c = pool_id_to_chain.get(pid, "?") + ax.scatter(beta_tvl_arr[i], beta_vol_arr[i], + color=chain_colors.get(c, "gray"), alpha=0.6, s=20, + edgecolors="white", linewidths=0.3) + + from matplotlib.lines import Line2D + handles = [Line2D([0], [0], marker="o", color="w", + markerfacecolor=chain_colors[c], markersize=8, + label=c) + for c in unique_chains if c in chain_colors] + ax.legend(handles=handles, fontsize=8, loc="best") + ax.set_xlabel("b_tvl (TVL elasticity)") + ax.set_ylabel("b_sigma (volatility sensitivity)") + ax.set_title("Pool-specific coefficients by chain") + ax.axhline(0, color="gray", linewidth=0.5, linestyle="--") + ax.axvline(0, color="gray", linewidth=0.5, linestyle="--") + plt.tight_layout() + path = os.path.join(output_dir, "beta_tvl_vs_beta_vol.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- 9. Model summary panel --- + B_samples = np.array(sample_dict["B"]) + B_median = np.median(B_samples, axis=0) # (K_coeff, K_cov) + sigma_theta_med = np.median(np.array(sample_dict["sigma_theta"]), axis=0) + df_med = np.median(np.array(sample_dict["df"])) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) + + col_names = data["covariate_names"] + + fig, ax = plt.subplots(figsize=(12, 8)) + ax.axis("off") + + summary = "Group-level regression B (posterior median):\n" + header = f" {'covariate':<20s}" + for cn in COEFF_NAMES: + header += f" {cn:>10s}" + summary += header + "\n" + summary += " " + "-" * (20 + 11 * K_COEFF) + "\n" + for j, name in enumerate(col_names): + line = f" {name:<20s}" + for k in range(K_COEFF): + line += f" {B_median[k, j]:>10.3f}" + summary += line + "\n" + + summary += f"\nsigma_theta: [{', '.join(f'{v:.3f}' for v in sigma_theta_med)}]\n" + summary += f"\nCorrelation matrix (Omega):\n" + for i in range(K_COEFF): + row = " [" + " ".join(f"{Omega_median[i, j]:>6.3f}" + for j in range(K_COEFF)) + "]\n" + summary += row + + tier_names = ["blue-chip", "mid-cap", "long-tail"] + sigma_eps_str = ", ".join(f"{tier_names[i]}={sigma_eps_med[i]:.3f}" + for i in range(len(sigma_eps_med))) + summary += f"\nsigma_eps: [{sigma_eps_str}]\n" + summary += f"df (Student-t): {df_med:.1f}\n" + summary += f"R^2: {r2:.3f}\n" + + ax.text(0.02, 0.98, summary, transform=ax.transAxes, + fontsize=7, verticalalignment="top", fontfamily="monospace") + ax.set_title("Model Summary") + plt.tight_layout() + path = os.path.join(output_dir, "model_summary.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") diff --git a/quantammsim/noise_calibration/postprocessing.py b/quantammsim/noise_calibration/postprocessing.py new file mode 100644 index 00000000..190f07c0 --- /dev/null +++ b/quantammsim/noise_calibration/postprocessing.py @@ -0,0 +1,659 @@ +"""Post-processing: extract params, predict, convergence, prior predictive.""" + +import numpy as np + +from .constants import K_COEFF, COEFF_NAMES, K_OBS_COEFF, OBS_COEFF_NAMES +from .token_classification import classify_token_tier +from .covariate_encoding import _tier_pair_idx +from .inference import _get_theta_samples, _build_model_kwargs +from .model import noise_model + + +def extract_noise_params(samples, data, use_median=True) -> list: + """Extract per-pool noise params from posterior samples. + + Handles both MCMC.get_samples() and SVI samples dict. + Applies weekend absorption: b_0_eff = b_0_raw + b_weekend * (2/7). + """ + # Get theta samples + if hasattr(samples, "get_samples"): + # MCMC object + sample_dict = samples.get_samples() + else: + sample_dict = samples + + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) # (S, N_pools, K_coeff) + + agg_fn = np.median if use_median else np.mean + theta_agg = agg_fn(theta_samples, axis=0) # (N_pools, K_coeff) + theta_std = np.std(theta_samples, axis=0) + + pool_ids = data["pool_ids"] + pool_meta = data["pool_meta"] + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + b_0_raw, b_tvl, b_sigma, b_weekend = theta_agg[i] + std_vals = theta_std[i] + + # Weekend absorption: simulator has no weekend indicator, + # so fold the expected weekend effect into the intercept. + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "theta_median": [float(x) for x in theta_agg[i]], + "theta_std": [float(x) for x in std_vals], + "b_weekend": float(b_weekend), + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(meta["swap_fee"]), + }, + }) + + return results + + +def predict_new_pool(samples, data, chain: str, tokens: list, + fee: float, feature_assignments=None) -> dict: + """Predict noise params for an unseen pool using population effects. + + Constructs z_new, computes mu_new = B @ z_new across all posterior samples, + returns point estimate + 90% credible intervals with weekend absorption. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Build z_new using data-driven column names + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b}": + z_new[i] = 1.0 + + # mu_new = B @ z_new across all posterior samples + B_samples = np.array(sample_dict["B"]) # (S, K_coeff, K_cov) + mu_samples = np.einsum("skd,d->sk", B_samples, z_new) # (S, K_coeff) + + # IBP path: add feature effects + is_ibp = "W" in sample_dict + if is_ibp: + W_samples = np.array(sample_dict["W"]) # (S, K_features, K_coeff) + if feature_assignments is not None: + # User-specified binary features + Z = np.array(feature_assignments) # (K_features,) + feature_effect = np.einsum("skj,k->sj", W_samples, Z) + prediction_source = "ibp_user_features" + else: + # Marginal: weight by prevalences pi = cumprod(v_ibp) + v_ibp = np.array(sample_dict["v_ibp"]) # (S, K_features) + pi = np.cumprod(v_ibp, axis=1) # (S, K_features) + feature_effect = np.einsum("skj,sk->sj", W_samples, pi) + prediction_source = "ibp_marginal" + mu_samples = mu_samples + feature_effect + else: + prediction_source = "population_level" + + mu_median = np.median(mu_samples, axis=0) + mu_q05 = np.percentile(mu_samples, 5, axis=0) + mu_q95 = np.percentile(mu_samples, 95, axis=0) + + # Weekend absorption + b_0_raw, b_tvl, b_sigma, b_weekend = mu_median + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + result = { + "chain": chain, + "tokens": tokens, + "fee": fee, + "prediction_source": prediction_source, + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(fee), + }, + "credible_intervals_90": { + name: { + "median": float(mu_median[k]), + "q05": float(mu_q05[k]), + "q95": float(mu_q95[k]), + } + for k, name in enumerate(COEFF_NAMES) + }, + } + + print(f"\n Predicted noise_params for {chain} {tokens} (fee={fee}):") + for name, ci in result["credible_intervals_90"].items(): + print(f" {name:12s}: {ci['median']:+.3f} " + f"[{ci['q05']:+.3f}, {ci['q95']:+.3f}]") + print(f"\n Effective b_0 (weekend-absorbed): {b_0_effective:.3f}") + + return result + + +def check_convergence(mcmc_or_losses, method="nuts") -> dict: + """Compute convergence diagnostics. + + For NUTS: R-hat, ESS, divergences. + For SVI: final ELBO, ELBO stability. + """ + if method == "svi": + losses = np.array(mcmc_or_losses) + return { + "method": "svi", + "final_elbo": float(losses[-1]), + "elbo_last_100_std": float(np.std(losses[-100:])), + "elbo_last_100_mean": float(np.mean(losses[-100:])), + } + + # NUTS diagnostics + import arviz as az + + mcmc = mcmc_or_losses + idata = az.from_numpyro(mcmc) + + n_chains = idata.posterior.sizes.get("chain", 1) + + rhat_max = float("nan") + if n_chains >= 2: + rhat = az.rhat(idata) + rhat_vals = [] + for var in rhat.data_vars: + if var == "theta": + continue + vals = rhat[var].values + rhat_vals.extend(vals.flatten()) + rhat_max = float(np.nanmax(rhat_vals)) if rhat_vals else float("nan") + + ess = az.ess(idata) + ess_vals = [] + for var in ess.data_vars: + if var == "theta": + continue + vals = ess[var].values + ess_vals.extend(vals.flatten()) + ess_min = float(np.nanmin(ess_vals)) if ess_vals else float("nan") + + divergences = int(idata.sample_stats["diverging"].sum().values) + + print(f"\n Convergence diagnostics:") + if n_chains >= 2: + print(f" R-hat max: {rhat_max:.4f} " + f"{'OK' if rhat_max < 1.05 else 'WARNING'}") + else: + print(f" R-hat max: N/A (need >= 2 chains)") + print(f" ESS min: {ess_min:.0f} " + f"{'OK' if ess_min > 400 else 'WARNING'}") + print(f" Divergences: {divergences} " + f"{'OK' if divergences == 0 else 'WARNING'}") + + return { + "method": "nuts", + "r_hat_max": rhat_max, + "ess_min": ess_min, + "divergences": divergences, + } + + +def assign_dp_clusters(samples, data) -> np.ndarray: + """Compute posterior MAP cluster assignments for DP mixture model. + + Uses median posterior samples for v->w, sigma_eps, df, theta to compute + per-pool-per-cluster log-likelihoods, then returns argmax assignments. + """ + from scipy.special import logsumexp + from .model import stick_breaking_weights + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + # Median posterior parameters + v_med = np.median(np.array(sample_dict["v"]), axis=0) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) + df_med = float(np.median(np.array(sample_dict["df"]))) + + # Compute w from v via stick-breaking (using numpy) + import jax.numpy as jnp + w = np.array(stick_breaking_weights(jnp.array(v_med))) + + # Reconstruct theta + theta_samples = _get_theta_samples( + sample_dict, np.array(data["X_pool"]), data=data + ) + theta_med = np.median(theta_samples, axis=0) # (N_pools, K_coeff) + + # Per-observation predicted means + pool_idx = np.array(data["pool_idx"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + K_clusters = len(sigma_eps_med) + + theta_obs = theta_med[pool_idx] + mu_obs = np.sum(theta_obs * x_obs, axis=1) # (N_obs,) + + # Log-likelihood per observation per cluster + from scipy.stats import t as t_dist + log_lik_per_k = np.zeros((len(y_obs), K_clusters)) + for k in range(K_clusters): + log_lik_per_k[:, k] = t_dist.logpdf( + y_obs, df_med, loc=mu_obs, scale=sigma_eps_med[k] + ) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, K_clusters)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik_per_k[i] + + # Posterior cluster probabilities: log p(z=k|data) = log w_k + sum log p(y|k) + log_posterior = np.log(w + 1e-30)[None, :] + pool_log_liks + # MAP assignment + assignments = np.argmax(log_posterior, axis=1).astype(np.int64) + return assignments + + +def assign_ibp_features(samples, data) -> np.ndarray: + """Compute MAP feature assignments for marginalized IBP model. + + Enumerates all 2^K binary feature configurations per pool, evaluates + per-pool log-posterior (log-prior + log-likelihood), returns argmax + config as (N_pools, K_features) binary ndarray. + + Uses median posterior parameters for B, W, v_ibp, sigma_eps, df. + """ + from scipy.stats import t as t_dist + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + B_med = np.median(np.array(sample_dict["B"]), axis=0) # (K_coeff, K_cov) + W_med = np.median(np.array(sample_dict["W"]), axis=0) # (K_features, K_coeff) + v_ibp_med = np.median(np.array(sample_dict["v_ibp"]), axis=0) # (K_features,) + sigma_eps_med = float(np.median(np.array(sample_dict["sigma_eps"]))) + df_med = float(np.median(np.array(sample_dict["df"]))) + + pi = np.cumprod(v_ibp_med) # (K_features,) + K_features = len(pi) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + + # Enumerate all 2^K configs + n_configs = 2 ** K_features + configs = ( + (np.arange(n_configs)[:, None] >> np.arange(K_features)[None, :]) & 1 + ).astype(float) # (n_configs, K_features) + + # Log-prior per config + log_pi = np.log(pi + 1e-30) + log_1mpi = np.log(1.0 - pi + 1e-30) + log_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Population mean + mu_pop = X_pool @ B_med.T # (N_pools, K_coeff) + + # Feature effects per config + feature_effects = configs @ W_med # (n_configs, K_coeff) + + # Per-obs means: mu_pop_obs + feature_mu + mu_pop_obs = np.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config + log_lik = t_dist.logpdf( + y_obs[:, None], df_med, loc=mu_obs, scale=sigma_eps_med + ) # (N_obs, n_configs) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, n_configs)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik[i] + + # Posterior = log_prior + pool_log_liks; MAP config per pool + log_posterior = log_prior[None, :] + pool_log_liks # (N_pools, n_configs) + best_config_idx = np.argmax(log_posterior, axis=1) # (N_pools,) + + return configs[best_config_idx].astype(int) # (N_pools, K_features) + + +def assign_ibp_dp_joint(samples, data) -> tuple: + """Compute MAP joint (feature, cluster) assignments for hybrid IBP+DP model. + + Enumerates all (2^K_features × K_clusters) joint configurations per pool, + evaluates joint log-posterior, returns argmax assignments. + + Returns: + (feature_assignments, cluster_assignments): + feature_assignments: (N_pools, K_features) binary ndarray + cluster_assignments: (N_pools,) int ndarray + """ + from scipy.stats import t as t_dist + from .model import stick_breaking_weights + + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + B_med = np.median(np.array(sample_dict["B"]), axis=0) # (K_coeff, K_cov) + W_med = np.median(np.array(sample_dict["W"]), axis=0) # (K_features, K_coeff) + v_ibp_med = np.median(np.array(sample_dict["v_ibp"]), axis=0) # (K_features,) + v_med = np.median(np.array(sample_dict["v"]), axis=0) # (K_clusters-1,) + sigma_eps_med = np.median(np.array(sample_dict["sigma_eps"]), axis=0) # (K_clusters,) + df_med = float(np.median(np.array(sample_dict["df"]))) + + pi = np.cumprod(v_ibp_med) # (K_features,) + K_features = len(pi) + K_clusters = len(sigma_eps_med) + + import jax.numpy as jnp + w = np.array(stick_breaking_weights(jnp.array(v_med))) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + y_obs = np.array(data["y_obs"]) + N_pools = data["N_pools"] + + # Enumerate all 2^K configs + n_configs = 2 ** K_features + configs = ( + (np.arange(n_configs)[:, None] >> np.arange(K_features)[None, :]) & 1 + ).astype(float) # (n_configs, K_features) + + # IBP log-prior per config + log_pi = np.log(pi + 1e-30) + log_1mpi = np.log(1.0 - pi + 1e-30) + log_ibp_prior = configs @ log_pi + (1.0 - configs) @ log_1mpi # (n_configs,) + + # Joint log-prior: IBP config × DP cluster + log_joint_prior = log_ibp_prior[:, None] + np.log(w + 1e-30)[None, :] # (n_configs, K_clusters) + + # Population mean + mu_pop = X_pool @ B_med.T # (N_pools, K_coeff) + feature_effects = configs @ W_med # (n_configs, K_coeff) + + # Per-obs means + mu_pop_obs = np.sum(mu_pop[pool_idx] * x_obs, axis=1) # (N_obs,) + feature_mu = x_obs @ feature_effects.T # (N_obs, n_configs) + mu_obs = mu_pop_obs[:, None] + feature_mu # (N_obs, n_configs) + + # Log-likelihood per obs per config per cluster + log_lik = np.zeros((len(y_obs), n_configs, K_clusters)) + for k in range(K_clusters): + log_lik[:, :, k] = t_dist.logpdf( + y_obs[:, None], df_med, loc=mu_obs, scale=sigma_eps_med[k] + ) + + # Sum within pools + pool_log_liks = np.zeros((N_pools, n_configs, K_clusters)) + for i in range(len(y_obs)): + pool_log_liks[pool_idx[i]] += log_lik[i] + + # Joint posterior: log_joint_prior + pool_log_liks + log_posterior = log_joint_prior[None, :, :] + pool_log_liks # (N_pools, n_configs, K_clusters) + + # Flatten to (N_pools, n_configs * K_clusters), argmax, unravel + flat = log_posterior.reshape(N_pools, -1) + best_flat_idx = np.argmax(flat, axis=1) + best_config_idx = best_flat_idx // K_clusters + best_cluster_idx = best_flat_idx % K_clusters + + feature_assignments = configs[best_config_idx].astype(int) # (N_pools, K_features) + cluster_assignments = best_cluster_idx.astype(np.int64) # (N_pools,) + + return feature_assignments, cluster_assignments + + +def extract_structural_params(samples, data, use_median=True) -> list: + """Extract per-pool arb frequency and noise coefficients from structural model. + + Parameters + ---------- + samples : dict + Posterior samples from SVI/NUTS with structural_noise_model. + data : dict + Output of encode_covariates_structural(). + + Returns + ------- + list of dict + Per-pool dicts with: pool_id, chain, tokens, arb_frequency, noise_params. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + agg_fn = np.median if use_median else np.mean + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # Per-pool theta from hierarchical model + # theta = X_pool @ B.T + eta @ L_Sigma.T + B = agg_fn(np.array(sample_dict["B"]), axis=0) + eta = agg_fn(np.array(sample_dict["eta"]), axis=0) + sigma_theta = agg_fn(np.array(sample_dict["sigma_theta"]), axis=0) + L_Omega = agg_fn(np.array(sample_dict["L_Omega"]), axis=0) + + X_pool = np.array(data["X_pool"]) + L_Sigma = np.diag(sigma_theta) @ L_Omega + theta = X_pool @ B.T + eta @ L_Sigma.T # (N_pools, K_obs_coeff) + + chain_idx = np.array(data["chain_idx"]) + tier_idx = np.array(data["tier_idx"]) + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + + # Per-pool cadence + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + # Per-pool log_tvl (median across observations) + pool_idx_arr = np.array(data["pool_idx"]) + lag_log_tvl = np.array(data["lag_log_tvl"]) + N_pools = data["N_pools"] + + pool_tvl_median = np.zeros(N_pools) + for p in range(N_pools): + mask = pool_idx_arr == p + if mask.any(): + pool_tvl_median[p] = np.median(lag_log_tvl[mask]) + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + log_cadence = ( + alpha_0 + + padded_chain[chain_idx[i]] + + padded_tier[tier_idx[i]] + + alpha_tvl * pool_tvl_median[i] + ) + cadence = np.exp(np.clip(log_cadence, -2.0, 6.0)) + arb_freq = int(np.clip(np.round(cadence), 1, 60)) + + noise_coeffs = { + name: float(theta[i, k]) + for k, name in enumerate(OBS_COEFF_NAMES) + } + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "arb_frequency": arb_freq, + "noise_params": noise_coeffs, + }) + + return results + + +def predict_new_pool_structural( + samples, data, chain: str, tokens: list, fee: float, tvl_est: float, +) -> dict: + """Predict cadence and noise coefficients for a hypothetical pool. + + Uses the structural model's arb cadence parameters and hierarchical B + regression for noise (population mean, no pool-specific random effect). + + Parameters + ---------- + samples : dict + Posterior samples from structural_noise_model. + data : dict + Output of encode_covariates_structural(). + chain : str + Chain name. + tokens : list of str + Token symbols. + fee : float + Swap fee (fraction). + tvl_est : float + Estimated TVL in USD. + """ + if hasattr(samples, "get_samples"): + sample_dict = samples.get_samples() + else: + sample_dict = samples + + agg_fn = np.median + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # Hierarchical B for noise population mean + B = agg_fn(np.array(sample_dict["B"]), axis=0) + + # Construct chain and tier indices for the new pool + chains = data["chains"] + chain_to_idx = {c: i for i, c in enumerate(chains)} + c_idx = chain_to_idx.get(chain, 0) # fallback to reference + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tier_a + t_idx = _tier_pair_idx(tier_a, tier_b) + + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + log_tvl = np.log(max(tvl_est, 1.0)) + log_cadence = ( + alpha_0 + + padded_chain[c_idx] + + padded_tier[t_idx] + + alpha_tvl * log_tvl + ) + cadence = np.exp(np.clip(log_cadence, -2.0, 6.0)) + arb_freq = int(np.clip(np.round(cadence), 1, 60)) + + # Construct X_pool_new for hierarchical regression + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + tier_a_str = str(tier_a) + tier_b_str = str(tier_b) + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a_str}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b_str}": + z_new[i] = 1.0 + + # Population mean theta for new pool (no random effect) + theta_new = z_new @ B.T # (K_obs_coeff,) + + noise_coeffs = { + name: float(theta_new[k]) + for k, name in enumerate(OBS_COEFF_NAMES) + } + + return { + "chain": chain, + "tokens": tokens, + "fee": fee, + "tvl_est": tvl_est, + "arb_frequency": arb_freq, + "noise_params": noise_coeffs, + } + + +def run_prior_predictive(data, num_samples=500, model_fn=None) -> dict: + """Run prior predictive check (no observations).""" + import jax + from numpyro.infer import Predictive + + if model_fn is None: + model_fn = noise_model + + model_kwargs = _build_model_kwargs(data, model_fn=model_fn) + model_kwargs["y_obs"] = None # no observations + + predictive = Predictive(model_fn, num_samples=num_samples) + rng_key = jax.random.PRNGKey(99) + prior_samples = predictive(rng_key, **model_kwargs) + prior_samples = {k: np.array(v) for k, v in prior_samples.items()} + + print(f" Prior predictive: drew {num_samples} samples") + y_prior = prior_samples.get("y", None) + if y_prior is not None: + print(f" Prior log-volume range: " + f"[{np.percentile(y_prior, 1):.1f}, " + f"{np.percentile(y_prior, 99):.1f}]") + print(f" Observed log-volume range: " + f"[{data['y_obs'].min():.1f}, {data['y_obs'].max():.1f}]") + + return prior_samples diff --git a/quantammsim/noise_calibration/token_classification.py b/quantammsim/noise_calibration/token_classification.py new file mode 100644 index 00000000..aad9caa7 --- /dev/null +++ b/quantammsim/noise_calibration/token_classification.py @@ -0,0 +1,23 @@ +"""Token tier classification.""" + +from .constants import _TIER_0, _TIER_1 + + +def _normalise_symbol(symbol: str) -> str: + """Normalise wrapped/bridged variants to canonical form.""" + s = symbol.strip() + mapping = { + "WETH": "WETH", "WBTC": "WBTC", "cbBTC": "cbBTC", + "WMATIC": "WMATIC", "WAVAX": "WAVAX", "WXDAI": "WXDAI", "wS": "wS", + } + return mapping.get(s, s) + + +def classify_token_tier(symbol: str) -> int: + """Classify a token symbol into tier 0/1/2.""" + s = _normalise_symbol(symbol) + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 diff --git a/tests/noise/__init__.py b/tests/noise/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/noise/conftest.py b/tests/noise/conftest.py new file mode 100644 index 00000000..5bfeadc6 --- /dev/null +++ b/tests/noise/conftest.py @@ -0,0 +1,274 @@ +"""Shared fixtures for noise calibration tests.""" + +from datetime import date, timedelta + +import numpy as np +import pandas as pd +import pytest + + +# --------------------------------------------------------------------------- +# synthetic_panel: 3 pools × 10 days = 30 obs (before lag drop → 27 obs) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_panel() -> pd.DataFrame: + """3 pools × 10 days with known structure. + + Pool A: MAINNET, WETH/USDC, tier_A=0, tier_B=0, fee=0.003 + Pool B: ARBITRUM, BAL/WETH, tier_A=0, tier_B=1, fee=0.01 + Pool C: BASE, RATS/WETH, tier_A=0, tier_B=2, fee=0.005 + + Dates: 2026-01-01 to 2026-01-10 + Weekend flags: Sat 2026-01-03 and Sun 2026-01-04 are weekends. + (2026-01-01 is Thursday, ..., 01-03 Sat, 01-04 Sun, 01-05 Mon, ...) + """ + np.random.seed(42) + + pools = [ + ("pool_A", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("pool_B", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("pool_C", "BASE", "RATS,WETH", 0.005, 0, 2), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + + records = [] + for pool_id, chain, tokens, fee, tier_a, tier_b in pools: + log_tvl_base = 14.0 + np.random.randn() * 0.5 + for d in dates: + log_tvl = log_tvl_base + np.random.randn() * 0.1 + log_vol = log_tvl - 2.0 + np.random.randn() * 0.3 + vol = 0.3 + np.random.rand() * 0.2 + is_weekend = 1.0 if d.weekday() >= 5 else 0.0 + + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": d, + "log_volume": log_vol, + "log_tvl": log_tvl, + "volatility": vol, + "weekend": is_weekend, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": tokens, + }) + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + # Structural model covariates + panel["log_sigma"] = np.log(np.maximum(panel["volatility"].values, 1e-6)) + dow = panel["date"].apply( + lambda d: d.weekday() if hasattr(d, "weekday") else pd.Timestamp(d).weekday() + ) + panel["dow_sin"] = np.sin(2.0 * np.pi * dow / 7.0) + panel["dow_cos"] = np.cos(2.0 * np.pi * dow / 7.0) + panel["tvl_x_sigma"] = panel["log_tvl_lag1"] * panel["log_sigma"] + panel["tvl_x_fee"] = panel["log_tvl_lag1"] * panel["log_fee"] + panel["sigma_x_fee"] = panel["log_sigma"] * panel["log_fee"] + + return panel + + +# --------------------------------------------------------------------------- +# synthetic_encoded_data: output of encode_covariates(synthetic_panel) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_encoded_data(synthetic_panel): + from quantammsim.noise_calibration import encode_covariates + return encode_covariates(synthetic_panel) + + +# --------------------------------------------------------------------------- +# synthetic_samples: deterministic posterior-like dict +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_samples(synthetic_encoded_data): + """Deterministic posterior samples with eta=0, L_Omega=I. + + With this structure: theta = X_pool @ B^T exactly. + """ + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_coeff = 4 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + sigma_theta = np.ones((S, K_coeff)) + L_Omega = np.tile(np.eye(K_coeff), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_coeff)) + df = np.full((S,), 5.0) + sigma_eps = np.tile([0.5, 0.8, 0.6], (S, 1)) + + return { + "B": B, + "sigma_theta": sigma_theta, + "L_Omega": L_Omega, + "eta": eta, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_pools_df: matches enumerate_balancer_pools output schema +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_pools_df() -> pd.DataFrame: + return pd.DataFrame([ + { + "pool_id": "pool_A", + "chain": "MAINNET", + "pool_type": "WEIGHTED", + "tokens": ["WETH", "USDC"], + "token_addresses": ["0xweth", "0xusdc"], + "swap_fee": 0.003, + "current_tvl": 1_000_000, + }, + { + "pool_id": "pool_B", + "chain": "ARBITRUM", + "pool_type": "WEIGHTED", + "tokens": ["BAL", "WETH"], + "token_addresses": ["0xbal", "0xweth"], + "swap_fee": 0.01, + "current_tvl": 500_000, + }, + { + "pool_id": "pool_C", + "chain": "BASE", + "pool_type": "WEIGHTED", + "tokens": ["RATS", "WETH"], + "token_addresses": ["0xrats", "0xweth"], + "swap_fee": 0.005, + "current_tvl": 100_000, + }, + ]) + + +# --------------------------------------------------------------------------- +# synthetic_ibp_samples: IBP posterior-like dict (no eta, L_Omega, sigma_theta) +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_ibp_samples(synthetic_encoded_data): + """Deterministic IBP posterior samples (marginalized model). + + Contains B, W, v_ibp, alpha_ibp — no z_logit, eta, L_Omega, sigma_theta. + Z is analytically marginalized; MAP assignments are computed from data. + """ + data = synthetic_encoded_data + K_cov = data["K_cov"] + K_coeff = 4 + K_features = 6 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + W = np.random.randn(S, K_features, K_coeff) * 0.3 + v_ibp = np.random.beta(2, 1, size=(S, K_features)) + alpha_ibp = np.full((S,), 2.0) + sigma_w = np.full((S,), 1.0) + df = np.full((S,), 5.0) + sigma_eps = np.full((S,), 0.5) + + return { + "B": B, + "W": W, + "v_ibp": v_ibp, + "alpha_ibp": alpha_ibp, + "sigma_w": sigma_w, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_ibp_dp_samples: hybrid IBP+DP posterior-like dict +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_ibp_dp_samples(synthetic_encoded_data): + """Deterministic hybrid IBP+DP posterior samples. + + Contains both IBP keys (B, W, v_ibp, alpha_ibp, sigma_w) and + DP keys (v, alpha_dp, sigma_eps as vector). No z_logit, eta, L_Omega, + sigma_theta. + """ + data = synthetic_encoded_data + K_cov = data["K_cov"] + K_coeff = 4 + K_features = 6 + K_clusters = 6 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_coeff, K_cov) * 0.5 + W = np.random.randn(S, K_features, K_coeff) * 0.3 + v_ibp = np.random.beta(2, 1, size=(S, K_features)) + alpha_ibp = np.full((S,), 2.0) + sigma_w = np.full((S,), 1.0) + v = np.random.beta(1, 2, size=(S, K_clusters - 1)) + alpha_dp = np.full((S,), 1.0) + df = np.full((S,), 5.0) + sigma_eps = np.abs(np.random.randn(S, K_clusters)) + 0.1 + + return { + "B": B, + "W": W, + "v_ibp": v_ibp, + "alpha_ibp": alpha_ibp, + "sigma_w": sigma_w, + "v": v, + "alpha_dp": alpha_dp, + "df": df, + "sigma_eps": sigma_eps, + } + + +# --------------------------------------------------------------------------- +# synthetic_snapshots_df: matches fetch_all_snapshots output schema +# --------------------------------------------------------------------------- + +# --------------------------------------------------------------------------- +# synthetic_structural_data: output of encode_covariates_structural() +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_structural_data(synthetic_panel): + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + return encode_covariates_structural(synthetic_panel) + + +# --------------------------------------------------------------------------- +# synthetic_snapshots_df: matches fetch_all_snapshots output schema +# --------------------------------------------------------------------------- + +@pytest.fixture() +def synthetic_snapshots_df() -> pd.DataFrame: + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + records = [] + np.random.seed(42) + for pool_id, chain in [("pool_A", "MAINNET"), ("pool_B", "ARBITRUM"), + ("pool_C", "BASE")]: + for d in dates: + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": d, + "volume_usd": np.exp(10.0 + np.random.randn() * 0.5), + "total_liquidity_usd": np.exp(14.0 + np.random.randn() * 0.3), + }) + return pd.DataFrame(records) diff --git a/tests/noise/test_covariate_encoding.py b/tests/noise/test_covariate_encoding.py new file mode 100644 index 00000000..dececb5b --- /dev/null +++ b/tests/noise/test_covariate_encoding.py @@ -0,0 +1,326 @@ +"""Tests for encode_covariates and encode_covariates_structural.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import encode_covariates +from quantammsim.noise_calibration.constants import K_OBS_COEFF, OBS_COEFF_NAMES + + +class TestEncodeCovariates: + def test_x_pool_shape(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + N_pools = data["N_pools"] + K_cov = data["K_cov"] + assert data["X_pool"].shape == (N_pools, K_cov) + + def test_x_obs_shape(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + N_obs = len(synthetic_panel) + assert data["x_obs"].shape == (N_obs, 4) + + def test_x_obs_column_0_is_intercept(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal(data["x_obs"][:, 0], 1.0) + + def test_x_obs_column_1_is_lagged_tvl(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 1], + synthetic_panel["log_tvl_lag1"].values, + ) + + def test_x_obs_column_2_is_volatility(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 2], + synthetic_panel["volatility"].values, + ) + + def test_x_obs_column_3_is_weekend(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["x_obs"][:, 3], + synthetic_panel["weekend"].values, + ) + + def test_intercept_column_all_ones(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal(data["X_pool"][:, 0], 1.0) + + def test_chain_dummies_one_hot(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + chain_cols = [i for i, n in enumerate(col_names) if n.startswith("chain_")] + X = data["X_pool"] + + for row in range(X.shape[0]): + chain_vals = X[row, chain_cols] + assert chain_vals.sum() <= 1.0 + + def test_reference_chain_is_alphabetically_first(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + chains = sorted(synthetic_panel["chain"].unique()) + ref_chain = chains[0] # ARBITRUM + + pool_meta = data["pool_meta"] + ref_idx = pool_meta[pool_meta["chain"] == ref_chain].index[0] + col_names = data["covariate_names"] + chain_cols = [i for i, n in enumerate(col_names) if n.startswith("chain_")] + X = data["X_pool"] + assert all(X[ref_idx, c] == 0.0 for c in chain_cols) + + def test_tier_a_dummies_match(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + tier_a_cols = [ + (i, n) for i, n in enumerate(col_names) if n.startswith("tier_A_") + ] + pool_meta = data["pool_meta"] + X = data["X_pool"] + + for idx, row in pool_meta.iterrows(): + tier_a_str = str(row["tier_A"]) + for col_idx, col_name in tier_a_cols: + expected_tier = col_name.split("_")[-1] + expected = 1.0 if tier_a_str == expected_tier else 0.0 + assert X[idx, col_idx] == expected, ( + f"Pool {idx} tier_A={tier_a_str}, col {col_name}: " + f"expected {expected}, got {X[idx, col_idx]}" + ) + + def test_tier_b_dummies_match(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + tier_b_cols = [ + (i, n) for i, n in enumerate(col_names) if n.startswith("tier_B_") + ] + pool_meta = data["pool_meta"] + X = data["X_pool"] + + for idx, row in pool_meta.iterrows(): + tier_b_str = str(row["tier_B"]) + for col_idx, col_name in tier_b_cols: + expected_tier = col_name.split("_")[-1] + expected = 1.0 if tier_b_str == expected_tier else 0.0 + assert X[idx, col_idx] == expected + + def test_log_fee_column(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + col_names = data["covariate_names"] + fee_idx = col_names.index("log_fee") + pool_meta = data["pool_meta"] + + for idx, row in pool_meta.iterrows(): + expected = np.log(max(row["swap_fee"], 1e-6)) + np.testing.assert_allclose( + data["X_pool"][idx, fee_idx], expected, rtol=1e-10, + ) + + def test_pool_idx_maps_observations(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + pool_ids = data["pool_ids"] + pool_idx = data["pool_idx"] + + assert pool_idx.min() >= 0 + assert pool_idx.max() < len(pool_ids) + + for i, pid in enumerate(pool_ids): + mask = synthetic_panel["pool_id"] == pid + obs_indices = np.where(mask.values)[0] + assert (pool_idx[obs_indices] == i).all() + + def test_y_obs_matches_panel(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + np.testing.assert_array_equal( + data["y_obs"], + synthetic_panel["log_volume"].values, + ) + + def test_covariate_names_length(self, synthetic_panel): + data = encode_covariates(synthetic_panel) + assert len(data["covariate_names"]) == data["K_cov"] + + def test_covariate_column_ordering(self, synthetic_panel): + """X_pool columns must follow: intercept, chain dummies, tier_A dummies, + tier_B dummies, log_fee. This ordering is load-bearing because B is + indexed by column position.""" + data = encode_covariates(synthetic_panel) + names = data["covariate_names"] + + assert names[0] == "intercept" + assert names[-1] == "log_fee" + + # Find boundaries + chain_start = None + tier_a_start = None + tier_b_start = None + fee_idx = len(names) - 1 + + for i, n in enumerate(names): + if n.startswith("chain_") and chain_start is None: + chain_start = i + if n.startswith("tier_A_") and tier_a_start is None: + tier_a_start = i + if n.startswith("tier_B_") and tier_b_start is None: + tier_b_start = i + + # Verify ordering: intercept < chains < tier_A < tier_B < log_fee + if chain_start is not None: + assert chain_start > 0 # after intercept + if tier_a_start is not None and chain_start is not None: + assert tier_a_start > chain_start + if tier_b_start is not None and tier_a_start is not None: + assert tier_b_start > tier_a_start + if tier_b_start is not None: + assert fee_idx > tier_b_start + + # All chain dummies are contiguous + chain_names = [n for n in names if n.startswith("chain_")] + if chain_names: + chain_indices = [names.index(n) for n in chain_names] + assert chain_indices == list(range(min(chain_indices), + max(chain_indices) + 1)) + + def test_output_dict_has_all_required_keys(self, synthetic_panel): + """encode_covariates must return all keys consumed by downstream + functions (predict_new_pool, generate_output_json, _save_sample_cache).""" + data = encode_covariates(synthetic_panel) + required_keys = { + "pool_idx", "X_pool", "x_obs", "y_obs", "pool_ids", "pool_meta", + "covariate_names", "tier_A_per_pool", "N_pools", "K_cov", + "ref_chain", "ref_tier_a", "ref_tier_b", "chains", + } + assert required_keys.issubset(data.keys()), ( + f"Missing keys: {required_keys - data.keys()}" + ) + + +class TestEncodeCovariatesNoTiers: + """Tests for encode_covariates(include_tiers=False).""" + + def test_no_tier_columns_in_covariate_names(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + for name in data["covariate_names"]: + assert not name.startswith("tier_A_"), ( + f"Found tier_A column {name} with include_tiers=False" + ) + assert not name.startswith("tier_B_"), ( + f"Found tier_B column {name} with include_tiers=False" + ) + + def test_still_has_intercept_and_log_fee(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + assert "intercept" in data["covariate_names"] + assert "log_fee" in data["covariate_names"] + + def test_still_has_chain_dummies(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + chain_cols = [n for n in data["covariate_names"] + if n.startswith("chain_")] + assert len(chain_cols) > 0 + + def test_k_cov_smaller_than_with_tiers(self, synthetic_panel): + data_tiers = encode_covariates(synthetic_panel, include_tiers=True) + data_no_tiers = encode_covariates(synthetic_panel, include_tiers=False) + assert data_no_tiers["K_cov"] < data_tiers["K_cov"] + + def test_default_is_include_tiers_true(self, synthetic_panel): + data_default = encode_covariates(synthetic_panel) + data_explicit = encode_covariates(synthetic_panel, include_tiers=True) + assert data_default["K_cov"] == data_explicit["K_cov"] + + def test_tier_A_per_pool_still_present(self, synthetic_panel): + """tier_A_per_pool is still returned (for other downstream uses).""" + data = encode_covariates(synthetic_panel, include_tiers=False) + assert "tier_A_per_pool" in data + + def test_x_pool_shape_matches_k_cov(self, synthetic_panel): + data = encode_covariates(synthetic_panel, include_tiers=False) + assert data["X_pool"].shape == (data["N_pools"], data["K_cov"]) + + +class TestEncodeStructuralCovariates: + """Tests for encode_covariates_structural().""" + + @pytest.fixture() + def struct_data(self, synthetic_panel): + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + return encode_covariates_structural(synthetic_panel) + + def test_encode_structural_x_obs_shape(self, struct_data, synthetic_panel): + N_obs = len(synthetic_panel) + assert struct_data["x_obs"].shape == (N_obs, K_OBS_COEFF) + + def test_encode_structural_x_obs_columns(self, struct_data, synthetic_panel): + """Columns must match OBS_COEFF_NAMES ordering: + [1, lag_log_tvl, log_sigma, tvl_x_sigma, tvl_x_fee, sigma_x_fee, + dow_sin, dow_cos].""" + x = struct_data["x_obs"] + np.testing.assert_array_equal(x[:, 0], 1.0) # intercept + np.testing.assert_array_equal( + x[:, 1], synthetic_panel["log_tvl_lag1"].values, + ) + np.testing.assert_array_equal( + x[:, 2], synthetic_panel["log_sigma"].values, + ) + np.testing.assert_allclose( + x[:, 3], synthetic_panel["tvl_x_sigma"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 4], synthetic_panel["tvl_x_fee"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 5], synthetic_panel["sigma_x_fee"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 6], synthetic_panel["dow_sin"].values, rtol=1e-12, + ) + np.testing.assert_allclose( + x[:, 7], synthetic_panel["dow_cos"].values, rtol=1e-12, + ) + + def test_encode_structural_has_sigma_daily(self, struct_data): + """sigma_daily = volatility / sqrt(365), de-annualised.""" + assert "sigma_daily" in struct_data + assert len(struct_data["sigma_daily"]) > 0 + + def test_encode_structural_has_gas(self, struct_data): + assert "gas" in struct_data + assert len(struct_data["gas"]) > 0 + assert (struct_data["gas"] >= 0).all() + + def test_encode_structural_has_chain_idx_tier_idx(self, struct_data): + assert "chain_idx" in struct_data + assert "tier_idx" in struct_data + assert struct_data["chain_idx"].dtype in (np.int32, np.int64) + assert struct_data["tier_idx"].dtype in (np.int32, np.int64) + + def test_encode_structural_has_fee(self, struct_data): + """fee array is raw (not log), for the formula.""" + assert "fee" in struct_data + assert (struct_data["fee"] > 0).all() + assert (struct_data["fee"] < 1).all() # fees are fractions + + def test_encode_structural_tier_idx_is_pair(self, struct_data): + """tier_idx encodes the (tier_A, tier_B) PAIR, not individual tokens. + (0,0)->0, (0,1)->1, (0,2)->2, (1,1)->3, (1,2)->4, (2,2)->5.""" + tier_idx = struct_data["tier_idx"] + pool_meta = struct_data["pool_meta"] + + for i, row in pool_meta.iterrows(): + a, b = int(row["tier_A"]), int(row["tier_B"]) + expected = a * (5 - a) // 2 + b - a + assert tier_idx[i] == expected, ( + f"Pool {i}: tier ({a},{b}) expected idx {expected}, " + f"got {tier_idx[i]}" + ) + + def test_encode_structural_n_chains_n_tiers(self, struct_data): + """n_chains and n_tiers computed from data.""" + assert "n_chains" in struct_data + assert "n_tiers" in struct_data + assert struct_data["n_chains"] >= 1 + assert struct_data["n_tiers"] >= 1 diff --git a/tests/noise/test_formula_arb.py b/tests/noise/test_formula_arb.py new file mode 100644 index 00000000..f862d516 --- /dev/null +++ b/tests/noise/test_formula_arb.py @@ -0,0 +1,142 @@ +"""Tests for JAX-differentiable LVR formula.""" + +import numpy as np +import pytest + + +class TestFormulaArbJax: + @pytest.fixture(autouse=True) + def _import(self): + from quantammsim.noise_calibration.formula_arb import ( + formula_arb_volume_daily_jax, + ) + self.formula = formula_arb_volume_daily_jax + + def test_formula_arb_zero_vol_returns_zero(self): + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.0), + tvl=jnp.float64(1e6), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-10) + + def test_formula_arb_zero_tvl_returns_zero(self): + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.03), + tvl=jnp.float64(0.0), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-10) + + def test_formula_arb_quadratic_in_sigma(self): + """V_arb(2σ) / V_arb(σ) ≈ 4 for small gas (correction ≈ 1).""" + import jax.numpy as jnp + sigma = jnp.float64(0.01) + tvl = jnp.float64(1e8) # large TVL so gas is negligible + fee = jnp.float64(0.003) + gas = jnp.float64(0.001) # tiny gas + cadence = jnp.float64(0.01) # very fast arb + + v1 = float(self.formula(sigma, tvl, fee, gas, cadence)) + v2 = float(self.formula(2.0 * sigma, tvl, fee, gas, cadence)) + assert v1 > 0 + ratio = v2 / v1 + assert ratio == pytest.approx(4.0, rel=0.1) + + def test_formula_arb_linear_in_tvl(self): + """V_arb(2V) / V_arb(V) ≈ 2 for small gas.""" + import jax.numpy as jnp + sigma = jnp.float64(0.02) + tvl = jnp.float64(1e8) + fee = jnp.float64(0.003) + gas = jnp.float64(0.0001) + cadence = jnp.float64(0.01) + + v1 = float(self.formula(sigma, tvl, fee, gas, cadence)) + v2 = float(self.formula(sigma, 2.0 * tvl, fee, gas, cadence)) + assert v1 > 0 + ratio = v2 / v1 + assert ratio == pytest.approx(2.0, rel=0.1) + + def test_formula_arb_gas_kills_small_pools(self): + """High gas, small TVL → V_arb ≈ 0.""" + import jax.numpy as jnp + result = self.formula( + sigma_daily=jnp.float64(0.02), + tvl=jnp.float64(1000.0), + fee=jnp.float64(0.003), + gas_usd=jnp.float64(1000.0), + cadence_minutes=jnp.float64(1.0), + ) + assert float(result) == pytest.approx(0.0, abs=1e-6) + + def test_formula_arb_matches_numpy_reference(self): + """Compare to the numpy formula in plot_formula_arb_vs_real.py.""" + import jax.numpy as jnp + + # Reference implementation (from plot_formula_arb_vs_real.py:58) + def ref(sigma_daily, tvl, fee, block_time_s, gas_usd): + if tvl <= 0 or fee <= 0 or sigma_daily <= 0: + return 0.0 + gamma = fee + delta = 2.0 * np.sqrt(2.0 * gas_usd / tvl) if gas_usd > 0 else 0.0 + bLVR = sigma_daily**2 * tvl / 8.0 + sqrt_s2_2l = sigma_daily * np.sqrt(block_time_s / (2.0 * 86400.0)) + bFEE = bLVR * max( + 1.0 - delta / (2.0 * gamma) - sqrt_s2_2l / (gamma + delta / 2.0), + 0.0, + ) + return bFEE / gamma + + test_cases = [ + (0.03, 1e6, 0.003, 1.0, 1.0), + (0.05, 5e5, 0.01, 0.5, 0.005), + (0.01, 1e7, 0.005, 2.0, 0.01), + (0.1, 1e4, 0.03, 10.0, 5.0), + (0.02, 1e5, 0.001, 1.0, 0.001), + ] + + for sigma, tvl, fee, cadence_min, gas in test_cases: + block_time_s = cadence_min * 60.0 + expected = ref(sigma, tvl, fee, block_time_s, gas) + actual = float(self.formula( + jnp.float64(sigma), jnp.float64(tvl), + jnp.float64(fee), jnp.float64(gas), + jnp.float64(cadence_min), + )) + np.testing.assert_allclose( + actual, expected, rtol=1e-10, + err_msg=f"Mismatch for sigma={sigma}, tvl={tvl}, " + f"fee={fee}, cadence={cadence_min}, gas={gas}", + ) + + def test_formula_arb_is_jax_differentiable(self): + """jax.grad w.r.t. sigma should run without error.""" + import jax + import jax.numpy as jnp + + grad_fn = jax.grad(self.formula, argnums=0) + result = grad_fn( + jnp.float64(0.03), jnp.float64(1e6), + jnp.float64(0.003), jnp.float64(1.0), + jnp.float64(1.0), + ) + assert np.isfinite(float(result)) + + def test_formula_arb_cadence_reduces_volume(self): + """Higher cadence → less frequent arb → lower volume.""" + import jax.numpy as jnp + sigma = jnp.float64(0.03) + tvl = jnp.float64(1e6) + fee = jnp.float64(0.003) + gas = jnp.float64(1.0) + + v_fast = float(self.formula(sigma, tvl, fee, gas, jnp.float64(1.0))) + v_slow = float(self.formula(sigma, tvl, fee, gas, jnp.float64(10.0))) + assert v_fast > v_slow diff --git a/tests/noise/test_model_and_inference.py b/tests/noise/test_model_and_inference.py new file mode 100644 index 00000000..ee1fda5c --- /dev/null +++ b/tests/noise/test_model_and_inference.py @@ -0,0 +1,328 @@ +"""Tests for noise_model, _get_theta_samples, _build_model_kwargs, and SVI smoke.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + noise_model, + _get_theta_samples, + _build_model_kwargs, + K_COEFF, +) + + +# =========================================================================== +# TestNoiseModelDefinition +# =========================================================================== + + +class TestNoiseModelDefinition: + def test_model_traces_without_error(self, synthetic_encoded_data): + import jax + import jax.numpy as jnp + import numpyro + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + assert trace is not None + + def test_required_sites_present(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + required = {"B", "sigma_theta", "L_Omega", "df", "sigma_eps", "eta", "y"} + assert required.issubset(trace.keys()) + + def test_site_shapes(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + N_pools = data["N_pools"] + K_cov = data["K_cov"] + + assert trace["B"]["value"].shape == (K_COEFF, K_cov) + assert trace["sigma_theta"]["value"].shape == (K_COEFF,) + assert trace["L_Omega"]["value"].shape == (K_COEFF, K_COEFF) + assert trace["df"]["value"].shape == () + assert trace["sigma_eps"]["value"].shape == (3,) + assert trace["eta"]["value"].shape == (N_pools, K_COEFF) + + def test_theta_deterministic_site(self, synthetic_encoded_data): + import jax + import numpyro.handlers as handlers + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace(handlers.seed(noise_model, rng_key)).get_trace( + **kwargs + ) + assert "theta" in trace + assert trace["theta"]["value"].shape == (data["N_pools"], K_COEFF) + + def test_prior_predictive_produces_y(self, synthetic_encoded_data): + import jax + from numpyro.infer import Predictive + + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + kwargs["y_obs"] = None + + predictive = Predictive(noise_model, num_samples=5) + rng_key = jax.random.PRNGKey(42) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 5 + + +# =========================================================================== +# TestGetThetaSamples +# =========================================================================== + + +class TestGetThetaSamples: + def test_returns_theta_directly_when_present(self, synthetic_encoded_data): + data = synthetic_encoded_data + theta_direct = np.random.randn(10, data["N_pools"], K_COEFF) + sample_dict = {"theta": theta_direct} + result = _get_theta_samples(sample_dict, data["X_pool"]) + np.testing.assert_array_equal(result, theta_direct) + + def test_reconstructs_from_non_centered( + self, synthetic_encoded_data, synthetic_samples + ): + data = synthetic_encoded_data + result = _get_theta_samples(synthetic_samples, data["X_pool"]) + assert result.shape == (10, data["N_pools"], K_COEFF) + + def test_eta_zero_identity_gives_mu( + self, synthetic_encoded_data, synthetic_samples + ): + """With eta=0 and L_Omega=I, theta = X_pool @ B^T.""" + data = synthetic_encoded_data + X_pool = data["X_pool"] + B = synthetic_samples["B"] + + result = _get_theta_samples(synthetic_samples, X_pool) + + # Expected: mu[s,p,j] = sum_d X_pool[p,d] * B[s,j,d] + expected = np.einsum("pd,sjd->spj", X_pool, B) + np.testing.assert_allclose(result, expected, atol=1e-12) + + def test_output_shape(self, synthetic_encoded_data, synthetic_samples): + data = synthetic_encoded_data + result = _get_theta_samples(synthetic_samples, data["X_pool"]) + S = synthetic_samples["B"].shape[0] + assert result.shape == (S, data["N_pools"], K_COEFF) + + def test_reconstructed_matches_direct(self, synthetic_encoded_data): + """When both theta and raw params are present, reconstruction matches.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + S = 8 + X_pool = data["X_pool"] + + np.random.seed(123) + B = np.random.randn(S, K_COEFF, K_cov) * 0.3 + sigma_theta = np.abs(np.random.randn(S, K_COEFF)) + 0.1 + # Random lower-triangular L + L_raw = np.zeros((S, K_COEFF, K_COEFF)) + for s in range(S): + A = np.random.randn(K_COEFF, K_COEFF) + L_raw[s] = np.linalg.cholesky(A @ A.T + np.eye(K_COEFF)) + eta = np.random.randn(S, N_pools, K_COEFF) + + sample_dict = { + "B": B, "sigma_theta": sigma_theta, + "L_Omega": L_raw, "eta": eta, + } + + theta_recon = _get_theta_samples(sample_dict, X_pool) + + # Compute expected directly + mu = np.einsum("pd,sjd->spj", X_pool, B) + L_Sigma = sigma_theta[:, :, None] * L_raw + offset = np.einsum("spi,sji->spj", eta, L_Sigma) + expected = mu + offset + + np.testing.assert_allclose(theta_recon, expected, atol=1e-10) + + +# =========================================================================== +# TestBuildModelKwargs +# =========================================================================== + + +class TestBuildModelKwargs: + def test_all_outputs_are_jnp(self, synthetic_encoded_data): + import jax.numpy as jnp + + kwargs = _build_model_kwargs(synthetic_encoded_data) + for key in ["pool_idx", "X_pool", "x_obs", "y_obs", "tier_A_per_pool"]: + assert isinstance(kwargs[key], jnp.ndarray), ( + f"{key} should be jnp array" + ) + + def test_shapes_preserved(self, synthetic_encoded_data): + data = synthetic_encoded_data + kwargs = _build_model_kwargs(data) + assert kwargs["pool_idx"].shape == data["pool_idx"].shape + assert kwargs["X_pool"].shape == data["X_pool"].shape + assert kwargs["x_obs"].shape == data["x_obs"].shape + assert kwargs["y_obs"].shape == data["y_obs"].shape + assert kwargs["N_pools"] == data["N_pools"] + assert kwargs["K_cov"] == data["K_cov"] + + def test_dp_model_kwargs_exclude_tier_A(self, synthetic_encoded_data): + """When model_fn is noise_model_dp_sigma, tier_A_per_pool is excluded + and K_clusters is included.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + data = dict(synthetic_encoded_data) + data["K_clusters"] = 6 + kwargs = _build_model_kwargs(data, model_fn=noise_model_dp_sigma) + assert "tier_A_per_pool" not in kwargs + assert kwargs["K_clusters"] == 6 + + def test_tier_model_kwargs_include_tier_A(self, synthetic_encoded_data): + """When model_fn is noise_model (default), tier_A_per_pool is included + and K_clusters is not.""" + kwargs = _build_model_kwargs(synthetic_encoded_data) + assert "tier_A_per_pool" in kwargs + assert "K_clusters" not in kwargs + + def test_default_model_fn_is_noise_model(self, synthetic_encoded_data): + """Calling without model_fn should behave identically to model_fn=noise_model.""" + kwargs_default = _build_model_kwargs(synthetic_encoded_data) + kwargs_explicit = _build_model_kwargs( + synthetic_encoded_data, model_fn=noise_model + ) + assert set(kwargs_default.keys()) == set(kwargs_explicit.keys()) + + +# =========================================================================== +# TestSVISmoke +# =========================================================================== + + +class TestSVISmoke: + @pytest.mark.slow + def test_svi_converges_small_data(self): + """SVI on tiny synthetic data: ELBO should decrease.""" + import jax + import numpyro + + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + + # Build minimal panel: 5 pools × 20 days + np.random.seed(42) + from datetime import date, timedelta + + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("p2", "BASE", "RATS,WETH", 0.005, 0, 2), + ("p3", "MAINNET", "LINK,WETH", 0.005, 0, 1), + ("p4", "ARBITRUM", "AAVE,USDC", 0.003, 0, 1), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(21)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + + import pandas as pd + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + data = encode_covariates(panel) + samples, losses = run_svi(data, num_steps=2000, lr=1e-3, seed=0, + num_samples=50) + + # ELBO should decrease: mean of last 100 < mean of first 100 + assert np.mean(losses[-100:]) < np.mean(losses[:100]) + + @pytest.mark.slow + def test_svi_samples_have_required_keys(self): + """SVI output dict has all expected latent variable keys.""" + import jax + import numpyro + + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + + np.random.seed(42) + from datetime import date, timedelta + import pandas as pd + + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(15)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + data = encode_covariates(panel) + samples, _ = run_svi(data, num_steps=500, lr=1e-3, seed=0, + num_samples=10) + + required = {"B", "sigma_theta", "L_Omega", "eta", "df", "sigma_eps"} + assert required.issubset(samples.keys()) diff --git a/tests/noise/test_model_dp_sigma.py b/tests/noise/test_model_dp_sigma.py new file mode 100644 index 00000000..3d2be214 --- /dev/null +++ b/tests/noise/test_model_dp_sigma.py @@ -0,0 +1,305 @@ +"""Tests for stick_breaking_weights and noise_model_dp_sigma.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import K_COEFF +from quantammsim.noise_calibration.constants import K_CLUSTERS_DEFAULT + + +# =========================================================================== +# TestStickBreakingWeights +# =========================================================================== + + +class TestStickBreakingWeights: + def test_sums_to_one(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.5, 0.3, 0.4, 0.6, 0.2]) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(jnp.sum(w)), 1.0, atol=1e-6) + + def test_correct_length(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + K = 7 + v = jnp.ones(K - 1) * 0.3 + w = stick_breaking_weights(v) + assert w.shape == (K,) + + def test_non_negative(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.1, 0.9, 0.5, 0.7, 0.3]) + w = stick_breaking_weights(v) + assert jnp.all(w >= 0.0) + + def test_first_weight_equals_first_v(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.array([0.7, 0.4, 0.2]) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(w[0]), 0.7, atol=1e-6) + + def test_jit_compatible(self): + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax + import jax.numpy as jnp + + v = jnp.array([0.5, 0.3, 0.4]) + w_eager = stick_breaking_weights(v) + w_jit = jax.jit(stick_breaking_weights)(v) + np.testing.assert_allclose( + np.array(w_eager), np.array(w_jit), atol=1e-6 + ) + + def test_all_v_one_concentrates_on_first(self): + """If v = [1, 1, ...], all mass goes to first component.""" + from quantammsim.noise_calibration.model import stick_breaking_weights + import jax.numpy as jnp + + v = jnp.ones(5) + w = stick_breaking_weights(v) + np.testing.assert_allclose(float(w[0]), 1.0, atol=1e-6) + np.testing.assert_allclose(float(jnp.sum(w[1:])), 0.0, atol=1e-6) + + +# =========================================================================== +# TestDPModelDefinition +# =========================================================================== + + +class TestDPModelDefinition: + def _get_dp_model_kwargs(self, data, K_clusters=6): + """Build kwargs for noise_model_dp_sigma from encoded data.""" + import jax.numpy as jnp + + return dict( + pool_idx=jnp.array(data["pool_idx"]), + X_pool=jnp.array(data["X_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K_coeff=K_COEFF, + K_cov=data["K_cov"], + K_clusters=K_clusters, + ) + + def test_model_traces_without_error(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + assert trace is not None + + def test_has_dp_sites(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + dp_sites = {"alpha_dp", "v", "sigma_eps", "log_lik"} + assert dp_sites.issubset(trace.keys()), ( + f"Missing DP sites: {dp_sites - trace.keys()}" + ) + + def test_no_y_site_when_obs_provided(self, synthetic_encoded_data): + """With y_obs provided, the marginalized model uses factor, not obs.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + assert "y" not in trace, ( + "DP model should use numpyro.factor, not obs=y_obs" + ) + + def test_shared_mean_structure_sites(self, synthetic_encoded_data): + """The DP model must share the same mean structure as the tier model.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + shared_sites = {"B", "sigma_theta", "L_Omega", "df", "eta", "theta"} + assert shared_sites.issubset(trace.keys()), ( + f"Missing shared sites: {shared_sites - trace.keys()}" + ) + + def test_correct_shapes(self, synthetic_encoded_data): + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + K_clusters = 6 + kwargs = self._get_dp_model_kwargs( + synthetic_encoded_data, K_clusters=K_clusters + ) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + N_pools = synthetic_encoded_data["N_pools"] + K_cov = synthetic_encoded_data["K_cov"] + + assert trace["B"]["value"].shape == (K_COEFF, K_cov) + assert trace["sigma_theta"]["value"].shape == (K_COEFF,) + assert trace["L_Omega"]["value"].shape == (K_COEFF, K_COEFF) + assert trace["df"]["value"].shape == () + assert trace["sigma_eps"]["value"].shape == (K_clusters,) + assert trace["eta"]["value"].shape == (N_pools, K_COEFF) + assert trace["v"]["value"].shape == (K_clusters - 1,) + assert trace["alpha_dp"]["value"].shape == () + + def test_no_tier_A_per_pool_in_signature(self): + """The DP model should not accept tier_A_per_pool.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import inspect + + sig = inspect.signature(noise_model_dp_sigma) + assert "tier_A_per_pool" not in sig.parameters + + def test_prior_predictive_produces_y(self, synthetic_encoded_data): + """With y_obs=None, the model should sample y explicitly.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + from numpyro.infer import Predictive + + kwargs = self._get_dp_model_kwargs(synthetic_encoded_data) + kwargs["y_obs"] = None + + predictive = Predictive(noise_model_dp_sigma, num_samples=5) + rng_key = jax.random.PRNGKey(42) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 5 + + def test_k_clusters_configurable(self, synthetic_encoded_data): + """K_clusters=4 should produce different sigma_eps shape.""" + from quantammsim.noise_calibration.model import noise_model_dp_sigma + import jax + import numpyro.handlers as handlers + + K_clusters = 4 + kwargs = self._get_dp_model_kwargs( + synthetic_encoded_data, K_clusters=K_clusters + ) + rng_key = jax.random.PRNGKey(0) + + trace = handlers.trace( + handlers.seed(noise_model_dp_sigma, rng_key) + ).get_trace(**kwargs) + + assert trace["sigma_eps"]["value"].shape == (K_clusters,) + assert trace["v"]["value"].shape == (K_clusters - 1,) + + +# =========================================================================== +# TestDPModelSVISmoke +# =========================================================================== + + +class TestDPModelSVISmoke: + def _build_dp_panel(self): + """Build a minimal panel for DP SVI testing.""" + from datetime import date, timedelta + import pandas as pd + + np.random.seed(42) + pools_spec = [ + ("p0", "MAINNET", "WETH,USDC", 0.003, 0, 0), + ("p1", "ARBITRUM", "BAL,WETH", 0.01, 0, 1), + ("p2", "BASE", "RATS,WETH", 0.005, 0, 2), + ] + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(15)] + records = [] + for pid, chain, tokens, fee, ta, tb in pools_spec: + base_tvl = 14 + np.random.randn() * 0.5 + for d in dates: + tvl = base_tvl + np.random.randn() * 0.1 + vol = tvl - 2 + np.random.randn() * 0.3 + records.append({ + "pool_id": pid, "chain": chain, "date": d, + "log_volume": vol, "log_tvl": tvl, + "volatility": 0.3 + np.random.rand() * 0.2, + "weekend": 1.0 if d.weekday() >= 5 else 0.0, + "log_fee": np.log(max(fee, 1e-6)), + "swap_fee": fee, "tier_A": ta, "tier_B": tb, + "tokens": tokens, + }) + panel = pd.DataFrame(records) + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + return panel + + def test_svi_converges_dp_model(self): + """SVI on DP model with tiny data: ELBO should decrease.""" + import numpyro + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + panel = self._build_dp_panel() + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = 4 + + samples, losses = run_svi( + data, num_steps=2000, lr=1e-3, seed=0, + num_samples=50, model_fn=noise_model_dp_sigma, + ) + assert np.mean(losses[-100:]) < np.mean(losses[:100]) + + def test_svi_samples_have_dp_keys(self): + """SVI output for DP model has v, alpha_dp, sigma_eps.""" + import numpyro + numpyro.enable_x64() + + from quantammsim.noise_calibration import encode_covariates, run_svi + from quantammsim.noise_calibration.model import noise_model_dp_sigma + + panel = self._build_dp_panel() + data = encode_covariates(panel, include_tiers=False) + data["K_clusters"] = 4 + + samples, _ = run_svi( + data, num_steps=500, lr=1e-3, seed=0, + num_samples=10, model_fn=noise_model_dp_sigma, + ) + required = {"B", "sigma_theta", "L_Omega", "eta", "df", + "sigma_eps", "v", "alpha_dp"} + assert required.issubset(samples.keys()), ( + f"Missing keys: {required - samples.keys()}" + ) diff --git a/tests/noise/test_model_structural.py b/tests/noise/test_model_structural.py new file mode 100644 index 00000000..582cdd5b --- /dev/null +++ b/tests/noise/test_model_structural.py @@ -0,0 +1,190 @@ +"""Tests for structural_noise_model definition and SVI integration.""" + +import numpy as np +import pytest + + +class TestStructuralModelDefinition: + """Tests for the structural_noise_model numpyro model.""" + + def test_model_traces_without_error(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + rng_key = jax.random.PRNGKey(0) + samples = predictive(rng_key, **kwargs) + assert "y" in samples + + def test_required_sites_present(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + required = { + "alpha_0", "alpha_chain", "alpha_tier", "alpha_tvl", + "B", "sigma_theta", "L_Omega", "eta", "theta", + "df", "sigma_eps", "y", + } + assert required.issubset(samples.keys()), ( + f"Missing sites: {required - samples.keys()}" + ) + + def test_no_moe_sites(self, synthetic_structural_data): + """MoE sites (W_gate, beta) should not be present.""" + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=2) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + for site in ("W_gate", "beta"): + assert site not in samples, f"MoE site '{site}' should not be present" + + def test_theta_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + N_pools = synthetic_structural_data["N_pools"] + K_obs = synthetic_structural_data["x_obs"].shape[1] + assert samples["theta"].shape == (3, N_pools, K_obs) + + def test_B_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + K_obs = synthetic_structural_data["x_obs"].shape[1] + K_cov = synthetic_structural_data["K_cov"] + assert samples["B"].shape == (3, K_obs, K_cov) + + def test_alpha_chain_shape(self, synthetic_structural_data): + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + predictive = Predictive(structural_noise_model, num_samples=3) + samples = predictive(jax.random.PRNGKey(0), **kwargs) + + n_chains = synthetic_structural_data["n_chains"] + assert samples["alpha_chain"].shape == (3, n_chains - 1) + + def test_prior_predictive_produces_y(self, synthetic_structural_data): + """y_obs=None path produces y samples.""" + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + kwargs["y_obs"] = None + predictive = Predictive(structural_noise_model, num_samples=10) + samples = predictive(jax.random.PRNGKey(42), **kwargs) + assert "y" in samples + assert samples["y"].shape[0] == 10 + + def test_prior_predictive_range_reasonable(self, synthetic_structural_data): + """Prior y should contain finite values in a plausible range.""" + import jax + from numpyro.infer import Predictive + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import _build_model_kwargs + + kwargs = _build_model_kwargs(synthetic_structural_data, + model_fn=structural_noise_model) + kwargs["y_obs"] = None + predictive = Predictive(structural_noise_model, num_samples=200) + samples = predictive(jax.random.PRNGKey(42), **kwargs) + y = np.array(samples["y"]) + finite = y[np.isfinite(y)] + assert len(finite) > 0.5 * y.size, "Too many non-finite prior samples" + median = np.median(finite) + assert median > -50, f"Prior y median too low: {median}" + assert median < 100, f"Prior y median too high: {median}" + + +class TestSVIStructural: + """SVI convergence tests for structural model.""" + + @pytest.mark.slow + def test_svi_structural_converges(self, synthetic_structural_data): + """2000 SVI steps, ELBO should decrease.""" + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, losses = run_svi( + synthetic_structural_data, + num_steps=2000, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + assert losses[-100:].mean() < losses[:100].mean() + + @pytest.mark.slow + def test_svi_structural_samples_have_required_keys( + self, synthetic_structural_data, + ): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + required = {"B", "sigma_theta", "L_Omega", "eta", + "alpha_0", "alpha_chain", + "alpha_tier", "alpha_tvl", "df", "sigma_eps"} + assert required.issubset(samples.keys()), ( + f"Missing keys: {required - samples.keys()}" + ) + + @pytest.mark.slow + def test_svi_structural_no_moe_keys(self, synthetic_structural_data): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + for key in ("W_gate", "beta"): + assert key not in samples, f"MoE key '{key}' should not be in samples" diff --git a/tests/noise/test_output.py b/tests/noise/test_output.py new file mode 100644 index 00000000..2246c984 --- /dev/null +++ b/tests/noise/test_output.py @@ -0,0 +1,305 @@ +"""Tests for generate_output_json and _save_sample_cache.""" + +import json +import os + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + extract_noise_params, + generate_output_json, + _save_sample_cache, + K_COEFF, +) + + +# =========================================================================== +# TestGenerateOutputJSON +# =========================================================================== + + +class TestGenerateOutputJSON: + @pytest.fixture() + def _output_setup(self, tmp_path, synthetic_samples, synthetic_encoded_data): + """Produce the JSON file and return (path, data dict).""" + pool_params = extract_noise_params( + synthetic_samples, synthetic_encoded_data + ) + output_path = str(tmp_path / "test_output.json") + convergence = {"method": "svi", "final_elbo": 1234.0} + inference_config = {"method": "svi", "svi_steps": 1000} + + generate_output_json( + pool_params, synthetic_samples, synthetic_encoded_data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + data = json.load(f) + return output_path, data + + def test_writes_valid_json(self, _output_setup): + path, data = _output_setup + assert isinstance(data, dict) + + def test_top_level_keys(self, _output_setup): + _, data = _output_setup + expected = { + "model", "model_spec", "inference", "population_effects", + "convergence", "n_pools", "n_obs", "pools", + } + assert expected.issubset(data.keys()) + + def test_model_spec_fields(self, _output_setup): + _, data = _output_setup + spec = data["model_spec"] + assert "K_coeff" in spec + assert "K_cov" in spec + assert "coeff_names" in spec + assert "covariate_names" in spec + assert "likelihood" in spec + assert spec["likelihood"] == "StudentT" + assert "tvl_lag" in spec + assert spec["tvl_lag"] == "log_tvl_lag1" + + def test_population_effects_fields(self, _output_setup): + _, data = _output_setup + pe = data["population_effects"] + assert "B" in pe + assert "sigma_theta" in pe + assert "sigma_eps" in pe + assert "df" in pe + assert "correlation_matrix" in pe + + def test_pool_entries(self, _output_setup, synthetic_encoded_data): + _, data = _output_setup + pools = data["pools"] + for pid in synthetic_encoded_data["pool_ids"]: + assert pid in pools + entry = pools[pid] + assert "chain" in entry + assert "tokens" in entry + assert "theta_median" in entry + assert "noise_params" in entry + + def test_correlation_matrix_symmetric_unit_diagonal(self, _output_setup): + _, data = _output_setup + Omega = np.array(data["population_effects"]["correlation_matrix"]) + np.testing.assert_allclose(Omega, Omega.T, atol=1e-10) + np.testing.assert_allclose(np.diag(Omega), 1.0, atol=1e-10) + + def test_n_pools_and_n_obs(self, _output_setup, synthetic_encoded_data): + _, data = _output_setup + assert data["n_pools"] == synthetic_encoded_data["N_pools"] + assert data["n_obs"] == len(synthetic_encoded_data["y_obs"]) + + def test_model_name(self, _output_setup): + _, data = _output_setup + assert data["model"] == "unified_hierarchical_student_t" + + +# =========================================================================== +# TestSaveSampleCache +# =========================================================================== + + +class TestSaveSampleCache: + def test_creates_npz_and_json( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + cache_dir = str(tmp_path / "cache") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + assert os.path.exists(os.path.join(cache_dir, "unified_samples.npz")) + assert os.path.exists(os.path.join(cache_dir, "unified_data.json")) + + def test_npz_excludes_y_and_theta( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """Both 'y' and 'theta' must be excluded from the npz cache.""" + samples_with_extras = dict(synthetic_samples) + samples_with_extras["y"] = np.random.randn(10, 27) + samples_with_extras["theta"] = np.random.randn(10, 3, 4) + + cache_dir = str(tmp_path / "cache2") + _save_sample_cache(samples_with_extras, synthetic_encoded_data, cache_dir) + + npz_path = os.path.join(cache_dir, "unified_samples.npz") + loaded = np.load(npz_path) + assert "y" not in loaded.files + assert "theta" not in loaded.files + + def test_npz_contains_required_keys( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """The npz must contain B, sigma_theta, L_Omega, eta, df, sigma_eps.""" + cache_dir = str(tmp_path / "cache3") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + npz_path = os.path.join(cache_dir, "unified_samples.npz") + loaded = np.load(npz_path) + required = {"B", "sigma_theta", "L_Omega", "eta", "df", "sigma_eps"} + assert required.issubset(set(loaded.files)) + + def test_json_contains_metadata_keys( + self, tmp_path, synthetic_samples, synthetic_encoded_data + ): + """The JSON cache must contain all keys needed by --predict.""" + cache_dir = str(tmp_path / "cache4") + _save_sample_cache(synthetic_samples, synthetic_encoded_data, cache_dir) + + json_path = os.path.join(cache_dir, "unified_data.json") + with open(json_path) as f: + meta = json.load(f) + required = { + "pool_ids", "covariate_names", "K_cov", "N_pools", + "ref_chain", "ref_tier_a", "ref_tier_b", "chains", + } + assert required.issubset(meta.keys()) + + +# =========================================================================== +# TestGenerateOutputJSONDP +# =========================================================================== + + +class TestGenerateOutputJSONDP: + @pytest.fixture() + def _dp_output_setup(self, tmp_path, synthetic_encoded_data): + """Produce DP model JSON output and return (path, data dict).""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_clusters = 4 + S = 10 + + np.random.seed(99) + dp_samples = { + "B": np.random.randn(S, K_COEFF, K_cov) * 0.5, + "sigma_theta": np.ones((S, K_COEFF)), + "L_Omega": np.tile(np.eye(K_COEFF), (S, 1, 1)), + "eta": np.zeros((S, N_pools, K_COEFF)), + "df": np.full((S,), 5.0), + "sigma_eps": np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)), + "v": np.tile([0.6, 0.3, 0.05], (S, 1)), + "alpha_dp": np.full((S,), 1.5), + } + + pool_params = extract_noise_params(dp_samples, data) + output_path = str(tmp_path / "dp_output.json") + convergence = {"method": "svi", "final_elbo": 1234.0} + inference_config = {"method": "svi", "svi_steps": 1000} + + generate_output_json( + pool_params, dp_samples, data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + result = json.load(f) + return output_path, result + + def test_sigma_eps_structure_is_dp_mixture(self, _dp_output_setup): + _, data = _dp_output_setup + assert data["model_spec"]["sigma_eps_structure"] == "dp_mixture" + + def test_model_name_includes_dp(self, _dp_output_setup): + _, data = _dp_output_setup + assert "dp_sigma" in data["model"] + + def test_has_cluster_weights(self, _dp_output_setup): + _, data = _dp_output_setup + assert "cluster_weights" in data["population_effects"] + + def test_sigma_eps_length_equals_k_clusters(self, _dp_output_setup): + _, data = _dp_output_setup + sigma_eps = data["population_effects"]["sigma_eps"] + assert len(sigma_eps) == 4 # K_clusters = 4 + + def test_cluster_weights_sum_to_one(self, _dp_output_setup): + _, data = _dp_output_setup + w = data["population_effects"]["cluster_weights"] + np.testing.assert_allclose(sum(w), 1.0, atol=1e-4) + + def test_still_has_standard_fields(self, _dp_output_setup): + _, data = _dp_output_setup + expected = { + "model", "model_spec", "inference", "population_effects", + "convergence", "n_pools", "n_obs", "pools", + } + assert expected.issubset(data.keys()) + + +# =========================================================================== +# TestGenerateOutputJSONStructural +# =========================================================================== + + +class TestGenerateOutputJSONStructural: + @pytest.fixture() + def _structural_output_setup(self, tmp_path, synthetic_structural_data): + """Produce structural model JSON output and return (path, data dict).""" + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + from quantammsim.noise_calibration.constants import K_OBS_COEFF + + data = synthetic_structural_data + K_cov = data["K_cov"] + N_pools = data["N_pools"] + n_chains = data["n_chains"] + n_tiers = data["n_tiers"] + S = 10 + + np.random.seed(77) + structural_samples = { + "alpha_0": np.random.randn(S) * 0.1 + 2.0, + "alpha_chain": np.random.randn(S, n_chains - 1) * 0.1, + "alpha_tier": np.random.randn(S, n_tiers - 1) * 0.1, + "alpha_tvl": np.random.randn(S) * 0.01, + "B": np.random.randn(S, K_OBS_COEFF, K_cov) * 0.5, + "sigma_theta": np.ones((S, K_OBS_COEFF)), + "L_Omega": np.tile(np.eye(K_OBS_COEFF), (S, 1, 1)), + "eta": np.zeros((S, N_pools, K_OBS_COEFF)), + "df": np.full((S,), 5.0), + "sigma_eps": np.tile([0.5, 0.8, 0.6], (S, 1)), + } + + pool_params = extract_structural_params(structural_samples, data) + output_path = str(tmp_path / "structural_output.json") + convergence = {"method": "svi", "final_elbo": 999.0} + inference_config = {"method": "svi", "svi_steps": 2000} + + generate_output_json( + pool_params, structural_samples, data, + convergence, output_path, inference_config, + ) + with open(output_path) as f: + result = json.load(f) + return output_path, result + + def test_output_model_name(self, _structural_output_setup): + _, data = _structural_output_setup + assert data["model"] == "structural_mixture" + + def test_output_has_arb_params(self, _structural_output_setup): + _, data = _structural_output_setup + pe = data["population_effects"] + assert "alpha_0" in pe + assert "alpha_chain" in pe + assert "alpha_tier" in pe + assert "alpha_tvl" in pe + + def test_output_has_hierarchical_noise_params(self, _structural_output_setup): + _, data = _structural_output_setup + pe = data["population_effects"] + assert "B" in pe + assert "sigma_theta" in pe + assert "correlation_matrix" in pe + + def test_output_pools_have_arb_frequency(self, _structural_output_setup): + _, data = _structural_output_setup + pools = data["pools"] + for pid, entry in pools.items(): + assert "arb_frequency" in entry + assert isinstance(entry["arb_frequency"], int) + assert 1 <= entry["arb_frequency"] <= 60 diff --git a/tests/noise/test_panel_assembly.py b/tests/noise/test_panel_assembly.py new file mode 100644 index 00000000..dd7cd315 --- /dev/null +++ b/tests/noise/test_panel_assembly.py @@ -0,0 +1,442 @@ +"""Tests for compute_pair_volatility, assemble_panel, validate_panel.""" + +from datetime import date, timedelta + +import numpy as np +import pandas as pd +import pytest + +from quantammsim.noise_calibration import ( + compute_pair_volatility, + assemble_panel, + validate_panel, +) + + +# =========================================================================== +# TestComputePairVolatility +# =========================================================================== + + +class TestComputePairVolatility: + @pytest.fixture() + def _snap_dates(self): + """10 unique dates for snapshot stub.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(10)] + return pd.DataFrame({"date": dates}) + + def test_stablecoin_pair_returns_001(self, _snap_dates): + pool_row = pd.Series({ + "tokens": ["USDC", "DAI"], + "chain": "MAINNET", + }) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert (vol == 0.01).all() + + def test_both_missing_non_stable(self, _snap_dates): + pool_row = pd.Series({ + "tokens": ["FOO", "BAR"], + "chain": "MAINNET", + }) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert (vol == 0.5).all() + + def test_one_missing_non_stable(self, _snap_dates): + np.random.seed(77) + pool_row = pd.Series({ + "tokens": ["WETH", "BAR"], + "chain": "MAINNET", + }) + prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": [1735689600 + i * 3600 for i in range(48)], + "price": [3000.0 + np.random.randn() * 10 for _ in range(48)], + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, prices) + assert (vol == 0.5).all() + + def test_synthetic_hourly_prices_positive_finite(self, _snap_dates): + np.random.seed(42) + n_hours = 240 # 10 days x 24 hours + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + prices_a = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 8) + prices_b = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 7) + + pool_row = pd.Series({ + "tokens": ["WETH", "LINK"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": prices_a, + }), + ("MAINNET", "LINK"): pd.DataFrame({ + "timestamp": timestamps, "price": prices_b, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + assert len(vol) > 0 + assert (vol > 0).all() + assert np.all(np.isfinite(vol)) + + def test_annualisation_uses_sqrt_24x365(self, _snap_dates): + """Verify the annualisation factor is sqrt(24*365).""" + np.random.seed(7) + n_hours = 240 + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + raw_prices = np.exp(np.cumsum(np.random.randn(n_hours) * 0.005) + 8) + + pool_row = pd.Series({ + "tokens": ["WETH", "USDC"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": raw_prices, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + + # Reconstruct manually + df = pd.DataFrame({"timestamp": timestamps, "price": raw_prices}) + df["datetime"] = pd.to_datetime(df["timestamp"], unit="s") + df["date"] = df["datetime"].dt.date + df["ratio"] = df["price"] # WETH vs stable => ratio = price + df["log_return"] = np.log(df["ratio"] / df["ratio"].shift(1)) + df = df.dropna(subset=["log_return"]) + daily_std = df.groupby("date")["log_return"].std() + expected = daily_std * np.sqrt(24 * 365) + + common = vol.index.intersection(expected.index) + assert len(common) > 0 + np.testing.assert_allclose( + vol.loc[common].values, expected.loc[common].values, rtol=1e-10, + ) + + def test_single_token_pool_returns_empty(self, _snap_dates): + pool_row = pd.Series({"tokens": ["WETH"], "chain": "MAINNET"}) + vol = compute_pair_volatility(_snap_dates, pool_row, {}) + assert len(vol) == 0 + + def test_no_overlapping_price_dates(self, _snap_dates): + """Two tokens with non-overlapping timestamps -> fallback 0.5.""" + pool_row = pd.Series({ + "tokens": ["WETH", "LINK"], + "chain": "MAINNET", + }) + token_prices = { + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": [1000000 + i for i in range(10)], + "price": [3000.0] * 10, + }), + ("MAINNET", "LINK"): pd.DataFrame({ + "timestamp": [9000000 + i for i in range(10)], + "price": [15.0] * 10, + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + assert (vol == 0.5).all() + + def test_cross_chain_price_fallback(self, _snap_dates): + """Prices keyed to a different chain SHOULD be used as fallback. + + Token prices are chain-agnostic (WETH is WETH regardless of chain), + so an ARBITRUM pool should use MAINNET WETH prices if ARBITRUM + prices aren't available. + """ + np.random.seed(55) + n_hours = 240 + ts_base = 1735689600 + timestamps = [ts_base + i * 3600 for i in range(n_hours)] + prices = np.exp(np.cumsum(np.random.randn(n_hours) * 0.01) + 8) + + pool_row = pd.Series({ + "tokens": ["WETH", "USDC"], + "chain": "ARBITRUM", + }) + token_prices = { + # Only MAINNET prices, but ARBITRUM pool should still use them + ("MAINNET", "WETH"): pd.DataFrame({ + "timestamp": timestamps, "price": prices.tolist(), + }), + } + vol = compute_pair_volatility(_snap_dates, pool_row, token_prices) + # Should get real volatility, NOT the 0.5 fallback + assert len(vol) > 0 + assert not (vol == 0.5).all(), ( + "Cross-chain price fallback should have produced real volatility" + ) + + +# =========================================================================== +# TestAssemblePanel +# =========================================================================== + + +class TestAssemblePanel: + def test_lagged_tvl_drops_first_obs_per_pool( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + counts = panel.groupby("pool_id").size() + assert (counts == 9).all() + + def test_lagged_tvl_exact_values( + self, synthetic_pools_df, synthetic_snapshots_df + ): + """log_tvl_lag1[t] must equal log_tvl[t-1] for same pool (shift(1)).""" + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + for pid in panel["pool_id"].unique(): + pool = panel[panel["pool_id"] == pid].sort_values("date") + tvl_vals = pool["log_tvl"].values + lag_vals = pool["log_tvl_lag1"].values + # After dropping the first obs, lag[i] = tvl[i-1] in the original + # pre-drop series. Since the panel is sorted by date, each + # lag value should equal the log_tvl of the chronologically + # preceding observation. We verify consecutive pairs: for rows + # i and i+1, lag[i+1] == tvl[i]. + for i in range(len(tvl_vals) - 1): + np.testing.assert_allclose( + lag_vals[i + 1], tvl_vals[i], rtol=1e-14, + err_msg=f"Pool {pid} row {i+1}: lag should equal previous tvl", + ) + + def test_weekend_flag(self, synthetic_pools_df, synthetic_snapshots_df): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + for _, row in panel.iterrows(): + d = row["date"] + if not isinstance(d, date): + d = pd.Timestamp(d).date() + expected = 1.0 if d.weekday() >= 5 else 0.0 + assert row["weekend"] == expected, f"Wrong weekend flag for {d}" + + def test_log_volume_is_natural_log( + self, synthetic_pools_df, synthetic_snapshots_df + ): + """log_volume must equal ln(volume_usd), not log10 or log2.""" + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + snaps = synthetic_snapshots_df.copy() + # Join snapshots to panel by pool_id + date to verify exact log values + for _, row in panel.iterrows(): + pid = row["pool_id"] + d = row["date"] + snap_match = snaps[ + (snaps["pool_id"] == pid) & (snaps["date"] == d) + ] + assert len(snap_match) == 1, f"No snapshot for {pid} on {d}" + expected = np.log(snap_match.iloc[0]["volume_usd"]) + np.testing.assert_allclose( + row["log_volume"], expected, rtol=1e-14, + err_msg=f"log_volume for {pid} on {d} should be ln(volume_usd)", + ) + + def test_tier_assignment_min_first( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + # Pool A: WETH(0), USDC(0) -> tier_A=0, tier_B=0 + pool_a = panel[panel["pool_id"] == "pool_A"].iloc[0] + assert pool_a["tier_A"] == 0 + assert pool_a["tier_B"] == 0 + + # Pool B: BAL(1), WETH(0) -> tier_A=0, tier_B=1 + pool_b = panel[panel["pool_id"] == "pool_B"].iloc[0] + assert pool_b["tier_A"] == 0 + assert pool_b["tier_B"] == 1 + + # Pool C: RATS(2), WETH(0) -> tier_A=0, tier_B=2 + pool_c = panel[panel["pool_id"] == "pool_C"].iloc[0] + assert pool_c["tier_A"] == 0 + assert pool_c["tier_B"] == 2 + + def test_log_fee(self, synthetic_pools_df, synthetic_snapshots_df): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + pool_a = panel[panel["pool_id"] == "pool_A"].iloc[0] + assert np.isclose(pool_a["log_fee"], np.log(0.003)) + + def test_all_expected_columns( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + expected_cols = { + "pool_id", "chain", "date", "log_volume", "log_tvl", + "log_tvl_lag1", "volatility", "weekend", "log_fee", + "swap_fee", "tier_A", "tier_B", "tokens", + } + assert expected_cols.issubset(set(panel.columns)) + + def test_zero_volume_rows_dropped(self, synthetic_pools_df): + """Rows with volume_usd <= 0 must be excluded from the panel.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(5)] + records = [] + for d in dates: + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", "date": d, + "volume_usd": 1000.0, + "total_liquidity_usd": 100000.0, + }) + # Add a zero-volume row + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", + "date": date(2026, 1, 6), + "volume_usd": 0.0, + "total_liquidity_usd": 100000.0, + }) + snaps = pd.DataFrame(records) + panel = assemble_panel(synthetic_pools_df, snaps, {}) + pool_a = panel[panel["pool_id"] == "pool_A"] + # 5 valid rows, minus 1 for lag = 4 (the zero-volume row is skipped) + assert len(pool_a) == 4 + + def test_zero_tvl_rows_dropped(self, synthetic_pools_df): + """Rows with total_liquidity_usd <= 0 must be excluded.""" + dates = [date(2026, 1, 1) + timedelta(days=i) for i in range(5)] + records = [] + for d in dates: + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", "date": d, + "volume_usd": 1000.0, + "total_liquidity_usd": 100000.0, + }) + # Add a zero-TVL row + records.append({ + "pool_id": "pool_A", "chain": "MAINNET", + "date": date(2026, 1, 6), + "volume_usd": 1000.0, + "total_liquidity_usd": 0.0, + }) + snaps = pd.DataFrame(records) + panel = assemble_panel(synthetic_pools_df, snaps, {}) + pool_a = panel[panel["pool_id"] == "pool_A"] + assert len(pool_a) == 4 + + +# =========================================================================== +# TestSyntheticPanelColumns — structural model covariates in the fixture +# =========================================================================== + + +class TestSyntheticPanelColumns: + """Tests that the synthetic_panel fixture has new structural columns.""" + + def test_panel_has_log_sigma(self, synthetic_panel): + assert "log_sigma" in synthetic_panel.columns + expected = np.log(np.maximum(synthetic_panel["volatility"].values, 1e-6)) + np.testing.assert_allclose( + synthetic_panel["log_sigma"].values, expected, rtol=1e-12, + ) + + def test_panel_has_dow_harmonics(self, synthetic_panel): + assert "dow_sin" in synthetic_panel.columns + assert "dow_cos" in synthetic_panel.columns + assert (synthetic_panel["dow_sin"] >= -1.0).all() + assert (synthetic_panel["dow_sin"] <= 1.0).all() + assert (synthetic_panel["dow_cos"] >= -1.0).all() + assert (synthetic_panel["dow_cos"] <= 1.0).all() + + def test_panel_has_interactions(self, synthetic_panel): + assert "tvl_x_sigma" in synthetic_panel.columns + assert "tvl_x_fee" in synthetic_panel.columns + assert "sigma_x_fee" in synthetic_panel.columns + + def test_dow_harmonics_correct_for_known_date(self, synthetic_panel): + """2026-01-03 is Saturday (weekday=5), so dow=5.""" + sat_rows = synthetic_panel[ + synthetic_panel["date"] == date(2026, 1, 3) + ] + if len(sat_rows) == 0: + pytest.skip("No Saturday rows in fixture") + expected_sin = np.sin(2 * np.pi * 5 / 7) + expected_cos = np.cos(2 * np.pi * 5 / 7) + np.testing.assert_allclose( + sat_rows["dow_sin"].values[0], expected_sin, atol=1e-12, + ) + np.testing.assert_allclose( + sat_rows["dow_cos"].values[0], expected_cos, atol=1e-12, + ) + + def test_interactions_use_lagged_tvl(self, synthetic_panel): + """tvl_x_sigma must use log_tvl_lag1, not log_tvl.""" + expected = ( + synthetic_panel["log_tvl_lag1"].values + * synthetic_panel["log_sigma"].values + ) + np.testing.assert_allclose( + synthetic_panel["tvl_x_sigma"].values, expected, rtol=1e-12, + ) + + +# =========================================================================== +# TestAssemblePanelStructuralColumns — new columns from real pipeline +# =========================================================================== + + +class TestAssemblePanelStructuralColumns: + """Tests that assemble_panel() produces new structural columns.""" + + def test_assemble_panel_has_log_sigma( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "log_sigma" in panel.columns + expected = np.log(np.maximum(panel["volatility"].values, 1e-6)) + np.testing.assert_allclose( + panel["log_sigma"].values, expected, rtol=1e-12, + ) + + def test_assemble_panel_has_dow_harmonics( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "dow_sin" in panel.columns + assert "dow_cos" in panel.columns + + def test_assemble_panel_has_interactions( + self, synthetic_pools_df, synthetic_snapshots_df + ): + panel = assemble_panel(synthetic_pools_df, synthetic_snapshots_df, {}) + assert "tvl_x_sigma" in panel.columns + assert "tvl_x_fee" in panel.columns + assert "sigma_x_fee" in panel.columns + # Interactions use lagged TVL + expected = panel["log_tvl_lag1"].values * panel["log_sigma"].values + np.testing.assert_allclose( + panel["tvl_x_sigma"].values, expected, rtol=1e-12, + ) + + +# =========================================================================== +# TestValidatePanel +# =========================================================================== + + +class TestValidatePanel: + def test_flags_constant_volume(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + mask = panel["pool_id"] == "pool_A" + panel.loc[mask, "log_volume"] = 10.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "near-constant" in captured.out or "constant" in captured.out.lower() + + def test_flags_tvl_jumps(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + idx = panel[panel["pool_id"] == "pool_A"].index[1] + panel.loc[idx, "log_tvl"] = panel.loc[idx, "log_tvl"] + 5.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "TVL jumps" in captured.out + + def test_flags_volume_exceeds_tvl(self, synthetic_panel, capsys): + panel = synthetic_panel.copy() + panel["log_volume"] = panel["log_tvl"] + 1.0 + validate_panel(panel) + captured = capsys.readouterr() + assert "volume > TVL" in captured.out + + def test_returns_dataframe_unchanged(self, synthetic_panel): + result = validate_panel(synthetic_panel) + pd.testing.assert_frame_equal(result, synthetic_panel) diff --git a/tests/noise/test_postprocessing.py b/tests/noise/test_postprocessing.py new file mode 100644 index 00000000..c40e5dd9 --- /dev/null +++ b/tests/noise/test_postprocessing.py @@ -0,0 +1,533 @@ +"""Tests for extract_noise_params, predict_new_pool, check_convergence, +assign_dp_clusters, and structural model post-processing.""" + +import numpy as np +import pytest + +from quantammsim.noise_calibration import ( + extract_noise_params, + predict_new_pool, + check_convergence, + classify_token_tier, + _get_theta_samples, + K_COEFF, + COEFF_NAMES, +) +from quantammsim.noise_calibration.constants import K_OBS_COEFF, OBS_COEFF_NAMES + + +# =========================================================================== +# TestExtractNoiseParams +# =========================================================================== + + +class TestExtractNoiseParams: + def test_output_length(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + assert len(result) == synthetic_encoded_data["N_pools"] + + def test_weekend_absorption(self, synthetic_samples, synthetic_encoded_data): + """b_0_eff = b_0_raw + b_weekend * (2/7).""" + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + + theta = _get_theta_samples( + synthetic_samples, synthetic_encoded_data["X_pool"] + ) + theta_med = np.median(theta, axis=0) + + for i, p in enumerate(result): + b_0_raw = theta_med[i, 0] + b_weekend = theta_med[i, 3] + expected_b_0 = b_0_raw + b_weekend * (2.0 / 7.0) + np.testing.assert_allclose( + p["noise_params"]["b_0"], expected_b_0, atol=1e-10, + ) + + def test_noise_params_keys(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + expected_keys = {"b_0", "b_sigma", "b_c", "b_weekend", "base_fee"} + for p in result: + assert set(p["noise_params"].keys()) == expected_keys + + def test_theta_median_length(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + assert len(p["theta_median"]) == K_COEFF + + def test_b_c_equals_theta_1(self, synthetic_samples, synthetic_encoded_data): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + np.testing.assert_allclose( + p["noise_params"]["b_c"], p["theta_median"][1], atol=1e-10, + ) + + def test_b_sigma_equals_theta_2( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + for p in result: + np.testing.assert_allclose( + p["noise_params"]["b_sigma"], p["theta_median"][2], atol=1e-10, + ) + + def test_base_fee_matches_pool( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + pool_meta = synthetic_encoded_data["pool_meta"] + for i, p in enumerate(result): + expected_fee = pool_meta.iloc[i]["swap_fee"] + np.testing.assert_allclose( + p["noise_params"]["base_fee"], expected_fee, atol=1e-10, + ) + + def test_pool_id_and_chain_preserved( + self, synthetic_samples, synthetic_encoded_data + ): + result = extract_noise_params(synthetic_samples, synthetic_encoded_data) + pool_ids = synthetic_encoded_data["pool_ids"] + pool_meta = synthetic_encoded_data["pool_meta"] + for i, p in enumerate(result): + assert p["pool_id"] == pool_ids[i] + assert p["chain"] == str(pool_meta.iloc[i]["chain"]) + + def test_use_median_false_uses_mean( + self, synthetic_samples, synthetic_encoded_data + ): + """use_median=False must produce different values than use_median=True.""" + result_med = extract_noise_params( + synthetic_samples, synthetic_encoded_data, use_median=True + ) + result_mean = extract_noise_params( + synthetic_samples, synthetic_encoded_data, use_median=False + ) + assert len(result_mean) == len(result_med) + + # With random B samples (S=10, seed 99), median != mean for at least + # one pool. Check that at least one theta_median value differs. + any_differ = False + for pm, pn in zip(result_med, result_mean): + for tm, tn in zip(pm["theta_median"], pn["theta_median"]): + if not np.isclose(tm, tn, atol=1e-14): + any_differ = True + break + if any_differ: + break + assert any_differ, "median and mean paths produced identical values" + + +# =========================================================================== +# TestPredictNewPool +# =========================================================================== + + +class TestPredictNewPool: + def _build_z_new(self, data, chain, tokens, fee): + """Reconstruct z_new the same way predict_new_pool does internally.""" + col_names = data["covariate_names"] + z_new = np.zeros(len(col_names), dtype=np.float64) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + z_new[i] = 1.0 + elif name == "log_fee": + z_new[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + z_new[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z_new[i] = 1.0 + elif name == f"tier_B_{tier_b}": + z_new[i] = 1.0 + return z_new + + def test_known_chain_sets_dummy( + self, synthetic_samples, synthetic_encoded_data + ): + """ARBITRUM dummy must be 1 and must affect the prediction vs MAINNET.""" + data = synthetic_encoded_data + + result_arb = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "USDC"], fee=0.003, + ) + result_main = predict_new_pool( + synthetic_samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], fee=0.003, + ) + # MAINNET is the reference chain (alphabetically: ARBITRUM < BASE < MAINNET). + # Wait — ARBITRUM is alphabetically first, so ARBITRUM is the reference. + # MAINNET has a chain_MAINNET dummy. Both should produce different mu. + # If chain dummies are ignored, these would be identical. + arb_b0 = result_arb["noise_params"]["b_0"] + main_b0 = result_main["noise_params"]["b_0"] + assert not np.isclose(arb_b0, main_b0, atol=1e-10), ( + "Different chains should produce different predictions" + ) + + def test_tier_assignment_affects_prediction( + self, synthetic_samples, synthetic_encoded_data + ): + """WETH/RATS (tier 0,2) vs WETH/USDC (tier 0,0) must differ.""" + data = synthetic_encoded_data + + result_rats = predict_new_pool( + synthetic_samples, data, + chain="BASE", tokens=["WETH", "RATS"], fee=0.005, + ) + result_usdc = predict_new_pool( + synthetic_samples, data, + chain="BASE", tokens=["WETH", "USDC"], fee=0.005, + ) + # Different tier_B dummies should give different mu + rats_b0 = result_rats["noise_params"]["b_0"] + usdc_b0 = result_usdc["noise_params"]["b_0"] + assert not np.isclose(rats_b0, usdc_b0, atol=1e-10), ( + "Different tier assignments should produce different predictions" + ) + + def test_weekend_absorption_arithmetic( + self, synthetic_samples, synthetic_encoded_data + ): + """b_0 in noise_params must equal mu_median[0] + mu_median[3] * (2/7).""" + data = synthetic_encoded_data + result = predict_new_pool( + synthetic_samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], fee=0.003, + ) + # Reconstruct mu_median independently + z_new = self._build_z_new(data, "MAINNET", ["WETH", "USDC"], 0.003) + B = synthetic_samples["B"] + mu_samples = np.einsum("skd,d->sk", B, z_new) + mu_median = np.median(mu_samples, axis=0) + + b_0_raw = mu_median[0] + b_weekend = mu_median[3] + expected_b_0 = b_0_raw + b_weekend * (2.0 / 7.0) + np.testing.assert_allclose( + result["noise_params"]["b_0"], expected_b_0, atol=1e-10, + ) + + def test_mu_equals_b_at_z_new( + self, synthetic_samples, synthetic_encoded_data + ): + """With known B samples, verify credible interval medians = median(B @ z_new).""" + data = synthetic_encoded_data + z_new = self._build_z_new(data, "ARBITRUM", ["WETH", "RATS"], 0.005) + + B = synthetic_samples["B"] + mu_expected = np.einsum("skd,d->sk", B, z_new) + mu_median_expected = np.median(mu_expected, axis=0) + + result = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "RATS"], fee=0.005, + ) + for k, name in enumerate(COEFF_NAMES): + np.testing.assert_allclose( + result["credible_intervals_90"][name]["median"], + mu_median_expected[k], + atol=1e-10, + ) + + def test_unseen_chain_uses_reference( + self, synthetic_samples, synthetic_encoded_data + ): + """A chain not in training data should get reference-chain prediction + (all chain dummies = 0), not raise an error.""" + data = synthetic_encoded_data + result = predict_new_pool( + synthetic_samples, data, + chain="SONIC", tokens=["WETH", "USDC"], fee=0.003, + ) + # Should be same as ARBITRUM (the reference chain, all dummies 0) + result_ref = predict_new_pool( + synthetic_samples, data, + chain="ARBITRUM", tokens=["WETH", "USDC"], fee=0.003, + ) + np.testing.assert_allclose( + result["noise_params"]["b_0"], + result_ref["noise_params"]["b_0"], + atol=1e-10, + ) + + +# =========================================================================== +# TestCheckConvergence +# =========================================================================== + + +class TestCheckConvergence: + def test_svi_returns_expected_keys(self): + losses = np.random.randn(1000).cumsum() + 5000 + result = check_convergence(losses, method="svi") + assert "final_elbo" in result + assert "elbo_last_100_std" in result + assert "elbo_last_100_mean" in result + + def test_svi_method_key(self): + losses = np.linspace(5000, 1000, 500) + result = check_convergence(losses, method="svi") + assert result["method"] == "svi" + + def test_svi_elbo_last_100_std_correct(self): + np.random.seed(42) + losses = np.random.randn(500) * 10 + 1000 + result = check_convergence(losses, method="svi") + expected_std = float(np.std(losses[-100:])) + np.testing.assert_allclose( + result["elbo_last_100_std"], expected_std, atol=1e-10, + ) + + def test_svi_final_elbo_is_last_loss(self): + losses = np.array([100.0, 50.0, 25.0, 12.5]) + result = check_convergence(losses, method="svi") + assert result["final_elbo"] == 12.5 + + +# =========================================================================== +# TestAssignDPClusters +# =========================================================================== + + +class TestAssignDPClusters: + @pytest.fixture() + def dp_samples_and_data(self, synthetic_encoded_data): + """Synthetic DP posterior samples with known cluster structure.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + K_clusters = 4 + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_COEFF, K_cov) * 0.5 + sigma_theta = np.ones((S, K_COEFF)) + L_Omega = np.tile(np.eye(K_COEFF), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_COEFF)) + df = np.full((S,), 5.0) + + # Well-separated sigma_eps clusters + sigma_eps = np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)) + # Stick-breaking weights: mostly on cluster 0 + v = np.tile([0.6, 0.3, 0.05], (S, 1)) + + samples = { + "B": B, "sigma_theta": sigma_theta, "L_Omega": L_Omega, + "eta": eta, "df": df, "sigma_eps": sigma_eps, "v": v, + } + data_with_k = dict(data) + data_with_k["K_clusters"] = K_clusters + return samples, data_with_k + + def test_returns_correct_length(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + assert len(assignments) == data["N_pools"] + + def test_valid_cluster_indices(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + K = data["K_clusters"] + assert all(0 <= a < K for a in assignments) + + def test_returns_integer_array(self, dp_samples_and_data): + from quantammsim.noise_calibration.postprocessing import assign_dp_clusters + + samples, data = dp_samples_and_data + assignments = assign_dp_clusters(samples, data) + assert assignments.dtype in (np.int32, np.int64, int) + + +# =========================================================================== +# TestExtractNoiseParamsDP +# =========================================================================== + + +class TestExtractNoiseParamsDP: + @pytest.fixture() + def dp_samples_and_data(self, synthetic_encoded_data): + """Same as above for extract_noise_params testing.""" + data = synthetic_encoded_data + N_pools = data["N_pools"] + K_cov = data["K_cov"] + S = 10 + + np.random.seed(99) + B = np.random.randn(S, K_COEFF, K_cov) * 0.5 + sigma_theta = np.ones((S, K_COEFF)) + L_Omega = np.tile(np.eye(K_COEFF), (S, 1, 1)) + eta = np.zeros((S, N_pools, K_COEFF)) + df = np.full((S,), 5.0) + sigma_eps = np.tile([0.3, 0.8, 1.5, 2.5], (S, 1)) + v = np.tile([0.6, 0.3, 0.05], (S, 1)) + + samples = { + "B": B, "sigma_theta": sigma_theta, "L_Omega": L_Omega, + "eta": eta, "df": df, "sigma_eps": sigma_eps, "v": v, + } + return samples, data + + def test_output_length(self, dp_samples_and_data): + samples, data = dp_samples_and_data + result = extract_noise_params(samples, data) + assert len(result) == data["N_pools"] + + def test_noise_params_keys_present(self, dp_samples_and_data): + samples, data = dp_samples_and_data + result = extract_noise_params(samples, data) + expected_keys = {"b_0", "b_sigma", "b_c", "b_weekend", "base_fee"} + for p in result: + assert set(p["noise_params"].keys()) == expected_keys + + +# =========================================================================== +# TestExtractStructuralParams +# =========================================================================== + + +class TestExtractStructuralParams: + """Tests for extract_structural_params().""" + + @pytest.fixture() + def structural_samples_and_data(self, synthetic_structural_data): + """Run a quick SVI fit to get structural samples.""" + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + return samples, synthetic_structural_data + + def test_extract_returns_arb_params(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + samples, data = structural_samples_and_data + result = extract_structural_params(samples, data) + assert len(result) == data["N_pools"] + for p in result: + assert "arb_frequency" in p + assert isinstance(p["arb_frequency"], int) + assert 1 <= p["arb_frequency"] <= 60 + + def test_extract_returns_noise_params(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + extract_structural_params, + ) + samples, data = structural_samples_and_data + result = extract_structural_params(samples, data) + for p in result: + assert "noise_params" in p + coeffs = p["noise_params"] + assert len(coeffs) == K_OBS_COEFF + for name in OBS_COEFF_NAMES: + assert name in coeffs, f"Missing coefficient: {name}" + + +# =========================================================================== +# TestPredictStructural +# =========================================================================== + + +class TestPredictStructural: + @pytest.fixture() + def structural_samples_and_data(self, synthetic_structural_data): + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.model import structural_noise_model + + samples, _ = run_svi( + synthetic_structural_data, + num_steps=500, + lr=5e-3, + seed=0, + num_samples=10, + model_fn=structural_noise_model, + ) + return samples, synthetic_structural_data + + def test_predict_returns_cadence(self, structural_samples_and_data): + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + result = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + assert "arb_frequency" in result + assert isinstance(result["arb_frequency"], int) + assert 1 <= result["arb_frequency"] <= 60 + + def test_predict_returns_noise_coefficients( + self, structural_samples_and_data, + ): + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + result = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + assert "noise_params" in result + assert len(result["noise_params"]) == K_OBS_COEFF + + def test_predict_uses_B_regression(self, structural_samples_and_data): + """Different (chain, tier) → different predictions via B regression.""" + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + r1 = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + r2 = predict_new_pool_structural( + samples, data, + chain="ARBITRUM", tokens=["BAL", "WETH"], + fee=0.01, tvl_est=5e5, + ) + # Different pool characteristics should produce different noise params + assert r1["noise_params"] != r2["noise_params"] + + def test_predict_cadence_higher_for_longtail( + self, structural_samples_and_data, + ): + """Long-tail pools should have higher cadence (less efficient arb).""" + from quantammsim.noise_calibration.postprocessing import ( + predict_new_pool_structural, + ) + samples, data = structural_samples_and_data + # This tests the structural relationship — it may not hold with + # random SVI samples on synthetic data, so we just check it doesn't + # crash and returns valid values + r1 = predict_new_pool_structural( + samples, data, + chain="MAINNET", tokens=["WETH", "USDC"], + fee=0.003, tvl_est=1e6, + ) + r2 = predict_new_pool_structural( + samples, data, + chain="BASE", tokens=["RATS", "WETH"], + fee=0.005, tvl_est=1e5, + ) + # Both should be valid + assert 1 <= r1["arb_frequency"] <= 60 + assert 1 <= r2["arb_frequency"] <= 60 diff --git a/tests/noise/test_token_classification.py b/tests/noise/test_token_classification.py new file mode 100644 index 00000000..dd540f84 --- /dev/null +++ b/tests/noise/test_token_classification.py @@ -0,0 +1,66 @@ +"""Tests for _normalise_symbol and classify_token_tier.""" + +import pytest +from quantammsim.noise_calibration import _normalise_symbol, classify_token_tier + + +# =========================================================================== +# TestNormaliseSymbol +# =========================================================================== + + +class TestNormaliseSymbol: + def test_passthrough_unknown(self): + assert _normalise_symbol("FOO") == "FOO" + + def test_known_mapping_preserved(self): + assert _normalise_symbol("WETH") == "WETH" + assert _normalise_symbol("WBTC") == "WBTC" + assert _normalise_symbol("cbBTC") == "cbBTC" + + def test_whitespace_stripped(self): + assert _normalise_symbol(" ETH ") == "ETH" + assert _normalise_symbol(" WETH ") == "WETH" + + def test_case_sensitivity_preserved(self): + # Lowercase is NOT normalised to uppercase + assert _normalise_symbol("weth") == "weth" + + +# =========================================================================== +# TestClassifyTokenTier +# =========================================================================== + + +class TestClassifyTokenTier: + def test_tier0_native_tokens(self): + for sym in ["ETH", "BTC"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_wrapped(self): + for sym in ["WETH", "WBTC", "cbBTC"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_stablecoins(self): + for sym in ["USDC", "USDT", "DAI"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier0_chain_natives(self): + for sym in ["MATIC", "AVAX", "GNO", "S", "wS"]: + assert classify_token_tier(sym) == 0, f"{sym} should be tier 0" + + def test_tier1_defi_bluechips(self): + for sym in ["AAVE", "BAL", "COW", "LINK", "ARB"]: + assert classify_token_tier(sym) == 1, f"{sym} should be tier 1" + + def test_tier2_unknown_tokens(self): + for sym in ["RATS", "PEPE"]: + assert classify_token_tier(sym) == 2, f"{sym} should be tier 2" + + def test_tier2_empty_string(self): + assert classify_token_tier("") == 2 + + def test_wrapped_variant_normalisation(self): + # "wS" is in the mapping table AND in _TIER_0 + assert classify_token_tier("wS") == 0 + assert classify_token_tier(" wS ") == 0 From 6c48e035ca3ac7349b670902a0ac964889601101 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:44:02 +0000 Subject: [PATCH 027/115] feat: add calibration framework: per-pool fit, joint fit, grid interpolation General-purpose pool parameter calibration with loss functions, per-pool and joint optimisation, learned parameter mappings, and grid-based interpolation. Independent of noise calibration. --- quantammsim/calibration/__init__.py | 41 ++ quantammsim/calibration/grid_interpolation.py | 463 +++++++++++++++++ quantammsim/calibration/joint_fit.py | 479 ++++++++++++++++++ quantammsim/calibration/learned_mapping.py | 126 +++++ quantammsim/calibration/loss.py | 74 +++ quantammsim/calibration/per_pool_fit.py | 127 +++++ quantammsim/calibration/pool_data.py | 302 +++++++++++ tests/calibration/__init__.py | 0 tests/calibration/conftest.py | 134 +++++ tests/calibration/test_joint_fit.py | 190 +++++++ tests/calibration/test_learned_mapping.py | 152 ++++++ tests/calibration/test_loss.py | 239 +++++++++ tests/calibration/test_per_pool_fit.py | 170 +++++++ tests/calibration/test_pool_data.py | 303 +++++++++++ 14 files changed, 2800 insertions(+) create mode 100644 quantammsim/calibration/__init__.py create mode 100644 quantammsim/calibration/grid_interpolation.py create mode 100644 quantammsim/calibration/joint_fit.py create mode 100644 quantammsim/calibration/learned_mapping.py create mode 100644 quantammsim/calibration/loss.py create mode 100644 quantammsim/calibration/per_pool_fit.py create mode 100644 quantammsim/calibration/pool_data.py create mode 100644 tests/calibration/__init__.py create mode 100644 tests/calibration/conftest.py create mode 100644 tests/calibration/test_joint_fit.py create mode 100644 tests/calibration/test_learned_mapping.py create mode 100644 tests/calibration/test_loss.py create mode 100644 tests/calibration/test_per_pool_fit.py create mode 100644 tests/calibration/test_pool_data.py diff --git a/quantammsim/calibration/__init__.py b/quantammsim/calibration/__init__.py new file mode 100644 index 00000000..a8887583 --- /dev/null +++ b/quantammsim/calibration/__init__.py @@ -0,0 +1,41 @@ +from quantammsim.calibration.grid_interpolation import ( + PoolCoeffs, + PoolCoeffsDaily, + PoolGridInterpolator, + build_scipy_interpolator, + interpolate_pool, + interpolate_pool_daily, + load_daily_grid, + load_valid_pool_grids, + pivot_grid, + precompute_pool_coeffs, + precompute_pool_coeffs_daily, +) +from quantammsim.calibration.joint_fit import ( + JointData, + fit_joint, + predict_new_pool_joint, +) +from quantammsim.calibration.learned_mapping import ( + build_targets, + cross_validate_loo, + fit_mapping, + predict_pool, +) +from quantammsim.calibration.loss import ( + K_OBS, + noise_volume, + pack_params, + pool_loss, + unpack_params, +) +from quantammsim.calibration.per_pool_fit import ( + fit_all_pools, + fit_single_pool, + make_initial_guess, +) +from quantammsim.calibration.pool_data import ( + build_pool_attributes, + build_x_obs, + match_grids_to_panel, +) diff --git a/quantammsim/calibration/grid_interpolation.py b/quantammsim/calibration/grid_interpolation.py new file mode 100644 index 00000000..bf0c55f4 --- /dev/null +++ b/quantammsim/calibration/grid_interpolation.py @@ -0,0 +1,463 @@ +"""PCHIP interpolation layer for precomputed arb-volume grids. + +Two grid formats: + - v1 (scalar): cadence x gas_cost -> median daily V_arb (single scalar) + - v2 (daily): cadence x gas_cost x day -> per-day V_arb (vector output) + +Interpolation in (log(cadence), gas_cost) space using PCHIP (monotone +piecewise cubic Hermite), which avoids Runge oscillation on non-uniform grids. + +Two interfaces: + 1. scipy-based: RegularGridInterpolator(method='pchip') for validation/plotting + 2. JAX-compatible: precomputed slopes + Hermite cubic eval, fully differentiable + +The JAX path uses tensor-product evaluation: + - Along cadence: Hermite cubic with scipy-precomputed PCHIP slopes + - Along gas: PCHIP slopes computed on the fly from intermediate values +""" + +import os +from typing import Dict, NamedTuple, Tuple + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd +from scipy.interpolate import PchipInterpolator, RegularGridInterpolator + +GRID_DIR = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))), + "results", + "pool_grids", +) + +GRID_DIR_V2 = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))), + "results", + "pool_grids_v2", +) + + +# ── Grid loading ───────────────────────────────────────────────────────── + + +def load_pool_grid(pool_id_prefix: str, grid_dir: str = GRID_DIR) -> pd.DataFrame: + """Load a single pool's grid CSV.""" + path = os.path.join(grid_dir, f"{pool_id_prefix}_grid.csv") + return pd.read_csv(path) + + +def load_valid_pool_grids(grid_dir: str = GRID_DIR) -> Dict[str, pd.DataFrame]: + """Load all pool grid CSVs that have valid (non-NaN) data.""" + grids = {} + for f in sorted(os.listdir(grid_dir)): + if f.endswith("_grid.csv") and f != "grid_summary.csv": + prefix = f.replace("_grid.csv", "") + df = pd.read_csv(os.path.join(grid_dir, f)) + if df["median_daily_arb_volume"].notna().any(): + grids[prefix] = df + return grids + + +def pivot_grid( + df: pd.DataFrame, value_col: str = "median_daily_arb_volume" +) -> Tuple[np.ndarray, np.ndarray, np.ndarray]: + """Pivot grid DataFrame to (log_cadences, gas_costs, values) arrays.""" + pivot = df.pivot(index="cadence", columns="gas_cost", values=value_col) + cadences = pivot.index.values.astype(float) + gas_costs = pivot.columns.values.astype(float) + values = pivot.values.astype(float) + return np.log(cadences), gas_costs, values + + +# ── Scipy interpolation ───────────────────────────────────────────────── + + +def build_scipy_interpolator( + df: pd.DataFrame, value_col: str = "median_daily_arb_volume" +) -> RegularGridInterpolator: + """Build a scipy RegularGridInterpolator with PCHIP method.""" + log_cadences, gas_costs, values = pivot_grid(df, value_col) + return RegularGridInterpolator( + (log_cadences, gas_costs), + values, + method="pchip", + bounds_error=False, + fill_value=None, + ) + + +def query_scipy( + interp: RegularGridInterpolator, cadence: float, gas_cost: float +) -> float: + """Query scipy interpolator at (cadence, gas_cost). Cadence in minutes.""" + log_cad = np.log(np.clip(cadence, 1.0, 60.0)) + return float(interp(np.array([[log_cad, gas_cost]]))[0]) + + +# ── JAX-compatible PCHIP ──────────────────────────────────────────────── + + +class PoolCoeffs(NamedTuple): + """Precomputed coefficients for one pool's 2D PCHIP interpolation.""" + + log_cadences: jnp.ndarray # (n_cad,) + gas_costs: jnp.ndarray # (n_gas,) + values: jnp.ndarray # (n_cad, n_gas) + slopes_cad: jnp.ndarray # (n_cad, n_gas) PCHIP slopes along cadence axis + + +def precompute_pool_coeffs( + df: pd.DataFrame, value_col: str = "median_daily_arb_volume" +) -> PoolCoeffs: + """Precompute PCHIP slopes along cadence axis using scipy. + + These slopes are used by the JAX evaluation function for the first + interpolation axis (cadence). The second axis (gas) is computed on + the fly in JAX to maintain full differentiability. + """ + log_cadences, gas_costs, values = pivot_grid(df, value_col) + + n_gas = values.shape[1] + slopes = np.zeros_like(values) + for j in range(n_gas): + pchip = PchipInterpolator(log_cadences, values[:, j]) + slopes[:, j] = pchip.derivative()(log_cadences) + + return PoolCoeffs( + log_cadences=jnp.array(log_cadences), + gas_costs=jnp.array(gas_costs), + values=jnp.array(values), + slopes_cad=jnp.array(slopes), + ) + + +@jax.jit +def _pchip_slopes(x: jnp.ndarray, y: jnp.ndarray) -> jnp.ndarray: + """Compute PCHIP slopes via Fritsch-Carlson method. JAX-compatible. + + x: (n,) sorted knot positions + y: (n,) values at knots + Returns: (n,) slopes at knots + """ + h = x[1:] - x[:-1] + delta = (y[1:] - y[:-1]) / h + + # Interior points: weighted harmonic mean of neighboring secants + w1 = 2 * h[1:] + h[:-1] + w2 = h[1:] + 2 * h[:-1] + + sign_agree = (delta[:-1] * delta[1:]) > 0 + + # When sign_agree is False, d_mid is masked to 0. But the harmonic mean + # can produce Inf when deltas have opposite signs and w1==w2 (denominator + # cancels). JAX's where can't mask the NaN gradient of Inf (0*NaN=NaN). + # Fix: replace deltas with 1.0 when sign_agree is False, ensuring hm + # is always finite. The value is irrelevant since it gets masked. + d0 = jnp.where(sign_agree, delta[:-1], 1.0) + d1 = jnp.where(sign_agree, delta[1:], 1.0) + d0 = jnp.where(d0 == 0, 1e-30, d0) + d1 = jnp.where(d1 == 0, 1e-30, d1) + + hm = (w1 + w2) / (w1 / d0 + w2 / d1) + hm = jnp.where(jnp.isfinite(hm), hm, 0.0) + d_mid = jnp.where(sign_agree, hm, 0.0) + + # Endpoints: one-sided shape-preserving + d0 = ((2 * h[0] + h[1]) * delta[0] - h[0] * delta[1]) / (h[0] + h[1]) + d0 = jnp.where(d0 * delta[0] <= 0, 0.0, d0) + d0 = jnp.where( + (delta[0] * delta[1] < 0) & (jnp.abs(d0) > 3 * jnp.abs(delta[0])), + 3 * delta[0], + d0, + ) + + dn = ((2 * h[-1] + h[-2]) * delta[-1] - h[-1] * delta[-2]) / (h[-1] + h[-2]) + dn = jnp.where(dn * delta[-1] <= 0, 0.0, dn) + dn = jnp.where( + (delta[-1] * delta[-2] < 0) & (jnp.abs(dn) > 3 * jnp.abs(delta[-1])), + 3 * delta[-1], + dn, + ) + + return jnp.concatenate([d0[None], d_mid, dn[None]]) + + +@jax.jit +def interpolate_pool( + coeffs: PoolCoeffs, log_cadence: jnp.ndarray, gas_cost: jnp.ndarray +) -> jnp.ndarray: + """Evaluate 2D PCHIP at (log_cadence, gas_cost). JAX-differentiable. + + Tensor-product approach: + 1. Hermite cubic along cadence for all gas columns (precomputed slopes) + 2. PCHIP slopes along gas through intermediate values (computed on the fly) + 3. Hermite cubic along gas to final value + + Args: + coeffs: PoolCoeffs from precompute_pool_coeffs + log_cadence: scalar, log of cadence in minutes + gas_cost: scalar, effective profit threshold in USD + Returns: + V_arb: scalar, interpolated median daily arb volume + """ + log_cads = coeffs.log_cadences + gas = coeffs.gas_costs + vals = coeffs.values + sl_cad = coeffs.slopes_cad + + # Clamp to grid bounds + log_cadence = jnp.clip(log_cadence, log_cads[0], log_cads[-1]) + gas_cost = jnp.clip(gas_cost, gas[0], gas[-1]) + + # ── Step 1: Hermite along cadence for all gas columns ── + idx = jnp.searchsorted(log_cads, log_cadence) - 1 + idx = jnp.clip(idx, 0, log_cads.shape[0] - 2) + + h = log_cads[idx + 1] - log_cads[idx] + t = (log_cadence - log_cads[idx]) / h + t2 = t * t + t3 = t2 * t + + h00 = 2 * t3 - 3 * t2 + 1 + h10 = t3 - 2 * t2 + t + h01 = -2 * t3 + 3 * t2 + h11 = t3 - t2 + + v_at_gas = ( + h00 * vals[idx, :] + + h01 * vals[idx + 1, :] + + h * (h10 * sl_cad[idx, :] + h11 * sl_cad[idx + 1, :]) + ) + + # ── Step 2: PCHIP slopes along gas ── + gas_slopes = _pchip_slopes(gas, v_at_gas) + + # ── Step 3: Hermite along gas ── + jdx = jnp.searchsorted(gas, gas_cost) - 1 + jdx = jnp.clip(jdx, 0, gas.shape[0] - 2) + + hg = gas[jdx + 1] - gas[jdx] + s = (gas_cost - gas[jdx]) / hg + s2 = s * s + s3 = s2 * s + + g00 = 2 * s3 - 3 * s2 + 1 + g10 = s3 - 2 * s2 + s + g01 = -2 * s3 + 3 * s2 + g11 = s3 - s2 + + return ( + g00 * v_at_gas[jdx] + + g01 * v_at_gas[jdx + 1] + + hg * (g10 * gas_slopes[jdx] + g11 * gas_slopes[jdx + 1]) + ) + + +# ── Per-day (v2) grid support ────────────────────────────────────────────── + + +class PoolCoeffsDaily(NamedTuple): + """Precomputed coefficients for per-day 2D PCHIP interpolation. + + Like PoolCoeffs but values/slopes have a day dimension: + values: (n_cad, n_gas, n_days) + slopes_cad: (n_cad, n_gas, n_days) + dates: (n_days,) ordinal dates for alignment with panel + """ + + log_cadences: jnp.ndarray # (n_cad,) + gas_costs: jnp.ndarray # (n_gas,) + values: jnp.ndarray # (n_cad, n_gas, n_days) + slopes_cad: jnp.ndarray # (n_cad, n_gas, n_days) + dates: jnp.ndarray # (n_days,) ordinal dates + + +def load_daily_grid( + pool_id_prefix: str, grid_dir: str = GRID_DIR_V2 +) -> pd.DataFrame: + """Load a pool's per-day grid parquet.""" + path = os.path.join(grid_dir, f"{pool_id_prefix}_daily.parquet") + return pd.read_parquet(path) + + +def precompute_pool_coeffs_daily(df: pd.DataFrame) -> PoolCoeffsDaily: + """Build PoolCoeffsDaily from per-day grid DataFrame. + + Args: + df: DataFrame with columns [cadence, gas_cost, date, daily_arb_volume] + + Returns: + PoolCoeffsDaily with 3D values (n_cad, n_gas, n_days) + """ + df = df.copy() + df["date"] = pd.to_datetime(df["date"]) + + cadences = np.array(sorted(df["cadence"].unique()), dtype=float) + gas_costs = np.array(sorted(df["gas_cost"].unique()), dtype=float) + dates = np.array(sorted(df["date"].unique())) + log_cadences = np.log(cadences) + + n_cad = len(cadences) + n_gas = len(gas_costs) + n_days = len(dates) + + # Build 3D array: cadence x gas x day + values = np.zeros((n_cad, n_gas, n_days)) + cad_idx = {c: i for i, c in enumerate(cadences)} + gas_idx = {g: i for i, g in enumerate(gas_costs)} + date_idx = {d: i for i, d in enumerate(dates)} + + for _, row in df.iterrows(): + ci = cad_idx.get(float(row["cadence"])) + gi = gas_idx.get(float(row["gas_cost"])) + di = date_idx.get(row["date"]) + if ci is not None and gi is not None and di is not None: + values[ci, gi, di] = row["daily_arb_volume"] + + # Compute PCHIP slopes along cadence axis for each (gas, day) + slopes = np.zeros_like(values) + for j in range(n_gas): + for k in range(n_days): + col = values[:, j, k] + if np.all(np.isfinite(col)): + pchip = PchipInterpolator(log_cadences, col) + slopes[:, j, k] = pchip.derivative()(log_cadences) + + # Convert dates to ordinals for JAX + date_ordinals = np.array([ + pd.Timestamp(d).toordinal() for d in dates + ], dtype=np.int32) + + return PoolCoeffsDaily( + log_cadences=jnp.array(log_cadences), + gas_costs=jnp.array(gas_costs), + values=jnp.array(values), + slopes_cad=jnp.array(slopes), + dates=jnp.array(date_ordinals), + ) + + +@jax.jit +def interpolate_pool_daily( + coeffs: PoolCoeffsDaily, + log_cadence: jnp.ndarray, + gas_cost: jnp.ndarray, +) -> jnp.ndarray: + """Evaluate 2D PCHIP at (log_cadence, gas_cost) for all days. + + Same tensor-product approach as interpolate_pool, but values are 3D + (n_cad, n_gas, n_days) so the output is (n_days,). + + The Hermite basis coefficients are scalars that broadcast over the + day dimension of values. + + Args: + coeffs: PoolCoeffsDaily from precompute_pool_coeffs_daily + log_cadence: scalar, log of cadence in minutes + gas_cost: scalar, effective profit threshold in USD + Returns: + V_arb: (n_days,) interpolated daily arb volume + """ + log_cads = coeffs.log_cadences + gas = coeffs.gas_costs + vals = coeffs.values # (n_cad, n_gas, n_days) + sl_cad = coeffs.slopes_cad # (n_cad, n_gas, n_days) + + # Clamp to grid bounds + log_cadence = jnp.clip(log_cadence, log_cads[0], log_cads[-1]) + gas_cost = jnp.clip(gas_cost, gas[0], gas[-1]) + + # ── Step 1: Hermite along cadence for all gas columns, all days ── + idx = jnp.searchsorted(log_cads, log_cadence) - 1 + idx = jnp.clip(idx, 0, log_cads.shape[0] - 2) + + h = log_cads[idx + 1] - log_cads[idx] + t = (log_cadence - log_cads[idx]) / h + t2 = t * t + t3 = t2 * t + + h00 = 2 * t3 - 3 * t2 + 1 + h10 = t3 - 2 * t2 + t + h01 = -2 * t3 + 3 * t2 + h11 = t3 - t2 + + # vals[idx, :, :] is (n_gas, n_days) — scalars broadcast + v_at_gas = ( + h00 * vals[idx, :, :] + + h01 * vals[idx + 1, :, :] + + h * (h10 * sl_cad[idx, :, :] + h11 * sl_cad[idx + 1, :, :]) + ) # (n_gas, n_days) + + # ── Step 2: PCHIP slopes along gas, vmapped over days ── + gas_slopes = jax.vmap( + lambda y_col: _pchip_slopes(gas, y_col), + in_axes=1, out_axes=1, + )(v_at_gas) # (n_gas, n_days) + + # ── Step 3: Hermite along gas ── + jdx = jnp.searchsorted(gas, gas_cost) - 1 + jdx = jnp.clip(jdx, 0, gas.shape[0] - 2) + + hg = gas[jdx + 1] - gas[jdx] + s = (gas_cost - gas[jdx]) / hg + s2 = s * s + s3 = s2 * s + + g00 = 2 * s3 - 3 * s2 + 1 + g10 = s3 - 2 * s2 + s + g01 = -2 * s3 + 3 * s2 + g11 = s3 - s2 + + # v_at_gas[jdx] is (n_days,) — scalars broadcast + return ( + g00 * v_at_gas[jdx] + + g01 * v_at_gas[jdx + 1] + + hg * (g10 * gas_slopes[jdx] + g11 * gas_slopes[jdx + 1]) + ) # (n_days,) + + +# ── Convenience class ──────────────────────────────────────────────────── + + +class PoolGridInterpolator: + """Collection of PCHIP interpolators for all valid pool grids. + + Provides both scipy (for validation) and JAX (for optimization) access. + """ + + def __init__(self, grid_dir: str = GRID_DIR): + self.grid_dir = grid_dir + grids = load_valid_pool_grids(grid_dir) + + self._scipy_interps = {} + self._jax_coeffs = {} + self._pool_ids = sorted(grids.keys()) + + for pid, df in grids.items(): + self._scipy_interps[pid] = build_scipy_interpolator(df) + self._jax_coeffs[pid] = precompute_pool_coeffs(df) + + @property + def pool_ids(self): + return list(self._pool_ids) + + @property + def n_pools(self): + return len(self._pool_ids) + + def query_scipy(self, pool_id: str, cadence: float, gas_cost: float) -> float: + """Query scipy PCHIP at (cadence_minutes, gas_cost_usd).""" + return query_scipy(self._scipy_interps[pool_id], cadence, gas_cost) + + def query_jax(self, pool_id: str, log_cadence, gas_cost): + """Query JAX PCHIP at (log_cadence, gas_cost). Differentiable.""" + return interpolate_pool(self._jax_coeffs[pool_id], log_cadence, gas_cost) + + def get_coeffs(self, pool_id: str) -> PoolCoeffs: + """Get precomputed JAX coefficients for a single pool.""" + return self._jax_coeffs[pool_id] + + def get_scipy(self, pool_id: str) -> RegularGridInterpolator: + """Get scipy interpolator for a single pool.""" + return self._scipy_interps[pool_id] diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py new file mode 100644 index 00000000..4f891c6d --- /dev/null +++ b/quantammsim/calibration/joint_fit.py @@ -0,0 +1,479 @@ +"""Joint end-to-end optimization (Option A) for the direct calibration pipeline. + +A parametric f_params maps pool_attributes → (cadence, gas, noise_coeffs), +optimized simultaneously across all pools through the grid interpolation loss. + +Two noise modes: + - "per_pool_noise": each pool has independent noise_coeffs (most flexible) + - "shared_noise": noise_coeffs = bias_noise + x_attr @ W_noise (generalizes) + +The cadence/gas mapping is always shared: + log_cadence = bias_cad + x_attr @ W_cad + log_gas = bias_gas + x_attr @ W_gas +""" + +from typing import Dict, List, NamedTuple, Optional + +import jax +import jax.numpy as jnp +import numpy as np +import scipy.optimize + +from quantammsim.calibration.grid_interpolation import ( + PoolCoeffsDaily, + interpolate_pool_daily, +) +from quantammsim.calibration.loss import K_OBS +from quantammsim.calibration.pool_data import build_pool_attributes, build_x_obs + + +class JointData(NamedTuple): + """Batched data for joint optimization.""" + pool_data: list # list of dicts with coeffs, x_obs, y_obs, day_indices + x_attr: jnp.ndarray # (n_pools, K_attr) pool attributes (no intercept) + pool_ids: list # list of pool_id prefixes + attr_names: list # attribute column names + + +def prepare_joint_data( + matched: Dict[str, dict], + drop_chain_dummies: bool = False, +) -> JointData: + """Build batched JAX arrays from matched pool data. + + Args: + matched: dict from match_grids_to_panel + drop_chain_dummies: if True, remove chain_* columns from attributes + (reduces feature count for small n) + + Returns: + JointData with per-pool JAX arrays and shared attribute matrix. + """ + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + + if drop_chain_dummies: + keep = [i for i, name in enumerate(attr_names) + if not name.startswith("chain_")] + X_attr = X_attr[:, keep] + attr_names = [attr_names[i] for i in keep] + + pool_data = [] + for pid in pool_ids: + entry = matched[pid] + panel = entry["panel"] + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + + pool_data.append({ + "coeffs": entry["coeffs"], + "x_obs": jnp.array(x_obs), + "y_obs": jnp.array(y_obs), + "day_indices": jnp.array(entry["day_indices"]), + }) + + return JointData( + pool_data=pool_data, + x_attr=jnp.array(X_attr), + pool_ids=pool_ids, + attr_names=attr_names, + ) + + +def pack_joint_params( + bias_cad: float, + bias_gas: float, + W_cad: jnp.ndarray, + W_gas: jnp.ndarray, + noise_params: jnp.ndarray, +) -> jnp.ndarray: + """Pack joint params into flat array. + + Layout: [bias_cad, bias_gas, W_cad(k_attr), W_gas(k_attr), noise_params...] + + noise_params is either: + - (n_pools, K_OBS) for per_pool_noise mode + - (1 + K_attr, K_OBS) for shared_noise mode (row 0 = noise bias) + """ + return jnp.concatenate([ + jnp.array([bias_cad, bias_gas]), + W_cad.ravel(), + W_gas.ravel(), + noise_params.ravel(), + ]) + + +def unpack_joint_params( + flat: jnp.ndarray, config: dict +) -> dict: + """Unpack flat array to structured params. + + config must have: k_attr, n_pools, mode + """ + k_attr = config["k_attr"] + mode = config["mode"] + + bias_cad = flat[0] + bias_gas = flat[1] + W_cad = flat[2:2 + k_attr] + W_gas = flat[2 + k_attr:2 + 2 * k_attr] + rest = flat[2 + 2 * k_attr:] + + if mode == "per_pool_noise": + n_pools = config["n_pools"] + noise_coeffs = rest.reshape(n_pools, K_OBS) + return { + "bias_cad": bias_cad, "bias_gas": bias_gas, + "W_cad": W_cad, "W_gas": W_gas, + "noise_coeffs": noise_coeffs, + } + else: # shared_noise + # noise_params: (1 + k_attr, K_OBS) — row 0 is bias + W_noise_full = rest.reshape(1 + k_attr, K_OBS) + return { + "bias_cad": bias_cad, "bias_gas": bias_gas, + "W_cad": W_cad, "W_gas": W_gas, + "bias_noise": W_noise_full[0], # (K_OBS,) + "W_noise": W_noise_full[1:], # (k_attr, K_OBS) + } + + +def _make_pool_loss_fn( + pool_idx: int, + pool_data_i: dict, + x_attr_i: jnp.ndarray, + config: dict, +): + """Create a JIT'd loss function for a single pool. + + Closes over pool-specific data; takes only params_flat as input. + Each pool gets its own small JIT'd computation graph. + """ + coeffs = pool_data_i["coeffs"] + x_obs = pool_data_i["x_obs"] + y_obs = pool_data_i["y_obs"] + day_indices = pool_data_i["day_indices"] + mode = config["mode"] + i = pool_idx + + @jax.jit + def pool_loss_fn(params_flat): + params = unpack_joint_params(params_flat, config) + log_cad = params["bias_cad"] + jnp.dot(x_attr_i, params["W_cad"]) + log_gas = params["bias_gas"] + jnp.dot(x_attr_i, params["W_gas"]) + + if mode == "per_pool_noise": + noise_c = params["noise_coeffs"][i] + else: + noise_c = params["bias_noise"] + jnp.dot(x_attr_i, params["W_noise"]) + + v_arb_all = interpolate_pool_daily(coeffs, log_cad, jnp.exp(log_gas)) + v_arb = v_arb_all[day_indices] + v_noise = jnp.exp(x_obs @ noise_c) + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + return jnp.mean((log_v_pred - y_obs) ** 2) + + return pool_loss_fn + + +def make_joint_loss_fn( + jdata: JointData, + mode: str = "per_pool_noise", + alpha_cad: float = 0.01, + alpha_gas: float = 0.01, +): + """Create per-pool JIT'd loss functions and a Python-level aggregator. + + Each pool gets its own small JIT'd computation graph (compiled + independently), avoiding a massive unrolled trace. The outer + function sums per-pool losses in Python and adds regularization. + + Loss averages over pools (not observations), giving equal weight + to each pool regardless of observation count. + + L2 regularization is applied to W_cad and W_gas only (not biases). + + Args: + jdata: JointData from prepare_joint_data + mode: "per_pool_noise" or "shared_noise" + alpha_cad: L2 regularization on W_cad + alpha_gas: L2 regularization on W_gas + + Returns: + loss_fn(params_flat) -> scalar loss + """ + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode} + + # Build per-pool JIT'd loss functions + pool_loss_fns = [] + pool_val_and_grad_fns = [] + for i in range(n_pools): + fn = _make_pool_loss_fn(i, jdata.pool_data[i], jdata.x_attr[i], config) + pool_loss_fns.append(fn) + pool_val_and_grad_fns.append(jax.value_and_grad(fn)) + + def loss_fn(params_flat): + total = sum(fn(params_flat) for fn in pool_loss_fns) + data_loss = total / n_pools + + params = unpack_joint_params(params_flat, config) + reg = alpha_cad * jnp.sum(params["W_cad"] ** 2) + \ + alpha_gas * jnp.sum(params["W_gas"] ** 2) + return data_loss + reg + + # Attach per-pool functions for the value_and_grad wrapper + loss_fn._pool_val_and_grad_fns = pool_val_and_grad_fns + loss_fn._n_pools = n_pools + loss_fn._config = config + loss_fn._alpha_cad = alpha_cad + loss_fn._alpha_gas = alpha_gas + + return loss_fn + + +def make_initial_joint_params( + jdata: JointData, + mode: str = "per_pool_noise", + init_from_option_c: Optional[Dict[str, dict]] = None, +) -> jnp.ndarray: + """Create initial parameter vector. + + If init_from_option_c is provided, warm-start from Option C per-pool fits: + - bias_cad, W_cad from OLS on per-pool fitted log_cadence + - bias_gas, W_gas from OLS on per-pool fitted log_gas + - noise_coeffs from per-pool fits + + Otherwise, use defaults: cadence=12min, gas=$1 for all pools. + """ + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + x_attr_np = np.array(jdata.x_attr) + + if init_from_option_c is not None: + pool_ids = jdata.pool_ids + # Filter out pools with NaN losses from warm start + valid = {p: init_from_option_c[p] for p in pool_ids + if p in init_from_option_c + and np.isfinite(init_from_option_c[p].get("loss", float("nan")))} + + if len(valid) < len(pool_ids): + default_lc = np.log(12.0) + default_lg = np.log(1.0) + for p in pool_ids: + if p not in valid: + valid[p] = { + "log_cadence": default_lc, + "log_gas": default_lg, + "noise_coeffs": np.zeros(K_OBS), + } + + log_cads = np.array([valid[p]["log_cadence"] for p in pool_ids]) + log_gases = np.array([valid[p]["log_gas"] for p in pool_ids]) + noise_all = np.array([valid[p]["noise_coeffs"] for p in pool_ids]) + + # OLS with intercept: X_aug = [1, x_attr]; solve for [bias, W] + X_aug = np.column_stack([np.ones(n_pools), x_attr_np]) + cad_params, _, _, _ = np.linalg.lstsq(X_aug, log_cads, rcond=None) + gas_params, _, _, _ = np.linalg.lstsq(X_aug, log_gases, rcond=None) + bias_cad, W_cad = cad_params[0], cad_params[1:] + bias_gas, W_gas = gas_params[0], gas_params[1:] + + if mode == "per_pool_noise": + noise_params = noise_all + else: + # OLS with intercept for noise mapping + noise_aug, _, _, _ = np.linalg.lstsq(X_aug, noise_all, rcond=None) + # noise_aug: (1+k_attr, K_OBS) — row 0 is bias + noise_params = noise_aug + else: + # Default: all pools get cadence=12min, gas=$1 + bias_cad = np.log(12.0) + bias_gas = np.log(1.0) # = 0.0 + W_cad = np.zeros(k_attr) + W_gas = np.zeros(k_attr) + + if mode == "per_pool_noise": + # Initialize noise via OLS per pool + noise_params = np.zeros((n_pools, K_OBS)) + for i, pd in enumerate(jdata.pool_data): + x_obs_np = np.array(pd["x_obs"]) + y_obs_np = np.array(pd["y_obs"]) + c, _, _, _ = np.linalg.lstsq(x_obs_np, y_obs_np, rcond=None) + noise_params[i] = c + else: + # Initialize shared noise from pooled OLS + all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) + all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) + c, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) + # (1+k_attr, K_OBS): bias row + zero weight rows + noise_params = np.zeros((1 + k_attr, K_OBS)) + noise_params[0, :] = c + + return pack_joint_params( + float(bias_cad), + float(bias_gas), + jnp.array(W_cad), + jnp.array(W_gas), + jnp.array(noise_params), + ) + + +def _make_bounds(k_attr, n_pools, mode): + """Build scipy bounds for joint params.""" + # bias_cad, bias_gas: unbounded + bounds = [(None, None)] * 2 + # W_cad, W_gas: unbounded + bounds += [(None, None)] * (2 * k_attr) + + if mode == "per_pool_noise": + bounds += [(None, None)] * (n_pools * K_OBS) + else: + bounds += [(None, None)] * ((1 + k_attr) * K_OBS) + + return bounds + + +def fit_joint( + matched: Dict[str, dict], + mode: str = "per_pool_noise", + init_from_option_c: Optional[Dict[str, dict]] = None, + maxiter: int = 500, + alpha_cad: float = 0.01, + alpha_gas: float = 0.01, + drop_chain_dummies: bool = False, +) -> dict: + """Joint end-to-end optimization across all pools. + + Args: + matched: dict from match_grids_to_panel + mode: "per_pool_noise" or "shared_noise" + init_from_option_c: Optional Option C results for warm start. + Pools with NaN losses are silently excluded from warm start. + maxiter: max L-BFGS-B iterations + alpha_cad: L2 regularization on W_cad (not bias) + alpha_gas: L2 regularization on W_gas (not bias) + drop_chain_dummies: if True, remove chain_* columns from attributes + + Returns dict with fitted params and diagnostics. + """ + jdata = prepare_joint_data(matched, drop_chain_dummies=drop_chain_dummies) + loss_fn = make_joint_loss_fn(jdata, mode=mode, + alpha_cad=alpha_cad, alpha_gas=alpha_gas) + init = make_initial_joint_params(jdata, mode=mode, + init_from_option_c=init_from_option_c) + + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode} + bounds = _make_bounds(k_attr, n_pools, mode) + + # Per-pool value_and_grad — each pool has its own small JIT graph + pool_vg_fns = loss_fn._pool_val_and_grad_fns + + # Indices for W_cad and W_gas in the flat param vector (for reg gradient) + w_cad_start = 2 + w_cad_end = 2 + k_attr + w_gas_start = 2 + k_attr + w_gas_end = 2 + 2 * k_attr + + def scipy_wrapper(params_np): + params_j = jnp.array(params_np) + + # Sum per-pool losses and gradients + total_val = 0.0 + total_grad = jnp.zeros_like(params_j) + for vg_fn in pool_vg_fns: + v, g = vg_fn(params_j) + total_val += float(v) + total_grad = total_grad + g + + data_loss = total_val / n_pools + data_grad = total_grad / n_pools + + # Regularization on W_cad and W_gas (not biases) + reg = (alpha_cad * float(jnp.sum(params_j[w_cad_start:w_cad_end] ** 2)) + + alpha_gas * float(jnp.sum(params_j[w_gas_start:w_gas_end] ** 2))) + + reg_grad = jnp.zeros_like(params_j) + reg_grad = reg_grad.at[w_cad_start:w_cad_end].set( + 2 * alpha_cad * params_j[w_cad_start:w_cad_end]) + reg_grad = reg_grad.at[w_gas_start:w_gas_end].set( + 2 * alpha_gas * params_j[w_gas_start:w_gas_end]) + + val = data_loss + reg + grad = data_grad + reg_grad + return val, np.array(grad, dtype=np.float64) + + init_np = np.array(init, dtype=np.float64) + init_loss = float(loss_fn(jnp.array(init_np))) + + result = scipy.optimize.minimize( + scipy_wrapper, + init_np, + method="L-BFGS-B", + jac=True, + bounds=bounds, + options={"maxiter": maxiter, "ftol": 1e-10, "gtol": 1e-8}, + ) + + params = unpack_joint_params(jnp.array(result.x), config) + + out = { + "init_loss": init_loss, + "bias_cad": float(params["bias_cad"]), + "bias_gas": float(params["bias_gas"]), + "W_cad": np.array(params["W_cad"]), + "W_gas": np.array(params["W_gas"]), + "loss": float(result.fun), + "converged": result.success, + "mode": mode, + "k_attr": k_attr, + "pool_ids": jdata.pool_ids, + "attr_names": jdata.attr_names, + } + + if mode == "per_pool_noise": + out["noise_coeffs"] = np.array(params["noise_coeffs"]) + else: + out["bias_noise"] = np.array(params["bias_noise"]) + out["W_noise"] = np.array(params["W_noise"]) + + return out + + +def predict_new_pool_joint( + result: dict, + x_attr: np.ndarray, +) -> dict: + """Predict simulator settings for a new pool using joint-fitted mapping. + + In per_pool_noise mode, only cadence and gas are predicted (noise + coefficients are per-pool and can't generalize). Use shared_noise + mode for full deployment predictions including noise_coeffs. + + Args: + result: dict from fit_joint + x_attr: (K_attr,) pool attribute vector — must match the k_attr + from training (check result['attr_names'] for the feature order). + No intercept — just the real features. + + Returns dict with cadence_minutes, gas_usd, and noise_coeffs (shared_noise only). + """ + x = np.asarray(x_attr) + log_cadence = result["bias_cad"] + float(x @ result["W_cad"]) + log_gas = result["bias_gas"] + float(x @ result["W_gas"]) + + out = { + "log_cadence": log_cadence, + "log_gas": log_gas, + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": float(np.exp(log_gas)), + } + + if result["mode"] == "shared_noise": + out["noise_coeffs"] = np.array( + result["bias_noise"] + x @ result["W_noise"] + ) + + return out diff --git a/quantammsim/calibration/learned_mapping.py b/quantammsim/calibration/learned_mapping.py new file mode 100644 index 00000000..294d0082 --- /dev/null +++ b/quantammsim/calibration/learned_mapping.py @@ -0,0 +1,126 @@ +"""Learned mapping: pool attributes -> (cadence, gas, noise_coeffs). + +Ridge regression from pool-level features to per-pool fitted parameters. +Trained on per-pool fit results from per_pool_fit.py; used to predict +simulator settings for new/hypothetical pools. +""" + +from typing import Dict, List + +import numpy as np + +from quantammsim.calibration.loss import K_OBS + + +def build_targets( + fit_results: Dict[str, dict], + pool_order: List[str], +) -> np.ndarray: + """Stack per-pool fitted params into (n_pools, 2+K_OBS) target matrix. + + Columns: [log_cadence, log_gas, noise_coeffs...] + Row ordering matches pool_order. + """ + n_pools = len(pool_order) + Y = np.zeros((n_pools, 2 + K_OBS)) + for i, pid in enumerate(pool_order): + r = fit_results[pid] + Y[i, 0] = r["log_cadence"] + Y[i, 1] = r["log_gas"] + Y[i, 2:] = r["noise_coeffs"] + return Y + + +def fit_mapping( + X_attr: np.ndarray, + Y_target: np.ndarray, + alpha: float = 1.0, +) -> dict: + """Fit Ridge regression: X_attr -> Y_target. + + Multi-output Ridge: one shared regularization strength across all + target columns. + + Returns dict with weights, intercept, and diagnostics. + """ + # Ridge with intercept: center Y, solve on centered data + n, k = X_attr.shape + Y_mean = Y_target.mean(axis=0) + X_mean = X_attr.mean(axis=0) + Xc = X_attr - X_mean + Yc = Y_target - Y_mean + + # W = (Xc^T Xc + alpha * I)^{-1} Xc^T Yc + A = Xc.T @ Xc + alpha * np.eye(k) + W = np.linalg.solve(A, Xc.T @ Yc) # (K_attr, K_target) + intercept = Y_mean - X_mean @ W # (K_target,) + + Y_pred = X_attr @ W + intercept + ss_res = np.sum((Y_target - Y_pred) ** 2) + ss_tot = np.sum((Y_target - Y_target.mean(axis=0)) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + return { + "weights": W, # (K_attr, K_target) + "intercept": intercept, # (K_target,) + "alpha": alpha, + "r2_train": float(r2), + } + + +def predict_pool( + mapping: dict, + x_attr: np.ndarray, +) -> dict: + """Predict simulator settings for a new pool. + + Args: + mapping: dict from fit_mapping + x_attr: (K_attr,) single pool attribute vector + + Returns dict with cadence_minutes, gas_usd, noise_coeffs, etc. + """ + y = x_attr @ mapping["weights"] + mapping["intercept"] + + log_cadence = float(y[0]) + log_gas = float(y[1]) + noise_coeffs = np.array(y[2:]) + + return { + "log_cadence": log_cadence, + "log_gas": log_gas, + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": float(np.exp(log_gas)), + "noise_coeffs": noise_coeffs, + } + + +def cross_validate_loo( + X_attr: np.ndarray, + Y_target: np.ndarray, + alpha: float = 1.0, +) -> dict: + """Leave-one-out cross-validation. + + Returns per-pool prediction errors and summary statistics. + """ + n = X_attr.shape[0] + errors = np.zeros((n, Y_target.shape[1])) + + for i in range(n): + mask = np.ones(n, dtype=bool) + mask[i] = False + X_train = X_attr[mask] + Y_train = Y_target[mask] + + model = fit_mapping(X_train, Y_train, alpha=alpha) + y_pred = X_attr[i] @ model["weights"] + model["intercept"] + errors[i] = Y_target[i] - y_pred + + mse = np.mean(errors ** 2, axis=0) + return { + "per_pool_errors": errors, + "mse_per_target": mse, + "rmse_per_target": np.sqrt(mse), + "mean_rmse": float(np.mean(np.sqrt(mse))), + } diff --git a/quantammsim/calibration/loss.py b/quantammsim/calibration/loss.py new file mode 100644 index 00000000..13b2e650 --- /dev/null +++ b/quantammsim/calibration/loss.py @@ -0,0 +1,74 @@ +"""Per-pool loss function for the direct calibration pipeline. + +JAX-differentiable loss: sum((log(V_arb_i + V_noise_i) - log(V_obs_i))^2) +where V_arb comes from per-day grid interpolation and V_noise from +log-linear regression on observation covariates. +""" + +from typing import Tuple + +import jax.numpy as jnp + +from quantammsim.calibration.grid_interpolation import ( + PoolCoeffsDaily, + interpolate_pool_daily, +) + +K_OBS = 8 # observation-level covariates + + +def noise_volume( + noise_coeffs: jnp.ndarray, x_obs: jnp.ndarray +) -> jnp.ndarray: + """V_noise = exp(x_obs @ noise_coeffs). Shape: (n_obs,).""" + return jnp.exp(x_obs @ noise_coeffs) + + +def pack_params( + log_cadence: float, log_gas: float, noise_coeffs: jnp.ndarray +) -> jnp.ndarray: + """Pack into flat array: [log_cadence, log_gas, noise_coeffs...].""" + return jnp.concatenate([ + jnp.array([log_cadence, log_gas]), + jnp.asarray(noise_coeffs), + ]) + + +def unpack_params( + flat: jnp.ndarray, +) -> Tuple[float, float, jnp.ndarray]: + """Unpack flat array to (log_cadence, log_gas, noise_coeffs).""" + return flat[0], flat[1], flat[2:] + + +def pool_loss( + params_flat: jnp.ndarray, + coeffs: PoolCoeffsDaily, + x_obs: jnp.ndarray, + y_obs: jnp.ndarray, + day_indices: jnp.ndarray, +) -> jnp.ndarray: + """Per-pool log-space L2 loss with per-day V_arb. + + Args: + params_flat: [log_cadence, log_gas, noise_coeffs...] from pack_params + coeffs: PoolCoeffsDaily with per-day grid values + x_obs: (n_obs, K_OBS) observation covariates + y_obs: (n_obs,) log(V_obs) — observed log volume + day_indices: (n_obs,) int indices mapping panel rows to grid days + + Returns: + Scalar mean squared error in log space. + """ + log_cadence, log_gas, noise_coeffs = unpack_params(params_flat) + + # Per-day V_arb from grid interpolation + v_arb_all = interpolate_pool_daily(coeffs, log_cadence, jnp.exp(log_gas)) # (n_days,) + v_arb = v_arb_all[day_indices] # (n_obs,) + + # Per-day V_noise from covariates + v_noise = noise_volume(noise_coeffs, x_obs) # (n_obs,) + + # Log-space L2 loss + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + return jnp.mean((log_v_pred - y_obs) ** 2) diff --git a/quantammsim/calibration/per_pool_fit.py b/quantammsim/calibration/per_pool_fit.py new file mode 100644 index 00000000..037a124d --- /dev/null +++ b/quantammsim/calibration/per_pool_fit.py @@ -0,0 +1,127 @@ +"""Per-pool fitting via L-BFGS-B for the direct calibration pipeline. + +Fits (log_cadence, log_gas, noise_coeffs) per pool by minimizing +the log-space L2 loss using scipy.optimize.minimize with JAX gradients. +""" + +from typing import Dict, Optional + +import jax +import jax.numpy as jnp +import numpy as np +import scipy.optimize + +from quantammsim.calibration.grid_interpolation import PoolCoeffsDaily +from quantammsim.calibration.loss import K_OBS, pack_params, pool_loss +from quantammsim.calibration.pool_data import build_x_obs + + +def make_initial_guess(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: + """Initial params: cadence=12min, gas=$1, noise_coeffs from OLS. + + OLS: noise_coeffs = lstsq(x_obs, y_obs) — assumes all volume is noise. + This overestimates noise but gives a reasonable starting point. + """ + noise_coeffs, _, _, _ = np.linalg.lstsq(x_obs, y_obs, rcond=None) + init = np.zeros(2 + K_OBS) + init[0] = np.log(12.0) # log_cadence + init[1] = np.log(1.0) # log_gas (= 0.0) + init[2:] = noise_coeffs + return init + + +def fit_single_pool( + coeffs: PoolCoeffsDaily, + x_obs: np.ndarray, + y_obs: np.ndarray, + day_indices: np.ndarray, + init: Optional[np.ndarray] = None, + bounds: Optional[dict] = None, +) -> dict: + """Fit (log_cadence, log_gas, noise_coeffs) for one pool via L-BFGS-B. + + Returns dict with fitted params, loss, and convergence status. + """ + if init is None: + init = make_initial_guess(x_obs, y_obs) + + # Default bounds + if bounds is None: + bounds = {} + log_cad_bounds = bounds.get("log_cadence", (np.log(1.0), np.log(60.0))) + log_gas_bounds = bounds.get("log_gas", (np.log(0.001), np.log(50.0))) + noise_bounds = bounds.get("noise_coeffs", (-20.0, 20.0)) + + scipy_bounds = [ + log_cad_bounds, + log_gas_bounds, + ] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + + # Convert to JAX arrays + x_obs_j = jnp.array(x_obs) + y_obs_j = jnp.array(y_obs) + day_idx_j = jnp.array(day_indices) + + # Value and gradient function + @jax.jit + def loss_and_grad(params_flat): + loss = pool_loss(params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j) + grad = jax.grad(pool_loss, argnums=0)( + params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j + ) + return loss, grad + + def scipy_wrapper(params_np): + params_j = jnp.array(params_np) + loss, grad = loss_and_grad(params_j) + return float(loss), np.array(grad, dtype=np.float64) + + result = scipy.optimize.minimize( + scipy_wrapper, + init, + method="L-BFGS-B", + jac=True, + bounds=scipy_bounds, + options={"maxiter": 500, "ftol": 1e-10, "gtol": 1e-8}, + ) + + log_cadence = float(result.x[0]) + log_gas = float(result.x[1]) + noise_coeffs = np.array(result.x[2:]) + + return { + "log_cadence": log_cadence, + "log_gas": log_gas, + "noise_coeffs": noise_coeffs, + "loss": float(result.fun), + "converged": result.success, + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": float(np.exp(log_gas)), + } + + +def fit_all_pools( + matched: Dict[str, dict], + n_workers: int = 1, +) -> Dict[str, dict]: + """Fit all matched pools. Returns prefix -> fit_result with metadata.""" + results = {} + + for prefix, entry in matched.items(): + panel = entry["panel"] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + + result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + + # Add metadata + result["chain"] = entry["chain"] + result["fee"] = entry["fee"] + result["tokens"] = entry["tokens"] + + results[prefix] = result + + return results diff --git a/quantammsim/calibration/pool_data.py b/quantammsim/calibration/pool_data.py new file mode 100644 index 00000000..039880bc --- /dev/null +++ b/quantammsim/calibration/pool_data.py @@ -0,0 +1,302 @@ +"""Data assembly for the direct calibration pipeline. + +Matches precomputed per-day arb grids to panel observations and builds +model-ready arrays for the loss function. +""" + +import json +import os +from typing import Dict, List, Tuple + +import numpy as np +import pandas as pd + +from quantammsim.calibration.grid_interpolation import ( + PoolCoeffsDaily, + load_daily_grid, + precompute_pool_coeffs_daily, +) + +K_OBS = 8 # observation-level covariates + +# Default path for cached token market caps +_MCAP_PATH = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "local_data", "noise_calibration", "token_mcaps.json", +) + +# Asset type classification (fallback if not in mcap JSON) +_STABLECOINS = { + "USDC", "USDT", "DAI", "WXDAI", "xDAI", "GHO", "LUSD", "crvUSD", + "FRAX", "sDAI", "scUSD", "DOLA", + "waBasUSDC", "waEthUSDC", +} +_NATIVE_LST = { + "WETH", "ETH", "wstETH", "stETH", "rETH", "cbETH", + "WBTC", "BTC", "cbBTC", + "WMATIC", "MATIC", "POL", "wPOL", + "WAVAX", "AVAX", + "GNO", "S", "wS", "stS", + "JitoSOL", + "waEthLidoWETH", "waEthLidowstETH", + "waBasWETH", "waGnoGNO", "waGnowstETH", +} + + +def _load_token_mcaps(path: str = None) -> dict: + """Load cached token market caps. Returns {} if file missing.""" + path = path or _MCAP_PATH + if os.path.exists(path): + with open(path) as f: + return json.load(f) + return {} + + +def _get_asset_type(symbol: str, mcaps: dict) -> int: + """Return asset type: 0=stable, 1=native/LST, 2=volatile.""" + if symbol in mcaps and "asset_type" in mcaps[symbol]: + t = mcaps[symbol]["asset_type"] + return {"stable": 0, "native_lst": 1, "volatile": 2}.get(t, 2) + if symbol in _STABLECOINS: + return 0 + if symbol in _NATIVE_LST: + return 1 + return 2 + + +def _parse_tokens(tokens_str: str) -> List[str]: + """Parse comma-separated token string into list.""" + if isinstance(tokens_str, (list, tuple)): + return list(tokens_str) + return [t.strip() for t in tokens_str.split(",")] + + +def match_grids_to_panel( + grid_dir: str, panel: pd.DataFrame, pools_path: str = None, +) -> Dict[str, dict]: + """Match grid parquets to panel rows by pool_id prefix. + + For each _daily.parquet in grid_dir, find the panel pool whose + pool_id starts with the same 16-char prefix. Build PoolCoeffsDaily + and compute day_indices mapping panel dates to grid date indices. + + Args: + grid_dir: directory containing {prefix}_daily.parquet files + panel: panel DataFrame with pool observations + pools_path: path to pools.parquet for weight metadata. + Defaults to local_data/noise_calibration/pools.parquet. + + Returns dict: prefix -> { + 'panel': DataFrame (obs for this pool), + 'coeffs': PoolCoeffsDaily (per-day), + 'day_indices': np.ndarray (panel date -> grid day index), + 'pool_id': full pool_id from panel, + 'chain': str, 'fee': float, 'tokens': str, + 'weights': list of float (pool weights, e.g. [0.5, 0.5]), + } + """ + # Discover grid files + grid_prefixes = [] + for f in sorted(os.listdir(grid_dir)): + if f.endswith("_daily.parquet"): + prefix = f.replace("_daily.parquet", "") + grid_prefixes.append(prefix) + + if not grid_prefixes: + return {} + + # Load pool metadata for weights + if pools_path is None: + pools_path = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(__file__))), + "local_data", "noise_calibration", "pools.parquet", + ) + pools_meta = {} + if os.path.exists(pools_path): + pools_df = pd.read_parquet(pools_path) + pools_df["_prefix"] = pools_df["pool_id"].str[:16] + for _, row in pools_df.iterrows(): + w = row.get("weights") + if w is not None: + try: + weights = [float(x) for x in w] + except (TypeError, ValueError): + weights = [0.5, 0.5] + else: + weights = [0.5, 0.5] + pools_meta[row["_prefix"]] = {"weights": weights} + + # Ensure date column + panel = panel.copy() + if "date" in panel.columns: + panel["date"] = pd.to_datetime(panel["date"]) + + # Build prefix -> panel rows mapping + panel["_prefix"] = panel["pool_id"].str[:16] + + matched = {} + for prefix in grid_prefixes: + pool_rows = panel[panel["_prefix"] == prefix] + if len(pool_rows) == 0: + continue + + # Load and precompute grid + grid_df = load_daily_grid(prefix, grid_dir) + coeffs = precompute_pool_coeffs_daily(grid_df) + + # Build date alignment: panel date ordinals -> grid day indices + grid_ordinals = np.array(coeffs.dates) + grid_ord_to_idx = {int(o): i for i, o in enumerate(grid_ordinals)} + + panel_dates = pd.to_datetime(pool_rows["date"]) + panel_ordinals = np.array([d.toordinal() for d in panel_dates]) + + # Filter to dates present in both panel and grid + valid_mask = np.array([int(o) in grid_ord_to_idx for o in panel_ordinals]) + pool_rows = pool_rows[valid_mask].copy() + panel_ordinals = panel_ordinals[valid_mask] + + if len(pool_rows) == 0: + continue + + day_indices = np.array([grid_ord_to_idx[int(o)] for o in panel_ordinals]) + + row0 = pool_rows.iloc[0] + weights = pools_meta.get(prefix, {}).get("weights", [0.5, 0.5]) + + matched[prefix] = { + "panel": pool_rows.reset_index(drop=True), + "coeffs": coeffs, + "day_indices": day_indices, + "pool_id": row0["pool_id"], + "chain": row0["chain"], + "fee": float(np.exp(row0["log_fee"])) if "swap_fee" not in pool_rows.columns + else float(row0.get("swap_fee", np.exp(row0["log_fee"]))), + "tokens": row0["tokens"], + "weights": weights, + } + + return matched + + +def build_x_obs(panel_rows: pd.DataFrame) -> np.ndarray: + """Build (n_obs, 8) observation covariate matrix from panel rows. + + Columns: [1, log_tvl_lag1, log_sigma, tvl*sigma, tvl*fee, + sigma*fee, dow_sin, dow_cos] + + Where: + log_sigma = log(max(volatility, 1e-6)) + tvl = log_tvl_lag1 + fee = log_fee + dow_sin = sin(2*pi*weekday/7), dow_cos = cos(2*pi*weekday/7) + weekday: Monday=0, ..., Sunday=6 + """ + n = len(panel_rows) + x = np.zeros((n, K_OBS)) + + tvl = panel_rows["log_tvl_lag1"].values.astype(float) + sigma = np.log(np.maximum(panel_rows["volatility"].values.astype(float), 1e-6)) + fee = panel_rows["log_fee"].values.astype(float) + weekdays = pd.to_datetime(panel_rows["date"]).dt.weekday.values.astype(float) + + x[:, 0] = 1.0 # intercept + x[:, 1] = tvl # log_tvl_lag1 + x[:, 2] = sigma # log_sigma + x[:, 3] = tvl * sigma # tvl × sigma + x[:, 4] = tvl * fee # tvl × fee + x[:, 5] = sigma * fee # sigma × fee + x[:, 6] = np.sin(2 * np.pi * weekdays / 7) # dow_sin + x[:, 7] = np.cos(2 * np.pi * weekdays / 7) # dow_cos + + return x + + +def build_pool_attributes( + matched: Dict[str, dict], + mcap_path: str = None, +) -> Tuple[np.ndarray, List[str], List[str]]: + """Build (n_pools, K_attr) pool attribute matrix. + + Columns: + chain_dummies..., log_fee, mean_log_tvl, + log_mcap_product, has_stable, same_asset_type, weight_imbalance + + No intercept column — bias terms are handled by the model internally. + Chain dummies: one-hot with the first chain (alphabetically) as reference. + Market caps loaded from cached JSON (run scripts/fetch_token_mcaps.py). + + Returns: (X_attr, attr_names, pool_ids) + """ + mcaps = _load_token_mcaps(mcap_path) + + pool_ids = sorted(matched.keys()) + n_pools = len(pool_ids) + + # Collect per-pool attributes + chains = [] + log_fees = [] + mean_tvls = [] + log_mcap_products = [] + has_stables = [] + same_asset_types = [] + weight_imbalances = [] + + for pid in pool_ids: + entry = matched[pid] + chains.append(entry["chain"]) + log_fee = entry["panel"]["log_fee"].values[0] + log_fees.append(float(log_fee)) + mean_tvls.append(float(entry["panel"]["log_tvl_lag1"].mean())) + + # Token-level features + tokens = _parse_tokens(entry["tokens"]) + tok_a = tokens[0] if len(tokens) > 0 else "UNKNOWN" + tok_b = tokens[1] if len(tokens) > 1 else "UNKNOWN" + + # Market cap product + mcap_a = mcaps.get(tok_a, {}).get("mcap_usd", 1e6) # $1M fallback + mcap_b = mcaps.get(tok_b, {}).get("mcap_usd", 1e6) + log_mcap_products.append(np.log(max(mcap_a, 1.0) * max(mcap_b, 1.0))) + + # Asset type: 0=stable, 1=native/LST, 2=volatile + type_a = _get_asset_type(tok_a, mcaps) + type_b = _get_asset_type(tok_b, mcaps) + has_stables.append(1.0 if (type_a == 0 or type_b == 0) else 0.0) + same_asset_types.append(1.0 if type_a == type_b else 0.0) + + # Weight imbalance: 0 for 50/50, 0.3 for 80/20 + weights = entry.get("weights", [0.5, 0.5]) + if len(weights) >= 2: + weight_imbalances.append(abs(weights[0] - weights[1])) + else: + weight_imbalances.append(0.0) + + # Chain dummies (first alphabetically is reference) + unique_chains = sorted(set(chains)) + chain_dummies = unique_chains[1:] # drop reference + + attr_names = ( + [f"chain_{c}" for c in chain_dummies] + + [ + "log_fee", "mean_log_tvl", "log_mcap_product", + "has_stable", "same_asset_type", "weight_imbalance", + ] + ) + k_attr = len(attr_names) + + X = np.zeros((n_pools, k_attr)) + for i, pid in enumerate(pool_ids): + chain = chains[i] + for j, cd in enumerate(chain_dummies): + if chain == cd: + X[i, j] = 1.0 + base = len(chain_dummies) + X[i, base] = log_fees[i] + X[i, base + 1] = mean_tvls[i] + X[i, base + 2] = log_mcap_products[i] + X[i, base + 3] = has_stables[i] + X[i, base + 4] = same_asset_types[i] + X[i, base + 5] = weight_imbalances[i] + + return X, attr_names, pool_ids diff --git a/tests/calibration/__init__.py b/tests/calibration/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/calibration/conftest.py b/tests/calibration/conftest.py new file mode 100644 index 00000000..2c171c01 --- /dev/null +++ b/tests/calibration/conftest.py @@ -0,0 +1,134 @@ +"""Fixtures for calibration pipeline tests.""" + +import numpy as np +import pandas as pd +import pytest + + +# ── Constants ────────────────────────────────────────────────────────────── + +N_CADENCES = 3 +N_GAS = 3 +N_DAYS = 15 +N_POOLS = 2 +K_OBS = 8 + +CADENCES = np.array([1.0, 12.0, 60.0]) +GAS_COSTS = np.array([0.0, 1.0, 5.0]) + +# Pool ID prefixes (16 chars) that map to full 66-char pool IDs +POOL_PREFIXES = ["0xaaaa11112222aa", "0xbbbb33334444bb"] +POOL_IDS_FULL = [ + "0xaaaa11112222aa63ae5d458857e731c129069f29000200000000000000000588", + "0xbbbb33334444bb9c8ef030ab642b10820db8f56000200000000000000000014", +] + + +# ── Fixtures ────────────────────────────────────────────────────────────── + + +@pytest.fixture +def synthetic_daily_grid(): + """Small per-day grid DataFrame: 3 cadences x 3 gas_costs x 15 days. + + V_arb decreasing in cadence and gas, with daily sinusoidal variation. + """ + np.random.seed(42) + dates = pd.date_range("2025-12-01", periods=N_DAYS, freq="D") + + rows = [] + for ci, cad in enumerate(CADENCES): + for gi, gas in enumerate(GAS_COSTS): + base = 10000.0 / (1 + 0.3 * ci) / (1 + 0.5 * gi) + for di, date in enumerate(dates): + daily_var = 1 + 0.1 * np.sin(2 * np.pi * di / 7) + vol = base * daily_var + np.random.normal(0, base * 0.01) + rows.append({ + "cadence": cad, + "gas_cost": gas, + "date": date, + "daily_arb_volume": max(vol, 0), + }) + + return pd.DataFrame(rows) + + +@pytest.fixture +def synthetic_panel(): + """Minimal panel DataFrame: 2 pools x 15 days. + + Columns match the real panel.parquet schema. + """ + np.random.seed(42) + dates = pd.date_range("2025-12-01", periods=N_DAYS, freq="D") + + rows = [] + for pi, (prefix, full_id) in enumerate(zip(POOL_PREFIXES, POOL_IDS_FULL)): + chain = "MAINNET" if pi == 0 else "ARBITRUM" + base_tvl = 12.0 + pi # log TVL + base_vol = 9.0 + pi + fee = 0.003 if pi == 0 else 0.01 + + for di, date in enumerate(dates): + tvl = base_tvl + 0.05 * np.sin(2 * np.pi * di / 30) + vol = base_vol + 0.3 * np.random.randn() + sigma = 0.4 + 0.1 * np.random.randn() + rows.append({ + "pool_id": full_id, + "chain": chain, + "date": date, + "log_volume": vol, + "log_tvl": tvl, + "log_tvl_lag1": tvl - 0.01 if di > 0 else np.nan, + "volatility": max(sigma, 0.01), + "weekend": 1 if date.weekday() >= 5 else 0, + "log_fee": np.log(fee), + "swap_fee": fee, + "tier_A": "major" if pi == 0 else "mid", + "tier_B": "major", + "tokens": "BTC,ETH" if pi == 0 else "AAVE,ETH", + "total_shares": 1e6 * (1 + 0.01 * di), + }) + + df = pd.DataFrame(rows) + # Drop rows where log_tvl_lag1 is NaN (first day per pool) + df = df.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + return df + + +@pytest.fixture +def synthetic_pool_coeffs(synthetic_daily_grid): + """PoolCoeffsDaily built from synthetic_daily_grid.""" + from quantammsim.calibration.grid_interpolation import precompute_pool_coeffs_daily + return precompute_pool_coeffs_daily(synthetic_daily_grid) + + +@pytest.fixture +def synthetic_x_obs(synthetic_panel): + """NumPy array (n_obs, K_OBS) from synthetic panel for one pool.""" + from quantammsim.calibration.pool_data import build_x_obs + pool0 = synthetic_panel[ + synthetic_panel["pool_id"] == POOL_IDS_FULL[0] + ] + return build_x_obs(pool0) + + +@pytest.fixture +def synthetic_pool_fit_result(): + """Dict with per-pool fitted params for testing learned mapping.""" + np.random.seed(42) + results = {} + for prefix in POOL_PREFIXES: + results[prefix] = { + "log_cadence": np.log(12.0) + 0.1 * np.random.randn(), + "log_gas": np.log(1.0) + 0.1 * np.random.randn(), + "noise_coeffs": np.random.randn(K_OBS) * 0.1, + "loss": 0.5 + 0.1 * np.random.rand(), + "converged": True, + "cadence_minutes": 12.0, + "gas_usd": 1.0, + "chain": "MAINNET" if prefix == POOL_PREFIXES[0] else "ARBITRUM", + "fee": 0.003 if prefix == POOL_PREFIXES[0] else 0.01, + "tokens": "BTC/ETH" if prefix == POOL_PREFIXES[0] else "AAVE/ETH", + } + return results diff --git a/tests/calibration/test_joint_fit.py b/tests/calibration/test_joint_fit.py new file mode 100644 index 00000000..50d77307 --- /dev/null +++ b/tests/calibration/test_joint_fit.py @@ -0,0 +1,190 @@ +"""Tests for quantammsim.calibration.joint_fit — joint end-to-end optimization (Option A).""" + +import jax +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, POOL_IDS_FULL, POOL_PREFIXES + + +@pytest.fixture +def matched_data(synthetic_daily_grid, synthetic_panel, tmp_path): + """Build matched data dict from synthetic fixtures.""" + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + +class TestPrepareJointData: + """Test prepare_joint_data: build batched arrays for joint optimization.""" + + def test_returns_expected_structure(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data) + assert hasattr(jdata, "pool_data") + assert hasattr(jdata, "x_attr") + assert hasattr(jdata, "pool_ids") + + def test_pool_count(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data) + assert len(jdata.pool_data) == len(matched_data) + + def test_pool_data_has_jax_arrays(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data) + for pd in jdata.pool_data: + assert isinstance(pd["x_obs"], jnp.ndarray) + assert isinstance(pd["y_obs"], jnp.ndarray) + assert isinstance(pd["day_indices"], jnp.ndarray) + + def test_x_attr_shape(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data) + n_pools = len(matched_data) + assert jdata.x_attr.shape[0] == n_pools + assert jdata.x_attr.shape[1] > 0 # K_attr + + +class TestJointLoss: + """Test joint_loss: end-to-end loss over all pools.""" + + def _make_loss_fn(self, matched_data): + from quantammsim.calibration.joint_fit import ( + make_joint_loss_fn, + make_initial_joint_params, + prepare_joint_data, + ) + + jdata = prepare_joint_data(matched_data) + init = make_initial_joint_params(jdata, mode="per_pool_noise") + loss_fn = make_joint_loss_fn(jdata, mode="per_pool_noise") + return loss_fn, init, jdata + + def test_loss_scalar(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + loss = loss_fn(init) + assert loss.shape == () + + def test_loss_positive(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + loss = loss_fn(init) + assert float(loss) >= 0 + + def test_loss_differentiable(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + grad = jax.grad(loss_fn)(init) + assert grad.shape == init.shape + assert jnp.all(jnp.isfinite(grad)) + + def test_loss_grad_nonzero(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + grad = jax.grad(loss_fn)(init) + assert float(jnp.sum(jnp.abs(grad))) > 0 + + def test_shared_cadence_gas_affects_all_pools(self, matched_data): + """Changing W_cad should affect losses from all pools.""" + from quantammsim.calibration.joint_fit import ( + make_joint_loss_fn, + make_initial_joint_params, + prepare_joint_data, + unpack_joint_params, + ) + + jdata = prepare_joint_data(matched_data) + init = make_initial_joint_params(jdata, mode="per_pool_noise") + loss_fn = make_joint_loss_fn(jdata, mode="per_pool_noise") + + # Gradient w.r.t. W_cad should be nonzero + grad = jax.grad(loss_fn)(init) + config = {"n_pools": len(jdata.pool_data), + "k_attr": jdata.x_attr.shape[1], + "mode": "per_pool_noise"} + params = unpack_joint_params(grad, config) + assert float(jnp.sum(jnp.abs(params["W_cad"]))) > 0 + + +class TestFitJoint: + """Test fit_joint: L-BFGS-B joint optimization.""" + + def test_returns_result(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=20) + assert isinstance(result, dict) + for key in ["bias_cad", "bias_gas", "W_cad", "W_gas", "loss", "converged"]: + assert key in result, f"Missing key: {key}" + + def test_loss_decreases(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=50) + assert result["loss"] <= result["init_loss"] + + def test_predict_new_pool(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=20) + # Predict for a new pool — k_attr must match training + k_attr = result["W_cad"].shape[0] + x_attr_new = np.zeros(k_attr) + x_attr_new[0] = 1.0 # intercept + pred = predict_new_pool_joint(result, x_attr_new) + assert "cadence_minutes" in pred + assert "gas_usd" in pred + assert pred["cadence_minutes"] > 0 + assert pred["gas_usd"] > 0 + + def test_init_from_option_c(self, matched_data): + """Warm-starting from Option C per-pool fits should work.""" + from quantammsim.calibration.joint_fit import fit_joint + from quantammsim.calibration.per_pool_fit import fit_all_pools + + option_c = fit_all_pools(matched_data) + result = fit_joint( + matched_data, mode="per_pool_noise", + init_from_option_c=option_c, maxiter=20, + ) + assert result["loss"] >= 0 + + +class TestModes: + """Test per_pool_noise vs shared_noise modes.""" + + def test_per_pool_noise_mode(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=20) + assert "noise_coeffs" in result + n_pools = len(matched_data) + assert result["noise_coeffs"].shape == (n_pools, K_OBS) + + def test_shared_noise_mode(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="shared_noise", maxiter=20) + assert "W_noise" in result + k_attr = result["W_cad"].shape[0] + assert result["W_noise"].shape == (k_attr, K_OBS) + + def test_shared_noise_predict(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint(matched_data, mode="shared_noise", maxiter=20) + k_attr = result["W_cad"].shape[0] + x_attr_new = np.zeros(k_attr) + x_attr_new[0] = 1.0 # intercept + pred = predict_new_pool_joint(result, x_attr_new) + assert "noise_coeffs" in pred + assert len(pred["noise_coeffs"]) == K_OBS diff --git a/tests/calibration/test_learned_mapping.py b/tests/calibration/test_learned_mapping.py new file mode 100644 index 00000000..8fbf16e2 --- /dev/null +++ b/tests/calibration/test_learned_mapping.py @@ -0,0 +1,152 @@ +"""Tests for quantammsim.calibration.learned_mapping — attribute -> params mapping.""" + +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, POOL_PREFIXES + + +class TestBuildTargets: + """Test build_targets: stack per-pool fitted params into target matrix.""" + + def test_target_matrix_shape(self, synthetic_pool_fit_result): + from quantammsim.calibration.learned_mapping import build_targets + + pool_order = sorted(synthetic_pool_fit_result.keys()) + Y = build_targets(synthetic_pool_fit_result, pool_order) + assert Y.shape == (len(pool_order), 2 + K_OBS) + + def test_target_ordering_matches_attributes(self, synthetic_pool_fit_result): + from quantammsim.calibration.learned_mapping import build_targets + + pool_order = sorted(synthetic_pool_fit_result.keys()) + Y = build_targets(synthetic_pool_fit_result, pool_order) + # First pool's log_cadence should be in first row + expected_lc = synthetic_pool_fit_result[pool_order[0]]["log_cadence"] + np.testing.assert_allclose(Y[0, 0], expected_lc) + + +class TestFitMapping: + """Test fit_mapping: Ridge regression from attributes to params.""" + + def _make_data(self, n_pools=10): + np.random.seed(42) + k_attr = 4 + X = np.random.randn(n_pools, k_attr) + X[:, 0] = 1.0 # intercept + W_true = np.random.randn(k_attr, 2 + K_OBS) * 0.5 + Y = X @ W_true + np.random.randn(n_pools, 2 + K_OBS) * 0.01 + return X, Y + + def test_returns_model(self): + from quantammsim.calibration.learned_mapping import fit_mapping + + X, Y = self._make_data() + model = fit_mapping(X, Y) + assert isinstance(model, dict) + assert "weights" in model + assert "intercept" in model + + def test_predict_shape(self): + from quantammsim.calibration.learned_mapping import fit_mapping + + X, Y = self._make_data() + model = fit_mapping(X, Y) + Y_pred = X @ model["weights"] + model["intercept"] + assert Y_pred.shape == Y.shape + + def test_predict_reasonable_range(self): + from quantammsim.calibration.learned_mapping import fit_mapping, predict_pool + + X, Y = self._make_data() + # Constrain Y targets to reasonable range + Y[:, 0] = np.log(np.random.uniform(1, 60, len(Y))) # log_cadence + Y[:, 1] = np.log(np.random.uniform(0.01, 10, len(Y))) # log_gas + model = fit_mapping(X, Y) + + result = predict_pool(model, X[0]) + assert 0.5 <= result["cadence_minutes"] <= 120.0 + assert result["gas_usd"] > 0 + + def test_overfit_on_training_data(self): + from quantammsim.calibration.learned_mapping import fit_mapping + + X, Y = self._make_data(n_pools=10) + model = fit_mapping(X, Y, alpha=0.001) + Y_pred = X @ model["weights"] + model["intercept"] + ss_res = np.sum((Y - Y_pred) ** 2) + ss_tot = np.sum((Y - Y.mean(axis=0)) ** 2) + r2 = 1 - ss_res / ss_tot + assert r2 > 0.8 + + def test_leave_one_out_runs(self): + from quantammsim.calibration.learned_mapping import cross_validate_loo + + X, Y = self._make_data(n_pools=10) + cv_result = cross_validate_loo(X, Y) + assert "per_pool_errors" in cv_result + assert len(cv_result["per_pool_errors"]) == 10 + + +class TestPredictNewPool: + """Test predict_pool: predict params for a single pool.""" + + def test_predict_single_pool(self): + from quantammsim.calibration.learned_mapping import fit_mapping, predict_pool + + np.random.seed(42) + X = np.random.randn(5, 3) + X[:, 0] = 1.0 + Y = np.random.randn(5, 2 + K_OBS) + model = fit_mapping(X, Y) + result = predict_pool(model, X[0]) + assert isinstance(result, dict) + + def test_predict_cadence_and_gas(self): + from quantammsim.calibration.learned_mapping import fit_mapping, predict_pool + + np.random.seed(42) + X = np.random.randn(5, 3) + X[:, 0] = 1.0 + Y = np.random.randn(5, 2 + K_OBS) + model = fit_mapping(X, Y) + result = predict_pool(model, X[0]) + assert "cadence_minutes" in result + assert "gas_usd" in result + assert "log_cadence" in result + assert "log_gas" in result + + def test_predict_noise_coeffs(self): + from quantammsim.calibration.learned_mapping import fit_mapping, predict_pool + + np.random.seed(42) + X = np.random.randn(5, 3) + X[:, 0] = 1.0 + Y = np.random.randn(5, 2 + K_OBS) + model = fit_mapping(X, Y) + result = predict_pool(model, X[0]) + assert "noise_coeffs" in result + assert len(result["noise_coeffs"]) == K_OBS + + def test_different_chains_different_predictions(self): + from quantammsim.calibration.learned_mapping import fit_mapping, predict_pool + + np.random.seed(42) + # X with chain dummy in col 1 + X = np.random.randn(10, 4) + X[:, 0] = 1.0 + X[:5, 1] = 1.0 # chain A + X[5:, 1] = 0.0 # chain B + Y = np.random.randn(10, 2 + K_OBS) + Y[:5, :] += 1.0 # chain A has different targets + + model = fit_mapping(X, Y, alpha=0.01) + + x_a = X[0].copy() + x_b = X[0].copy() + x_b[1] = 0.0 # flip chain + + pred_a = predict_pool(model, x_a) + pred_b = predict_pool(model, x_b) + # Predictions should differ + assert pred_a["log_cadence"] != pred_b["log_cadence"] diff --git a/tests/calibration/test_loss.py b/tests/calibration/test_loss.py new file mode 100644 index 00000000..8fd29580 --- /dev/null +++ b/tests/calibration/test_loss.py @@ -0,0 +1,239 @@ +"""Tests for quantammsim.calibration.loss — per-pool loss function.""" + +import jax +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, N_DAYS + + +class TestNoiseVolume: + """Test noise_volume: V_noise = exp(x_obs @ noise_coeffs).""" + + def test_noise_volume_shape(self, synthetic_x_obs): + from quantammsim.calibration.loss import noise_volume + + coeffs = jnp.zeros(K_OBS) + v = noise_volume(coeffs, jnp.array(synthetic_x_obs)) + assert v.shape == (synthetic_x_obs.shape[0],) + + def test_noise_volume_positive(self, synthetic_x_obs): + from quantammsim.calibration.loss import noise_volume + + coeffs = jnp.ones(K_OBS) * 0.1 + v = noise_volume(coeffs, jnp.array(synthetic_x_obs)) + assert jnp.all(v > 0) + + def test_noise_volume_intercept_only(self, synthetic_x_obs): + from quantammsim.calibration.loss import noise_volume + + c = 5.0 + coeffs = jnp.zeros(K_OBS).at[0].set(c) + v = noise_volume(coeffs, jnp.array(synthetic_x_obs)) + # x_obs[:, 0] is all 1s, so V_noise should be close to exp(c) + # (not exactly, because other columns are nonzero and contribute 0*x) + np.testing.assert_allclose(v, jnp.exp(c), rtol=1e-5) + + def test_noise_volume_tvl_effect(self, synthetic_x_obs): + from quantammsim.calibration.loss import noise_volume + + # Positive TVL coefficient: higher TVL → higher noise + coeffs = jnp.zeros(K_OBS).at[1].set(1.0) + v = noise_volume(coeffs, jnp.array(synthetic_x_obs)) + tvl_col = synthetic_x_obs[:, 1] + # Sort by TVL, check volume is monotone + order = np.argsort(tvl_col) + assert np.all(np.diff(np.array(v[order])) >= -1e-6) + + +class TestPoolLoss: + """Test pool_loss: per-pool log-space L2 loss with per-day V_arb.""" + + def _make_params(self, log_cad=None, log_gas=None, noise_coeffs=None): + from quantammsim.calibration.loss import pack_params + + if log_cad is None: + log_cad = float(jnp.log(jnp.array(12.0))) + if log_gas is None: + log_gas = float(jnp.log(jnp.array(1.0))) + if noise_coeffs is None: + noise_coeffs = jnp.zeros(K_OBS).at[0].set(8.0) + return pack_params(log_cad, log_gas, noise_coeffs) + + def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = jnp.array(np.arange(n_obs) % n_days) + y_obs = jnp.ones(n_obs) * 9.0 + return jnp.array(synthetic_x_obs), y_obs, day_indices + + def test_loss_scalar_output(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + loss = pool_loss(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert loss.shape == () + + def test_loss_zero_when_perfect(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.loss import noise_volume, pack_params, pool_loss + + log_cad = jnp.log(jnp.array(12.0)) + log_gas = jnp.log(jnp.array(1.0)) + noise_coeffs = jnp.zeros(K_OBS).at[0].set(8.0) + + # Compute what V_pred would be + v_arb_all = interpolate_pool_daily(synthetic_pool_coeffs, log_cad, jnp.exp(log_gas)) + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = jnp.array(np.arange(n_obs) % n_days) + v_arb = v_arb_all[day_indices] + v_noise = noise_volume(noise_coeffs, jnp.array(synthetic_x_obs)) + y_obs = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + + params = pack_params(float(log_cad), float(log_gas), noise_coeffs) + loss = pool_loss(params, synthetic_pool_coeffs, jnp.array(synthetic_x_obs), + y_obs, day_indices) + assert float(loss) < 1e-6 + + def test_loss_positive(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + loss = pool_loss(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert float(loss) >= 0 + + def test_loss_increases_with_error(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + params_ok = self._make_params(noise_coeffs=jnp.zeros(K_OBS).at[0].set(8.0)) + params_bad = self._make_params(noise_coeffs=jnp.zeros(K_OBS).at[0].set(20.0)) + + loss_ok = pool_loss(params_ok, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + loss_bad = pool_loss(params_bad, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert float(loss_bad) > float(loss_ok) + + def test_loss_differentiable(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + grad_fn = jax.grad(pool_loss, argnums=0) + g = grad_fn(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert g.shape == params.shape + + def test_loss_grad_nonzero(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + grad_fn = jax.grad(pool_loss, argnums=0) + g = grad_fn(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert float(jnp.sum(jnp.abs(g))) > 0 + + def test_loss_grad_wrt_log_cadence(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + grad_fn = jax.grad(pool_loss, argnums=0) + g = grad_fn(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert jnp.isfinite(g[0]) # log_cadence gradient + + def test_loss_grad_wrt_log_gas(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + grad_fn = jax.grad(pool_loss, argnums=0) + g = grad_fn(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert jnp.isfinite(g[1]) # log_gas gradient + + def test_loss_grad_wrt_noise_coeffs(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + grad_fn = jax.grad(pool_loss, argnums=0) + g = grad_fn(params, synthetic_pool_coeffs, x_obs, y_obs, day_indices) + assert jnp.all(jnp.isfinite(g[2:])) # noise_coeffs gradients + + def test_loss_uses_per_day_varb(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.loss import pool_loss, unpack_params + + params = self._make_params() + log_cad, log_gas, _ = unpack_params(params) + v_arb = interpolate_pool_daily( + synthetic_pool_coeffs, jnp.array(log_cad), jnp.exp(jnp.array(log_gas)) + ) + # Per-day V_arb should vary across days + assert float(jnp.std(v_arb)) > 0 + + def test_day_indices_alignment(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss + + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + params = self._make_params() + x_obs = jnp.array(synthetic_x_obs) + y_obs = jnp.ones(n_obs) * 9.0 + + # Different day_indices should give different loss + day_idx_a = jnp.zeros(n_obs, dtype=jnp.int32) # all same day + day_idx_b = jnp.array(np.arange(n_obs) % n_days) # varying days + + loss_a = pool_loss(params, synthetic_pool_coeffs, x_obs, y_obs, day_idx_a) + loss_b = pool_loss(params, synthetic_pool_coeffs, x_obs, y_obs, day_idx_b) + # Different day mappings → different losses + assert float(loss_a) != float(loss_b) + + +class TestPackUnpack: + """Test pack/unpack parameter roundtrip.""" + + def test_roundtrip(self): + from quantammsim.calibration.loss import pack_params, unpack_params + + log_cad = 2.5 + log_gas = -0.3 + noise_coeffs = jnp.array([1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0]) + + flat = pack_params(log_cad, log_gas, noise_coeffs) + lc, lg, nc = unpack_params(flat) + + np.testing.assert_allclose(lc, log_cad) + np.testing.assert_allclose(lg, log_gas) + np.testing.assert_allclose(nc, noise_coeffs) + + def test_pack_shape(self): + from quantammsim.calibration.loss import pack_params + + flat = pack_params(1.0, 2.0, jnp.zeros(K_OBS)) + assert flat.shape == (2 + K_OBS,) diff --git a/tests/calibration/test_per_pool_fit.py b/tests/calibration/test_per_pool_fit.py new file mode 100644 index 00000000..32d63ab2 --- /dev/null +++ b/tests/calibration/test_per_pool_fit.py @@ -0,0 +1,170 @@ +"""Tests for quantammsim.calibration.per_pool_fit — L-BFGS-B per-pool fitting.""" + +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, N_DAYS, POOL_IDS_FULL, POOL_PREFIXES + + +class TestFitSinglePool: + """Test fit_single_pool: L-BFGS-B optimization for one pool.""" + + def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = np.arange(n_obs) % n_days + y_obs = np.ones(n_obs) * 9.0 # log(V_obs) + return synthetic_x_obs, y_obs, day_indices + + def test_returns_result_dict(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + assert isinstance(result, dict) + for key in ["log_cadence", "log_gas", "noise_coeffs", "loss", "converged"]: + assert key in result, f"Missing key: {key}" + + def test_cadence_in_range(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + cadence = np.exp(result["log_cadence"]) + assert 1.0 <= cadence <= 60.0 + + def test_gas_positive(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + assert np.exp(result["log_gas"]) > 0 + + def test_noise_coeffs_length(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + assert len(result["noise_coeffs"]) == K_OBS + + def test_loss_decreases_from_init(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pack_params, pool_loss + from quantammsim.calibration.per_pool_fit import ( + fit_single_pool, + make_initial_guess, + ) + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + init = make_initial_guess(x_obs, y_obs) + init_loss = float(pool_loss( + jnp.array(init), synthetic_pool_coeffs, + jnp.array(x_obs), jnp.array(y_obs), jnp.array(day_idx), + )) + + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + assert result["loss"] <= init_loss + + def test_converged_flag(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool(synthetic_pool_coeffs, x_obs, y_obs, day_idx) + assert isinstance(result["converged"], bool) + + +class TestFitAllPools: + """Test fit_all_pools: fit all matched pools.""" + + def test_returns_dict_per_pool( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + results = fit_all_pools(matched) + assert isinstance(results, dict) + + def test_all_pools_have_results( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + results = fit_all_pools(matched) + for prefix in matched: + assert prefix in results + + def test_results_have_metadata( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + results = fit_all_pools(matched) + for prefix, res in results.items(): + assert "chain" in res + assert "fee" in res + assert "tokens" in res + + +class TestInitialGuess: + """Test make_initial_guess: reasonable starting point.""" + + def test_default_init_reasonable(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import make_initial_guess + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + init = make_initial_guess(synthetic_x_obs, y_obs) + assert len(init) == 2 + K_OBS + # log_cadence ~ log(12) + np.testing.assert_allclose(init[0], np.log(12.0), atol=0.1) + # log_gas ~ log(1.0) + np.testing.assert_allclose(init[1], np.log(1.0), atol=0.1) + + def test_init_noise_from_ols(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import make_initial_guess + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + init = make_initial_guess(synthetic_x_obs, y_obs) + noise_coeffs = init[2:] + # OLS should give finite values + assert np.all(np.isfinite(noise_coeffs)) diff --git a/tests/calibration/test_pool_data.py b/tests/calibration/test_pool_data.py new file mode 100644 index 00000000..b3a9cf76 --- /dev/null +++ b/tests/calibration/test_pool_data.py @@ -0,0 +1,303 @@ +"""Tests for quantammsim.calibration.pool_data — data assembly.""" + +import numpy as np +import pandas as pd +import pytest + +from tests.calibration.conftest import ( + K_OBS, + N_DAYS, + POOL_IDS_FULL, + POOL_PREFIXES, +) + + +class TestMatchGridsToPanel: + """Test match_grids_to_panel: match grid parquets to panel rows.""" + + def test_match_returns_dict_per_pool( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + # Write grid for pool 0 + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + assert isinstance(matched, dict) + assert POOL_PREFIXES[0] in matched + + def test_match_filters_to_grid_pools_only( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + # Only write grid for pool 0 — pool 1 should be excluded + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + assert POOL_PREFIXES[0] in matched + assert POOL_PREFIXES[1] not in matched + + def test_match_includes_panel_obs( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + entry = matched[POOL_PREFIXES[0]] + assert "panel" in entry + assert isinstance(entry["panel"], pd.DataFrame) + assert len(entry["panel"]) > 0 + + def test_match_includes_coeffs( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.grid_interpolation import PoolCoeffsDaily + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + assert "coeffs" in matched[POOL_PREFIXES[0]] + assert isinstance(matched[POOL_PREFIXES[0]]["coeffs"], PoolCoeffsDaily) + + def test_pool_id_prefix_matching( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + entry = matched[POOL_PREFIXES[0]] + assert entry["pool_id"] == POOL_IDS_FULL[0] + + def test_match_includes_day_indices( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + entry = matched[POOL_PREFIXES[0]] + assert "day_indices" in entry + assert len(entry["day_indices"]) == len(entry["panel"]) + + def test_day_indices_align_dates( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + synthetic_daily_grid.to_parquet( + grid_dir / f"{POOL_PREFIXES[0]}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + entry = matched[POOL_PREFIXES[0]] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + + # Panel dates should map to grid dates via ordinals + panel_dates = pd.to_datetime(entry["panel"]["date"]) + panel_ordinals = np.array([d.toordinal() for d in panel_dates]) + grid_ordinals = np.array(coeffs.dates) + + for i, panel_ord in enumerate(panel_ordinals): + grid_idx = day_indices[i] + assert grid_ordinals[grid_idx] == panel_ord + + +class TestBuildXObs: + """Test build_x_obs: observation covariate matrix.""" + + def test_x_obs_shape(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + assert x.shape == (len(pool0), K_OBS) + + def test_x_obs_intercept_column(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + np.testing.assert_array_equal(x[:, 0], 1.0) + + def test_x_obs_lagged_tvl(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + np.testing.assert_allclose(x[:, 1], pool0["log_tvl_lag1"].values) + + def test_x_obs_log_sigma(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + expected = np.log(np.maximum(pool0["volatility"].values, 1e-6)) + np.testing.assert_allclose(x[:, 2], expected) + + def test_x_obs_interactions(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + tvl = pool0["log_tvl_lag1"].values + sigma = np.log(np.maximum(pool0["volatility"].values, 1e-6)) + fee = pool0["log_fee"].values + np.testing.assert_allclose(x[:, 3], tvl * sigma) + np.testing.assert_allclose(x[:, 4], tvl * fee) + np.testing.assert_allclose(x[:, 5], sigma * fee) + + def test_x_obs_dow_harmonics(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + weekdays = pd.to_datetime(pool0["date"]).dt.weekday.values + expected_sin = np.sin(2 * np.pi * weekdays / 7) + expected_cos = np.cos(2 * np.pi * weekdays / 7) + np.testing.assert_allclose(x[:, 6], expected_sin, atol=1e-10) + np.testing.assert_allclose(x[:, 7], expected_cos, atol=1e-10) + + def test_x_obs_no_nans(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + assert not np.any(np.isnan(x)) + + +class TestBuildPoolAttributes: + """Test build_pool_attributes: pool-level feature matrix.""" + + def test_attributes_shape( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + match_grids_to_panel, + ) + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + assert X_attr.shape[0] == len(matched) + assert X_attr.shape[1] == len(attr_names) + + def test_attributes_has_chain( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + match_grids_to_panel, + ) + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + chain_cols = [n for n in attr_names if n.startswith("chain_")] + assert len(chain_cols) > 0 + + def test_attributes_has_log_fee( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + match_grids_to_panel, + ) + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + assert "log_fee" in attr_names + + def test_attributes_has_log_tvl( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + match_grids_to_panel, + ) + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + assert "mean_log_tvl" in attr_names + + def test_attributes_returns_pool_order( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + match_grids_to_panel, + ) + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + assert isinstance(pool_ids, list) + assert len(pool_ids) == len(matched) + assert set(pool_ids) == set(matched.keys()) From 953144ea8b2b4513107247d6751d32aa144ca5b3 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:44:30 +0000 Subject: [PATCH 028/115] feat: add analysis scripts, experiments, and CLI parameter wiring - tune_reclamm_params: wire noise_trader_ratio, date ranges, bout_offset, val_fraction, overfitting_penalty from CLI args - compare_exposures: pool exposure comparison experiment - scripts: noise calibration runners, reclamm benchmarks, plotting utilities, demo simulation, token mcap fetcher - historic_data_utils: minor cleanup --- experiments/compare_exposures.py | 141 ++ experiments/tune_reclamm_params.py | 21 +- .../data_processing/historic_data_utils.py | 3 - scripts/benchmark_reclamm_interpolation.py | 648 +++++++ scripts/build_pool_grids.py | 695 ++++++++ scripts/calibrate_noise_bayesian.py | 853 +++++++++ scripts/calibrate_noise_hierarchical.py | 1556 +++++++++++++++++ scripts/calibrate_noise_unified.py | 6 + scripts/calibrate_reclamm_noise.py | 842 +++++++++ scripts/compare_reclamm_thermostats.py | 379 ++++ scripts/demo_run_reclamm.py | 207 +++ scripts/fetch_token_mcaps.py | 196 +++ scripts/plot_predicted_vs_real_volume.py | 151 ++ scripts/plot_reclamm_optuna_result.py | 451 +++++ scripts/plot_top50_predicted_vs_real.py | 521 ++++++ scripts/run_direct_calibration_top50.py | 852 +++++++++ scripts/run_structural_top50.py | 448 +++++ 17 files changed, 7964 insertions(+), 6 deletions(-) create mode 100644 experiments/compare_exposures.py create mode 100644 scripts/benchmark_reclamm_interpolation.py create mode 100644 scripts/build_pool_grids.py create mode 100644 scripts/calibrate_noise_bayesian.py create mode 100644 scripts/calibrate_noise_hierarchical.py create mode 100644 scripts/calibrate_noise_unified.py create mode 100644 scripts/calibrate_reclamm_noise.py create mode 100644 scripts/compare_reclamm_thermostats.py create mode 100644 scripts/demo_run_reclamm.py create mode 100644 scripts/fetch_token_mcaps.py create mode 100644 scripts/plot_predicted_vs_real_volume.py create mode 100644 scripts/plot_reclamm_optuna_result.py create mode 100644 scripts/plot_top50_predicted_vs_real.py create mode 100644 scripts/run_direct_calibration_top50.py create mode 100644 scripts/run_structural_top50.py diff --git a/experiments/compare_exposures.py b/experiments/compare_exposures.py new file mode 100644 index 00000000..14fdd06a --- /dev/null +++ b/experiments/compare_exposures.py @@ -0,0 +1,141 @@ +"""Compare effective weight/exposure trajectories: reClAMM vs Balancer 50/50. + +Prints weight stats and saves a plot of weight[AAVE] over time for both pools. +""" + +import jax.numpy as jnp +import numpy as np +import matplotlib.pyplot as plt +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(exponent): + return 1.0 - exponent / 124649.0 + + +TOKENS = ["AAVE", "ETH"] +START = "2024-06-01 00:00:00" +END = "2025-06-01 00:00:00" + +CONFIGS = { + "reClAMM on-chain (pr=1.5)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(0.1)), + }, + }, + "reClAMM wide (pr=4)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(1.0)), + }, + }, + "reClAMM Phase 2 (pr=4, m=0.1)": { + "fingerprint": { + "tokens": TOKENS, "rule": "reclamm", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.1), + "daily_price_shift_base": jnp.array(to_daily_price_shift_base(0.001)), + }, + }, + "Balancer 50/50": { + "fingerprint": { + "tokens": TOKENS, "rule": "balancer", + "startDateString": START, "endDateString": END, + "initial_pool_value": 1_000_000.0, "do_arb": True, + "fees": 0.0, "gas_cost": 0.0, "arb_fees": 0.0, + "chunk_period": 60, "weight_interpolation_period": 60, + }, + "params": { + "initial_weights_logits": jnp.zeros(2), + }, + }, +} + +results = {} +for name, cfg in CONFIGS.items(): + print(f"Running {name}...") + r = do_run_on_historic_data( + run_fingerprint=cfg["fingerprint"], params=cfg["params"] + ) + results[name] = r + +# Compute effective weights (value fraction in token 0 = AAVE) +print("\n" + "=" * 90) +print(f" {'Config':<35s} {'w_AAVE mean':>10s} {'w_AAVE std':>10s} " + f"{'w_AAVE min':>10s} {'w_AAVE max':>10s} {'vs HODL':>10s}") +print("-" * 90) + +daily = 1440 # subsample to daily for stats and plotting +fig, axes = plt.subplots(3, 1, figsize=(14, 10), sharex=True) + +for name, r in results.items(): + reserves = np.array(r["reserves"]) + prices = np.array(r["prices"]) + values = reserves * prices # (T, 2) + total = values.sum(axis=1, keepdims=True) + weights = values / np.clip(total, 1e-10, None) # (T, 2) + w_aave = weights[::daily, 0] + + hodl_value = float((reserves[0] * prices[-1]).sum()) + vs_hodl = r["final_value"] / hodl_value - 1.0 + + print(f" {name:<35s} {w_aave.mean():>10.4f} {w_aave.std():>10.4f} " + f"{w_aave.min():>10.4f} {w_aave.max():>10.4f} {vs_hodl * 100:>9.2f}%") + + days = np.arange(len(w_aave)) + axes[0].plot(days, w_aave, label=name, alpha=0.8) + + # Pool value over time + pool_val = np.array(r["value"])[::daily] + axes[1].plot(days[:len(pool_val)], pool_val / 1e6, label=name, alpha=0.8) + +print("=" * 90) + +# HODL line +r0 = results[list(results.keys())[0]] +prices_daily = np.array(r0["prices"])[::daily] +reserves_0 = np.array(r0["reserves"])[0] +hodl_val = (reserves_0 * prices_daily).sum(axis=1) / 1e6 +axes[1].plot(np.arange(len(hodl_val)), hodl_val, label="HODL", ls="--", color="gray", alpha=0.7) + +# Price ratio (AAVE/ETH) on third axis +price_ratio_series = prices_daily[:, 0] / prices_daily[:, 1] +axes[2].plot(np.arange(len(price_ratio_series)), price_ratio_series, color="black", alpha=0.7) +axes[2].set_ylabel("AAVE/ETH price") +axes[2].set_xlabel("Days") + +axes[0].set_ylabel("AAVE weight (value fraction)") +axes[0].axhline(0.5, ls="--", color="gray", alpha=0.5) +axes[0].legend(fontsize=8) +axes[0].set_title("Effective AAVE exposure over time") + +axes[1].set_ylabel("Pool value ($M)") +axes[1].legend(fontsize=8) +axes[1].set_title("Pool value over time") + +plt.tight_layout() +plt.savefig("reclamm_exposure_comparison.png", dpi=150) +print("\nSaved reclamm_exposure_comparison.png") diff --git a/experiments/tune_reclamm_params.py b/experiments/tune_reclamm_params.py index 0951e2c1..49699230 100644 --- a/experiments/tune_reclamm_params.py +++ b/experiments/tune_reclamm_params.py @@ -38,6 +38,17 @@ def main(): parser.add_argument("--interpolation", default="geometric", choices=["geometric", "constant_arc_length"]) parser.add_argument("--centeredness-scaling", action="store_true") + parser.add_argument("--noise-trader-ratio", type=float, default=0.0) + parser.add_argument("--start-date", default="2024-06-01 00:00:00") + parser.add_argument("--end-date", default="2025-01-01 00:00:00", + help="End of training / start of test") + parser.add_argument("--end-test-date", default="2025-06-01 00:00:00") + parser.add_argument("--bout-offset", type=int, default=None, + help="bout_offset in minutes (default: 10080 = 7 days)") + parser.add_argument("--val-fraction", type=float, default=None, + help="Validation holdout fraction (default: 0.2, use 0 to disable)") + parser.add_argument("--overfitting-penalty", type=float, default=None, + help="Overfitting penalty weight (default: 0.2)") args = parser.parse_args() learn_speed = args.interpolation == "constant_arc_length" @@ -48,29 +59,33 @@ def main(): fp = { "rule": "reclamm", "tokens": ["AAVE", "ETH"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-01-01 00:00:00", - "endTestDateString": "2025-06-01 00:00:00", + "startDateString": args.start_date, + "endDateString": args.end_date, + "endTestDateString": args.end_test_date, "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": args.fees, "gas_cost": args.gas_cost, "arb_fees": 0.0, "protocol_fee_split": 0.5, + "noise_trader_ratio": args.noise_trader_ratio, "return_val": args.objective, "reclamm_interpolation_method": args.interpolation, "reclamm_centeredness_scaling": args.centeredness_scaling, "reclamm_learn_arc_length_speed": learn_speed, "reclamm_use_shift_exponent": True, + **({"bout_offset": args.bout_offset} if args.bout_offset is not None else {}), "optimisation_settings": { "method": "optuna", "n_parameter_sets": 1, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), "optuna_settings": { "make_scalar": True, "expand_around": False, "n_trials": args.n_trials, "multi_objective": False, "parameter_config": param_config, + **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), }, }, } diff --git a/quantammsim/utils/data_processing/historic_data_utils.py b/quantammsim/utils/data_processing/historic_data_utils.py index 31fb76c3..b83a97b7 100644 --- a/quantammsim/utils/data_processing/historic_data_utils.py +++ b/quantammsim/utils/data_processing/historic_data_utils.py @@ -83,9 +83,6 @@ def start_and_end_calcs( if oracle_values is not None: oracle_values = oracle_values[remainder_idx:] - print("start_date: ", start_date) - print("end_date: ", end_date) - print("unix_values: ", unix_values) start_idx = np.where(unix_values == start_date)[0][0] end_idx = np.where(unix_values == end_date)[0][0] + 1 else: diff --git a/scripts/benchmark_reclamm_interpolation.py b/scripts/benchmark_reclamm_interpolation.py new file mode 100644 index 00000000..f462bbf0 --- /dev/null +++ b/scripts/benchmark_reclamm_interpolation.py @@ -0,0 +1,648 @@ +"""Benchmark reClAMM range shift interpolation: current vs optimal midpoint. + +Compares total arb loss during a range shift under different interpolation methods: + Geometric VB -- exponential decay of overvalued virtual (what contracts do) + Linear VB -- uniform steps in VB + Linear Z -- uniform steps in Z = sqrt(P)*VA - VB/sqrt(P) (optimal, from note) + Optimal 2-step -- exact midpoint via quadratic formula (Section 5 of note) + Brute-force optimal -- JAX gradient-optimised Z-target sequence + +Key result: per-step loss ~ (DeltaZ)^2 / (4X). Equal Z-increments minimise +total loss, analogous to TFMM optimal intermediate for G3M weight changes. + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/benchmark_reclamm_interpolation.py +""" + +import numpy as np +import matplotlib.pyplot as plt + +import jax +import jax.numpy as jnp +from scipy.optimize import minimize as scipy_minimize + +jax.config.update("jax_enable_x64", True) + + +# ── Core reClAMM mechanics ───────────────────────────────────────────────── + + +def compute_VA_from_VB(RA, RB, VB, Q): + """Contract rule (eq 15): VA = RA*(VB + RB) / ((Q-1)*VB - RB).""" + return RA * (VB + RB) / ((Q - 1) * VB - RB) + + +def compute_Z(VA, VB, P): + """Z = sqrt(P)*VA - VB/sqrt(P) (eq 12).""" + sqP = np.sqrt(P) + return sqP * VA - VB / sqP + + +def pool_value(RA, RB, P): + """Real pool value: P*RA + RB (eq 3).""" + return P * RA + RB + + +def micro_step(RA, RB, VA_new, VB_new, P): + """Virtual-balance update then arb to equilibrium Y/X = P. + + Returns (RA_new, RB_new, arb_loss). + """ + val_before = pool_value(RA, RB, P) + X = RA + VA_new + Y = RB + VB_new + L = X * Y + X_eq = np.sqrt(L / P) + Y_eq = P * X_eq + RA_new = X_eq - VA_new + RB_new = Y_eq - VB_new + return RA_new, RB_new, val_before - pool_value(RA_new, RB_new, P) + + +def solve_VB_for_Z(RA, RB, Z_star, Q, P): + """Solve quadratic for VB achieving Z(VB) = Z_star. + + Derived by substituting VA = RA*(VB+RB)/((Q-1)*VB-RB) into + Z = sqrt(P)*VA - VB/sqrt(P), then collecting terms in VB. + + NOTE: The research note (eq 28) has a sign error: the RB/sqrt(P) + term in b should be positive, not negative. Re-derived here from + scratch. + + Returns the physically valid root (VB > RB/(Q-1), positive). + Raises ValueError if no valid root exists. + """ + sqP = np.sqrt(P) + a = -(Q - 1) / sqP + b = sqP * RA + RB / sqP - (Q - 1) * Z_star # +RB/sqP, not minus + c = sqP * RA * RB + Z_star * RB + disc = b * b - 4 * a * c + if disc < -1e-6: + raise ValueError(f"negative discriminant: {disc:.4e}") + disc = max(disc, 0.0) + sd = np.sqrt(disc) + r1, r2 = (-b + sd) / (2 * a), (-b - sd) / (2 * a) + floor = RB / (Q - 1) + 1e-12 + ok = [r for r in (r1, r2) if r > floor] + if not ok: + raise ValueError(f"no valid root: r1={r1:.4f}, r2={r2:.4f}, floor={floor:.4f}") + return min(ok) + + +# ── Interpolation methods ────────────────────────────────────────────────── + + +def run_shift(RA, RB, VA_stale, VB_start, VB_end, Q, P, N, schedule): + """Execute N-step range shift (B overvalued, VB decreasing). + + schedule: "geometric" | "linear_VB" | "linear_Z" + + VA_stale: the current (possibly stale) VA -- used only for Z_start + in the linear_Z schedule. All micro-steps compute VA from the + contract rule with current reserves. + """ + # For linear_Z, precompute Z endpoints using contract-rule VA + if schedule == "linear_Z": + VA_start_cr = compute_VA_from_VB(RA, RB, VB_start, Q) + Z0 = compute_Z(VA_start_cr, VB_start, P) + VA_end_approx = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end = compute_Z(VA_end_approx, VB_end, P) + + total_loss = 0.0 + RA_c, RB_c = RA, RB + + for i in range(1, N + 1): + frac = i / N + if schedule == "geometric": + VB_i = VB_start * (VB_end / VB_start) ** frac + elif schedule == "linear_VB": + VB_i = VB_start + frac * (VB_end - VB_start) + elif schedule == "linear_Z": + Z_i = Z0 + frac * (Z_end - Z0) + VB_i = solve_VB_for_Z(RA_c, RB_c, Z_i, Q, P) + else: + raise ValueError(schedule) + + VA_i = compute_VA_from_VB(RA_c, RB_c, VB_i, Q) + RA_c, RB_c, loss = micro_step(RA_c, RB_c, VA_i, VB_i, P) + total_loss += loss + + return total_loss, RA_c, RB_c + + +def run_shift_optimal_2step(RA, RB, VA_stale, VB_start, VB_end, Q, P): + """Exact 2-step optimal midpoint (Section 5 of the note). + + Computes Z* = (Z_start + Z_end) / 2, solves quadratic for VB_mid. + """ + VA_start_cr = compute_VA_from_VB(RA, RB, VB_start, Q) + Z0 = compute_Z(VA_start_cr, VB_start, P) + VA_end_approx = compute_VA_from_VB(RA, RB, VB_end, Q) + Z2 = compute_Z(VA_end_approx, VB_end, P) + Z_star = (Z0 + Z2) / 2.0 + + # Step 1: jump to Z-midpoint + VB_mid = solve_VB_for_Z(RA, RB, Z_star, Q, P) + VA_mid = compute_VA_from_VB(RA, RB, VB_mid, Q) + RA1, RB1, loss1 = micro_step(RA, RB, VA_mid, VB_mid, P) + + # Step 2: jump to endpoint + VA_end = compute_VA_from_VB(RA1, RB1, VB_end, Q) + RA2, RB2, loss2 = micro_step(RA1, RB1, VA_end, VB_end, P) + + return loss1 + loss2, RA2, RB2 + + +# ── Scenario setup ───────────────────────────────────────────────────────── + + +def setup_centered_pool(P, price_ratio, R_scale=10000.0): + """Centered pool at price P with contract-rule-consistent virtuals. + + Returns (RA, RB, VA, VB, Q). + """ + Q = np.sqrt(price_ratio) + q4 = price_ratio ** 0.25 + + RA = R_scale + RB = P * R_scale + VA = RA / (q4 - 1) + VB = RB / (q4 - 1) + + return RA, RB, VA, VB, Q + + +def setup_decentered_pool(P_init, P_final, price_ratio, R_scale=10000.0): + """Centered pool at P_init, arb to P_final, then refresh virtuals. + + The refresh applies the contract rule to get consistent (VA, VB) at + the post-arb reserves, then arbs once more. This gives a decentered + but fully consistent state (equilibrium + contract rule). + + Returns (RA, RB, VA, VB, Q). + """ + Q = np.sqrt(price_ratio) + q4 = price_ratio ** 0.25 + + RA0 = R_scale + RB0 = P_init * R_scale + VA0 = RA0 / (q4 - 1) + VB0 = RB0 / (q4 - 1) + + # Arb to P_final (L preserved, virtuals stale) + X0 = RA0 + VA0 + Y0 = RB0 + VB0 + L = X0 * Y0 + X_new = np.sqrt(L / P_final) + Y_new = np.sqrt(L * P_final) + RA = X_new - VA0 + RB = Y_new - VB0 + + # Refresh: apply contract rule for current VB, then arb + VB = VB0 + VA = compute_VA_from_VB(RA, RB, VB, Q) + RA, RB, _ = micro_step(RA, RB, VA, VB, P_final) + + return RA, RB, VA, VB, Q + + +# ── JAX-differentiable versions for brute-force optimisation ────────────── + + +def _compute_VA_from_VB_jax(RA, RB, VB, Q): + return RA * (VB + RB) / ((Q - 1) * VB - RB) + + +def _compute_Z_jax(VA, VB, P): + sqP = jnp.sqrt(P) + return sqP * VA - VB / sqP + + +def _pool_value_jax(RA, RB, P): + return P * RA + RB + + +def _micro_step_jax(RA, RB, VA, VB, P): + val_before = _pool_value_jax(RA, RB, P) + X = RA + VA + Y = RB + VB + L = X * Y + X_eq = jnp.sqrt(L / P) + Y_eq = P * X_eq + RA_new = X_eq - VA + RB_new = Y_eq - VB + return RA_new, RB_new, val_before - _pool_value_jax(RA_new, RB_new, P) + + +def _solve_VB_for_Z_jax(RA, RB, Z_star, Q, P): + sqP = jnp.sqrt(P) + a = -(Q - 1) / sqP + b = sqP * RA + RB / sqP - (Q - 1) * Z_star + c = sqP * RA * RB + Z_star * RB + disc = jnp.maximum(b * b - 4 * a * c, 1e-30) + sd = jnp.sqrt(disc) + r1 = (-b + sd) / (2 * a) + r2 = (-b - sd) / (2 * a) + floor = RB / (Q - 1) + 1e-8 + return jnp.where(r2 > floor, r2, r1) + + +def _z_targets_from_raw(raw_params, Z_start, Z_end): + """Map unconstrained params -> sorted Z targets via softplus gaps.""" + gaps = jax.nn.softplus(raw_params) + gaps = gaps / jnp.sum(gaps) * (Z_end - Z_start) + return Z_start + jnp.cumsum(gaps) + + +def _make_loss_fn(N): + """Build a JIT-compiled loss function for a given N (unrolled loop).""" + + def total_loss(raw_params, RA, RB, Q, P, Z_start, Z_end): + Z_all = _z_targets_from_raw(raw_params, Z_start, Z_end) + RA_c, RB_c = RA, RB + total = 0.0 + for i in range(N): + VB_i = _solve_VB_for_Z_jax(RA_c, RB_c, Z_all[i], Q, P) + VA_i = _compute_VA_from_VB_jax(RA_c, RB_c, VB_i, Q) + RA_c, RB_c, loss = _micro_step_jax(RA_c, RB_c, VA_i, VB_i, P) + total = total + loss + return total + + return jax.jit(jax.value_and_grad(total_loss)) + + +def optimise_z_targets(RA, RB, Q, P, Z_start, Z_end, N, verbose=False): + """Find the Z-target sequence minimising total arb loss. + + Returns (optimal_loss, optimal_Z_targets_array_of_length_N). + """ + loss_and_grad_fn = _make_loss_fn(N) + RA_j = jnp.float64(RA) + RB_j = jnp.float64(RB) + Q_j = jnp.float64(Q) + P_j = jnp.float64(P) + Zs_j = jnp.float64(Z_start) + Ze_j = jnp.float64(Z_end) + + def objective(x): + val, grad = loss_and_grad_fn( + jnp.array(x, dtype=jnp.float64), RA_j, RB_j, Q_j, P_j, Zs_j, Ze_j + ) + return float(val), np.array(grad, dtype=np.float64) + + x0 = np.zeros(N) # softplus(0) = ln2, uniform gaps → linear Z init + result = scipy_minimize(objective, x0, jac=True, method="L-BFGS-B") + + optimal_Z = np.array( + _z_targets_from_raw(jnp.array(result.x), Zs_j, Ze_j) + ) + if verbose: + print(f" N={N}: loss={result.fun:.6f} " + f"nit={result.nit} success={result.success}") + return result.fun, optimal_Z + + +# ── Experiments ──────────────────────────────────────────────────────────── + + +def main(): + # --- Scenario: centered pool, moderate VB decay --- + P = 2.0 # token A costs 2 units of token B + price_ratio = 4.0 # rho, so Q = sqrt(4) = 2 + R_scale = 10000.0 + decay_fraction = 0.90 # VB_end = 0.90 * VB_start (10% decay) + + RA, RB, VA, VB, Q = setup_centered_pool(P, price_ratio, R_scale) + VB_start = VB + VB_end = VB * decay_fraction + + # Diagnostics + C = min(RA * VB, RB * VA) / max(RA * VB, RB * VA) + is_above = RA * VB > RB * VA + X = RA + VA + print("=" * 72) + print(f"Scenario: centered pool at P={P}, price_ratio={price_ratio}, Q={Q:.4f}") + print(f" RA={RA:.2f} RB={RB:.2f} VA={VA:.2f} VB={VB:.2f}") + print(f" Effective X={X:.2f} Pool value = {pool_value(RA, RB, P):.2f}") + print(f" Centeredness = {C:.4f} is_above = {is_above}") + print(f" VB shift: {VB_start:.2f} -> {VB_end:.2f} ({decay_fraction:.0%})") + VB_floor = RB / (Q - 1) + print(f" VB floor (denominator > 0): {VB_floor:.2f}") + Z_start = compute_Z(VA, VB, P) + VA_end_cr = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end = compute_Z(VA_end_cr, VB_end, P) + print(f" Z_start = {Z_start:.4f} Z_end = {Z_end:.4f}") + print(f" Approx 1-step loss ~ (DeltaZ)^2/(4X) = {(Z_end-Z_start)**2/(4*X):.2f}") + print("=" * 72) + + # ── Experiment 1: Loss vs N ──────────────────────────────────────── + + N_values = [1, 2, 3, 4, 6, 8, 12, 16, 24, 32, 48, 64, 96, 128] + schedules = ["geometric", "linear_VB", "linear_Z"] + results = {s: [] for s in schedules} + + for N in N_values: + for sched in schedules: + try: + loss, _, _ = run_shift( + RA, RB, VA, VB_start, VB_end, Q, P, N, sched + ) + except (ValueError, AssertionError) as e: + loss = np.nan + results[sched].append(loss) + + # Optimal 2-step (single point) + try: + loss_opt2, _, _ = run_shift_optimal_2step( + RA, RB, VA, VB_start, VB_end, Q, P + ) + except (ValueError, AssertionError): + loss_opt2 = np.nan + + # Table + loss_1 = results["geometric"][0] + print(f"\n{'N':>5s} {'Geo VB':>12s} {'Lin VB':>12s} {'Lin Z':>12s}" + f" {'Geo/1step':>9s} {'LinZ/1step':>10s} {'LinZ/Geo':>9s}") + print("-" * 80) + for j, N in enumerate(N_values): + g = results["geometric"][j] + lv = results["linear_VB"][j] + lz = results["linear_Z"][j] + print(f"{N:>5d} {g:>12.6f} {lv:>12.6f} {lz:>12.6f}" + f" {g / loss_1:>9.4f} {lz / loss_1:>10.4f} {lz / g:>9.4f}") + + print(f"\n Optimal 2-step loss: {loss_opt2:.6f}") + print(f" Geometric N=2 loss: {results['geometric'][1]:.6f}" + f" (opt/geo = {loss_opt2 / results['geometric'][1]:.4f})") + print(f" Linear Z N=2 loss: {results['linear_Z'][1]:.6f}" + f" (opt/linZ = {loss_opt2 / results['linear_Z'][1]:.4f})") + + # ── Experiment 2: Z and VB trajectories at N=8 ───────────────────── + + N_viz = 8 + traj_data = {} + for sched in schedules: + VB_traj, Z_traj, loss_traj = [VB_start], [], [] + VA_s = VA # stale + Z_traj.append(compute_Z(VA_s, VB_start, P)) + + RA_c, RB_c = RA, RB + if sched == "linear_Z": + Z0 = Z_traj[0] + VA_end_a = compute_VA_from_VB(RA, RB, VB_end, Q) + Z_end_val = compute_Z(VA_end_a, VB_end, P) + + for i in range(1, N_viz + 1): + frac = i / N_viz + if sched == "geometric": + VB_i = VB_start * (VB_end / VB_start) ** frac + elif sched == "linear_VB": + VB_i = VB_start + frac * (VB_end - VB_start) + else: + Z_i = Z0 + frac * (Z_end_val - Z0) + VB_i = solve_VB_for_Z(RA_c, RB_c, Z_i, Q, P) + + try: + VA_i = compute_VA_from_VB(RA_c, RB_c, VB_i, Q) + VB_traj.append(VB_i) + Z_traj.append(compute_Z(VA_i, VB_i, P)) + RA_c, RB_c, loss = micro_step(RA_c, RB_c, VA_i, VB_i, P) + loss_traj.append(loss) + except (ValueError, AssertionError): + break + + traj_data[sched] = { + "VB": np.array(VB_traj), + "Z": np.array(Z_traj), + "loss": np.array(loss_traj), + } + + # ── Experiment 3: sweep shift size at N=2 ────────────────────────── + + decay_sweep = np.linspace(0.80, 0.99, 30) + sweep = {s: [] for s in ["geometric", "linear_Z", "optimal_2step"]} + for df in decay_sweep: + VB_e = VB * df + try: + g, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "geometric") + lz, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "linear_Z") + o2, _, _ = run_shift_optimal_2step(RA, RB, VA, VB_start, VB_e, Q, P) + except (AssertionError, ValueError): + g = lz = o2 = np.nan + sweep["geometric"].append(g) + sweep["linear_Z"].append(lz) + sweep["optimal_2step"].append(o2) + + # ── Plots ────────────────────────────────────────────────────────── + + colours = {"geometric": "C0", "linear_VB": "C1", "linear_Z": "C2"} + labels = { + "geometric": "Geometric VB (contract)", + "linear_VB": "Linear VB", + "linear_Z": "Linear Z (optimal)", + } + + fig, axes = plt.subplots(2, 2, figsize=(13, 10)) + + # (0,0) Loss vs N + ax = axes[0, 0] + for s in schedules: + ax.plot(N_values, results[s], "o-", ms=4, color=colours[s], label=labels[s]) + ax.axhline(loss_opt2, color="C3", ls=":", label=f"Optimal 2-step = {loss_opt2:.4f}") + ax.set_xlabel("Steps N") + ax.set_ylabel("Total arb loss") + ax.set_title("Arb loss vs interpolation steps") + ax.set_xscale("log") + ax.set_yscale("log") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (0,1) Ratio linear_Z / geometric + ax = axes[0, 1] + ratios = np.array(results["linear_Z"]) / np.array(results["geometric"]) + ax.plot(N_values, ratios, "o-", color="C2") + ax.axhline(1.0, color="gray", ls="--", alpha=0.5) + ax.set_xlabel("Steps N") + ax.set_ylabel("Loss(Linear Z) / Loss(Geometric VB)") + ax.set_title("Relative improvement of Z-optimal") + ax.grid(True, alpha=0.3) + + # (1,0) Z trajectories at N=8 + ax = axes[1, 0] + steps = np.arange(N_viz + 1) + for s in schedules: + ax.plot(steps, traj_data[s]["Z"], "o-", ms=4, color=colours[s], label=labels[s]) + ax.set_xlabel("Step") + ax.set_ylabel("Z = sqrt(P)*VA - VB/sqrt(P)") + ax.set_title(f"Z trajectory (N={N_viz})") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (1,1) 2-step loss vs shift size + ax = axes[1, 1] + shift_pct = (1 - decay_sweep) * 100 + ax.plot(shift_pct, sweep["geometric"], color="C0", label="Geometric VB (N=2)") + ax.plot(shift_pct, sweep["linear_Z"], color="C2", label="Linear Z (N=2)") + ax.plot(shift_pct, sweep["optimal_2step"], ":", color="C3", label="Optimal 2-step") + ax.set_xlabel("Shift size (% VB decay)") + ax.set_ylabel("Arb loss") + ax.set_title("2-step loss vs shift magnitude") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_interpolation_benchmark.png", dpi=150) + print("\nSaved reclamm_interpolation_benchmark.png") + + # ── Per-step loss bar chart for N=8 ──────────────────────────────── + + fig2, ax = plt.subplots(figsize=(10, 5)) + x = np.arange(1, N_viz + 1) + w = 0.25 + for i, s in enumerate(schedules): + ax.bar(x + i * w, traj_data[s]["loss"], w, color=colours[s], label=labels[s]) + ax.set_xlabel("Step") + ax.set_ylabel("Per-step arb loss") + ax.set_title(f"Per-step loss distribution (N={N_viz})") + ax.legend(fontsize=8) + ax.set_xticks(x + w) + plt.tight_layout() + plt.savefig("reclamm_interpolation_perstep.png", dpi=150) + print("Saved reclamm_interpolation_perstep.png") + + # ── Experiment 4: small-shift regime (paper's approximation valid) ─── + + print("\n" + "=" * 72) + print("Experiment 4: Optimal 2-step vs Geometric N=2 at small shifts") + print(" (reserves nearly constant → paper's analysis should hold)") + print("-" * 72) + print(f" {'Decay %':>8s} {'Geo N=2':>12s} {'LinZ N=2':>12s} " + f"{'Opt2':>12s} {'Opt2/Geo':>9s} {'Opt2/LinZ':>9s}") + print("-" * 72) + + small_decays = [0.999, 0.998, 0.995, 0.99, 0.98, 0.95, 0.90, 0.80] + for df in small_decays: + VB_e = VB * df + try: + g, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "geometric") + lz, _, _ = run_shift(RA, RB, VA, VB_start, VB_e, Q, P, 2, "linear_Z") + o2, _, _ = run_shift_optimal_2step( + RA, RB, VA, VB_start, VB_e, Q, P + ) + except (ValueError, AssertionError) as e: + print(f" {(1-df)*100:>7.1f}% FAILED: {e}") + continue + print(f" {(1-df)*100:>7.1f}% {g:>12.6f} {lz:>12.6f} " + f"{o2:>12.6f} {o2/g:>9.6f} {o2/lz:>9.6f}") + + print("=" * 72) + + # ── Experiment 5: brute-force JAX-optimised Z targets ──────────────── + + print("\n" + "=" * 72) + print("Experiment 5: Brute-force optimal Z targets (JAX + L-BFGS-B)") + print(" Parameterisation: softplus gaps → sorted Z targets") + print(" Initialised at linear Z (uniform gaps)") + print("-" * 72) + + opt_N_values = [2, 3, 4, 6, 8, 12, 16, 24, 32] + opt_losses = {} + opt_Z_trajs = {} + + for N in opt_N_values: + loss_bf, Z_bf = optimise_z_targets( + RA, RB, Q, P, Z_start, Z_end, N, verbose=True + ) + opt_losses[N] = loss_bf + opt_Z_trajs[N] = Z_bf + + # Comparison table + print(f"\n {'N':>5s} {'Geometric':>12s} {'Linear Z':>12s} " + f"{'BF Optimal':>12s} {'BF/LinZ':>9s} {'BF/Geo':>9s}") + print("-" * 72) + for N in opt_N_values: + idx = N_values.index(N) if N in N_values else None + g = results["geometric"][idx] if idx is not None else np.nan + lz = results["linear_Z"][idx] if idx is not None else np.nan + bf = opt_losses[N] + print(f" {N:>5d} {g:>12.6f} {lz:>12.6f} " + f"{bf:>12.6f} {bf/lz:>9.6f} {bf/g:>9.6f}") + + # ── Plot: overlay brute-force on the main loss-vs-N chart ──────────── + + fig3, axes3 = plt.subplots(1, 2, figsize=(14, 5)) + + # (left) Loss vs N with brute-force overlay + ax = axes3[0] + for s in schedules: + ax.plot(N_values, results[s], "o-", ms=4, color=colours[s], + label=labels[s]) + bf_Ns = sorted(opt_losses.keys()) + bf_vals = [opt_losses[n] for n in bf_Ns] + ax.plot(bf_Ns, bf_vals, "s--", ms=5, color="C3", label="BF Optimal (JAX)") + ax.set_xlabel("Steps N") + ax.set_ylabel("Total arb loss") + ax.set_title("Arb loss vs interpolation steps (with BF optimal)") + ax.set_xscale("log") + ax.set_yscale("log") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + # (right) Z trajectory comparison at N=8 + ax = axes3[1] + N_cmp = 8 + steps_cmp = np.arange(N_cmp + 1) + + # Geometric: compute Z trajectory from VB + z_geo = [Z_start] + RA_t, RB_t = RA, RB + for i in range(1, N_cmp + 1): + frac = i / N_cmp + VB_i = VB_start * (VB_end / VB_start) ** frac + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_geo.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + # Linear Z + z_linz = [Z_start] + RA_t, RB_t = RA, RB + for i in range(1, N_cmp + 1): + frac = i / N_cmp + Z_i = Z_start + frac * (Z_end - Z_start) + VB_i = solve_VB_for_Z(RA_t, RB_t, Z_i, Q, P) + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_linz.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + # BF optimal + z_bf = [Z_start] + list(opt_Z_trajs[N_cmp]) + # Trace actual Z achieved after arb at each step + z_bf_actual = [Z_start] + RA_t, RB_t = RA, RB + for i in range(N_cmp): + VB_i = solve_VB_for_Z(RA_t, RB_t, opt_Z_trajs[N_cmp][i], Q, P) + VA_i = compute_VA_from_VB(RA_t, RB_t, VB_i, Q) + z_bf_actual.append(compute_Z(VA_i, VB_i, P)) + RA_t, RB_t, _ = micro_step(RA_t, RB_t, VA_i, VB_i, P) + + ax.plot(steps_cmp, z_geo, "o-", ms=4, color="C0", label="Geometric VB") + ax.plot(steps_cmp, z_linz, "o-", ms=4, color="C2", label="Linear Z") + ax.plot(steps_cmp, z_bf_actual, "s--", ms=5, color="C3", + label="BF Optimal") + ax.plot(steps_cmp, np.linspace(Z_start, Z_end, N_cmp + 1), + ":", color="gray", alpha=0.5, label="Ideal linear Z") + ax.set_xlabel("Step") + ax.set_ylabel("Z = sqrt(P)*VA - VB/sqrt(P)") + ax.set_title(f"Z trajectory comparison (N={N_cmp})") + ax.legend(fontsize=7) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_interpolation_bruteforce.png", dpi=150) + print("\nSaved reclamm_interpolation_bruteforce.png") + + +if __name__ == "__main__": + main() diff --git a/scripts/build_pool_grids.py b/scripts/build_pool_grids.py new file mode 100644 index 00000000..9b710c26 --- /dev/null +++ b/scripts/build_pool_grids.py @@ -0,0 +1,695 @@ +"""Build 2D arb-volume grids (cadence x gas) for all Binance-matchable pools. + +v2: Per-day daily arb volumes at each grid point, correct pool weights, + pool type dispatch (WEIGHTED vs RECLAMM). + +For each real Balancer pool where both tokens have Binance minute data: + - Uses actual LP supply trajectory (BPT totalShares from panel) + - Uses actual initial TVL from panel + - Uses correct pool weights from pools.parquet + - Dispatches reCLAMM pools with on-chain params from pools_history.db + - Sweeps cadence x gas_cost as scalar grid + - Stores per-day V_arb at each grid point (not aggregated) + +Output: results/pool_grids_v2/{pool_id_prefix}_daily.parquet + summary CSV. + +Usage: + python scripts/build_pool_grids.py + python scripts/build_pool_grids.py --workers 6 --train-days 90 +""" + +import os +os.environ.setdefault("JAX_PLATFORMS", "cpu") + +import argparse +import ast +import sqlite3 +import time +from concurrent.futures import ProcessPoolExecutor, as_completed + +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +from quantammsim.runners.jax_runners import do_run_on_historic_data +from quantammsim.utils.data_processing.historic_data_utils import get_historic_parquet_data + +# ── Output ──────────────────────────────────────────────────────────────── +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "results", "pool_grids_v2", +) + +# ── Grid ────────────────────────────────────────────────────────────────── +CADENCES = [1, 2, 3, 5, 8, 12, 20, 30, 45, 60] + +# Finer gas grids — concentrated in $0.1-$3.0 where most fitted values land. +# Mainnet: 19 points (was 10). Fills the $0.25→$1.50 gap that caused +# zero-arb artifacts on low-volatility days. +GAS_COSTS_MAINNET = [ + 0.0, 0.05, 0.1, 0.15, 0.25, 0.35, 0.5, 0.65, 0.8, + 1.0, 1.25, 1.5, 2.0, 3.0, 5.0, 8.0, 12.0, 20.0, 50.0, +] +# L2: 14 points (was 10). Fills $0.05→$0.50 range. +GAS_COSTS_L2 = [ + 0.0, 0.001, 0.003, 0.005, 0.01, 0.02, 0.05, + 0.1, 0.15, 0.25, 0.5, 0.75, 1.0, 2.0, +] + +# ── Token mapping ───────────────────────────────────────────────────────── +# Maps Balancer pool token symbols to Binance trading symbols. +# Validated via Balancer hourly vs Binance minute price comparison: +# all wrappers below have daily return correlation > 0.75 with their +# underlying, or are stablecoins (basis < 0.5%). +TOKEN_MAP = { + # Wrapped natives (corr > 0.96) + "WBTC": "BTC", "WETH": "ETH", "cbBTC": "BTC", + # ETH LSTs / Aave wrappers (corr 0.87-0.97) + "wstETH": "ETH", "stETH": "ETH", "rETH": "ETH", "cbETH": "ETH", + "waEthLidoWETH": "ETH", "waEthLidowstETH": "ETH", + "waBasWETH": "ETH", # Aave wrapped WETH on Base (corr 0.975) + "waGnowstETH": "ETH", # Aave wrapped wstETH on Gnosis (corr 0.976) + # GNO wrappers (corr 0.66-0.98) + "waGnoGNO": "GNO", # Aave wrapped GNO on Gnosis (corr 0.979) + "osGNO": "GNO", # StakeWise staked GNO (corr 0.755) + # S (Sonic) wrappers (corr 0.945) + "wS": "S", + "stS": "S", # Staked Sonic + # SOL LSTs (corr 0.922) + "JitoSOL": "SOL", # Jito staked SOL + # POL/MATIC variants (corr 0.96) + "wPOL": "POL", "WMATIC": "POL", "MATIC": "POL", + # Stablecoin equivalents (all ~$1.00, basis < 0.5%) + "USDC.e": "USDC", "USDbC": "USDC", "waBasUSDC": "USDC", + "DAI": "USDC", # Stale Binance data; basis < 10bps vs USDC + "WXDAI": "USDC", "sDAI": "USDC", # Gnosis DAI variants → USDC + "USDT": "USDC", + "DOLA": "USDC", + "scUSD": "USDC", +} + +# ── reCLAMM pool → DB table mapping ────────────────────────────────────── +RECLAMM_DB_PATH = "/Users/matthew/Projects/reclamm-simulations/data/pools_history.db" +RECLAMM_POOL_TABLE_MAP = { + "0x9d1fcf346ea1b0": "AAVE_WETH", +} + + +def _get_binance_tokens(): + """Get set of tokens with Binance minute parquets.""" + data_dir = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "quantammsim", "data", + ) + tokens = set() + for f in os.listdir(data_dir): + if f.endswith("_USD.parquet"): + tokens.add(f.replace("_USD.parquet", "")) + return tokens + + +def _map_token(tok, binance_tokens): + """Map a Balancer token symbol to Binance symbol, or None.""" + mapped = TOKEN_MAP.get(tok, tok) + return mapped if mapped in binance_tokens else None + + +def gas_costs_for_chain(chain): + return GAS_COSTS_L2 if chain != "MAINNET" else GAS_COSTS_MAINNET + + +def _parse_weights(weights_raw): + """Parse weights from pools.parquet (numpy array of strings or list).""" + if weights_raw is None: + return None + if isinstance(weights_raw, str): + weights_raw = ast.literal_eval(weights_raw) + if hasattr(weights_raw, 'tolist'): + weights_raw = weights_raw.tolist() + parsed = [] + for w in weights_raw: + if w is None or str(w).lower() == 'none': + return None + parsed.append(float(w)) + return parsed + + +def _parse_tokens(tokens_raw): + """Parse tokens from pools.parquet.""" + if isinstance(tokens_raw, str): + try: + return ast.literal_eval(tokens_raw) + except (ValueError, SyntaxError): + return [t.strip() for t in tokens_raw.split(",")] + if hasattr(tokens_raw, 'tolist'): + return tokens_raw.tolist() + return list(tokens_raw) + + +# ── Pool metadata loading ───────────────────────────────────────────────── + +def load_pools_metadata(): + """Load pools.parquet to get weights, pool_type, and true token count.""" + pools_path = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "pools.parquet", + ) + pools = pd.read_parquet(pools_path) + meta = {} + for _, row in pools.iterrows(): + pool_id = row["pool_id"] + prefix = pool_id[:16] + tokens = _parse_tokens(row["tokens"]) + weights = _parse_weights(row["weights"]) + meta[prefix] = { + "pool_id": pool_id, + "pool_type": row["pool_type"], + "tokens_full": tokens, + "n_tokens": len(tokens), + "weights": weights, + } + return meta + + +def load_reclamm_params(pool_id_prefix): + """Load reCLAMM on-chain params from pools_history.db.""" + table_name = RECLAMM_POOL_TABLE_MAP.get(pool_id_prefix) + if table_name is None: + return None + if not os.path.exists(RECLAMM_DB_PATH): + return None + conn = sqlite3.connect(RECLAMM_DB_PATH) + try: + df = pd.read_sql(f"SELECT * FROM [{table_name}] ORDER BY timestamp DESC LIMIT 1", conn) + if len(df) == 0: + return None + row = df.iloc[0] + return { + "price_ratio": float(row["price_ratio"]), + "centeredness_margin": float(row["margin"]), + "shift_exponent": float(row["shift_rate"]), + "swap_fee": float(row["swap_fee"]), + } + except Exception: + return None + finally: + conn.close() + + +# ── Panel loading and pool matching ─────────────────────────────────────── + +def load_panel_and_match(train_days): + """Load panel, filter to last N days, find matchable 2-token pools.""" + panel_path = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", + ) + panel = pd.read_parquet(panel_path) + + if "obs_date" not in panel.columns and "date" in panel.columns: + panel = panel.rename(columns={"date": "obs_date"}) + panel["obs_date"] = pd.to_datetime(panel["obs_date"]) + + if "tvl" not in panel.columns and "log_tvl" in panel.columns: + panel["tvl"] = np.exp(panel["log_tvl"]) + + cutoff = panel["obs_date"].max() - pd.Timedelta(days=train_days) + panel = panel[panel["obs_date"] >= cutoff].copy() + + binance_tokens = _get_binance_tokens() + pools_meta = load_pools_metadata() + + pools = [] + for pool_id, grp in panel.groupby("pool_id"): + prefix = pool_id[:16] + meta = pools_meta.get(prefix) + if meta is None: + continue + + # Skip multi-token pools + if meta["n_tokens"] > 2: + continue + + pool_type = meta["pool_type"] + + # Get tokens for Binance matching from panel + row = grp.iloc[0] + tokens_str = row["tokens"] + toks = [t.strip() for t in tokens_str.split(",")] + if len(toks) != 2: + continue + + mapped = [] + for t in toks: + m = _map_token(t, binance_tokens) + if m is None: + break + mapped.append(m) + else: + if mapped[0] == mapped[1]: + continue + + chain = row["chain"] + fee = row.get("swap_fee", np.exp(row["log_fee"])) + + # Get weights + weights = meta["weights"] + if pool_type == "WEIGHTED" and weights is None: + weights = [0.5, 0.5] + + # reCLAMM params + reclamm_params = None + if pool_type == "RECLAMM": + reclamm_params = load_reclamm_params(prefix) + if reclamm_params is None: + print(f" SKIP reCLAMM {'/'.join(mapped)} ({chain}): " + f"no DB params for {prefix}") + continue + + pools.append({ + "pool_id": pool_id, + "pool_id_prefix": prefix, + "tokens": mapped, + "chain": chain, + "fee": float(fee), + "panel_data": grp, + "pool_type": pool_type, + "weights": weights, + "reclamm_params": reclamm_params, + }) + + return pools + + +def build_lp_supply_df(panel_pool): + """Build lp_supply_df from daily BPT supply. Returns (df, initial_tvl).""" + panel_pool = panel_pool.sort_values("obs_date") + + has_bpt = ( + "total_shares" in panel_pool.columns + and not panel_pool["total_shares"].isna().all() + and (panel_pool["total_shares"] > 0).any() + ) + + initial_tvl = float(panel_pool["tvl"].iloc[0]) if len(panel_pool) > 0 else 0.0 + + if not has_bpt: + return None, initial_tvl + + bpt = panel_pool[["obs_date", "total_shares", "tvl"]].drop_duplicates("obs_date") + initial_bpt = bpt["total_shares"].iloc[0] + + if initial_bpt <= 0 or initial_tvl <= 0: + return None, initial_tvl + + unix_ms = bpt["obs_date"].apply( + lambda d: int(pd.Timestamp(d).timestamp() * 1000) + ).values + lp_supply = (bpt["total_shares"].values / initial_bpt).astype(float) + + return pd.DataFrame({"unix": unix_ms, "lp_supply": lp_supply}), initial_tvl + + +def get_date_range(panel_data): + """Get start/end date strings from panel data.""" + dates = panel_data["obs_date"].sort_values() + start = dates.iloc[0].strftime("%Y-%m-%d %H:%M:%S") + end = dates.iloc[-1].strftime("%Y-%m-%d %H:%M:%S") + return start, end + + +# ── Simulation ──────────────────────────────────────────────────────────── + +def run_arb_sim(tokens, fee, initial_tvl, start, end, cadence, gas_cost, + lp_supply_df=None, weights=None, pool_type="WEIGHTED", + reclamm_params=None, price_data=None): + """Run arb-only sim at one (cadence, gas_cost) point. + + Returns: pd.Series with date index → daily arb volume. + """ + fp = { + "tokens": tokens, + "startDateString": start, + "endDateString": end, + "initial_pool_value": initial_tvl, + "fees": fee, + "gas_cost": float(gas_cost), + "arb_fees": 0.0, + "do_arb": True, + "noise_trader_ratio": 0.0, + "arb_frequency": int(cadence), + "chunk_period": 1440, + "weight_interpolation_period": 1440, + } + + if pool_type == "RECLAMM" and reclamm_params is not None: + fp["rule"] = "reclamm" + fp["reclamm_use_shift_exponent"] = True + fp["fees"] = reclamm_params["swap_fee"] + params = { + "price_ratio": jnp.array(reclamm_params["price_ratio"]), + "centeredness_margin": jnp.array(reclamm_params["centeredness_margin"]), + "shift_exponent": jnp.array(reclamm_params["shift_exponent"]), + } + else: + fp["rule"] = "balancer" + if weights is not None and len(weights) == 2: + logits = np.log(np.array(weights, dtype=float)) + params = {"initial_weights_logits": jnp.array(logits)} + else: + params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + + result = do_run_on_historic_data( + fp, params, lp_supply_df=lp_supply_df, verbose=False, + price_data=price_data, + ) + + reserves = np.array(result["reserves"]) + prices = np.array(result["data_dict"]["prices"]) + unix_ms = np.array(result["data_dict"]["unix_values"]) + start_idx = int(result["data_dict"]["start_idx"]) + + T = reserves.shape[0] - 1 + prices_window = prices[start_idx:start_idx + T + 1] + delta_r = np.diff(reserves, axis=0) + step_vol = np.sum(np.abs(delta_r * prices_window[1:]), axis=1) / 2.0 + + dates = pd.to_datetime( + unix_ms[start_idx + 1:start_idx + T + 1], unit="ms", + ).normalize() + daily = pd.DataFrame( + {"date": dates, "volume": step_vol}, + ).groupby("date")["volume"].sum() + + return daily + + +def _run_cadence_sweep(pool_info, cadence, gas_costs): + """Worker: sweep all gas costs for one (pool, cadence). + + Returns list of dicts with per-day data: one dict per (gas, date) pair, + plus a summary list. + """ + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + tokens = pool_info["tokens"] + fee = pool_info["fee"] + initial_tvl = pool_info["initial_tvl"] + start = pool_info["start"] + end = pool_info["end"] + lp_supply_df = pool_info["lp_supply_df"] + weights = pool_info.get("weights") + pool_type = pool_info.get("pool_type", "WEIGHTED") + reclamm_params = pool_info.get("reclamm_params") + + price_data = pool_info.get("price_data") + + daily_rows = [] + summary_rows = [] + + for gas in gas_costs: + try: + daily = run_arb_sim( + tokens, fee, initial_tvl, start, end, cadence, gas, + lp_supply_df=lp_supply_df, + weights=weights, + pool_type=pool_type, + reclamm_params=reclamm_params, + price_data=price_data, + ) + # Store per-day data + for date_val, vol in daily.items(): + daily_rows.append({ + "cadence": cadence, + "gas_cost": gas, + "date": date_val, + "daily_arb_volume": vol, + }) + # Summary for diagnostics + summary_rows.append({ + "cadence": cadence, + "gas_cost": gas, + "total_arb_volume": daily.sum(), + "median_daily_arb_volume": daily.median(), + "mean_daily_arb_volume": daily.mean(), + "n_days": len(daily), + }) + except Exception as e: + summary_rows.append({ + "cadence": cadence, + "gas_cost": gas, + "total_arb_volume": np.nan, + "median_daily_arb_volume": np.nan, + "mean_daily_arb_volume": np.nan, + "n_days": 0, + "error": str(e), + }) + + return daily_rows, summary_rows + + +# ── Plotting ────────────────────────────────────────────────────────────── + +def plot_pool_grid(summary_df, pool_id, tokens, chain, fee, tvl, pool_type, + gas_costs, output_dir): + """Simple 2-panel diagnostic: V_arb vs cadence, gas attenuation.""" + fig, axes = plt.subplots(1, 3, figsize=(16, 4.5)) + df = summary_df + + ax = axes[0] + for gas in gas_costs: + sub = df[df["gas_cost"] == gas].sort_values("cadence") + if len(sub) > 0: + ax.plot(sub["cadence"], sub["median_daily_arb_volume"], + "o-", label=f"${gas}", markersize=3) + ax.set_xlabel("Cadence (min)") + ax.set_ylabel("Median daily V_arb ($)") + ax.set_title("V_arb vs cadence") + ax.legend(fontsize=5, ncol=2, title="gas") + + ax = axes[1] + for cadence in CADENCES: + sub = df[df["cadence"] == cadence].sort_values("gas_cost") + v0 = sub[sub["gas_cost"] == 0.0]["median_daily_arb_volume"].values + if len(v0) > 0 and v0[0] > 0: + ratio = sub["median_daily_arb_volume"].values / v0[0] + ax.plot(sub["gas_cost"].values, ratio, "o-", + label=f"{cadence}min", markersize=3) + ax.set_xlabel("Gas cost ($)") + ax.set_ylabel("V_arb / V_arb(gas=0)") + ax.set_title("Gas attenuation") + ax.set_ylim(-0.05, 1.05) + ax.legend(fontsize=5, ncol=2) + + ax = axes[2] + for gas in gas_costs: + sub = df[df["gas_cost"] == gas].sort_values("cadence") + vals = sub["median_daily_arb_volume"].values + if len(vals) > 0 and np.all(vals > 0): + ax.plot(sub["cadence"], vals, "o-", label=f"${gas}", markersize=3) + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_xlabel("Cadence (min)") + ax.set_ylabel("Median daily V_arb ($)") + ax.set_title("Log-log") + ax.legend(fontsize=5, ncol=2, title="gas") + + tok_str = "/".join(tokens) + type_str = f" [{pool_type}]" if pool_type != "WEIGHTED" else "" + fig.suptitle( + f"{tok_str} ({chain}, fee={fee:.2%}, TVL=${tvl:,.0f}){type_str}\n" + f"{pool_id[:16]}", + fontsize=10, + ) + fig.tight_layout() + path = os.path.join(output_dir, f"{pool_id[:16]}_grid.png") + fig.savefig(path, dpi=120, bbox_inches="tight") + plt.close() + + +# ── Main ────────────────────────────────────────────────────────────────── + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--workers", type=int, default=6) + parser.add_argument("--train-days", type=int, default=90) + parser.add_argument("--pools", type=str, default=None, + help="Comma-separated pool_id prefixes to run (default: all)") + args = parser.parse_args() + + os.makedirs(OUTPUT_DIR, exist_ok=True) + + print("Loading panel and matching pools...") + pools = load_panel_and_match(args.train_days) + print(f"Found {len(pools)} matchable 2-token pools\n") + + for p in pools: + lp_df, tvl = build_lp_supply_df(p["panel_data"]) + p["lp_supply_df"] = lp_df + p["initial_tvl"] = tvl + p["start"], p["end"] = get_date_range(p["panel_data"]) + + pools = [p for p in pools if p["panel_data"]["obs_date"].nunique() >= 14] + pools.sort(key=lambda p: p["initial_tvl"], reverse=True) + + # Filter to specific pools if requested + if args.pools: + requested = set(args.pools.split(",")) + pools = [p for p in pools if p["pool_id_prefix"] in requested] + print(f"Filtered to {len(pools)} requested pools\n") + + # Print pool summary + print(f"{'#':>3} {'Tokens':<12} {'Chain':<10} {'Type':<9} {'Weights':<10} " + f"{'Fee':>6} {'TVL':>12} {'Days':>5} {'Pool ID':<18}") + print("-" * 95) + for i, p in enumerate(pools): + n_days = p["panel_data"]["obs_date"].nunique() + w_str = "/".join(f"{w:.0%}" for w in p["weights"]) if p["weights"] else "N/A" + print(f"{i+1:3d} {'/'.join(p['tokens']):<12} {p['chain']:<10} " + f"{p['pool_type']:<9} {w_str:<10} " + f"{p['fee']:5.2%} ${p['initial_tvl']:>10,.0f} " + f"{n_days:5d} {p['pool_id'][:16]}") + + all_summaries = [] + t_total = time.time() + + for pool_idx, pool in enumerate(pools): + pool_id = pool["pool_id"] + prefix = pool["pool_id_prefix"] + tokens = pool["tokens"] + chain = pool["chain"] + fee = pool["fee"] + tvl = pool["initial_tvl"] + pool_type = pool["pool_type"] + gas_costs = gas_costs_for_chain(chain) + n_runs = len(CADENCES) * len(gas_costs) + + if tvl <= 0: + print(f"\n SKIP {'/'.join(tokens)} ({chain}): TVL=0") + continue + + w_str = "/".join(f"{w:.0%}" for w in pool["weights"]) if pool["weights"] else "N/A" + print(f"\n{'='*60}") + print(f" [{pool_idx+1}/{len(pools)}] {'/'.join(tokens)} " + f"({chain}, {pool_type}, {w_str}, fee={fee:.2%}, TVL=${tvl:,.0f})") + print(f" Grid: {len(CADENCES)} cadences x {len(gas_costs)} gas = {n_runs} runs") + print(f"{'='*60}") + + t0 = time.time() + + # Preload price data once per pool (avoids re-reading parquet per run) + sorted_tokens = sorted(tokens) + price_data = get_historic_parquet_data(sorted_tokens, ["close"]) + + pool_info = { + "tokens": tokens, + "fee": fee, + "initial_tvl": tvl, + "start": pool["start"], + "end": pool["end"], + "lp_supply_df": pool["lp_supply_df"], + "weights": pool["weights"], + "pool_type": pool_type, + "reclamm_params": pool.get("reclamm_params"), + "price_data": price_data, + } + + all_daily_rows = [] + all_summary_rows = [] + + if args.workers <= 1: + for cadence in CADENCES: + daily_rows, summary_rows = _run_cadence_sweep( + pool_info, cadence, gas_costs, + ) + all_daily_rows.extend(daily_rows) + all_summary_rows.extend(summary_rows) + for r in summary_rows: + print(f" cad={r['cadence']:3d} gas=${r['gas_cost']:6.3f} -> " + f"median=${r['median_daily_arb_volume']:,.0f}/day") + else: + futures = {} + with ProcessPoolExecutor(max_workers=args.workers) as executor: + for cadence in CADENCES: + fut = executor.submit( + _run_cadence_sweep, pool_info, cadence, gas_costs, + ) + futures[fut] = cadence + + done = 0 + for fut in as_completed(futures): + cadence = futures[fut] + done += 1 + try: + daily_rows, summary_rows = fut.result() + all_daily_rows.extend(daily_rows) + all_summary_rows.extend(summary_rows) + medians = [r["median_daily_arb_volume"] for r in summary_rows + if not np.isnan(r.get("median_daily_arb_volume", np.nan))] + if medians: + print(f" [{done:2d}/{len(CADENCES)}] cad={cadence:3d} — " + f"V_arb: ${min(medians):,.0f} – ${max(medians):,.0f}/day") + else: + print(f" [{done:2d}/{len(CADENCES)}] cad={cadence:3d} — " + f"all failed") + except Exception as e: + print(f" [{done:2d}/{len(CADENCES)}] cad={cadence:3d} FAILED: {e}") + + elapsed = time.time() - t0 + print(f" {n_runs} runs in {elapsed:.1f}s ({elapsed/max(n_runs,1):.2f}s/run)") + + # Save per-day parquet + if all_daily_rows: + daily_df = pd.DataFrame(all_daily_rows) + daily_df["date"] = pd.to_datetime(daily_df["date"]) + parquet_path = os.path.join(OUTPUT_DIR, f"{prefix}_daily.parquet") + daily_df.to_parquet(parquet_path, index=False) + print(f" Saved {len(daily_df)} daily rows -> {parquet_path}") + + # Save summary CSV for diagnostics + summary_df = pd.DataFrame(all_summary_rows) + csv_path = os.path.join(OUTPUT_DIR, f"{prefix}_summary.csv") + summary_df.to_csv(csv_path, index=False) + + # Plot + if len(summary_df) > 0: + plot_pool_grid(summary_df, pool_id, tokens, chain, fee, tvl, + pool_type, gas_costs, OUTPUT_DIR) + + # Global summary + g0 = summary_df[summary_df["gas_cost"] == 0.0] if len(summary_df) > 0 else pd.DataFrame() + all_summaries.append({ + "pool_id": prefix, + "tokens": "/".join(tokens), + "chain": chain, + "pool_type": pool_type, + "weights": str(pool["weights"]), + "fee": fee, + "tvl": tvl, + "n_days": summary_df["n_days"].max() if len(summary_df) > 0 else 0, + "n_daily_rows": len(all_daily_rows), + "v_arb_cad1_gas0": g0[g0["cadence"] == 1]["median_daily_arb_volume"].values[0] + if len(g0[g0["cadence"] == 1]) > 0 else np.nan, + "v_arb_cad60_gas0": g0[g0["cadence"] == 60]["median_daily_arb_volume"].values[0] + if len(g0[g0["cadence"] == 60]) > 0 else np.nan, + "elapsed_s": elapsed, + }) + + total_elapsed = time.time() - t_total + + if all_summaries: + summary_df = pd.DataFrame(all_summaries) + summary_path = os.path.join(OUTPUT_DIR, "grid_summary.csv") + summary_df.to_csv(summary_path, index=False) + print(f"\n{'='*60}") + print(f" Summary saved: {summary_path}") + print(f" Total: {len(all_summaries)} pools, " + f"{total_elapsed:.0f}s ({total_elapsed/60:.1f} min)") + print(f"{'='*60}") + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_bayesian.py b/scripts/calibrate_noise_bayesian.py new file mode 100644 index 00000000..efdfb4bb --- /dev/null +++ b/scripts/calibrate_noise_bayesian.py @@ -0,0 +1,853 @@ +"""Bayesian hierarchical noise volume model across Balancer pools. + +Full Bayesian version of the noise calibration: ALL K=4 per-pool +coefficients (intercept, TVL elasticity, volatility response, weekend +effect) vary per pool with pool-level covariates modulating their priors, +and an LKJ-decomposed covariance capturing correlations between +coefficients. + +Generative model: + For pool i with pool-level covariates z_i, day t: + + mu_i = B . z_i # K-vector population mean + eta_i ~ N(0, I_K) # non-centered offsets + theta_i = mu_i + diag(sigma) . L . eta_i # per-pool coefficients + + log(V_{i,t}) ~ N(theta_i . x_{i,t}, sigma_eps^2) + + theta_i = [intercept_i, b_tvl_i, b_sigma_i, b_weekend_i] + x_{i,t} = [1, log_tvl, volatility, weekend] + z_i = [1, chain_dummies(6), tier_A_dummies(2), tier_B_dummies(2), log_fee] + +Priors: + B_{k,d} ~ N(0, 5^2) + sigma_k ~ HalfNormal(2.0) + L ~ LKJCholesky(K=4, eta=2) + sigma_eps ~ HalfNormal(3.0) + +Usage: + # Full pipeline: fit + output + diagnostics + python scripts/calibrate_noise_bayesian.py \\ + --fit --output results/bayesian_noise_params.json --plot + + # Predict for an unseen pool + python scripts/calibrate_noise_bayesian.py \\ + --predict --chain BASE --tokens ETH USDC --fee 0.003 + + # Custom NUTS settings + python scripts/calibrate_noise_bayesian.py \\ + --fit --num-warmup 2000 --num-samples 4000 --num-chains 4 +""" + +import argparse +import json +import os +import sys + +import numpy as np +import pandas as pd + +# arviz 0.17.x imports scipy.signal.gaussian which was removed in scipy 1.13+. +# Patch it back from scipy.signal.windows before any arviz import. +try: + from scipy.signal import gaussian as _ # noqa: F401 +except ImportError: + from scipy.signal.windows import gaussian as _gauss + import scipy.signal + scipy.signal.gaussian = _gauss + +# --------------------------------------------------------------------------- +# Reuse constants and helpers from the frequentist script +# --------------------------------------------------------------------------- + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "local_data", "noise_calibration" +) + +# Reference levels for dummy coding (dropped categories) +REF_CHAIN = "ARBITRUM" +REF_TIER = 0 + +# Ordered non-reference chains (alphabetical excluding REF_CHAIN) +CHAIN_ORDER = ["BASE", "GNOSIS", "MAINNET", "OPTIMISM", "POLYGON", "SONIC"] + +K = 4 # number of per-pool coefficients +D = 12 # pool-level covariate dimension: 1 + 6 chains + 2 tier_A + 2 tier_B + 1 log_fee + +COEFF_NAMES = ["intercept", "b_tvl", "b_sigma", "b_weekend"] + + +# --------------------------------------------------------------------------- +# Token tier helpers (duplicated to avoid import fragility) +# --------------------------------------------------------------------------- + +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", +} + +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} + + +def classify_token_tier(symbol: str) -> int: + s = symbol.strip() + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 + + +# --------------------------------------------------------------------------- +# Data preparation +# --------------------------------------------------------------------------- + +def load_panel(cache_dir: str = CACHE_DIR) -> pd.DataFrame: + """Load the cached panel parquet produced by calibrate_noise_hierarchical.py --fetch.""" + panel_path = os.path.join(cache_dir, "panel.parquet") + if not os.path.exists(panel_path): + print(f"ERROR: Panel cache not found at {panel_path}", file=sys.stderr) + print("Run: python scripts/calibrate_noise_hierarchical.py --fetch", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_path) + + # Filter pools with < 10 observations + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid_pools)].copy() + + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + return panel + + +def _build_z_pool(pool_meta: pd.DataFrame) -> np.ndarray: + """Build (N_pools, D) pool-level covariate matrix. + + Columns: [1, chain_BASE, ..., chain_SONIC (6), + tier_A_1, tier_A_2, tier_B_1, tier_B_2, log_fee] + """ + N = len(pool_meta) + z = np.zeros((N, D), dtype=np.float64) + + # Intercept + z[:, 0] = 1.0 + + # Chain dummies (columns 1-6) + for j, chain in enumerate(CHAIN_ORDER): + z[:, 1 + j] = (pool_meta["chain"].values == chain).astype(float) + + # tier_A dummies (columns 7-8): tiers 1 and 2, reference = 0 + tier_a = pool_meta["tier_A"].values.astype(int) + z[:, 7] = (tier_a == 1).astype(float) + z[:, 8] = (tier_a == 2).astype(float) + + # tier_B dummies (columns 9-10): tiers 1 and 2, reference = 0 + tier_b = pool_meta["tier_B"].values.astype(int) + z[:, 9] = (tier_b == 1).astype(float) + z[:, 10] = (tier_b == 2).astype(float) + + # log_fee (column 11) + z[:, 11] = np.log(np.maximum(pool_meta["swap_fee"].values.astype(float), 1e-6)) + + return z + + +def prepare_data(panel: pd.DataFrame) -> dict: + """Construct JAX-ready arrays from the panel DataFrame. + + Returns dict with: + pool_idx : (N_obs,) int32 — pool index per observation + z_pool : (N_pools, D) float64 — pool-level covariates + x_obs : (N_obs, K) float64 — within-day regressors + y_obs : (N_obs,) float64 — log_volume + pool_ids : list — ordered pool IDs + pool_meta : DataFrame — per-pool metadata (indexed same as z_pool rows) + """ + # Stable pool ordering + pool_ids = sorted(panel["pool_id"].unique()) + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + N_pools = len(pool_ids) + + # Pool-level metadata (one row per pool) + pool_meta = panel.drop_duplicates("pool_id").set_index("pool_id").loc[pool_ids].reset_index() + z_pool = _build_z_pool(pool_meta) + + # Observation-level arrays + pool_idx = panel["pool_id"].map(pool_id_to_idx).values.astype(np.int32) + x_obs = np.column_stack([ + np.ones(len(panel)), + panel["log_tvl"].values, + panel["volatility"].values, + panel["weekend"].values, + ]).astype(np.float64) + y_obs = panel["log_volume"].values.astype(np.float64) + + print(f" Prepared: N_obs={len(y_obs)}, N_pools={N_pools}, K={K}, D={D}") + print(f" z_pool range check — log_fee: [{z_pool[:, 11].min():.2f}, {z_pool[:, 11].max():.2f}]") + + return { + "pool_idx": pool_idx, + "z_pool": z_pool, + "x_obs": x_obs, + "y_obs": y_obs, + "pool_ids": pool_ids, + "pool_meta": pool_meta, + "N_pools": N_pools, + } + + +# --------------------------------------------------------------------------- +# NumPyro model +# --------------------------------------------------------------------------- + +def hierarchical_noise_model(pool_idx, z_pool, x_obs, y_obs=None, + N_pools=None, K=4, D=12): + """Bayesian hierarchical noise volume model. + + Non-centered parameterization with LKJ correlation prior. + """ + import jax.numpy as jnp + import numpyro + import numpyro.distributions as dist + + N_obs = pool_idx.shape[0] + + # --- Population coefficient matrix B: (K, D) --- + B = numpyro.sample("B", dist.Normal(0.0, 5.0).expand([K, D]).to_event(2)) + + # --- Per-pool scale and correlation --- + sigma = numpyro.sample("sigma", dist.HalfNormal(2.0).expand([K]).to_event(1)) + L_Omega = numpyro.sample("L_Omega", dist.LKJCholesky(K, concentration=2.0)) + + # Cholesky factor of covariance: diag(sigma) @ L_Omega + L_Sigma = jnp.diag(sigma) @ L_Omega # (K, K) + + # --- Non-centered pool effects --- + with numpyro.plate("pools", N_pools): + eta = numpyro.sample("eta", dist.Normal(0.0, 1.0).expand([K]).to_event(1)) + + # theta_i = B @ z_i + L_Sigma @ eta_i for each pool i + # mu: (N_pools, K) = z_pool @ B^T + mu = z_pool @ B.T # (N_pools, K) + theta = mu + eta @ L_Sigma.T # (N_pools, K) + + # --- Observation model --- + sigma_eps = numpyro.sample("sigma_eps", dist.HalfNormal(3.0)) + + # Predicted log-volume: theta[pool_idx] . x_obs (dot product per obs) + theta_obs = theta[pool_idx] # (N_obs, K) + mu_obs = jnp.sum(theta_obs * x_obs, axis=1) # (N_obs,) + + with numpyro.plate("obs", N_obs): + numpyro.sample("y", dist.Normal(mu_obs, sigma_eps), obs=y_obs) + + # Deterministic: store theta for extraction + numpyro.deterministic("theta", theta) + + +# --------------------------------------------------------------------------- +# Inference +# --------------------------------------------------------------------------- + +def run_inference(data, num_warmup=1000, num_samples=2000, num_chains=4, + target_accept=0.85, max_tree_depth=10, seed=42): + """Run NUTS on the hierarchical model. + + Returns the MCMC object with samples. + """ + import jax + import jax.numpy as jnp + import numpyro + from numpyro.infer import MCMC, NUTS + + # Use all available CPU cores for chains + numpyro.set_host_device_count(min(num_chains, len(jax.devices("cpu")))) + + kernel = NUTS( + hierarchical_noise_model, + target_accept_prob=target_accept, + max_tree_depth=max_tree_depth, + ) + mcmc = MCMC( + kernel, + num_warmup=num_warmup, + num_samples=num_samples, + num_chains=num_chains, + progress_bar=True, + ) + + rng_key = jax.random.PRNGKey(seed) + + print(f"\n Running NUTS: {num_chains} chains x " + f"({num_warmup} warmup + {num_samples} samples)") + print(f" target_accept={target_accept}, max_tree_depth={max_tree_depth}") + + mcmc.run( + rng_key, + pool_idx=jnp.array(data["pool_idx"]), + z_pool=jnp.array(data["z_pool"]), + x_obs=jnp.array(data["x_obs"]), + y_obs=jnp.array(data["y_obs"]), + N_pools=data["N_pools"], + K=K, + D=D, + ) + + mcmc.print_summary(exclude_deterministic=True) + return mcmc + + +# --------------------------------------------------------------------------- +# Post-processing +# --------------------------------------------------------------------------- + +def extract_noise_params(mcmc, data) -> list: + """Extract per-pool noise params from MCMC posterior. + + Reconstructs theta from the non-centered parameterization, + takes posterior medians, and applies weekend absorption: + b_0_effective = b_0_raw + b_weekend * (2/7) + + Returns list of dicts compatible with reclamm_loglinear_noise_volume. + """ + samples = mcmc.get_samples() + theta_samples = samples["theta"] # (n_samples, N_pools, K) + + # Posterior median per pool + theta_median = np.median(theta_samples, axis=0) # (N_pools, K) + theta_std = np.std(theta_samples, axis=0) # (N_pools, K) + + pool_ids = data["pool_ids"] + pool_meta = data["pool_meta"] + + results = [] + for i, pool_id in enumerate(pool_ids): + meta = pool_meta.iloc[i] + b_0_raw, b_tvl, b_sigma, b_weekend = theta_median[i] + std_vals = theta_std[i] + + # Weekend absorption: simulator has no weekend indicator, + # so fold the expected weekend effect into the intercept. + # Weekend days = 2/7 of all days. + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + tokens = meta["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + + results.append({ + "pool_id": pool_id, + "chain": str(meta["chain"]), + "tokens": tokens, + "theta_median": [float(x) for x in theta_median[i]], + "theta_std": [float(x) for x in std_vals], + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(meta["swap_fee"]), + }, + }) + + return results + + +def predict_new_pool(mcmc, data, chain: str, tokens: list, fee: float) -> dict: + """Predict noise params for an unseen pool using population effects. + + Constructs z_new, computes mu_new = B @ z_new across posterior samples, + and returns median + 90% credible intervals. + """ + # Build z_new + z_new = np.zeros(D, dtype=np.float64) + z_new[0] = 1.0 # intercept + + # Chain dummies + if chain in CHAIN_ORDER: + j = CHAIN_ORDER.index(chain) + z_new[1 + j] = 1.0 + + # Tier dummies + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + if tier_a == 1: + z_new[7] = 1.0 + elif tier_a == 2: + z_new[8] = 1.0 + if tier_b == 1: + z_new[9] = 1.0 + elif tier_b == 2: + z_new[10] = 1.0 + + # log_fee + z_new[11] = np.log(max(fee, 1e-6)) + + # Compute mu_new = B @ z_new across all posterior samples + B_samples = np.array(mcmc.get_samples()["B"]) # (n_samples, K, D) + mu_samples = np.einsum("skd,d->sk", B_samples, z_new) # (n_samples, K) + + mu_median = np.median(mu_samples, axis=0) + mu_q05 = np.percentile(mu_samples, 5, axis=0) + mu_q95 = np.percentile(mu_samples, 95, axis=0) + + # Weekend absorption + b_0_raw, b_tvl, b_sigma, b_weekend = mu_median + b_0_effective = b_0_raw + b_weekend * (2.0 / 7.0) + + result = { + "chain": chain, + "tokens": tokens, + "fee": fee, + "prediction_source": "population_level", + "noise_params": { + "b_0": float(b_0_effective), + "b_sigma": float(b_sigma), + "b_c": float(b_tvl), + "b_weekend": float(b_weekend), + "base_fee": float(fee), + }, + "credible_intervals_90": { + name: { + "median": float(mu_median[k]), + "q05": float(mu_q05[k]), + "q95": float(mu_q95[k]), + } + for k, name in enumerate(COEFF_NAMES) + }, + } + + print(f"\n Predicted noise_params for {chain} {tokens} (fee={fee}):") + for name, ci in result["credible_intervals_90"].items(): + print(f" {name:12s}: {ci['median']:+.3f} " + f"[{ci['q05']:+.3f}, {ci['q95']:+.3f}]") + print(f"\n Effective b_0 (weekend-absorbed): {b_0_effective:.3f}") + + return result + + +# --------------------------------------------------------------------------- +# Diagnostics +# --------------------------------------------------------------------------- + +def check_convergence(mcmc) -> dict: + """Compute convergence diagnostics: R-hat, ESS, divergences.""" + import arviz as az + + idata = az.from_numpyro(mcmc) + + # R-hat and ESS for non-deterministic parameters + n_chains = idata.posterior.sizes.get("chain", 1) + + rhat_max = float("nan") + if n_chains >= 2: + rhat = az.rhat(idata) + rhat_vals = [] + for var in rhat.data_vars: + if var == "theta": + continue # deterministic + vals = rhat[var].values + rhat_vals.extend(vals.flatten()) + rhat_max = float(np.nanmax(rhat_vals)) if rhat_vals else float("nan") + + ess = az.ess(idata) + ess_vals = [] + for var in ess.data_vars: + if var == "theta": + continue + vals = ess[var].values + ess_vals.extend(vals.flatten()) + ess_min = float(np.nanmin(ess_vals)) if ess_vals else float("nan") + + # Divergences + divergences = int(idata.sample_stats["diverging"].sum().values) + + print(f"\n Convergence diagnostics:") + if n_chains >= 2: + print(f" R-hat max: {rhat_max:.4f} {'OK' if rhat_max < 1.05 else 'WARNING'}") + else: + print(f" R-hat max: N/A (need >= 2 chains)") + print(f" ESS min: {ess_min:.0f} {'OK' if ess_min > 400 else 'WARNING'}") + print(f" Divergences: {divergences} {'OK' if divergences == 0 else 'WARNING'}") + + return { + "r_hat_max": rhat_max, + "ess_min": ess_min, + "divergences": divergences, + } + + +def plot_bayesian_diagnostics(mcmc, data, output_dir="results"): + """Generate ArviZ diagnostic plots.""" + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + import arviz as az + + os.makedirs(output_dir, exist_ok=True) + idata = az.from_numpyro(mcmc) + samples = mcmc.get_samples() + + # --- 1. Trace plots for sigma, sigma_eps --- + axes = az.plot_trace(idata, var_names=["sigma", "sigma_eps"], compact=True) + fig1 = axes.ravel()[0].figure + fig1.set_size_inches(14, 8) + path1 = os.path.join(output_dir, "bayesian_trace_sigma.png") + fig1.savefig(path1, dpi=150, bbox_inches="tight") + plt.close(fig1) + print(f" Saved: {path1}") + + # --- 2. Posterior predictive: predicted vs observed --- + theta_samples = samples["theta"] # (S, N_pools, K) + sigma_eps_samples = np.array(samples["sigma_eps"]) # (S,) + theta_median = np.median(theta_samples, axis=0) # (N_pools, K) + + pool_idx = data["pool_idx"] + x_obs = data["x_obs"] + y_obs = data["y_obs"] + + theta_obs = theta_median[pool_idx] + y_pred = np.sum(theta_obs * x_obs, axis=1) + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + ax.scatter(y_obs, y_pred, alpha=0.1, s=4, color="steelblue") + lims = [min(y_obs.min(), y_pred.min()), max(y_obs.max(), y_pred.max())] + ax.plot(lims, lims, "r--", linewidth=1) + ax.set_xlabel("Observed log(volume)") + ax.set_ylabel("Predicted log(volume)") + ax.set_title("Posterior predictive check") + r2 = 1 - np.var(y_obs - y_pred) / np.var(y_obs) + ax.text(0.05, 0.95, f"R² = {r2:.3f}", transform=ax.transAxes, + fontsize=11, verticalalignment="top") + + ax = axes[1] + residuals = y_obs - y_pred + ax.hist(residuals, bins=60, color="steelblue", edgecolor="white", alpha=0.8) + ax.axvline(0, color="red", linestyle="--") + ax.set_xlabel("Residual") + ax.set_title(f"Residual distribution (σ_ε ≈ {np.median(sigma_eps_samples):.2f})") + + plt.tight_layout() + path2 = os.path.join(output_dir, "bayesian_posterior_predictive.png") + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + # --- 3. Per-pool b_c (TVL elasticity) by chain/tier --- + pool_meta = data["pool_meta"] + b_tvl_all = theta_median[:, 1] # index 1 = b_tvl + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + ax = axes[0] + chains_present = sorted(pool_meta["chain"].unique()) + chain_data = [] + chain_labels = [] + for c in chains_present: + mask = pool_meta["chain"].values == c + if mask.sum() > 0: + chain_data.append(b_tvl_all[mask]) + chain_labels.append(f"{c}\n(n={mask.sum()})") + ax.boxplot(chain_data, tick_labels=chain_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by chain") + + ax = axes[1] + # By tier_A + tier_a_vals = pool_meta["tier_A"].values.astype(int) + tier_labels_map = {0: "Blue-chip", 1: "Mid-cap", 2: "Long-tail"} + tier_data = [] + tier_labels = [] + for t in [0, 1, 2]: + mask = tier_a_vals == t + if mask.sum() > 0: + tier_data.append(b_tvl_all[mask]) + tier_labels.append(f"{tier_labels_map[t]}\n(n={mask.sum()})") + ax.boxplot(tier_data, tick_labels=tier_labels, vert=True) + ax.axhline(1.0, color="red", linestyle="--", linewidth=0.8, alpha=0.6) + ax.set_ylabel("Per-pool b_c (TVL elasticity)") + ax.set_title("TVL elasticity by token tier (best token)") + + plt.tight_layout() + path3 = os.path.join(output_dir, "bayesian_per_pool_b_c.png") + plt.savefig(path3, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path3}") + + # --- 4. Correlation matrix posterior --- + L_Omega_samples = np.array(samples["L_Omega"]) # (S, K, K) + # Correlation = L @ L^T + Omega_samples = np.einsum("sij,skj->sik", L_Omega_samples, L_Omega_samples) + Omega_median = np.median(Omega_samples, axis=0) + + fig, ax = plt.subplots(figsize=(7, 6)) + im = ax.imshow(Omega_median, vmin=-1, vmax=1, cmap="RdBu_r") + ax.set_xticks(range(K)) + ax.set_yticks(range(K)) + ax.set_xticklabels(COEFF_NAMES, rotation=45, ha="right") + ax.set_yticklabels(COEFF_NAMES) + for i in range(K): + for j in range(K): + ax.text(j, i, f"{Omega_median[i, j]:.2f}", ha="center", va="center", + fontsize=10, color="white" if abs(Omega_median[i, j]) > 0.5 else "black") + plt.colorbar(im, ax=ax, shrink=0.8) + ax.set_title("Posterior median correlation matrix (Ω)") + plt.tight_layout() + path4 = os.path.join(output_dir, "bayesian_correlation_matrix.png") + plt.savefig(path4, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path4}") + + # --- 5. Shrinkage plot: OLS b_c vs hierarchical b_c --- + # Compute per-pool OLS b_c for comparison + panel_meta = data["pool_meta"] + pool_idx_arr = data["pool_idx"] + pool_ids = data["pool_ids"] + + ols_b_c = np.zeros(len(pool_ids)) + for i, pid in enumerate(pool_ids): + mask = pool_idx_arr == i + if mask.sum() < 5: + ols_b_c[i] = np.nan + continue + x_i = data["x_obs"][mask] + y_i = data["y_obs"][mask] + # Simple OLS: y = X @ beta + try: + beta, _, _, _ = np.linalg.lstsq(x_i, y_i, rcond=None) + ols_b_c[i] = beta[1] # TVL coefficient + except np.linalg.LinAlgError: + ols_b_c[i] = np.nan + + hier_b_c = theta_median[:, 1] + valid = np.isfinite(ols_b_c) + + fig, ax = plt.subplots(figsize=(8, 8)) + ax.scatter(ols_b_c[valid], hier_b_c[valid], alpha=0.6, s=20, color="steelblue") + + # Population mean line + pop_b_c = np.median(hier_b_c) + ax.axhline(pop_b_c, color="red", linestyle="--", linewidth=0.8, + label=f"Population median = {pop_b_c:.3f}") + + # 45-degree line + lims = [min(np.nanmin(ols_b_c[valid]), hier_b_c[valid].min()) - 0.2, + max(np.nanmax(ols_b_c[valid]), hier_b_c[valid].max()) + 0.2] + ax.plot(lims, lims, "k:", linewidth=0.8, alpha=0.5) + ax.set_xlabel("Per-pool OLS b_c") + ax.set_ylabel("Hierarchical posterior median b_c") + ax.set_title("Shrinkage: OLS vs hierarchical TVL elasticity") + ax.legend() + plt.tight_layout() + path5 = os.path.join(output_dir, "bayesian_shrinkage_b_c.png") + plt.savefig(path5, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path5}") + + +# --------------------------------------------------------------------------- +# JSON output +# --------------------------------------------------------------------------- + +def generate_output_json(pool_params, mcmc, data, convergence, output_path, + num_warmup, num_samples, num_chains, target_accept): + """Write structured JSON output with population effects and per-pool params.""" + samples = mcmc.get_samples() + + B_median = np.median(np.array(samples["B"]), axis=0).tolist() + sigma_median = np.median(np.array(samples["sigma"]), axis=0).tolist() + sigma_eps_median = float(np.median(np.array(samples["sigma_eps"]))) + + output = { + "model": "bayesian_hierarchical_loglinear", + "inference": { + "method": "NUTS", + "num_warmup": num_warmup, + "num_samples": num_samples, + "num_chains": num_chains, + "target_accept_prob": target_accept, + }, + "population_effects": { + "B": B_median, + "sigma": sigma_median, + "sigma_eps": sigma_eps_median, + "coeff_names": COEFF_NAMES, + "covariate_names": ( + ["intercept"] + [f"chain_{c}" for c in CHAIN_ORDER] + + ["tier_A_1", "tier_A_2", "tier_B_1", "tier_B_2", "log_fee"] + ), + }, + "convergence": convergence, + "pools": { + p["pool_id"]: { + "chain": p["chain"], + "tokens": p["tokens"], + "theta_median": p["theta_median"], + "theta_std": p["theta_std"], + "noise_params": p["noise_params"], + } + for p in pool_params + }, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params -> {output_path}") + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Bayesian hierarchical noise volume model for Balancer pools" + ) + parser.add_argument( + "--fetch", action="store_true", + help="Fetch pool data (delegates to calibrate_noise_hierarchical.py --fetch)", + ) + parser.add_argument("--fit", action="store_true", help="Run NUTS inference") + parser.add_argument("--plot", action="store_true", help="Generate diagnostic plots") + parser.add_argument("--output", default=None, help="Output JSON path") + parser.add_argument("--output-dir", default="results", help="Plot output directory") + parser.add_argument("--predict", action="store_true", help="Predict for a new pool") + parser.add_argument("--chain", default=None, help="Chain for --predict") + parser.add_argument("--tokens", nargs="+", default=None, help="Tokens for --predict") + parser.add_argument("--fee", type=float, default=0.003, help="Fee for --predict") + parser.add_argument("--cache-dir", default=None, help="Cache directory") + + # NUTS hyperparameters + parser.add_argument("--num-warmup", type=int, default=1000) + parser.add_argument("--num-samples", type=int, default=2000) + parser.add_argument("--num-chains", type=int, default=4) + parser.add_argument("--target-accept", type=float, default=0.85) + parser.add_argument("--max-tree-depth", type=int, default=10) + parser.add_argument("--seed", type=int, default=42) + + args = parser.parse_args() + + cache_dir = args.cache_dir or CACHE_DIR + + if not any([args.fetch, args.fit, args.predict]): + parser.error("At least one of --fetch, --fit, --predict is required") + + # --- Fetch (delegate to existing script) --- + if args.fetch: + import subprocess + cmd = [ + sys.executable, "scripts/calibrate_noise_hierarchical.py", + "--fetch", "--cache-dir", cache_dir, + ] + print("Delegating data fetch to calibrate_noise_hierarchical.py...") + subprocess.run(cmd, check=True) + + # --- Fit --- + if args.fit: + print("\nBayesian Hierarchical Noise Volume Model") + print("=" * 60) + + panel = load_panel(cache_dir) + data = prepare_data(panel) + + mcmc = run_inference( + data, + num_warmup=args.num_warmup, + num_samples=args.num_samples, + num_chains=args.num_chains, + target_accept=args.target_accept, + max_tree_depth=args.max_tree_depth, + seed=args.seed, + ) + + convergence = check_convergence(mcmc) + pool_params = extract_noise_params(mcmc, data) + + # Print summary statistics + b_c_vals = [p["noise_params"]["b_c"] for p in pool_params] + b_0_vals = [p["noise_params"]["b_0"] for p in pool_params] + print(f"\n Per-pool b_c: mean={np.mean(b_c_vals):.3f}, " + f"std={np.std(b_c_vals):.3f}, " + f"range=[{np.min(b_c_vals):.3f}, {np.max(b_c_vals):.3f}]") + print(f" Per-pool b_0: mean={np.mean(b_0_vals):.3f}, " + f"std={np.std(b_0_vals):.3f}") + + if args.output: + generate_output_json( + pool_params, mcmc, data, convergence, args.output, + args.num_warmup, args.num_samples, args.num_chains, + args.target_accept, + ) + + if args.plot: + print("\nGenerating diagnostic plots...") + plot_bayesian_diagnostics(mcmc, data, output_dir=args.output_dir) + + # Save MCMC samples for --predict reuse + mcmc_cache = os.path.join(cache_dir, "bayesian_mcmc_samples.npz") + samples = mcmc.get_samples() + np.savez_compressed( + mcmc_cache, + **{k: np.array(v) for k, v in samples.items()}, + ) + # Also save data arrays for predict + data_cache = os.path.join(cache_dir, "bayesian_data.npz") + np.savez_compressed( + data_cache, + pool_idx=data["pool_idx"], + z_pool=data["z_pool"], + x_obs=data["x_obs"], + y_obs=data["y_obs"], + ) + # Save pool_ids list + with open(os.path.join(cache_dir, "bayesian_pool_ids.json"), "w") as f: + json.dump(data["pool_ids"], f) + print(f" Saved MCMC samples -> {mcmc_cache}") + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + parser.error("--predict requires --chain and --tokens") + + # Load cached MCMC samples + mcmc_cache = os.path.join(cache_dir, "bayesian_mcmc_samples.npz") + if not os.path.exists(mcmc_cache): + print(f"ERROR: MCMC cache not found at {mcmc_cache}", file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + # For predict, we only need B samples — create a minimal mock + cached = np.load(mcmc_cache) + + class _MockMCMC: + """Minimal interface to reuse predict_new_pool with cached samples.""" + def __init__(self, samples_dict): + self._samples = samples_dict + def get_samples(self): + return self._samples + + samples_dict = {k: cached[k] for k in cached.files} + mock_mcmc = _MockMCMC(samples_dict) + + result = predict_new_pool(mock_mcmc, None, args.chain, args.tokens, args.fee) + print(json.dumps(result, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_hierarchical.py b/scripts/calibrate_noise_hierarchical.py new file mode 100644 index 00000000..4fef642d --- /dev/null +++ b/scripts/calibrate_noise_hierarchical.py @@ -0,0 +1,1556 @@ +"""Bayesian hierarchical noise volume model across Balancer WEIGHTED + RECLAMM pools. + +Pools data cross-sectionally across all Balancer weighted/reCLAMM pools, +fits a Bayesian hierarchical model where pool covariates (chain, token tier, +fee) modulate all coefficients via group-level regression, with full +posterior inference via NumPyro. + +Model: + Hyperpriors: + Φ ~ Normal(0, 2) (K × 3) group-level regression + σ_θ ~ HalfNormal(2) (3,) per-coefficient scales + L_ω ~ LKJCholesky(3, η=2) correlation structure + β_weekend ~ Normal(0, 2) shared nuisance + σ_ε ~ HalfNormal(3) observation noise + + For each pool i: + x_i = [1, chain_dummies, tier_dummies, log_fee] (K,) covariates + z_i ~ N(0, I₃) non-centered + θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i (α_i, β_tvl_i, β_vol_i) + + For each observation (i, t): + log(V) ~ N(α_i + β_tvl_i·log_tvl + β_vol_i·vol + β_weekend·weekend, σ²_ε) + +Usage: + # Full pipeline: fetch data + fit model + output + python scripts/calibrate_noise_hierarchical.py \\ + --fetch --fit --output results/hierarchical_noise_params.json --plot + + # Use cached data, re-fit only + python scripts/calibrate_noise_hierarchical.py \\ + --fit --output results/hierarchical_noise_params.json + + # Predict for a new pool + python scripts/calibrate_noise_hierarchical.py \\ + --predict --chain BASE --tokens ETH BTC --fee 0.003 + + # Use NUTS instead of SVI + python scripts/calibrate_noise_hierarchical.py \\ + --fit --nuts --output results/hierarchical_noise_params.json +""" + +import argparse +import json +import os +import sys +import time +import urllib.request +from datetime import datetime, timezone + +import numpy as np +import pandas as pd + +import jax +import jax.numpy as jnp +import numpyro +import numpyro.distributions as dist +from numpyro.infer import SVI, MCMC, NUTS, Trace_ELBO, Predictive +from numpyro.infer.autoguide import AutoMultivariateNormal + +numpyro.enable_x64() + + +# --------------------------------------------------------------------------- +# Constants +# --------------------------------------------------------------------------- + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAINS = [ + "MAINNET", "POLYGON", "ARBITRUM", "GNOSIS", "BASE", "SONIC", "OPTIMISM", + "AVALANCHE", +] + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "local_data", "noise_calibration" +) + +# --------------------------------------------------------------------------- +# Token tier classification +# --------------------------------------------------------------------------- + +# Tier 0: blue-chip — top by volume, wrapped native, major stables +_TIER_0 = { + "ETH", "WETH", "BTC", "WBTC", "cbBTC", "USDC", "USDT", "DAI", + "wstETH", "stETH", "rETH", "cbETH", "WMATIC", "MATIC", "POL", + "WAVAX", "AVAX", "GNO", "WXDAI", "xDAI", + "S", "wS", # Sonic native +} + +# Tier 1: mid-cap DeFi blue-chips (approx CoinGecko rank < 200) +_TIER_1 = { + "AAVE", "LINK", "UNI", "BAL", "MKR", "CRV", "COMP", "SNX", + "LDO", "RPL", "SUSHI", "YFI", "1INCH", "ENS", "DYDX", + "FXS", "FRAX", "LUSD", "sDAI", "GHO", "crvUSD", + "ARB", "OP", "PENDLE", "ENA", "EIGEN", + "SAFE", "COW", +} + + +def _normalise_symbol(symbol: str) -> str: + """Normalise wrapped/bridged variants to canonical form.""" + s = symbol.strip() + # Common wrapped → unwrapped + mapping = { + "WETH": "WETH", # keep WETH as-is (it's in tier 0) + "WBTC": "WBTC", + "cbBTC": "cbBTC", + "WMATIC": "WMATIC", + "WAVAX": "WAVAX", + "WXDAI": "WXDAI", + "wS": "wS", + } + return mapping.get(s, s) + + +def classify_token_tier(symbol: str) -> int: + """Classify a token symbol into tier 0/1/2. + + Returns + ------- + int + 0 = blue-chip, 1 = mid-cap, 2 = long-tail + """ + s = _normalise_symbol(symbol) + if s in _TIER_0: + return 0 + if s in _TIER_1: + return 1 + return 2 + + +# --------------------------------------------------------------------------- +# Phase 1: API data ingestion +# --------------------------------------------------------------------------- + +def _graphql_request(query: dict, base_url: str = BALANCER_API_URL, + timeout: int = 30) -> dict: + """Send a GraphQL request to the Balancer V3 API.""" + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + with urllib.request.urlopen(req, timeout=timeout) as resp: + return json.loads(resp.read().decode("utf-8")) + + +def enumerate_balancer_pools( + chains: list = None, + pool_types: list = None, + min_tvl: float = 10000.0, +) -> pd.DataFrame: + """Enumerate all WEIGHTED + RECLAMM pools across chains from Balancer API. + + Parameters + ---------- + chains : list of str + API chain identifiers (e.g. ["MAINNET", "BASE"]). + pool_types : list of str + Pool type filters (e.g. ["WEIGHTED", "STABLE"]). + min_tvl : float + Minimum TVL in USD to include. + + Returns + ------- + pd.DataFrame + Columns: pool_id, chain, pool_type, tokens (list of symbols), + swap_fee, create_time, dynamic_data_tvl. + """ + if chains is None: + chains = BALANCER_API_CHAINS + if pool_types is None: + pool_types = ["WEIGHTED", "RECLAMM"] + + all_pools = [] + for chain in chains: + print(f" Querying {chain}...", end=" ", flush=True) + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { + chainIn: [$chain] + poolTypeIn: $types + minTvl: $minTvl + } + ) { + id + chain + type + createTime + protocolVersion + poolTokens { + symbol + weight + address + } + dynamicData { + totalLiquidity + swapFee + } + } + } + """, + "variables": { + "chain": chain, + "types": pool_types, + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f"FAILED ({e})") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + weights = [t.get("weight") for t in p.get("poolTokens", [])] + token_addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "protocol_version": p.get("protocolVersion", 0), + "tokens": tokens, + "token_addresses": token_addresses, + "weights": weights, + "swap_fee": fee, + "create_time": p.get("createTime", 0), + "current_tvl": tvl, + }) + + print(f"{len(pools)} pools") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools across {len(chains)} chains") + return df + + +def fetch_pool_snapshots(pool_id: str, chain: str, + base_url: str = BALANCER_API_URL) -> pd.DataFrame: + """Fetch ALL_TIME daily snapshots for a single pool. + + Returns + ------- + pd.DataFrame + Columns: timestamp, volume_usd, total_liquidity_usd. + """ + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + } + } + """, + "variables": { + "poolId": pool_id, + "chain": chain, + "range": "ALL_TIME", + }, + } + + body = _graphql_request(query) + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + + if not snapshots: + return pd.DataFrame(columns=["timestamp", "volume_usd", "total_liquidity_usd"]) + + records = [] + for snap in snapshots: + records.append({ + "timestamp": int(snap["timestamp"]), + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + }) + + df = pd.DataFrame(records) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + # Deduplicate by date (keep last snapshot per day) + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df + + +def fetch_all_snapshots(pools_df: pd.DataFrame, + cache_path: str = None) -> pd.DataFrame: + """Fetch daily snapshots for all pools, with caching. + + Parameters + ---------- + pools_df : pd.DataFrame + Pool enumeration from enumerate_balancer_pools. + cache_path : str, optional + Path to parquet cache. If it exists, only fetch missing pools. + + Returns + ------- + pd.DataFrame + Panel with columns: pool_id, chain, date, volume_usd, + total_liquidity_usd. + """ + # Load cache if exists + cached = pd.DataFrame() + cached_pool_ids = set() + if cache_path and os.path.exists(cache_path): + cached = pd.read_parquet(cache_path) + cached_pool_ids = set(cached["pool_id"].unique()) + print(f" Cache has {len(cached_pool_ids)} pools, " + f"{len(cached)} pool-days") + + # Determine which pools need fetching + if len(pools_df) == 0: + print(" No pools to fetch.") + return cached if len(cached) > 0 else pd.DataFrame( + columns=["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + ) + to_fetch = pools_df[~pools_df["pool_id"].isin(cached_pool_ids)] + print(f" Need to fetch {len(to_fetch)} new pools") + + new_records = [] + for i, (_, pool) in enumerate(to_fetch.iterrows()): + if (i + 1) % 10 == 0 or i == 0: + print(f" Fetching {i+1}/{len(to_fetch)}: {pool['pool_id'][:10]}... " + f"({pool['chain']})", flush=True) + try: + snap_df = fetch_pool_snapshots(pool["pool_id"], pool["chain"]) + if len(snap_df) > 0: + snap_df["pool_id"] = pool["pool_id"] + snap_df["chain"] = pool["chain"] + new_records.append(snap_df[ + ["pool_id", "chain", "date", "volume_usd", + "total_liquidity_usd"] + ]) + except Exception as e: + print(f" FAILED {pool['pool_id'][:10]}: {e}") + time.sleep(0.5) # Rate limit + + if new_records: + new_df = pd.concat(new_records, ignore_index=True) + combined = pd.concat([cached, new_df], ignore_index=True) + else: + combined = cached + + # Save cache + if cache_path and len(combined) > 0: + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + combined.to_parquet(cache_path, index=False) + print(f" Saved cache: {len(combined)} pool-days → {cache_path}") + + return combined + + +def fetch_token_prices(token_addresses_by_chain: dict, + cache_dir: str = None) -> dict: + """Fetch hourly token prices from Balancer API. + + Parameters + ---------- + token_addresses_by_chain : dict + {chain: {symbol: address, ...}, ...} + cache_dir : str, optional + Directory for per-token price caches. + + Returns + ------- + dict + {(chain, symbol): pd.DataFrame with columns [timestamp, price], ...} + """ + if cache_dir: + os.makedirs(cache_dir, exist_ok=True) + + prices = {} + + for chain, tokens in token_addresses_by_chain.items(): + # Check cache first, collect uncached addresses + uncached = {} + for symbol, address in tokens.items(): + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") if cache_dir else None + + if cp and os.path.exists(cp): + prices[(chain, symbol)] = pd.read_parquet(cp) + else: + uncached[symbol] = address + + if not uncached: + continue + + # Batch fetch: API supports multiple addresses per request + addr_to_symbol = {addr: sym for sym, addr in uncached.items()} + addresses = list(uncached.values()) + + print(f" Fetching {len(addresses)} prices on {chain}...", + flush=True) + + # Batch in groups of 20 to avoid oversized requests + batch_size = 20 + for batch_start in range(0, len(addresses), batch_size): + batch_addrs = addresses[batch_start:batch_start + batch_size] + query = { + "query": """ + query GetPrices($chain: GqlChain!, $addresses: [String!]!, + $range: GqlTokenChartDataRange!) { + tokenGetHistoricalPrices( + addresses: $addresses, chain: $chain, range: $range + ) { + address + prices { + timestamp + price + } + } + } + """, + "variables": { + "chain": chain, + "addresses": batch_addrs, + "range": "ONE_YEAR", + }, + } + + try: + body = _graphql_request(query, timeout=60) + results = body.get("data", {}).get( + "tokenGetHistoricalPrices", []) + for result in results: + addr = result.get("address", "") + price_list = result.get("prices", []) + symbol = addr_to_symbol.get(addr) + if symbol and price_list: + pdf = pd.DataFrame(price_list) + pdf["timestamp"] = pdf["timestamp"].astype(int) + pdf["price"] = pdf["price"].astype(float) + prices[(chain, symbol)] = pdf + if cache_dir: + cache_key = f"{chain}_{symbol}".replace("/", "_") + cp = os.path.join(cache_dir, f"{cache_key}.parquet") + pdf.to_parquet(cp, index=False) + except Exception as e: + print(f" FAILED batch on {chain}: {e}") + + time.sleep(0.5) + + print(f" Got prices for {len(prices)} token-chain pairs") + return prices + + +def compute_pair_volatility( + snapshots_df: pd.DataFrame, + pool_row: pd.Series, + token_prices: dict, +) -> pd.Series: + """Compute daily annualised volatility for a pool's pair ratio. + + Uses hourly prices from the API to compute daily realised volatility. + Falls back to a default of 0.5 if price data is insufficient. + + Parameters + ---------- + snapshots_df : pd.DataFrame + Pool's daily snapshots (need dates). + pool_row : pd.Series + Pool metadata row (need tokens, chain). + token_prices : dict + {(chain, symbol): DataFrame, ...} + + Returns + ------- + pd.Series + Indexed by date, values are annualised daily volatility. + """ + tokens = pool_row["tokens"] + chain = pool_row["chain"] + + if len(tokens) < 2: + return pd.Series(dtype=float) + + # Get price series for token[0] and token[1] + # Try chain-specific first, then any chain + def _get_price_df(symbol): + # Exact match + key = (chain, symbol) + if key in token_prices: + return token_prices[key] + # Any chain + for k, v in token_prices.items(): + if k[1] == symbol: + return v + return None + + p0_df = _get_price_df(tokens[0]) + p1_df = _get_price_df(tokens[1]) + + # If either is a stablecoin, use $1 + stables = {"USDC", "USDT", "DAI", "LUSD", "GHO", "crvUSD", "sDAI", + "WXDAI", "xDAI", "USDC.e", "USDbC"} + + if tokens[0] in stables and tokens[1] in stables: + # Stable-stable pair: near-zero vol + dates = snapshots_df["date"].unique() + return pd.Series(0.01, index=dates) + + if p0_df is None and tokens[0] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) # fallback + if p1_df is None and tokens[1] not in stables: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) # fallback + + # Build hourly price ratio + if tokens[0] in stables: + # ratio = 1 / p1 + if p1_df is None or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p1_df.copy() + ratio_df["ratio"] = 1.0 / ratio_df["price"] + elif tokens[1] in stables: + # ratio = p0 + if p0_df is None or len(p0_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = p0_df.copy() + ratio_df["ratio"] = ratio_df["price"] + else: + # Both non-stable: ratio = p0/p1 + if p0_df is None or p1_df is None or len(p0_df) == 0 or len(p1_df) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + # Merge on nearest timestamp + merged = pd.merge_asof( + p0_df.sort_values("timestamp"), + p1_df.sort_values("timestamp"), + on="timestamp", + suffixes=("_0", "_1"), + tolerance=7200, # 2 hour tolerance + ).dropna() + if len(merged) == 0: + dates = snapshots_df["date"].unique() + return pd.Series(0.5, index=dates) + ratio_df = merged.copy() + ratio_df["ratio"] = merged["price_0"] / merged["price_1"] + + ratio_df["datetime"] = pd.to_datetime(ratio_df["timestamp"], unit="s") + ratio_df["date"] = ratio_df["datetime"].dt.date + ratio_df = ratio_df.sort_values("timestamp") + + # Log returns + ratio_df["log_return"] = np.log( + ratio_df["ratio"] / ratio_df["ratio"].shift(1) + ) + ratio_df = ratio_df.dropna(subset=["log_return"]) + + # Daily vol from hourly returns, annualised + daily_vol = ratio_df.groupby("date")["log_return"].std() + # Hourly data → ~24 returns/day. Annualise: σ_daily * sqrt(365) + # But std() already gives daily std from hourly returns, so: + # σ_annual = σ_hourly * sqrt(24 * 365) + daily_vol_ann = daily_vol * np.sqrt(24 * 365) + + return daily_vol_ann + + +def assemble_panel( + pools_df: pd.DataFrame, + snapshots_df: pd.DataFrame, + token_prices: dict, +) -> pd.DataFrame: + """Assemble the full panel DataFrame for hierarchical estimation. + + Parameters + ---------- + pools_df : pd.DataFrame + Pool enumeration from enumerate_balancer_pools. + snapshots_df : pd.DataFrame + Daily snapshots from fetch_all_snapshots. + token_prices : dict + Token prices from fetch_token_prices. + + Returns + ------- + pd.DataFrame + Panel with columns: pool_id, chain, date, log_volume, log_tvl, + volatility, weekend, log_fee, tier_A, tier_B, tokens. + """ + records = [] + pool_ids = snapshots_df["pool_id"].unique() + n_pools = len(pool_ids) + + for i, pool_id in enumerate(pool_ids): + if (i + 1) % 20 == 0 or i == 0: + print(f" Assembling {i+1}/{n_pools}...", flush=True) + + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pool_id] + pool_meta = pools_df[pools_df["pool_id"] == pool_id] + if len(pool_meta) == 0: + continue + pool_row = pool_meta.iloc[0] + + tokens = pool_row["tokens"] + if len(tokens) < 2: + continue + + chain = pool_row["chain"] + swap_fee = pool_row["swap_fee"] + + # Token tiers: sort by tier (best tier first) + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] # best (lowest) tier + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + # Compute volatility for this pool's pair + vol_series = compute_pair_volatility(pool_snaps, pool_row, token_prices) + + for _, snap in pool_snaps.iterrows(): + date = snap["date"] + volume = snap["volume_usd"] + tvl = snap["total_liquidity_usd"] + + # Skip zero/negative TVL or volume + if tvl <= 0 or volume <= 0: + continue + + # Volatility lookup + if isinstance(vol_series, pd.Series) and date in vol_series.index: + vol = vol_series[date] + else: + vol = 0.5 # fallback + + if not np.isfinite(vol) or vol <= 0: + vol = 0.5 + + # Weekend indicator + if isinstance(date, datetime): + is_weekend = date.weekday() >= 5 + else: + is_weekend = pd.Timestamp(date).weekday() >= 5 + + records.append({ + "pool_id": pool_id, + "chain": chain, + "date": date, + "log_volume": np.log(volume), + "log_tvl": np.log(tvl), + "volatility": vol, + "weekend": 1.0 if is_weekend else 0.0, + "log_fee": np.log(max(swap_fee, 1e-6)), + "tier_A": tier_a, + "tier_B": tier_b, + "tokens": ",".join(tokens[:2]), + "swap_fee": swap_fee, + }) + + panel = pd.DataFrame(records) + print(f"\n Panel: {len(panel)} observations, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + return panel + + +# --------------------------------------------------------------------------- +# Phase 2: Bayesian hierarchical model +# --------------------------------------------------------------------------- + +def _encode_covariates(panel: pd.DataFrame) -> dict: + """Build NumPyro-ready arrays from the panel DataFrame. + + Returns + ------- + dict with keys: + pool_idx : (N_obs,) int array mapping each observation to its pool + X_pool : (N_pools, K) covariate matrix (intercept + dummies + log_fee) + log_tvl, volatility, weekend, log_volume : (N_obs,) float arrays + pool_ids : (N_pools,) pool ID strings + covariate_names : list of str, column names for X_pool + ref_chain, ref_tier_a, ref_tier_b : reference categories + chains : sorted list of all chains + pool_meta : DataFrame of per-pool metadata + """ + pool_meta = panel.drop_duplicates("pool_id").reset_index(drop=True) + pool_ids = pool_meta["pool_id"].values + pool_id_to_idx = {pid: i for i, pid in enumerate(pool_ids)} + + pool_idx = panel["pool_id"].map(pool_id_to_idx).values + + # Build X_pool columns + chains = sorted(panel["chain"].unique()) + ref_chain = chains[0] + chain_cols = [] + chain_names = [] + for c in chains[1:]: + chain_cols.append((pool_meta["chain"] == c).astype(float).values) + chain_names.append(f"chain_{c}") + + tier_a_vals = sorted(pool_meta["tier_A"].astype(str).unique()) + ref_tier_a = tier_a_vals[0] + tier_a_cols = [] + tier_a_names = [] + for t in tier_a_vals[1:]: + tier_a_cols.append( + (pool_meta["tier_A"].astype(str) == t).astype(float).values + ) + tier_a_names.append(f"tier_A_{t}") + + tier_b_vals = sorted(pool_meta["tier_B"].astype(str).unique()) + ref_tier_b = tier_b_vals[0] + tier_b_cols = [] + tier_b_names = [] + for t in tier_b_vals[1:]: + tier_b_cols.append( + (pool_meta["tier_B"].astype(str) == t).astype(float).values + ) + tier_b_names.append(f"tier_B_{t}") + + N_pools = len(pool_ids) + columns = [np.ones((N_pools, 1))] + col_names = ["intercept"] + + for arr, name in zip(chain_cols, chain_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_a_cols, tier_a_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + for arr, name in zip(tier_b_cols, tier_b_names): + columns.append(arr.reshape(-1, 1)) + col_names.append(name) + columns.append(pool_meta["log_fee"].values.reshape(-1, 1)) + col_names.append("log_fee") + + X_pool = np.hstack(columns) + + return { + "pool_idx": pool_idx.astype(np.int32), + "X_pool": X_pool.astype(np.float64), + "log_tvl": panel["log_tvl"].values.astype(np.float64), + "volatility": panel["volatility"].values.astype(np.float64), + "weekend": panel["weekend"].values.astype(np.float64), + "log_volume": panel["log_volume"].values.astype(np.float64), + "pool_ids": pool_ids, + "covariate_names": col_names, + "ref_chain": ref_chain, + "ref_tier_a": ref_tier_a, + "ref_tier_b": ref_tier_b, + "chains": chains, + "pool_meta": pool_meta, + } + + +def _hierarchical_noise_model( + pool_idx, X_pool, log_tvl, volatility, weekend, log_volume=None, +): + """NumPyro model: Bayesian hierarchical loglinear noise volume. + + All pool covariates modulate all three coefficients (α, β_tvl, β_vol) + through the group-level regression matrix Φ, with correlated random + effects via LKJ-Cholesky. + """ + N_pools = X_pool.shape[0] + K = X_pool.shape[1] + + # Hyperpriors + Phi = numpyro.sample("Phi", dist.Normal(0, 2).expand([K, 3]).to_event(2)) + sigma_theta = numpyro.sample( + "sigma_theta", dist.HalfNormal(2).expand([3]).to_event(1) + ) + L_omega = numpyro.sample("L_omega", dist.LKJCholesky(3, concentration=2)) + beta_weekend = numpyro.sample("beta_weekend", dist.Normal(0, 2)) + sigma_eps = numpyro.sample("sigma_eps", dist.HalfNormal(3)) + + # Non-centered pool random effects + with numpyro.plate("pools", N_pools): + z = numpyro.sample("z", dist.Normal(jnp.zeros(3), 1).to_event(1)) + + # θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i + mu = X_pool @ Phi # (N_pools, 3) + L_Sigma = sigma_theta[:, None] * L_omega # (3, 3) + theta = mu + z @ L_Sigma.T # (N_pools, 3) + + alpha = theta[:, 0] + beta_tvl = theta[:, 1] + beta_vol = theta[:, 2] + + # Observation model + loc = (alpha[pool_idx] + + beta_tvl[pool_idx] * log_tvl + + beta_vol[pool_idx] * volatility + + beta_weekend * weekend) + + with numpyro.plate("obs", pool_idx.shape[0]): + numpyro.sample("log_volume", dist.Normal(loc, sigma_eps), obs=log_volume) + + +def fit_bayesian_model( + panel: pd.DataFrame, use_nuts: bool = False, +) -> tuple: + """Fit the Bayesian hierarchical model via SVI or NUTS. + + Parameters + ---------- + panel : pd.DataFrame + Panel from assemble_panel. + use_nuts : bool + If True, use NUTS MCMC (slower, exact). Otherwise SVI with + AutoMultivariateNormal guide. + + Returns + ------- + samples : dict + Posterior samples keyed by parameter name. + encoding : dict + From _encode_covariates (needed downstream). + """ + encoding = _encode_covariates(panel) + + model_kwargs = dict( + pool_idx=jnp.array(encoding["pool_idx"]), + X_pool=jnp.array(encoding["X_pool"]), + log_tvl=jnp.array(encoding["log_tvl"]), + volatility=jnp.array(encoding["volatility"]), + weekend=jnp.array(encoding["weekend"]), + log_volume=jnp.array(encoding["log_volume"]), + ) + + N_pools = encoding["X_pool"].shape[0] + K = encoding["X_pool"].shape[1] + print(f" N obs = {len(encoding['pool_idx'])}, " + f"N pools = {N_pools}, K covariates = {K}") + print(f" Covariates: {encoding['covariate_names']}") + + rng_key = jax.random.PRNGKey(0) + + if use_nuts: + print(" Running NUTS (500 warmup + 1000 samples)...") + kernel = NUTS(_hierarchical_noise_model) + mcmc = MCMC(kernel, num_warmup=500, num_samples=1000, num_chains=1) + mcmc.run(rng_key, **model_kwargs) + samples = mcmc.get_samples() + print(" NUTS complete.") + else: + print(" Running SVI with AutoMultivariateNormal (20k steps)...") + guide = AutoMultivariateNormal(_hierarchical_noise_model) + optimizer = numpyro.optim.Adam(1e-3) + svi = SVI(_hierarchical_noise_model, guide, optimizer, + loss=Trace_ELBO()) + svi_result = svi.run(rng_key, 20_000, **model_kwargs) + print(f" SVI complete. Final ELBO loss: {svi_result.losses[-1]:.2f}") + + predictive = Predictive( + guide, params=svi_result.params, num_samples=1000, + ) + samples = predictive(jax.random.PRNGKey(1), **model_kwargs) + print(" Drew 1000 posterior samples.") + + return samples, encoding + + +def extract_posteriors(samples: dict, encoding: dict) -> dict: + """Reconstruct pool-specific coefficients from posterior samples. + + Computes θ_i = Φᵀx_i + diag(σ_θ)·L_ω·z_i for each posterior draw, + then returns posterior means and variance components. + + Returns + ------- + dict with keys: + pool_effects : {pool_id: {alpha, beta_tvl, beta_vol}} + Phi_mean : (K, 3) array + sigma_theta_mean : (3,) array + correlation_matrix : (3, 3) array + beta_weekend_mean : float + sigma_eps_mean : float + theta_samples : (S, N_pools, 3) array (for diagnostics) + """ + Phi = np.array(samples["Phi"]) # (S, K, 3) + sigma_theta = np.array(samples["sigma_theta"]) # (S, 3) + L_omega = np.array(samples["L_omega"]) # (S, 3, 3) + z = np.array(samples["z"]) # (S, N_pools, 3) + beta_weekend = np.array(samples["beta_weekend"]) # (S,) + sigma_eps = np.array(samples["sigma_eps"]) # (S,) + + X_pool = encoding["X_pool"] # (N_pools, K) + pool_ids = encoding["pool_ids"] + + # mu = X_pool @ Phi for each sample: (S, N_pools, 3) + mu = np.einsum("pk,skj->spj", X_pool, Phi) + + # L_Sigma = diag(sigma_theta) @ L_omega: (S, 3, 3) + L_Sigma = sigma_theta[:, :, None] * L_omega + + # offset = z @ L_Sigma^T: (S, N_pools, 3) + offset = np.einsum("spi,sji->spj", z, L_Sigma) + + theta = mu + offset # (S, N_pools, 3) + theta_mean = theta.mean(axis=0) # (N_pools, 3) + + pool_effects = {} + for i, pid in enumerate(pool_ids): + pool_effects[pid] = { + "alpha": float(theta_mean[i, 0]), + "beta_tvl": float(theta_mean[i, 1]), + "beta_vol": float(theta_mean[i, 2]), + } + + Phi_mean = Phi.mean(axis=0) # (K, 3) + + # Correlation matrix: R = L_omega @ L_omega^T, averaged over samples + R_samples = np.einsum("sij,skj->sik", L_omega, L_omega) + R_mean = R_samples.mean(axis=0) + + return { + "pool_effects": pool_effects, + "Phi_mean": Phi_mean, + "sigma_theta_mean": sigma_theta.mean(axis=0), + "correlation_matrix": R_mean, + "beta_weekend_mean": float(beta_weekend.mean()), + "sigma_eps_mean": float(sigma_eps.mean()), + "theta_samples": theta, + } + + +def compute_noise_params(posteriors: dict, panel: pd.DataFrame) -> list: + """Convert posterior pool effects to per-pool noise_params dicts. + + Each pool now has its own β_tvl and β_vol (from the hierarchical + posterior), rather than sharing global slopes. + + Parameters + ---------- + posteriors : dict + From extract_posteriors. + panel : pd.DataFrame + Panel data. + + Returns + ------- + list of dict + Each dict has: pool_id, chain, tokens, noise_params. + """ + pool_effects = posteriors["pool_effects"] + pool_meta = panel.drop_duplicates("pool_id").set_index("pool_id") + + results = [] + for pool_id, effects in pool_effects.items(): + if pool_id not in pool_meta.index: + continue + meta = pool_meta.loc[pool_id] + swap_fee = float(meta.get("swap_fee", 0.003)) + + results.append({ + "pool_id": pool_id, + "chain": meta["chain"], + "tokens": (meta["tokens"].split(",") + if isinstance(meta["tokens"], str) + else meta["tokens"]), + "noise_params": { + "b_0": effects["alpha"], + "b_sigma": effects["beta_vol"], + "b_c": effects["beta_tvl"], + "base_fee": swap_fee, + }, + }) + + return results + + +def _build_covariate_vector( + encoding: dict, chain: str, tokens: list, fee: float, +) -> np.ndarray: + """Construct a covariate vector x for a new pool. + + Matches the column order of X_pool from _encode_covariates so that + x @ Phi_mean gives population-level predictions for all 3 coefficients. + """ + col_names = encoding["covariate_names"] + x = np.zeros(len(col_names)) + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = str(tiers[0]) + tier_b = str(tiers[1]) if len(tiers) > 1 else tier_a + + for i, name in enumerate(col_names): + if name == "intercept": + x[i] = 1.0 + elif name == "log_fee": + x[i] = np.log(max(fee, 1e-6)) + elif name == f"chain_{chain}": + x[i] = 1.0 + elif name == f"tier_A_{tier_a}": + x[i] = 1.0 + elif name == f"tier_B_{tier_b}": + x[i] = 1.0 + + return x + + +def predict_new_pool( + posteriors: dict, + encoding: dict, + chain: str, + tokens: list, + fee: float, +) -> dict: + """Predict noise params for an unseen pool. + + Uses population-level estimates only (x @ Φ, no pool random effect). + + Parameters + ---------- + posteriors : dict + From extract_posteriors. + encoding : dict + From _encode_covariates (or loaded from cache). + chain : str + Chain API identifier (e.g. "BASE"). + tokens : list + Token symbols (e.g. ["ETH", "BTC"]). + fee : float + Swap fee rate. + + Returns + ------- + dict + noise_params dict with pool-predicted coefficients. + """ + x = _build_covariate_vector(encoding, chain, tokens, fee) + Phi_mean = posteriors["Phi_mean"] + + theta_pred = x @ Phi_mean # (3,) — population-level prediction + + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + tier_b = tiers[1] if len(tiers) > 1 else tiers[0] + + return { + "b_0": float(theta_pred[0]), + "b_sigma": float(theta_pred[2]), + "b_c": float(theta_pred[1]), + "base_fee": float(fee), + "_prediction_source": "population_level", + "_alpha": float(theta_pred[0]), + "_beta_tvl": float(theta_pred[1]), + "_beta_vol": float(theta_pred[2]), + "_tier_a": tier_a, + "_tier_b": tier_b, + } + + +# --------------------------------------------------------------------------- +# Phase 3: Diagnostics and output +# --------------------------------------------------------------------------- + +def plot_hierarchical_diagnostics( + panel: pd.DataFrame, + posteriors: dict, + encoding: dict, + output_dir: str = "results", +): + """Generate diagnostic plots for the Bayesian hierarchical model. + + Figure 1 (2x2): + (0,0) Pool-specific coefficient distributions (α, β_tvl, β_vol) + (0,1) Chain effects on all 3 coefficients + (1,0) Tier effects on all 3 coefficients + (1,1) Model summary (Φ, σ_θ, correlations, σ_ε) + + Figure 2: + β_tvl vs β_vol scatter colored by chain + """ + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + os.makedirs(output_dir, exist_ok=True) + + pool_effects = posteriors["pool_effects"] + Phi_mean = posteriors["Phi_mean"] + col_names = encoding["covariate_names"] + pool_meta = panel.drop_duplicates("pool_id") + + alphas = [e["alpha"] for e in pool_effects.values()] + beta_tvls = [e["beta_tvl"] for e in pool_effects.values()] + beta_vols = [e["beta_vol"] for e in pool_effects.values()] + + # --- Figure 1: Diagnostics 2x2 --- + fig, axes = plt.subplots(2, 2, figsize=(14, 10)) + + # (0,0) Pool-specific coefficient distributions + ax = axes[0, 0] + bins = 25 + ax.hist(alphas, bins=bins, alpha=0.6, label="α (intercept)", + color="steelblue", edgecolor="white") + ax.hist(beta_tvls, bins=bins, alpha=0.6, label="β_tvl", + color="coral", edgecolor="white") + ax.hist(beta_vols, bins=bins, alpha=0.6, label="β_vol", + color="seagreen", edgecolor="white") + ax.set_xlabel("Coefficient value") + ax.set_ylabel("Count") + ax.set_title(f"Pool-specific coefficients (n={len(alphas)})") + ax.legend() + + # (0,1) Chain effects on all 3 coefficients + ax = axes[0, 1] + chain_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("chain_")} + chain_labels = [] + chain_alpha_effects = [] + chain_tvl_effects = [] + chain_vol_effects = [] + # Reference chain (effect = 0) + ref_chain = encoding["ref_chain"] + chain_counts = pool_meta["chain"].value_counts() + chain_labels.append(f"{ref_chain}\n(n={chain_counts.get(ref_chain, 0)})") + chain_alpha_effects.append(0.0) + chain_tvl_effects.append(0.0) + chain_vol_effects.append(0.0) + for name, row_idx in sorted(chain_rows.items()): + chain_name = name.replace("chain_", "") + chain_labels.append( + f"{chain_name}\n(n={chain_counts.get(chain_name, 0)})" + ) + chain_alpha_effects.append(Phi_mean[row_idx, 0]) + chain_tvl_effects.append(Phi_mean[row_idx, 1]) + chain_vol_effects.append(Phi_mean[row_idx, 2]) + + y_pos = np.arange(len(chain_labels)) + bar_h = 0.25 + ax.barh(y_pos - bar_h, chain_alpha_effects, bar_h, label="α", + color="steelblue", alpha=0.8) + ax.barh(y_pos, chain_tvl_effects, bar_h, label="β_tvl", + color="coral", alpha=0.8) + ax.barh(y_pos + bar_h, chain_vol_effects, bar_h, label="β_vol", + color="seagreen", alpha=0.8) + ax.set_yticks(y_pos) + ax.set_yticklabels(chain_labels) + ax.axvline(0, color="red", linestyle="--", linewidth=1) + ax.set_xlabel("Effect (relative to reference)") + ax.set_title("Chain effects on all coefficients") + ax.legend(fontsize=8) + + # (1,0) Tier effects on all 3 coefficients + ax = axes[1, 0] + tier_names = ["0 (blue-chip)", "1 (mid-cap)", "2 (long-tail)"] + coeff_labels = ["α", "β_tvl", "β_vol"] + tier_a_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("tier_A_")} + tier_b_rows = {name: i for i, name in enumerate(col_names) + if name.startswith("tier_B_")} + + # Build effects matrix: (3 tiers) x (3 coefficients) x (A/B) + x_pos = np.arange(len(tier_names)) + width = 0.13 + for coeff_idx, (coeff_name, color) in enumerate( + zip(coeff_labels, ["steelblue", "coral", "seagreen"]) + ): + tier_a_vals = [0.0, 0.0, 0.0] # reference tier gets 0 + tier_b_vals = [0.0, 0.0, 0.0] + for name, row_idx in tier_a_rows.items(): + tier_val = name.replace("tier_A_", "") + if tier_val in ("0", "1", "2"): + tier_a_vals[int(tier_val)] = Phi_mean[row_idx, coeff_idx] + for name, row_idx in tier_b_rows.items(): + tier_val = name.replace("tier_B_", "") + if tier_val in ("0", "1", "2"): + tier_b_vals[int(tier_val)] = Phi_mean[row_idx, coeff_idx] + offset = (coeff_idx - 1) * width * 2 + ax.bar(x_pos + offset - width / 2, tier_a_vals, width, + label=f"{coeff_name} (A)" if coeff_idx == 0 else "", + color=color, alpha=0.7, edgecolor="white") + ax.bar(x_pos + offset + width / 2, tier_b_vals, width, + label=f"{coeff_name} (B)" if coeff_idx == 0 else "", + color=color, alpha=0.4, edgecolor="white", hatch="//") + + ax.set_xticks(x_pos) + ax.set_xticklabels(tier_names) + ax.set_ylabel("Effect on coefficient") + ax.set_title("Token tier effects (solid=A, hatched=B)") + ax.axhline(0, color="black", linewidth=0.5) + # Manual legend for coefficient colors + from matplotlib.patches import Patch + ax.legend(handles=[Patch(color=c, label=l) for c, l in + zip(["steelblue", "coral", "seagreen"], coeff_labels)], + fontsize=8) + + # (1,1) Model summary text + ax = axes[1, 1] + ax.axis("off") + sigma_theta = posteriors["sigma_theta_mean"] + R = posteriors["correlation_matrix"] + summary = "Group-level regression Φ (posterior mean):\n" + summary += f" {'covariate':<20s} {'α':>8s} {'β_tvl':>8s} {'β_vol':>8s}\n" + summary += " " + "-" * 46 + "\n" + for j, name in enumerate(col_names): + summary += (f" {name:<20s} {Phi_mean[j,0]:>8.3f} " + f"{Phi_mean[j,1]:>8.3f} {Phi_mean[j,2]:>8.3f}\n") + summary += f"\nσ_θ: [{sigma_theta[0]:.3f}, {sigma_theta[1]:.3f}, " + summary += f"{sigma_theta[2]:.3f}]\n" + summary += f"Correlation:\n" + for i in range(3): + summary += f" [{R[i,0]:>6.3f} {R[i,1]:>6.3f} {R[i,2]:>6.3f}]\n" + summary += f"β_weekend: {posteriors['beta_weekend_mean']:.4f}\n" + summary += f"σ_ε: {posteriors['sigma_eps_mean']:.4f}\n" + ax.text(0.02, 0.98, summary, transform=ax.transAxes, + fontsize=7, verticalalignment="top", fontfamily="monospace") + ax.set_title("Model Summary") + + fig.suptitle( + "Bayesian Hierarchical Noise Model — Diagnostics", fontsize=13, + ) + plt.tight_layout() + path = os.path.join(output_dir, "hierarchical_diagnostics.png") + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- Figure 2: β_tvl vs β_vol scatter colored by chain --- + fig2, ax2 = plt.subplots(figsize=(10, 7)) + pool_id_to_chain = dict( + zip(pool_meta["pool_id"], pool_meta["chain"]) + ) + chain_colors = {} + cmap = plt.cm.tab10 + unique_chains = sorted(pool_meta["chain"].unique()) + for i, c in enumerate(unique_chains): + chain_colors[c] = cmap(i % 10) + + for pid, effects in pool_effects.items(): + c = pool_id_to_chain.get(pid, "?") + ax2.scatter(effects["beta_tvl"], effects["beta_vol"], + color=chain_colors.get(c, "gray"), alpha=0.6, s=20, + edgecolors="white", linewidths=0.3) + + # Legend + from matplotlib.lines import Line2D + handles = [Line2D([0], [0], marker="o", color="w", + markerfacecolor=chain_colors[c], markersize=8, + label=c) + for c in unique_chains if c in chain_colors] + ax2.legend(handles=handles, fontsize=8, loc="best") + ax2.set_xlabel("β_tvl (TVL elasticity)") + ax2.set_ylabel("β_vol (volatility sensitivity)") + ax2.set_title("Pool-specific coefficients by chain") + ax2.axhline(0, color="gray", linewidth=0.5, linestyle="--") + ax2.axvline(0, color="gray", linewidth=0.5, linestyle="--") + plt.tight_layout() + path2 = os.path.join(output_dir, "beta_tvl_vs_beta_vol.png") + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + return path, path2 + + +def generate_noise_params_json( + pool_params: list, + posteriors: dict, + encoding: dict, + output_path: str, + inference_method: str = "svi", +): + """Write per-pool noise params to JSON. + + Parameters + ---------- + pool_params : list of dict + From compute_noise_params. + posteriors : dict + From extract_posteriors. + encoding : dict + From _encode_covariates. + output_path : str + Output JSON path. + inference_method : str + "svi" or "nuts". + """ + output = { + "model": "bayesian_hierarchical_loglinear", + "inference_method": inference_method, + "Phi": posteriors["Phi_mean"].tolist(), + "covariate_names": encoding["covariate_names"], + "sigma_theta": posteriors["sigma_theta_mean"].tolist(), + "correlation_matrix": posteriors["correlation_matrix"].tolist(), + "beta_weekend": posteriors["beta_weekend_mean"], + "sigma_eps": posteriors["sigma_eps_mean"], + "pools": pool_params, + } + + os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True) + with open(output_path, "w") as f: + json.dump(output, f, indent=2, default=str) + print(f" Wrote {len(pool_params)} pool params → {output_path}") + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Bayesian hierarchical noise volume model for Balancer pools" + ) + parser.add_argument( + "--fetch", action="store_true", + help="Fetch pool data from Balancer API (cached to local_data/)", + ) + parser.add_argument( + "--fit", action="store_true", + help="Fit Bayesian hierarchical model (requires fetched data)", + ) + parser.add_argument( + "--nuts", action="store_true", + help="Use NUTS MCMC instead of SVI (slower, exact posteriors)", + ) + parser.add_argument( + "--plot", action="store_true", + help="Generate diagnostic plots", + ) + parser.add_argument( + "--output", default=None, + help="Output JSON path for per-pool noise params", + ) + parser.add_argument( + "--output-dir", default="results", + help="Directory for diagnostic plots (default: results)", + ) + parser.add_argument( + "--predict", action="store_true", + help="Predict noise params for a new pool", + ) + parser.add_argument( + "--chain", default=None, + help="Chain for --predict (e.g. BASE, MAINNET)", + ) + parser.add_argument( + "--tokens", nargs="+", default=None, + help="Token symbols for --predict (e.g. ETH BTC)", + ) + parser.add_argument( + "--fee", type=float, default=0.003, + help="Swap fee for --predict", + ) + parser.add_argument( + "--min-tvl", type=float, default=10000.0, + help="Minimum TVL filter for pool enumeration", + ) + parser.add_argument( + "--cache-dir", default=None, + help="Cache directory (default: local_data/noise_calibration/)", + ) + args = parser.parse_args() + + cache_dir = args.cache_dir or CACHE_DIR + + if not any([args.fetch, args.fit, args.predict]): + parser.error("At least one of --fetch, --fit, --predict is required") + + # --- Fetch --- + pools_cache = os.path.join(cache_dir, "pools.parquet") + snaps_cache = os.path.join(cache_dir, "pool_snapshots.parquet") + prices_cache = os.path.join(cache_dir, "token_prices") + panel_cache = os.path.join(cache_dir, "panel.parquet") + + if args.fetch: + print("Phase 1: Fetching data from Balancer API") + print("=" * 60) + + # Step 1: Enumerate pools + print("\n1. Enumerating pools...") + pools_df = enumerate_balancer_pools(min_tvl=args.min_tvl) + os.makedirs(cache_dir, exist_ok=True) + pools_df.to_parquet(pools_cache, index=False) + print(f" Saved {len(pools_df)} pools → {pools_cache}") + + # Step 2: Fetch snapshots + print("\n2. Fetching daily snapshots...") + snapshots_df = fetch_all_snapshots(pools_df, cache_path=snaps_cache) + + # Step 3: Fetch token prices + print("\n3. Fetching token prices...") + token_addr_by_chain = {} + for _, pool in pools_df.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + # Step 4: Assemble panel + print("\n4. Assembling panel...") + panel = assemble_panel(pools_df, snapshots_df, token_prices) + panel.to_parquet(panel_cache, index=False) + print(f" Saved panel → {panel_cache}") + + print(f"\nFetch complete. Panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools") + + # --- Fit --- + if args.fit: + inference_method = "nuts" if args.nuts else "svi" + print(f"\nPhase 2: Fitting Bayesian hierarchical model ({inference_method})") + print("=" * 60) + + # Load panel + if not os.path.exists(panel_cache): + print(f"ERROR: Panel cache not found at {panel_cache}", + file=sys.stderr) + print("Run with --fetch first.", file=sys.stderr) + sys.exit(1) + + panel = pd.read_parquet(panel_cache) + print(f" Loaded panel: {len(panel)} obs, " + f"{panel['pool_id'].nunique()} pools, " + f"{panel['chain'].nunique()} chains") + + # Filter: need at least 10 days per pool for stable estimates + pool_counts = panel.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel_filtered = panel[panel["pool_id"].isin(valid_pools)] + print(f" After filtering (≥10 days): {len(panel_filtered)} obs, " + f"{panel_filtered['pool_id'].nunique()} pools") + + samples, encoding = fit_bayesian_model( + panel_filtered, use_nuts=args.nuts, + ) + posteriors = extract_posteriors(samples, encoding) + pool_params = compute_noise_params(posteriors, panel_filtered) + + # Print key diagnostics + Phi_mean = posteriors["Phi_mean"] + col_names = encoding["covariate_names"] + intercept_idx = col_names.index("intercept") + log_fee_idx = col_names.index("log_fee") + + print(f"\n Key results:") + print(f" Population intercept (Φ[intercept]):") + print(f" α: {Phi_mean[intercept_idx, 0]:.4f}") + print(f" β_tvl: {Phi_mean[intercept_idx, 1]:.4f}") + print(f" β_vol: {Phi_mean[intercept_idx, 2]:.4f}") + print(f" Fee effect (Φ[log_fee]):") + print(f" α: {Phi_mean[log_fee_idx, 0]:.4f}") + print(f" β_tvl: {Phi_mean[log_fee_idx, 1]:.4f}") + print(f" β_vol: {Phi_mean[log_fee_idx, 2]:.4f}") + print(f" β_weekend: {posteriors['beta_weekend_mean']:.4f}") + print(f" σ_θ: {posteriors['sigma_theta_mean']}") + print(f" σ_ε: {posteriors['sigma_eps_mean']:.4f}") + + # Verify pool-specific variation + b_sigmas = [p["noise_params"]["b_sigma"] for p in pool_params] + b_cs = [p["noise_params"]["b_c"] for p in pool_params] + print(f" b_sigma range: [{min(b_sigmas):.4f}, {max(b_sigmas):.4f}]") + print(f" b_c range: [{min(b_cs):.4f}, {max(b_cs):.4f}]") + + # Cache posteriors + encoding for --predict and --plot + posteriors_cache = os.path.join(cache_dir, "posteriors.json") + cache_data = { + "Phi_mean": posteriors["Phi_mean"].tolist(), + "sigma_theta_mean": posteriors["sigma_theta_mean"].tolist(), + "correlation_matrix": posteriors["correlation_matrix"].tolist(), + "beta_weekend_mean": posteriors["beta_weekend_mean"], + "sigma_eps_mean": posteriors["sigma_eps_mean"], + "pool_effects": posteriors["pool_effects"], + "covariate_names": encoding["covariate_names"], + "ref_chain": encoding["ref_chain"], + "ref_tier_a": encoding["ref_tier_a"], + "ref_tier_b": encoding["ref_tier_b"], + "chains": encoding["chains"], + "inference_method": inference_method, + } + with open(posteriors_cache, "w") as f: + json.dump(cache_data, f, indent=2, default=str) + print(f" Cached posteriors → {posteriors_cache}") + + if args.output: + generate_noise_params_json( + pool_params, posteriors, encoding, + args.output, inference_method=inference_method, + ) + + if args.plot: + print("\nPhase 3: Generating diagnostics") + print("=" * 60) + plot_hierarchical_diagnostics( + panel_filtered, posteriors, encoding, + output_dir=args.output_dir, + ) + + # --- Predict --- + if args.predict: + if args.chain is None or args.tokens is None: + parser.error("--predict requires --chain and --tokens") + + print(f"\nPredicting noise params for new pool:") + print(f" Chain: {args.chain}") + print(f" Tokens: {args.tokens}") + print(f" Fee: {args.fee}") + + # Load cached posteriors + encoding metadata + posteriors_cache = os.path.join(cache_dir, "posteriors.json") + if not os.path.exists(posteriors_cache): + print(f"ERROR: Posteriors cache not found at {posteriors_cache}", + file=sys.stderr) + print("Run with --fit first.", file=sys.stderr) + sys.exit(1) + + with open(posteriors_cache) as f: + cache_data = json.load(f) + + posteriors = { + "Phi_mean": np.array(cache_data["Phi_mean"]), + "sigma_theta_mean": np.array(cache_data["sigma_theta_mean"]), + "correlation_matrix": np.array(cache_data["correlation_matrix"]), + "beta_weekend_mean": cache_data["beta_weekend_mean"], + "sigma_eps_mean": cache_data["sigma_eps_mean"], + "pool_effects": cache_data["pool_effects"], + } + encoding = { + "covariate_names": cache_data["covariate_names"], + "ref_chain": cache_data["ref_chain"], + "ref_tier_a": cache_data["ref_tier_a"], + "ref_tier_b": cache_data["ref_tier_b"], + "chains": cache_data["chains"], + } + + params = predict_new_pool( + posteriors, encoding, args.chain, args.tokens, args.fee, + ) + print(f"\n Predicted noise_params:") + print(json.dumps(params, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_noise_unified.py b/scripts/calibrate_noise_unified.py new file mode 100644 index 00000000..0e355271 --- /dev/null +++ b/scripts/calibrate_noise_unified.py @@ -0,0 +1,6 @@ +"""Thin wrapper — all logic lives in quantammsim.noise_calibration.""" +from quantammsim.noise_calibration import * # noqa: F401, F403 +from quantammsim.noise_calibration.cli import main + +if __name__ == "__main__": + main() diff --git a/scripts/calibrate_reclamm_noise.py b/scripts/calibrate_reclamm_noise.py new file mode 100644 index 00000000..8d095841 --- /dev/null +++ b/scripts/calibrate_reclamm_noise.py @@ -0,0 +1,842 @@ +"""OLS calibration of the Tsoukalas noise volume model for reClAMM pools. + +Fits the structural volume equation: + V_daily/1e6 = a_0 + a_sigma*sigma + a_c*sqrt(c_eff/1e6) + +where c_eff = (Ra+Va)*pA + (Rb+Vb)*pB is the effective TVL (real + virtual). + +From daily pool snapshots (volume, TVL, volatility). Outputs a noise_params dict +compatible with run_fingerprint["reclamm_noise_params"]. + +Usage: + # From a pre-assembled CSV + python scripts/calibrate_reclamm_noise.py --csv daily_data.csv --base-fee 0.003 + + # End-to-end from API + DB + parquets + python scripts/calibrate_reclamm_noise.py --pool cbBTC_WETH +""" + +import argparse +import json +import os +import sqlite3 +import sys +import urllib.request +from datetime import datetime, timezone + +import numpy as np +import pandas as pd + + +# --------------------------------------------------------------------------- +# Balancer V3 API +# --------------------------------------------------------------------------- + +BALANCER_API_URL = "https://api-v3.balancer.fi/" + +BALANCER_API_CHAIN = { + "base": "BASE", + "ethereum": "MAINNET", + "gnosis": "GNOSIS", + "avalanche": "AVALANCHE", + "arbitrum": "ARBITRUM", + "polygon": "POLYGON", + "optimism": "OPTIMISM", + "sonic": "SONIC", +} + + +def fetch_balancer_snapshots(chain, pool_address, start_ts, end_ts, + base_url=BALANCER_API_URL): + """Fetch daily pool snapshots from Balancer V3 GraphQL API. + + Parameters + ---------- + chain : str + Chain name (e.g. 'base', 'ethereum'). + pool_address : str + Pool contract address (hex, no 0x prefix). + start_ts : int + Start unix timestamp (seconds). + end_ts : int + End unix timestamp (seconds). + base_url : str + Balancer API base URL. + + Returns + ------- + pd.DataFrame + Columns: date, volume_usd, total_liquidity_usd. Indexed by date string. + """ + api_chain = BALANCER_API_CHAIN.get(chain) + if api_chain is None: + raise ValueError(f"Unknown chain for Balancer API: {chain!r}") + + pool_id = f"0x{pool_address}" if not pool_address.startswith("0x") else pool_address + + # Paginate: API may limit results. Fetch in 90-day windows. + all_snapshots = [] + window = 90 * 86400 + cursor = start_ts + + while cursor < end_ts: + window_end = min(cursor + window, end_ts) + query = { + "query": """ + query GetSnapshots($poolId: String!, $chain: GqlChain!, + $range: GqlPoolSnapshotDataRange!) { + poolGetSnapshots(id: $poolId, chain: $chain, range: $range) { + timestamp + volume24h + totalLiquidity + } + } + """, + "variables": { + "poolId": pool_id, + "chain": api_chain, + "range": "ALL_TIME", + }, + } + + data = json.dumps(query).encode("utf-8") + req = urllib.request.Request( + base_url, + data=data, + headers={ + "Content-Type": "application/json", + "User-Agent": "quantammsim/1.0", + }, + ) + + with urllib.request.urlopen(req, timeout=30) as resp: + body = json.loads(resp.read().decode("utf-8")) + + snapshots = body.get("data", {}).get("poolGetSnapshots", []) + if not snapshots: + break + + for snap in snapshots: + ts = int(snap["timestamp"]) + if start_ts <= ts <= end_ts: + all_snapshots.append({ + "timestamp": ts, + "volume_usd": float(snap["volume24h"]), + "total_liquidity_usd": float(snap["totalLiquidity"]), + }) + + # The API returns ALL_TIME, so no need to paginate further + break + + if not all_snapshots: + raise ValueError( + f"No Balancer snapshots for {pool_id} on {chain} " + f"between {start_ts} and {end_ts}" + ) + + df = pd.DataFrame(all_snapshots) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + # Deduplicate by date (keep last snapshot per day) + df = df.sort_values("timestamp").drop_duplicates("date", keep="last") + return df.set_index("date") + + +# --------------------------------------------------------------------------- +# DB-based daily pool state +# --------------------------------------------------------------------------- + +def load_daily_pool_state(pool, db_path, data_root): + """Load daily pool state from pools_history.db, compute effective TVL. + + Parameters + ---------- + pool : PoolConfig + Pool configuration (from pool_registry). + db_path : str + Path to pools_history.db. + data_root : str + Directory containing {TICKER}_USD.parquet files. + + Returns + ------- + pd.DataFrame + Indexed by date, columns: effective_tvl_usd, real_tvl_usd. + """ + conn = sqlite3.connect(db_path) + cur = conn.cursor() + cur.execute( + f"""SELECT timestamp, balance_0, balance_1, virtual_0, virtual_1 + FROM {pool.db_label} + ORDER BY timestamp""" + ) + rows = cur.fetchall() + conn.close() + + if not rows: + raise ValueError(f"No DB data for {pool.db_label}") + + df = pd.DataFrame(rows, columns=["timestamp", "bal_0", "bal_1", "virt_0", "virt_1"]) + df["date"] = pd.to_datetime(df["timestamp"], unit="s").dt.date + + # Keep last snapshot per day + daily = df.sort_values("timestamp").drop_duplicates("date", keep="last").set_index("date") + + # Load USD prices for each token + if pool.reverse: + tickers_in_db_order = [pool.tokens[1], pool.tokens[0]] + else: + tickers_in_db_order = [pool.tokens[0], pool.tokens[1]] + + price_dfs = {} + for ticker in tickers_in_db_order: + if ticker == "USDC": + price_dfs[ticker] = None # constant $1 + else: + path = os.path.join(data_root, f"{ticker}_USD.parquet") + pdf = pd.read_parquet(path) + pdf["date"] = pd.to_datetime(pdf["unix"], unit="ms").dt.date + # Daily close: last price per day + price_dfs[ticker] = ( + pdf.sort_values("unix") + .drop_duplicates("date", keep="last") + .set_index("date")["close"] + ) + + # Compute USD prices at each daily snapshot + records = [] + for date, row in daily.iterrows(): + b0, b1, v0, v1 = row["bal_0"], row["bal_1"], row["virt_0"], row["virt_1"] + + p0 = 1.0 if tickers_in_db_order[0] == "USDC" else price_dfs[tickers_in_db_order[0]].get(date, np.nan) + p1 = 1.0 if tickers_in_db_order[1] == "USDC" else price_dfs[tickers_in_db_order[1]].get(date, np.nan) + + if np.isnan(p0) or np.isnan(p1): + continue + + real_tvl = b0 * p0 + b1 * p1 + effective_tvl = (b0 + v0) * p0 + (b1 + v1) * p1 + + records.append({ + "date": date, + "real_tvl_usd": real_tvl, + "effective_tvl_usd": effective_tvl, + }) + + result = pd.DataFrame(records).set_index("date") + return result + + +# --------------------------------------------------------------------------- +# Daily volatility from price parquets +# --------------------------------------------------------------------------- + +def compute_daily_volatility(tokens, data_root, start_ts, end_ts): + """Compute daily annualised volatility of the price ratio. + + Uses 5-minute subsampled log returns within each day, then + annualises with sqrt(365). + + Parameters + ---------- + tokens : list + Token tickers in quantammsim sorted order (e.g. ['BTC', 'ETH']). + data_root : str + Directory containing {TICKER}_USD.parquet files. + start_ts : int + Start unix timestamp (seconds). + end_ts : int + End unix timestamp (seconds). + + Returns + ------- + pd.Series + Indexed by date, values are annualised daily volatility. + """ + # Load minute-level prices for both tokens + prices = {} + for ticker in tokens: + if ticker == "USDC": + prices[ticker] = None + else: + path = os.path.join(data_root, f"{ticker}_USD.parquet") + df = pd.read_parquet(path) + df = df[(df["unix"] >= start_ts * 1000) & (df["unix"] <= end_ts * 1000)] + df["datetime"] = pd.to_datetime(df["unix"], unit="ms") + df = df.set_index("datetime")["close"] + prices[ticker] = df + + # Compute price ratio (token[0] / token[1]) + t0, t1 = tokens[0], tokens[1] + if prices[t0] is not None and prices[t1] is not None: + # Align on common timestamps + combined = pd.DataFrame({"p0": prices[t0], "p1": prices[t1]}).dropna() + ratio = combined["p0"] / combined["p1"] + elif prices[t0] is not None: + ratio = prices[t0] # t1 is USDC ($1) + elif prices[t1] is not None: + ratio = 1.0 / prices[t1] # t0 is USDC + else: + raise ValueError("Both tokens are USDC — cannot compute ratio") + + # Subsample to 5-min intervals + ratio_5m = ratio.resample("5min").last().dropna() + log_returns = np.log(ratio_5m / ratio_5m.shift(1)).dropna() + + # Group by date, compute daily vol + log_returns_df = log_returns.to_frame("lr") + log_returns_df["date"] = log_returns_df.index.date + + daily_vol = log_returns_df.groupby("date")["lr"].std() + # Annualise: each day has ~288 5-min periods, scale by sqrt(288 * 365) + daily_vol_ann = daily_vol * np.sqrt(288 * 365) + + return daily_vol_ann + + +# --------------------------------------------------------------------------- +# Calibration DataFrame assembly +# --------------------------------------------------------------------------- + +def build_calibration_df(pool, data_root=None): + """Build daily calibration DataFrame from Balancer API + price parquets. + + All pool state (volume, effective TVL) comes from the Balancer V3 API. + The API's ``totalLiquidity`` is the effective TVL: for a reClAMM pool + on Balancer V3, the router sees real + virtual reserves, and + ``totalLiquidity`` reflects that full depth. Only the volatility + computation requires price parquets. + + Parameters + ---------- + pool : PoolConfig + Pool configuration (must have pool_address field). + data_root : str, optional + Directory containing {TICKER}_USD.parquet price files. + + Returns + ------- + pd.DataFrame + Columns: volume_usd, effective_tvl_usd, volatility. Indexed by date. + """ + from experiments.pool_registry import ( + get_data_end_date, + _date_to_unix, + ) + + if data_root is None: + data_root = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "quantammsim", "data", + ) + + start_ts = _date_to_unix(pool.plausible_start) + end_str = get_data_end_date(pool.tokens, data_root) + end_ts = _date_to_unix(end_str) + + print(f" Fetching Balancer snapshots for {pool.label} " + f"({pool.chain}, {pool.pool_address})...") + api_df = fetch_balancer_snapshots( + pool.chain, pool.pool_address, start_ts, end_ts, + ) + print(f" Got {len(api_df)} daily snapshots from API") + print(f" TVL range: ${api_df['total_liquidity_usd'].min():,.0f} — " + f"${api_df['total_liquidity_usd'].max():,.0f}") + + print(f" Computing daily volatility from price parquets...") + vol_series = compute_daily_volatility(pool.tokens, data_root, start_ts, end_ts) + print(f" Got {len(vol_series)} daily volatility values") + + # Assemble: volume + TVL from API, volatility from parquets + combined = api_df[["volume_usd", "total_liquidity_usd"]].copy() + combined = combined.rename(columns={"total_liquidity_usd": "effective_tvl_usd"}) + combined["volatility"] = vol_series + combined = combined.dropna() + + print(f" Combined: {len(combined)} days after join") + return combined + + +# --------------------------------------------------------------------------- +# OLS calibration +# --------------------------------------------------------------------------- + +def run_ols_calibration(daily_df, base_fee, model="sqrt"): + """OLS regression for Tsoukalas model params. + + Parameters + ---------- + daily_df : pd.DataFrame + Must contain columns: volume_usd, volatility, effective_tvl_usd. + base_fee : float + Static swap fee (e.g. 0.003). + model : str + 'sqrt' or 'log' — TVL regressor transformation. + + Returns + ------- + noise_params : dict + Coefficients for run_fingerprint["reclamm_noise_params"]. + diagnostics : dict + Standard errors, R², residual summary. + """ + if model == "loglinear": + # Multiplicative model: log(V) = b_0 + b_sigma·σ + b_c·log(TVL) + # Implies: V = exp(b_0) · TVL^b_c · exp(b_sigma·σ) + mask = daily_df["volume_usd"].values > 0 + n_dropped = int((~mask).sum()) + df_fit = daily_df[mask] + + y_log = np.log(df_fit["volume_usd"].values) + X = np.column_stack([ + np.ones(len(df_fit)), + df_fit["volatility"].values, + np.log(df_fit["effective_tvl_usd"].values), + ]) + + beta, _, _, _ = np.linalg.lstsq(X, y_log, rcond=None) + b_0, b_sigma, b_c = beta + + residuals = y_log - X @ beta + n, k = X.shape + bread = np.linalg.inv(X.T @ X) + hc1_scale = n / max(n - k, 1) + meat = X.T @ np.diag(residuals**2 * hc1_scale) @ X + robust_cov = bread @ meat @ bread + se = np.sqrt(np.diag(robust_cov)) + + ss_res = np.sum(residuals**2) + ss_tot = np.sum((y_log - y_log.mean())**2) + r_squared = 1.0 - ss_res / max(ss_tot, 1e-30) + + # Pseudo-R² in levels (median predictor) + y_pred_level = np.exp(X @ beta) + y_actual_level = df_fit["volume_usd"].values + res_level = y_actual_level - y_pred_level + r_sq_level = 1.0 - np.sum(res_level**2) / max( + np.sum((y_actual_level - y_actual_level.mean())**2), 1e-30) + + noise_params = { + "b_0": float(b_0), "b_sigma": float(b_sigma), + "b_c": float(b_c), "base_fee": float(base_fee), + } + diagnostics = { + "se": {"b_0": float(se[0]), "b_sigma": float(se[1]), + "b_c": float(se[2])}, + "r_squared": float(r_squared), + "r_squared_level": float(r_sq_level), + "n_obs": int(n), + "n_dropped_zero": n_dropped, + "residual_mean": float(np.mean(residuals)), + "residual_std": float(np.std(residuals)), + "smearing_factor": float(np.exp(np.var(residuals, ddof=1) / 2)), + "model": "loglinear", + } + return noise_params, diagnostics + + # --- Linear models (sqrt / log) --- + y = daily_df["volume_usd"].values / 1e6 + + if model == "sqrt": + tvl_eff = np.sqrt(daily_df["effective_tvl_usd"].values / 1e6) + elif model == "log": + tvl_eff = np.log(np.maximum(daily_df["effective_tvl_usd"].values / 1e6, 1e-30)) + else: + raise ValueError(f"Unknown model: {model!r}. Use 'sqrt', 'log', or 'loglinear'.") + + X = np.column_stack([ + np.ones(len(daily_df)), # a_0 + daily_df["volatility"].values, # a_sigma + tvl_eff, # a_c + ]) + + beta, residuals_ss, rank, sv = np.linalg.lstsq(X, y, rcond=None) + a_0, a_sigma, a_c = beta + + # Heteroskedasticity-robust standard errors (HC1) + residuals = y - X @ beta + n, k = X.shape + bread = np.linalg.inv(X.T @ X) + hc1_scale = n / max(n - k, 1) + meat = X.T @ np.diag(residuals**2 * hc1_scale) @ X + robust_cov = bread @ meat @ bread + se = np.sqrt(np.diag(robust_cov)) + + # R-squared + ss_res = np.sum(residuals**2) + ss_tot = np.sum((y - np.mean(y))**2) + r_squared = 1.0 - ss_res / max(ss_tot, 1e-30) + + noise_params = { + "a_0_base": float(a_0), + "a_f": 0.0, # not identified with static fees + "a_sigma": float(a_sigma), + "a_c": float(a_c), + "base_fee": float(base_fee), + } + + diagnostics = { + "se": dict(zip( + ["a_0", "a_sigma", "a_c"], + se.tolist(), + )), + "r_squared": float(r_squared), + "n_obs": int(n), + "residual_mean": float(np.mean(residuals)), + "residual_std": float(np.std(residuals)), + "model": model, + } + + return noise_params, diagnostics + + +# --------------------------------------------------------------------------- +# Plotting +# --------------------------------------------------------------------------- + +def plot_calibration_diagnostics(daily_df, noise_params, diagnostics, + pool_label="", model="sqrt", + output_dir="results"): + """Generate diagnostic plots for the noise volume calibration. + + Produces a 2×2 figure: + Top-left: Time series — real vs predicted daily volume + effective TVL + Top-right: Scatter — predicted vs actual with 45° line + Bot-left: Residuals vs time + residuals vs fitted + Bot-right: Component decomposition (stacked contributions) + + Parameters + ---------- + daily_df : pd.DataFrame + Calibration DataFrame (indexed by date). + noise_params : dict + Fitted coefficients from run_ols_calibration. + diagnostics : dict + Diagnostics dict from run_ols_calibration. + pool_label : str + Pool name for titles. + model : str + 'sqrt' or 'log'. + output_dir : str + Directory for output PNGs. + """ + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + from matplotlib.dates import DateFormatter + import matplotlib.dates as mdates + + os.makedirs(output_dir, exist_ok=True) + + # Extract data + dates = pd.to_datetime(daily_df.index) + y_actual = daily_df["volume_usd"].values / 1e6 # in $M + eff_tvl = daily_df["effective_tvl_usd"].values + vol = daily_df["volatility"].values + + r2 = diagnostics["r_squared"] + n = diagnostics["n_obs"] + + if model == "loglinear": + b_0 = noise_params["b_0"] + b_sigma = noise_params["b_sigma"] + b_c = noise_params["b_c"] + log_tvl = np.log(np.maximum(eff_tvl, 1.0)) + y_pred_log = b_0 + b_sigma * vol + b_c * log_tvl + y_pred = np.exp(y_pred_log) / 1e6 # median prediction in $M + + # Log-space residuals + mask_pos = daily_df["volume_usd"].values > 0 + residuals = np.full(len(dates), np.nan) + residuals[mask_pos] = ( + np.log(daily_df["volume_usd"].values[mask_pos]) + - y_pred_log[mask_pos] + ) + resid_unit = "log scale" + r2_level = diagnostics.get("r_squared_level") + else: + a_0 = noise_params["a_0_base"] + a_sigma = noise_params["a_sigma"] + a_c = noise_params["a_c"] + + if model == "sqrt": + tvl_term = a_c * np.sqrt(eff_tvl / 1e6) + else: + tvl_term = a_c * np.log(np.maximum(eff_tvl / 1e6, 1e-30)) + + y_pred = a_0 + a_sigma * vol + tvl_term + residuals = y_actual - y_pred + resid_unit = "$M" + r2_level = None + + # --- Figure 1: Main diagnostics (2×2) --- + fig, axes = plt.subplots(2, 2, figsize=(16, 12)) + + # (0,0) Time series: real vs predicted + TVL on secondary axis + ax = axes[0, 0] + ax.plot(dates, y_actual, color="steelblue", alpha=0.7, linewidth=1, + label="Actual volume") + ax.plot(dates, y_pred, color="crimson", linewidth=1.5, + label="Predicted volume") + ax.set_ylabel("Daily volume ($M)", color="steelblue") + ax.tick_params(axis="y", labelcolor="steelblue") + ax.legend(loc="upper left", fontsize=8) + r2_str = (f"R²(log)={r2:.3f}, R²(level)={r2_level:.3f}" + if r2_level is not None else f"R²={r2:.3f}") + ax.set_title(f"Daily volume: actual vs predicted ({r2_str}, n={n})") + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + + ax2 = ax.twinx() + ax2.fill_between(dates, eff_tvl / 1e6, alpha=0.15, color="green", + label="Effective TVL ($M)") + ax2.set_ylabel("Effective TVL ($M)", color="green") + ax2.tick_params(axis="y", labelcolor="green") + ax2.legend(loc="upper right", fontsize=8) + + # (0,1) Scatter: predicted vs actual + ax = axes[0, 1] + ax.scatter(y_pred, y_actual, alpha=0.5, s=15, color="steelblue", + edgecolors="none") + lims = [min(y_pred.min(), y_actual.min()), max(y_pred.max(), y_actual.max())] + margin = (lims[1] - lims[0]) * 0.05 + lims = [lims[0] - margin, lims[1] + margin] + ax.plot(lims, lims, "k--", linewidth=0.8, alpha=0.5, label="45° line") + ax.set_xlabel("Predicted ($M)") + ax.set_ylabel("Actual ($M)") + ax.set_title("Predicted vs actual") + ax.legend(fontsize=8) + ax.set_aspect("equal", adjustable="box") + + # (1,0) Residuals: vs time (top) and vs fitted (bottom) + ax = axes[1, 0] + ax.scatter(dates, residuals, alpha=0.5, s=12, color="steelblue", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + # 7-day rolling mean of residuals + res_series = pd.Series(residuals, index=dates) + rolling_mean = res_series.rolling(7, min_periods=1).mean() + ax.plot(dates, rolling_mean, color="crimson", linewidth=1.5, + label="7-day rolling mean") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs time") + ax.legend(fontsize=8) + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + + # (1,1) Component decomposition + ax = axes[1, 1] + if model == "loglinear": + # Log-space additive decomposition as line plots + log_tvl_plot = np.log(np.maximum(eff_tvl, 1.0)) + comp_base = np.full(len(dates), b_0) + comp_tvl = b_c * log_tvl_plot + comp_vol = b_sigma * vol + total_log = comp_base + comp_tvl + comp_vol + vol_usd = daily_df["volume_usd"].values.copy() + vol_usd[vol_usd <= 0] = np.nan + actual_log = np.log(vol_usd) + ax.plot(dates, comp_base, color="grey", linestyle="--", linewidth=1, + label=f"b_0 = {b_0:.2f}") + ax.plot(dates, comp_base + comp_tvl, color="green", linewidth=1.5, + label=f"b_0 + b_c·log(TVL) (b_c={b_c:.4f})") + ax.plot(dates, total_log, color="crimson", linewidth=1.5, + label="Full prediction") + ax.scatter(dates, actual_log, color="steelblue", s=10, alpha=0.5, + label="Actual log(V)", zorder=5) + ax.set_ylabel("log(Volume, USD)") + ax.set_title("Component decomposition (log space)") + + fig.suptitle( + f"{pool_label} — noise calibration ({model})\n" + f"log(V) = {b_0:.2f} + {b_sigma:.4f}·σ + {b_c:.4f}·log(TVL)", + fontsize=11, + ) + else: + intercept_contrib = np.full(len(dates), a_0) + vol_contrib = a_sigma * vol + tvl_contrib = tvl_term + + ax.fill_between(dates, 0, intercept_contrib, alpha=0.3, color="grey", + label=f"a_0 = {a_0:.4f}") + ax.fill_between(dates, intercept_contrib, intercept_contrib + vol_contrib, + alpha=0.3, color="orange", + label=f"a_σ·σ (a_σ={a_sigma:.4f})") + ax.fill_between(dates, intercept_contrib + vol_contrib, + intercept_contrib + vol_contrib + tvl_contrib, + alpha=0.3, color="green", + label=f"a_c·{model}(TVL) (a_c={a_c:.4f})") + ax.plot(dates, y_actual, color="steelblue", linewidth=1, alpha=0.7, + label="Actual") + ax.set_ylabel("Volume ($M)") + ax.set_title("Component decomposition") + + fig.suptitle( + f"{pool_label} — Tsoukalas noise calibration ({model})\n" + f"V/1e6 = {a_0:.4f} + {a_sigma:.4f}·σ + {a_c:.4f}·{model}(TVL_eff/1e6)", + fontsize=11, + ) + ax.legend(fontsize=7, loc="upper left") + ax.xaxis.set_major_formatter(DateFormatter("%b %y")) + plt.tight_layout() + + fname = f"noise_calibration_{pool_label}_{model}.png" + path = os.path.join(output_dir, fname) + plt.savefig(path, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path}") + + # --- Figure 2: Residuals vs each regressor --- + fig2, axes2 = plt.subplots(1, 3, figsize=(16, 5)) + + # Residuals vs volatility + ax = axes2[0] + ax.scatter(vol, residuals, alpha=0.5, s=12, color="orange", edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Volatility (annualised)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs volatility") + + # Residuals vs effective TVL + ax = axes2[1] + ax.scatter(eff_tvl / 1e6, residuals, alpha=0.5, s=12, color="green", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Effective TVL ($M)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs effective TVL") + + # Residuals vs fitted + ax = axes2[2] + ax.scatter(y_pred, residuals, alpha=0.5, s=12, color="steelblue", + edgecolors="none") + ax.axhline(0, color="black", linewidth=0.8) + ax.set_xlabel("Fitted ($M)") + ax.set_ylabel(f"Residual ({resid_unit})") + ax.set_title("Residuals vs fitted") + + fig2.suptitle(f"{pool_label} — Residual diagnostics ({model})", fontsize=11) + plt.tight_layout() + + fname2 = f"noise_residuals_{pool_label}_{model}.png" + path2 = os.path.join(output_dir, fname2) + plt.savefig(path2, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {path2}") + + return path, path2 + + +# --------------------------------------------------------------------------- +# CLI +# --------------------------------------------------------------------------- + +def main(): + parser = argparse.ArgumentParser( + description="Calibrate Tsoukalas noise volume model for reClAMM" + ) + parser.add_argument( + "--csv", default=None, + help="Path to CSV with columns: volume_usd, volatility, effective_tvl_usd", + ) + parser.add_argument( + "--pool", default=None, + help="Pool label from pool_registry (e.g. cbBTC_WETH) for end-to-end calibration", + ) + parser.add_argument("--base-fee", type=float, default=None, + help="Override base fee (default: use pool's swap_fee)") + parser.add_argument("--model", choices=["sqrt", "log", "loglinear"], + default="sqrt") + parser.add_argument( + "--output", default=None, + help="Output JSON file path. Defaults to stdout.", + ) + parser.add_argument( + "--plot", action="store_true", + help="Generate diagnostic plots (saved to --output-dir)", + ) + parser.add_argument( + "--output-dir", default="results", + help="Directory for diagnostic plots (default: results)", + ) + args = parser.parse_args() + + if args.csv is None and args.pool is None: + parser.error("One of --csv or --pool is required") + + if args.pool is not None: + # End-to-end mode: fetch data, assemble, calibrate + from experiments.pool_registry import POOL_REGISTRY + + if args.pool not in POOL_REGISTRY: + print(f"Unknown pool: {args.pool}", file=sys.stderr) + print(f"Available: {list(POOL_REGISTRY.keys())}", file=sys.stderr) + sys.exit(1) + + pool = POOL_REGISTRY[args.pool] + base_fee = args.base_fee if args.base_fee is not None else pool.swap_fee + + print(f"Calibrating noise model for {pool.label} ({pool.chain})") + print(f" Swap fee: {base_fee}") + print(f" Model: {args.model}") + + df = build_calibration_df(pool) + else: + # CSV mode + df = pd.read_csv(args.csv) + required_cols = {"volume_usd", "volatility", "effective_tvl_usd"} + missing = required_cols - set(df.columns) + if missing: + print(f"Error: missing columns: {missing}", file=sys.stderr) + sys.exit(1) + base_fee = args.base_fee if args.base_fee is not None else 0.003 + + noise_params, diagnostics = run_ols_calibration(df, base_fee, args.model) + + # Print diagnostics + print(f"\n OLS Results ({args.model} model):") + print(f" R² = {diagnostics['r_squared']:.4f}") + if "r_squared_level" in diagnostics: + print(f" R²(level) = {diagnostics['r_squared_level']:.4f}") + if "n_dropped_zero" in diagnostics and diagnostics["n_dropped_zero"] > 0: + print(f" Dropped {diagnostics['n_dropped_zero']} zero-volume days") + if "smearing_factor" in diagnostics: + print(f" Smearing factor = {diagnostics['smearing_factor']:.4f} " + f"(E[V]/median[V])") + print(f" n = {diagnostics['n_obs']}") + print(f" Coefficients:") + if args.model == "loglinear": + coef_keys = ["b_0", "b_sigma", "b_c"] + else: + coef_keys = ["a_0", "a_sigma", "a_c"] + for key in coef_keys: + param_key = "a_0_base" if key == "a_0" else key + val = noise_params[param_key] + se = diagnostics["se"][key] + t_stat = val / se if se > 0 else float("inf") + print(f" {key:>8} = {val:>10.4f} (SE={se:.4f}, t={t_stat:.2f})") + print(f" Residual: mean={diagnostics['residual_mean']:.6f}, " + f"std={diagnostics['residual_std']:.4f}") + + # Plot diagnostics + if args.plot: + label = args.pool if args.pool else "custom" + plot_calibration_diagnostics( + df, noise_params, diagnostics, + pool_label=label, model=args.model, + output_dir=args.output_dir, + ) + + result = { + "noise_params": noise_params, + "diagnostics": diagnostics, + } + + output_str = json.dumps(result, indent=2) + if args.output: + with open(args.output, "w") as f: + f.write(output_str + "\n") + print(f"\nWrote calibration to {args.output}", file=sys.stderr) + else: + print(f"\n{output_str}") + + +if __name__ == "__main__": + main() diff --git a/scripts/compare_reclamm_thermostats.py b/scripts/compare_reclamm_thermostats.py new file mode 100644 index 00000000..8a2c374c --- /dev/null +++ b/scripts/compare_reclamm_thermostats.py @@ -0,0 +1,379 @@ +"""Compare geometric vs constant-arc-length thermostats on historic data. + +Runs AAVE/ETH reClAMM pool simulations with both interpolation methods. +Plots: pool value, cumulative LVR, price path, empirical weights, +value difference, LVR ratio, and per-step LVR distribution (∝ Δs²). + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/compare_reclamm_thermostats.py +""" + +import jax.numpy as jnp +import numpy as np +import matplotlib.pyplot as plt +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(daily_price_shift_exponent): + """Convert shift rate to daily price shift base (matches Solidity).""" + return 1.0 - daily_price_shift_exponent / 124649.0 + + +# Pool configurations to compare +CONFIGS = [ + { + "name": "AAVE/ETH on-chain (25bps, narrow range)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 1.5, + "centeredness_margin": 0.5, + "daily_price_shift_exponent": 0.1, + }, + { + "name": "AAVE/ETH wide range (25bps)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 4.0, + "centeredness_margin": 0.2, + "daily_price_shift_exponent": 1.0, + }, + { + "name": "AAVE/ETH zero fees (narrow)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0, + "price_ratio": 1.5, + "centeredness_margin": 0.5, + "daily_price_shift_exponent": 0.1, + }, +] + + +def make_fingerprint(cfg, interpolation_method, centeredness_scaling=False): + """Build run fingerprint for a given config and interpolation method.""" + return { + "tokens": cfg["tokens"], + "rule": "reclamm", + "startDateString": cfg["start"], + "endDateString": cfg["end"], + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": cfg["fees"], + "gas_cost": 0.0, + "arb_fees": 0.0, + "reclamm_interpolation_method": interpolation_method, + "reclamm_arc_length_speed": None, # auto-calibrate + "reclamm_centeredness_scaling": centeredness_scaling, + } + + +def make_params(cfg): + """Build pool params from config.""" + return { + "price_ratio": jnp.array(cfg["price_ratio"]), + "centeredness_margin": jnp.array(cfg["centeredness_margin"]), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(cfg["daily_price_shift_exponent"]) + ), + } + + +def run_comparison(cfg): + """Run all thermostat variants, return results dict.""" + params = make_params(cfg) + + results = {} + for method in ["geometric", "constant_arc_length"]: + fp = make_fingerprint(cfg, method) + results[method] = do_run_on_historic_data( + run_fingerprint=fp, params=params + ) + + # Geometric + centeredness-proportional scaling (scales decay duration) + fp_geo_scaled = make_fingerprint(cfg, "geometric", centeredness_scaling=True) + results["geometric_scaled"] = do_run_on_historic_data( + run_fingerprint=fp_geo_scaled, params=params + ) + + # Arc-length + centeredness-proportional scaling (scales speed) + fp_cal_scaled = make_fingerprint(cfg, "constant_arc_length", centeredness_scaling=True) + results["cal_scaled"] = do_run_on_historic_data( + run_fingerprint=fp_cal_scaled, params=params + ) + + return results + + +def print_comparison(cfg, results): + """Print text summary table.""" + methods = [ + ("Geometric", results["geometric"]), + ("Geo+Scaled", results["geometric_scaled"]), + ("Const Arc", results["constant_arc_length"]), + ("Arc+Scaled", results["cal_scaled"]), + ] + + hodl_value = float((methods[0][1]["reserves"][0] * methods[0][1]["prices"][-1]).sum()) + + print("=" * 105) + print(f" {cfg['name']}") + print(f" price_ratio={cfg['price_ratio']}, " + f"margin={cfg['centeredness_margin']}, " + f"shift_exp={cfg['daily_price_shift_exponent']}, " + f"fees={cfg['fees']}") + print("-" * 105) + header = " {:20s}".format("") + for name, _ in methods: + header += f" {name:>14s}" + print(header) + + row = " {:20s}".format("Final value") + for _, r in methods: + row += f" ${float(r['final_value']):>13,.0f}" + print(row) + + print(f" {'HODL value':20s} ${hodl_value:>13,.0f}") + + row = " {:20s}".format("LVR (HODL - final)") + for _, r in methods: + lvr = hodl_value - float(r["final_value"]) + row += f" ${lvr:>13,.0f}" + print(row) + + row = " {:20s}".format("Return") + for _, r in methods: + ret = (float(r["final_value"]) / float(r["value"][0]) - 1) * 100 + row += f" {ret:>13.2f}%" + print(row) + + row = " {:20s}".format("vs HODL") + for _, r in methods: + vs = (float(r["final_value"]) / hodl_value - 1) * 100 + row += f" {vs:>13.2f}%" + print(row) + print("=" * 105) + + +def plot_comparison(cfg, results, fig_idx): + """Plot 4-panel comparison for one config.""" + # Method name → (result dict, color, linestyle) + variants = { + "Geometric": (results["geometric"], "C0", "-"), + "Geo+Scaled": (results["geometric_scaled"], "C1", "-"), + "Const arc-len": (results["constant_arc_length"], "C2", "--"), + "Arc+Scaled": (results["cal_scaled"], "C3", "--"), + } + + geo = results["geometric"] + geo_prices = np.array(geo["prices"]) + geo_reserves = np.array(geo["reserves"]) + n_steps = len(np.array(geo["value"])) + t_days = np.arange(n_steps) / (60 * 24) + + hodl_traj = (geo_reserves[0] * geo_prices[:n_steps]).sum(axis=-1) + price_ratio_traj = geo_prices[:n_steps, 0] / geo_prices[:n_steps, 1] + + fig, axes = plt.subplots(2, 2, figsize=(14, 10)) + fig.suptitle(cfg["name"], fontsize=13, fontweight="bold") + + # (0,0) Pool value over time + ax = axes[0, 0] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + ax.plot(t_days, vals / 1e6, color=color, ls=ls, label=name, alpha=0.9) + ax.plot(t_days, np.array(hodl_traj) / 1e6, color="gray", ls=":", + alpha=0.5, label="HODL") + ax.set_xlabel("Days") + ax.set_ylabel("Pool value ($M)") + ax.set_title("Pool value") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (0,1) Cumulative LVR + ax = axes[0, 1] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + lvr = np.array(hodl_traj) - vals + ax.plot(t_days, lvr / 1e3, color=color, ls=ls, label=name, alpha=0.9) + ax.set_xlabel("Days") + ax.set_ylabel("Cumulative LVR ($K)") + ax.set_title("Cumulative LVR (HODL - pool value)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (1,0) Price ratio + ax = axes[1, 0] + ax.plot(t_days, price_ratio_traj, color="C4", alpha=0.7) + ax.set_xlabel("Days") + ax.set_ylabel(f"{cfg['tokens'][0]}/{cfg['tokens'][1]} price ratio") + ax.set_title("Price path") + ax.grid(True, alpha=0.3) + + # (1,1) Empirical weights + ax = axes[1, 1] + for name, (r, color, ls) in variants.items(): + w = np.array(r["weights"]) + n_w = min(len(w), n_steps) + t_w = np.arange(n_w) / (60 * 24) + ax.plot(t_w, w[:n_w, 0], color=color, ls=ls, label=name, alpha=0.9) + ax.set_xlabel("Days") + ax.set_ylabel(f"Weight ({cfg['tokens'][0]})") + ax.set_title("Empirical weight (token 0)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + fname = f"reclamm_thermostat_comparison_{fig_idx}.png" + plt.savefig(fname, dpi=150) + print(f"Saved {fname}") + plt.close(fig) + + # Second figure: diagnostics + geo_values = np.array(geo["value"]) + geo_lvr = np.array(hodl_traj) - geo_values + + fig2, axes2 = plt.subplots(1, 3, figsize=(18, 5)) + fig2.suptitle(f"{cfg['name']} — diagnostics", fontsize=13, fontweight="bold") + + # (left) Value difference vs geometric + ax = axes2[0] + for name, (r, color, ls) in variants.items(): + if name == "Geometric": + continue + vals = np.array(r["value"]) + ax.plot(t_days, (vals - geo_values) / 1e3, color=color, ls=ls, + label=name, alpha=0.9) + ax.axhline(0, color="gray", ls="--", alpha=0.5) + ax.set_xlabel("Days") + ax.set_ylabel("Value difference ($K)") + ax.set_title("Minus Geometric") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # (middle) LVR ratio over time + ax = axes2[1] + mask = np.abs(geo_lvr) > 100 + if mask.any(): + for name, (r, color, ls) in variants.items(): + if name == "Geometric": + continue + vals = np.array(r["value"]) + method_lvr = np.array(hodl_traj) - vals + ratio = np.full_like(geo_lvr, np.nan) + ratio[mask] = method_lvr[mask] / geo_lvr[mask] + ax.plot(t_days, ratio, color=color, ls=ls, alpha=0.7, label=name) + ax.axhline(1.0, color="gray", ls="--", alpha=0.5) + ax.set_ylabel("LVR ratio (method / geometric)") + ax.legend(fontsize=8) + else: + ax.text(0.5, 0.5, "LVR too small to compare", + transform=ax.transAxes, ha="center", va="center") + ax.set_xlabel("Days") + ax.set_title("Relative LVR") + ax.grid(True, alpha=0.3) + + # (right) Per-step LVR histogram + ax = axes2[2] + all_pos = [] + for name, (r, color, ls) in variants.items(): + vals = np.array(r["value"]) + method_lvr = np.array(hodl_traj) - vals + step_lvr = np.diff(method_lvr) + pos = step_lvr[step_lvr > 0] + all_pos.append((name, pos, color)) + has_data = [len(p) > 10 for _, p, _ in all_pos] + if any(has_data): + max_val = max(np.percentile(p, 99) for _, p, _ in all_pos if len(p) > 10) + bins = np.linspace(0, max_val, 50) + for name, pos, color in all_pos: + if len(pos) > 10: + ax.hist(pos, bins=bins, color=color, alpha=0.3, label=name, + density=True) + ax.set_xlabel("Per-step LVR ($)") + ax.set_ylabel("Density") + ax.legend(fontsize=8) + else: + ax.text(0.5, 0.5, "Too few thermostat steps", + transform=ax.transAxes, ha="center", va="center") + ax.set_title("Per-step LVR distribution") + ax.grid(True, alpha=0.3) + + plt.tight_layout() + fname2 = f"reclamm_thermostat_diff_{fig_idx}.png" + plt.savefig(fname2, dpi=150) + print(f"Saved {fname2}") + plt.close(fig2) + + +if __name__ == "__main__": + all_results = [] + for i, cfg in enumerate(CONFIGS): + print(f"\n>>> Running {cfg['name']}...") + try: + results = run_comparison(cfg) + print_comparison(cfg, results) + plot_comparison(cfg, results, i) + all_results.append((cfg, results)) + except Exception as e: + print(f" FAILED: {e}") + import traceback + traceback.print_exc() + + # Summary overlay: all configs on one figure (pool value normalised) + if len(all_results) > 1: + fig, axes = plt.subplots(1, 2, figsize=(16, 5)) + fig.suptitle("Cross-config comparison (normalised)", fontsize=13, + fontweight="bold") + + method_keys = [ + ("geometric", "geo", "-"), + ("geometric_scaled", "geo+s", "-."), + ("constant_arc_length", "arc", "--"), + ("cal_scaled", "arc+s", ":"), + ] + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + + for j, (key, suffix, ls) in enumerate(method_keys): + v = np.array(results[key]["value"]) + color_idx = i * len(method_keys) + j + + # (left) Normalised pool value + axes[0].plot(t, v / v[0], ls=ls, alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}") + + # (right) Value difference vs geometric (skip geo itself) + if key != "geometric": + pct_diff = (v - geo_v) / geo_v * 100 + axes[1].plot(t, pct_diff, ls=ls, alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}") + + axes[0].set_xlabel("Days") + axes[0].set_ylabel("Normalised pool value") + axes[0].set_title("Pool value (V/V0)") + axes[0].legend(fontsize=6, ncol=2) + axes[0].grid(True, alpha=0.3) + + axes[1].set_xlabel("Days") + axes[1].set_ylabel("(Method - Geo) / Geo (%)") + axes[1].set_title("Relative value difference vs Geometric") + axes[1].axhline(0, color="gray", ls="--", alpha=0.5) + axes[1].legend(fontsize=6, ncol=2) + axes[1].grid(True, alpha=0.3) + + plt.tight_layout() + plt.savefig("reclamm_thermostat_summary.png", dpi=150) + print("\nSaved reclamm_thermostat_summary.png") + plt.close(fig) diff --git a/scripts/demo_run_reclamm.py b/scripts/demo_run_reclamm.py new file mode 100644 index 00000000..3ea21ec6 --- /dev/null +++ b/scripts/demo_run_reclamm.py @@ -0,0 +1,207 @@ +"""Demo runs for reClAMM pools vs Balancer 50/50 baseline. + +Runs reClAMM pool simulations with parameters pulled from on-chain pools +(AAVE/ETH) and hypothetical configurations, each paired with a Balancer +50/50 constant-weight pool at the same fee level for comparison. + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm + python scripts/demo_run_reclamm.py +""" + +import jax.numpy as jnp +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +def to_daily_price_shift_base(daily_price_shift_exponent): + """Convert shift rate to daily price shift base (matches Solidity).""" + return 1.0 - daily_price_shift_exponent / 124649.0 + + +def balancer_fingerprint(tokens, start, end, fees): + """Build a Balancer 50/50 fingerprint matching the given reclamm config.""" + return { + "tokens": tokens, + "rule": "balancer", + "startDateString": start, + "endDateString": end, + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": fees, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + } + + +SCENARIOS = [ + { + "name": "AAVE/ETH on-chain (25bps)", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0025, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.1) + ), + }, + }, + }, + { + "name": "AAVE/ETH zero fees", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(1.5), + "centeredness_margin": jnp.array(0.5), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.1) + ), + }, + }, + }, + { + "name": "AAVE/ETH wide range (25bps)", + "reclamm": { + "fingerprint": { + "tokens": ["AAVE", "ETH"], + "rule": "reclamm", + "startDateString": "2024-06-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.0025, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(4.0), + "centeredness_margin": jnp.array(0.2), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(1.0) + ), + }, + }, + }, + { + "name": "BTC/ETH (10bps)", + "reclamm": { + "fingerprint": { + "tokens": ["BTC", "ETH"], + "rule": "reclamm", + "startDateString": "2024-01-01 00:00:00", + "endDateString": "2025-06-01 00:00:00", + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": 0.001, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + }, + "params": { + "price_ratio": jnp.array(2.0), + "centeredness_margin": jnp.array(0.3), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(0.5) + ), + }, + }, + }, +] + + +def run_scenario(scenario): + """Run a reClAMM config and its Balancer 50/50 baseline, print comparison.""" + rc = scenario["reclamm"] + fp = rc["fingerprint"] + + # Run reClAMM + reclamm_result = do_run_on_historic_data( + run_fingerprint=fp, params=rc["params"] + ) + + # Run Balancer 50/50 with same tokens, dates, fees + bal_fp = balancer_fingerprint( + fp["tokens"], fp["startDateString"], fp["endDateString"], fp["fees"] + ) + bal_params = { + "initial_weights_logits": jnp.zeros(len(fp["tokens"])), + } + balancer_result = do_run_on_historic_data( + run_fingerprint=bal_fp, params=bal_params + ) + + # HODL value (from reClAMM initial reserves at final prices) + hodl_value = float( + (reclamm_result["reserves"][0] * reclamm_result["prices"][-1]).sum() + ) + + rc_final = float(reclamm_result["final_value"]) + bal_final = float(balancer_result["final_value"]) + rc_init = float(reclamm_result["value"][0]) + bal_init = float(balancer_result["value"][0]) + + print("=" * 80) + print(f" {scenario['name']}") + print(f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']}") + print("-" * 80) + print(f" {'':30s} {'reClAMM':>14s} {'Balancer 50/50':>14s}") + print(f" {'Initial value':30s} ${rc_init:>13,.0f} ${bal_init:>13,.0f}") + print(f" {'Final value':30s} ${rc_final:>13,.0f} ${bal_final:>13,.0f}") + print( + f" {'Return':30s} " + f"{(rc_final / rc_init - 1) * 100:>13.2f}% " + f"{(bal_final / bal_init - 1) * 100:>13.2f}%" + ) + print( + f" {'vs HODL':30s} " + f"{(rc_final / hodl_value - 1) * 100:>13.2f}% " + f"{(bal_final / hodl_value - 1) * 100:>13.2f}%" + ) + print( + f" {'reClAMM vs Balancer':30s} " + f"{(rc_final / bal_final - 1) * 100:>13.2f}%" + ) + print("=" * 80) + + +if __name__ == "__main__": + for scenario in SCENARIOS: + print(f"\n>>> {scenario['name']}...") + try: + run_scenario(scenario) + except Exception as e: + print(f" FAILED: {e}") + import traceback + + traceback.print_exc() diff --git a/scripts/fetch_token_mcaps.py b/scripts/fetch_token_mcaps.py new file mode 100644 index 00000000..8e71d6f2 --- /dev/null +++ b/scripts/fetch_token_mcaps.py @@ -0,0 +1,196 @@ +"""Fetch token market caps from CoinGecko and cache locally. + +Usage: + python scripts/fetch_token_mcaps.py + +Output: + local_data/noise_calibration/token_mcaps.json +""" + +import json +import os +import sys +import time + +import requests + +OUTPUT_PATH = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "token_mcaps.json", +) + +# Token symbol -> CoinGecko ID mapping +# Covers all tokens appearing in our 26 matched pools + common Balancer tokens +COINGECKO_IDS = { + # Blue-chip / wrapped natives + "WETH": "ethereum", + "ETH": "ethereum", + "WBTC": "wrapped-bitcoin", + "BTC": "bitcoin", + "cbBTC": "bitcoin", # Coinbase wrapped BTC — use BTC mcap + "USDC": "usd-coin", + "USDT": "tether", + "DAI": "dai", + "wstETH": "wrapped-steth", + "stETH": "staked-ether", + "rETH": "rocket-pool-eth", + "cbETH": "coinbase-wrapped-staked-eth", + "WMATIC": "polygon-ecosystem-token", + "MATIC": "polygon-ecosystem-token", + "POL": "polygon-ecosystem-token", + "WAVAX": "avalanche-2", + "AVAX": "avalanche-2", + "GNO": "gnosis", + "WXDAI": "dai", # Wrapped xDAI ≈ DAI + "xDAI": "dai", + "S": "sonic-3", + "wS": "sonic-3", + # Mid-cap DeFi + "AAVE": "aave", + "LINK": "chainlink", + "UNI": "uniswap", + "BAL": "balancer", + "MKR": "maker", + "CRV": "curve-dao-token", + "COMP": "compound-governance-token", + "SNX": "havven", + "LDO": "lido-dao", + "RPL": "rocket-pool", + "SUSHI": "sushi", + "YFI": "yearn-finance", + "1INCH": "1inch", + "ENS": "ethereum-name-service", + "ARB": "arbitrum", + "OP": "optimism", + "PENDLE": "pendle", + "ENA": "ethena", + "EIGEN": "eigenlayer", + "COW": "cow-protocol", + "SAFE": "safe", + # Smaller / specific tokens in our pools + "ACX": "across-protocol", + "ALCX": "alchemix", + "QI": "benqi", + "QNT": "quant-network", + "RDNT": "radiant-capital", + # TREE not on CoinGecko — handled as fallback below + "XAI": "xai-blockchain", + # Wrapped aTokens — use underlying + "waEthLidoWETH": "ethereum", + "waEthLidowstETH": "wrapped-steth", + "waBasWETH": "ethereum", + "waBasUSDC": "usd-coin", + "waEthUSDC": "usd-coin", + "waGnoGNO": "gnosis", + "waGnowstETH": "wrapped-steth", + # Additional tokens from expanded pool set + "wPOL": "polygon-ecosystem-token", + "stS": "sonic-3", # Staked Sonic — use S mcap + "JitoSOL": "jito-governance-token", + "scUSD": "usd-coin", # Rings scUSD stablecoin — use USDC mcap as proxy + "DOLA": "dola-usd", +} + +# Asset type classification +STABLECOINS = { + "USDC", "USDT", "DAI", "WXDAI", "xDAI", "GHO", "LUSD", "crvUSD", + "FRAX", "sDAI", "scUSD", "DOLA", + "waBasUSDC", "waEthUSDC", +} +NATIVE_LST = { + "WETH", "ETH", "wstETH", "stETH", "rETH", "cbETH", + "WBTC", "BTC", "cbBTC", + "WMATIC", "MATIC", "POL", "wPOL", + "WAVAX", "AVAX", + "GNO", "S", "wS", "stS", + "JitoSOL", + "waEthLidoWETH", "waEthLidowstETH", + "waBasWETH", "waGnoGNO", "waGnowstETH", +} +# Everything else is VOLATILE (asset_type=2) + + +def fetch_mcaps(): + """Fetch market caps from CoinGecko in batches.""" + unique_ids = sorted(set(COINGECKO_IDS.values())) + print(f"Fetching market caps for {len(unique_ids)} unique CoinGecko IDs...") + + # CoinGecko allows up to 250 IDs per request + batch_size = 100 + all_data = {} + + for i in range(0, len(unique_ids), batch_size): + batch = unique_ids[i:i + batch_size] + ids_str = ",".join(batch) + url = ( + f"https://api.coingecko.com/api/v3/simple/price" + f"?ids={ids_str}&vs_currencies=usd&include_market_cap=true" + ) + resp = requests.get(url, timeout=30) + resp.raise_for_status() + data = resp.json() + all_data.update(data) + print(f" Batch {i // batch_size + 1}: {len(data)} tokens") + if i + batch_size < len(unique_ids): + time.sleep(1) # rate limit + + # Build symbol -> mcap mapping + mcaps = {} + missing = [] + for symbol, gecko_id in COINGECKO_IDS.items(): + if gecko_id in all_data and "usd_market_cap" in all_data[gecko_id]: + mcaps[symbol] = { + "mcap_usd": all_data[gecko_id]["usd_market_cap"], + "price_usd": all_data[gecko_id]["usd"], + "coingecko_id": gecko_id, + } + else: + missing.append((symbol, gecko_id)) + + if missing: + print(f"\n Missing from CoinGecko: {missing}") + + # Fallback for tokens not on CoinGecko (very small tokens) + FALLBACK_MCAPS = { + "TREE": 1_000_000, # ~$1M estimate for small governance token + } + for symbol, mcap_est in FALLBACK_MCAPS.items(): + if symbol not in mcaps: + mcaps[symbol] = { + "mcap_usd": mcap_est, + "price_usd": 0.0, + "coingecko_id": "fallback", + } + print(f" Fallback: {symbol} -> ${mcap_est:,.0f}") + + # Add asset type classification + for symbol in mcaps: + if symbol in STABLECOINS: + mcaps[symbol]["asset_type"] = "stable" + elif symbol in NATIVE_LST: + mcaps[symbol]["asset_type"] = "native_lst" + else: + mcaps[symbol]["asset_type"] = "volatile" + + return mcaps + + +def main(): + mcaps = fetch_mcaps() + + os.makedirs(os.path.dirname(OUTPUT_PATH), exist_ok=True) + with open(OUTPUT_PATH, "w") as f: + json.dump(mcaps, f, indent=2) + + print(f"\nSaved {len(mcaps)} tokens to {OUTPUT_PATH}") + + # Summary + print("\nSample entries:") + for sym in ["WETH", "AAVE", "USDC", "QI", "TREE"]: + if sym in mcaps: + m = mcaps[sym] + print(f" {sym}: ${m['mcap_usd']:,.0f} ({m['asset_type']})") + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_predicted_vs_real_volume.py b/scripts/plot_predicted_vs_real_volume.py new file mode 100644 index 00000000..83232ab1 --- /dev/null +++ b/scripts/plot_predicted_vs_real_volume.py @@ -0,0 +1,151 @@ +"""Plot predicted vs real daily volume for pool registry pools. + +Uses the fitted noise model (from calibrate_noise_unified.py) to compute +predicted daily log-volume for each pool in the registry, and overlays +the actual observed volume from the Balancer API panel data. +""" + +import json +import os +import sys + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "experiments")) +from pool_registry import POOL_REGISTRY, BALANCER_API_CHAIN + + +def main(): + fitted_path = "results/unified_full_90d.json" + panel_path = "local_data/noise_calibration/panel.parquet" + output_dir = "results/unified_full_90d" + os.makedirs(output_dir, exist_ok=True) + + with open(fitted_path) as f: + fitted = json.load(f) + + panel = pd.read_parquet(panel_path) + + # Deduplicate registry: multiple entries can share the same pool address + # (e.g. cbBTC_WETH and cbBTC_WETH_post_oct). Group by address. + unique_pools = {} + for label, pool in POOL_REGISTRY.items(): + addr = pool.pool_address.lower() + if addr not in unique_pools: + unique_pools[addr] = (label, pool) + + # Match to panel + matched = [] + for addr, (label, pool) in unique_pools.items(): + pid_matches = [ + pid for pid in fitted["pools"] + if addr in pid.lower() + ] + if pid_matches: + pid = pid_matches[0] + matched.append((label, pool, pid)) + else: + print(f" {label}: not in fitted model (skipping)") + + if not matched: + print("No registry pools found in the fitted model.") + return + + print(f"Plotting {len(matched)} pools: {[m[0] for m in matched]}") + + # Determine grid layout + n = len(matched) + ncols = min(n, 2) + nrows = (n + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(7 * ncols, 5 * nrows), + squeeze=False) + + for idx, (label, pool, pid) in enumerate(matched): + ax = axes[idx // ncols][idx % ncols] + pool_data = fitted["pools"][pid] + theta = np.array(pool_data["theta_median"]) + # theta = [intercept, b_tvl, b_sigma, b_weekend] + + # Get panel data for this pool + pool_panel = panel[panel["pool_id"] == pid].copy() + pool_panel = pool_panel.sort_values("date") + + if len(pool_panel) == 0: + ax.set_title(f"{label}: no panel data") + continue + + # Filter to last 90 days (matching training window) + max_date = panel["date"].max() + if hasattr(max_date, "date"): + max_date = max_date + from datetime import date, timedelta + if isinstance(max_date, date): + cutoff = max_date - timedelta(days=90) + else: + cutoff = pd.Timestamp(max_date) - pd.Timedelta(days=90) + pool_panel = pool_panel[ + pool_panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if len(pool_panel) < 5: + ax.set_title(f"{label}: <5 obs in 90d window") + continue + + # Build x_obs: [1, log_tvl_lag1, volatility, weekend] + x_obs = np.column_stack([ + np.ones(len(pool_panel)), + pool_panel["log_tvl_lag1"].values, + pool_panel["volatility"].values, + pool_panel["weekend"].values, + ]) + + predicted_log_vol = x_obs @ theta + actual_log_vol = pool_panel["log_volume"].values + + # Convert to USD volume for interpretability + predicted_vol = np.exp(predicted_log_vol) + actual_vol = np.exp(actual_log_vol) + + dates = pd.to_datetime(pool_panel["date"].values) + + # Plot + ax.plot(dates, actual_vol, "o-", color="steelblue", markersize=3, + linewidth=1, alpha=0.7, label="Actual") + ax.plot(dates, predicted_vol, "s--", color="orangered", markersize=3, + linewidth=1, alpha=0.7, label="Predicted") + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)") + ax.set_title(f"{label} ({pool_data['chain']})\n" + f"b_c={theta[1]:.2f} b_σ={theta[2]:.2f} " + f"b_wknd={theta[3]:.2f}") + ax.legend(fontsize=8) + ax.tick_params(axis="x", rotation=30) + + # Annotate R² for this pool + ss_res = np.sum((actual_log_vol - predicted_log_vol) ** 2) + ss_tot = np.sum((actual_log_vol - actual_log_vol.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + ax.text(0.02, 0.95, f"R²={r2:.3f}\nn={len(pool_panel)}", + transform=ax.transAxes, fontsize=8, va="top", + bbox=dict(boxstyle="round,pad=0.3", fc="white", alpha=0.8)) + + # Hide unused axes + for idx in range(n, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle("Noise model: predicted vs actual daily volume\n" + "(registry pools, 90-day training window)", fontsize=13) + fig.tight_layout() + out_path = os.path.join(output_dir, "registry_predicted_vs_real.png") + fig.savefig(out_path, dpi=150, bbox_inches="tight") + print(f"Saved: {out_path}") + plt.close(fig) + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_reclamm_optuna_result.py b/scripts/plot_reclamm_optuna_result.py new file mode 100644 index 00000000..35242dac --- /dev/null +++ b/scripts/plot_reclamm_optuna_result.py @@ -0,0 +1,451 @@ +#!/usr/bin/env python3 +"""Plot reClAMM pool performance from Optuna tuning results. + +Reads the SGD-compatible JSON output of tune_reclamm_params.py (or any Optuna +run), extracts the best trial's pool params, re-runs a forward pass over the +full train+test window, and produces a value-over-time plot with on-chain +baselines and cumulative fee revenue. + +Usage: + python scripts/plot_reclamm_optuna_result.py results/run_.json + python scripts/plot_reclamm_optuna_result.py results/run_.json --output my_plot.png + python scripts/plot_reclamm_optuna_result.py results/run_.json --top-k 3 +""" + +import argparse +import json +import sys + +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from datetime import datetime + +from quantammsim.runners.jax_runners import do_run_on_historic_data + +# ── On-chain baselines ──────────────────────────────────────────────────── +ONCHAIN_LAUNCH_PARAMS = { + "price_ratio": 1.5, "centeredness_margin": 0.5, "shift_exponent": 0.1, +} +ONCHAIN_CURRENT_PARAMS = { + "price_ratio": 4.0, "centeredness_margin": 0.1, "shift_exponent": 0.001, +} + +BG = "#162536" +TEXT_COLOR = "#E6CE97" +COLORS = [ + "#3498db", "#2ecc71", "#e74c3c", # top-k + "#f39c12", # on-chain launch + "#9b59b6", # on-chain current +] + + +def _plot_order(configs): + """Yield (name, meta, color_idx) with baselines first, optimized trials last.""" + optimized = [] + baselines = [] + for i, (name, meta) in enumerate(configs.items()): + if "On-Chain" in name: + baselines.append((name, meta, i)) + else: + optimized.append((name, meta, i)) + return baselines + optimized + + +def parse_args(): + p = argparse.ArgumentParser( + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter, + ) + p.add_argument("results_json", help="Path to run_.json from Optuna") + p.add_argument("--top-k", type=int, default=1, + help="Plot top K trials by objective (default 1)") + p.add_argument("--output", default=None, + help="Output PNG path (default: auto-generated)") + p.add_argument("--no-onchain", action="store_true", + help="Skip on-chain baseline runs") + p.add_argument("--end-test-date", default=None, + help="Override endTestDateString (e.g. '2026-02-15 00:00:00')") + p.add_argument("--noise-trader-ratio", type=float, default=None, + help="Override noise_trader_ratio from results config") + return p.parse_args() + + +def load_results(path): + """Load the double-encoded JSONL from Optuna results.""" + with open(path) as f: + raw = f.read() + data = json.loads(raw) + if isinstance(data, str): + data = json.loads(data) + if not isinstance(data, list) or len(data) < 2: + print(f"ERROR: Expected [config, trial1, trial2, ...], got {type(data)}") + sys.exit(1) + config = data[0] + trials = data[1:] + return config, trials + + +def extract_pool_params(trial, config): + """Extract reClAMM pool params from a trial entry.""" + param_keys = ["price_ratio", "centeredness_margin", "shift_exponent", + "arc_length_speed", "fees"] + params = {} + for k in param_keys: + if k in trial: + params[k] = trial[k] + return params + + +def run_full_period(params, config, fees_override=None): + """Run forward pass over the full train+test window.""" + fees = fees_override if fees_override is not None else config["fees"] + fp = { + "rule": "reclamm", + "tokens": config["tokens"], + "startDateString": config["startDateString"], + "endDateString": config["endTestDateString"], # full period + "initial_pool_value": config["initial_pool_value"], + "do_arb": config["do_arb"], + "fees": fees, + "gas_cost": config.get("gas_cost", 1.0), + "arb_fees": config.get("arb_fees", 0.0), + "protocol_fee_split": config.get("protocol_fee_split", 0.0), + "noise_trader_ratio": config.get("noise_trader_ratio", 0.0), + "reclamm_use_shift_exponent": config.get("reclamm_use_shift_exponent", True), + "reclamm_interpolation_method": config.get("reclamm_interpolation_method", "geometric"), + "reclamm_centeredness_scaling": config.get("reclamm_centeredness_scaling", False), + "reclamm_learn_arc_length_speed": config.get("reclamm_learn_arc_length_speed", False), + } + jax_params = {k: jnp.array(v) for k, v in params.items()} + return do_run_on_historic_data(run_fingerprint=fp, params=jax_params) + + +def plot_results(configs, time_series, hodl_values, config, args): + """Two-panel plot: value-over-time + cumulative fee revenue.""" + train_end_str = config["endDateString"] + train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range( + start=datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S"), + periods=n_minutes, freq="1min", + ) + step = 1440 + dates_daily = dates[::step] + + has_fee_revenue = any( + "fee_revenue" in time_series[n] and time_series[n]["fee_revenue"] is not None + for n in time_series + ) + n_panels = 2 if has_fee_revenue else 1 + fig, axes = plt.subplots( + n_panels, 1, figsize=(14, 5 * n_panels), + sharex=True, gridspec_kw={"height_ratios": [3, 1] if n_panels == 2 else [1]}, + ) + if n_panels == 1: + axes = [axes] + ax_val = axes[0] + + # ── Panel 1: Value over time ────────────────────────────────────── + for name, meta, ci in _plot_order(configs): + out = time_series[name] + vals = np.array(out["value"][::step]) / 1e6 + label = f"{name}" + if "test_objective" in meta: + obj_name = config.get("return_val", "objective") + label += f" (OOS {obj_name}={meta['test_objective']:.4f})" + is_optimized = "On-Chain" not in name + ax_val.plot(dates_daily[:len(vals)], vals, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=label, + zorder=3 if is_optimized else 2) + + hodl_daily = hodl_values[::step] / 1e6 + ax_val.plot(dates_daily[:len(hodl_daily)], hodl_daily, linewidth=2, + color="white", alpha=0.7, linestyle="--", label="HODL") + + ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax_val.get_ylim() + ax_val.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax_val.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") + + _style_axis(ax_val) + ax_val.set_ylabel("Pool Value ($M USD)", color=TEXT_COLOR, fontsize=12) + tokens_str = "/".join(config["tokens"]) + obj_name = config.get("return_val", "objective") + ntr = config.get("noise_trader_ratio", 0.0) + ax_val.set_title( + f"reClAMM Optuna-Optimized ({obj_name}, noise={ntr}) — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15, + ) + ax_val.legend(loc="upper left", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + # ── Panel 2: Cumulative fee revenue ─────────────────────────────── + if has_fee_revenue: + ax_fee = axes[1] + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + fr = out.get("fee_revenue") + if fr is None: + continue + fr = np.array(fr) + cumfee = np.cumsum(fr)[::step] / 1e3 + is_optimized = "On-Chain" not in name + ax_fee.plot(dates_daily[:len(cumfee)], cumfee, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=name, + zorder=3 if is_optimized else 2) + + ax_fee.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + _style_axis(ax_fee) + ax_fee.set_ylabel("Cumulative Fee Revenue ($K)", color=TEXT_COLOR, fontsize=12) + ax_fee.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax_fee.legend(loc="upper left", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + else: + ax_val.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + + output = args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png" + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"\nSaved plot to {output}") + plt.close() + + +def plot_test_only(configs, time_series, hodl_values, config, args): + """Test-period plot with all curves normalised to start at 1.0.""" + train_end_str = config["endDateString"] + train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") + start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") + + # Find the index of the train/test boundary + train_minutes = int((train_end_dt - start_dt).total_seconds() / 60) + test_start_idx = min(train_minutes, n_minutes - 1) + + step = 1440 + test_dates = dates[test_start_idx::step] + + fig, ax = plt.subplots(1, 1, figsize=(14, 6)) + + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + vals = np.array(out["value"]) + test_vals = vals[test_start_idx::step] + if len(test_vals) == 0: + continue + normalised = test_vals / test_vals[0] + is_optimized = "On-Chain" not in name + ax.plot(test_dates[:len(normalised)], normalised, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=name, + zorder=3 if is_optimized else 2) + + hodl_test = hodl_values[test_start_idx::step] + if len(hodl_test) > 0: + hodl_norm = hodl_test / hodl_test[0] + ax.plot(test_dates[:len(hodl_norm)], hodl_norm, linewidth=2, + color="white", alpha=0.7, linestyle="--", label="HODL") + + ax.axhline(1.0, color="white", linestyle=":", alpha=0.3, linewidth=1) + _style_axis(ax) + tokens_str = "/".join(config["tokens"]) + obj_name = config.get("return_val", "objective") + ntr = config.get("noise_trader_ratio", 0.0) + ax.set_title(f"Test Period Only (normalised) — {obj_name}, noise={ntr} — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15) + ax.set_ylabel("Normalised Value", color=TEXT_COLOR, fontsize=12) + ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax.legend(loc="best", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") + output = base.replace(".png", "_test_only.png") + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"Saved plot to {output}") + plt.close() + + +def plot_weights(configs, time_series, config, args): + """Effective weight (value fraction) of token 0 over time.""" + start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") + train_end_dt = datetime.strptime(config["endDateString"], "%Y-%m-%d %H:%M:%S") + + first_out = next(iter(time_series.values())) + n_minutes = len(first_out["value"]) + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") + step = 1440 + dates_daily = dates[::step] + + token_name = config["tokens"][0] + + fig, ax = plt.subplots(1, 1, figsize=(14, 5)) + + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + weights = np.array(out["weights"]) # (T, 2) + w0 = weights[::step, 0] + is_optimized = "On-Chain" not in name + ax.plot(dates_daily[:len(w0)], w0, + linewidth=2.0 if is_optimized else 1.5, + color=COLORS[ci % len(COLORS)], label=name, + alpha=0.9 if is_optimized else 0.7, + zorder=3 if is_optimized else 2) + + ax.axhline(0.5, color="white", linestyle="--", alpha=0.3, linewidth=1) + ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax.get_ylim() + ax.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") + + _style_axis(ax) + tokens_str = "/".join(config["tokens"]) + ax.set_title(f"Effective {token_name} Weight — {tokens_str}", + color=TEXT_COLOR, fontsize=13, pad=15) + ax.set_ylabel(f"{token_name} weight (value fraction)", color=TEXT_COLOR, fontsize=12) + ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax.legend(loc="best", fontsize=9, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") + output = base.replace(".png", "_weights.png") + plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) + print(f"Saved plot to {output}") + plt.close() + + +def _style_axis(ax): + ax.set_facecolor(BG) + ax.tick_params(colors=TEXT_COLOR) + for spine in ax.spines.values(): + spine.set_color(TEXT_COLOR) + spine.set_alpha(0.3) + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.grid(True, alpha=0.15, color=TEXT_COLOR) + + +def main(): + args = parse_args() + config, trials = load_results(args.results_json) + if args.end_test_date: + config["endTestDateString"] = args.end_test_date + if args.noise_trader_ratio is not None: + config["noise_trader_ratio"] = args.noise_trader_ratio + tokens = config["tokens"] + obj_name = config.get("return_val", "objective") + + # Sort trials by penalised objective + trials_sorted = sorted(trials, key=lambda t: t.get("objective", 0), reverse=True) + top_trials = trials_sorted[:args.top_k] + + print("=" * 80) + print(f"reClAMM Optuna Result Plotter — objective: {obj_name}") + print("=" * 80) + print(f" Results: {args.results_json}") + print(f" Tokens: {'/'.join(tokens)}") + print(f" Train: {config['startDateString']} → {config['endDateString']}") + print(f" Test: {config['endDateString']} → {config['endTestDateString']}") + print(f" Fees: {config['fees']}, Gas: {config.get('gas_cost', 1.0)}") + print(f" Trials: {len(trials)} total, plotting top {len(top_trials)}") + + configs = {} + for i, trial in enumerate(top_trials): + params = extract_pool_params(trial, config) + name = f"#{trial.get('optuna_trial_number', i)} (rank {i+1})" + configs[name] = { + "params": params, + "objective": trial.get("objective", 0), + "train_objective": trial.get("train_objective", 0), + "test_objective": trial.get("test_objective", 0), + "train_sharpe": trial.get("train_sharpe", 0), + "validation_sharpe": trial.get("validation_sharpe", 0), + } + print(f"\n {name}:") + print(f" {obj_name}: train={trial.get('train_objective', 0):.4f} " + f"test={trial.get('test_objective', 0):.4f} " + f"penalised={trial.get('objective', 0):.4f}") + print(f" sharpe: train={trial.get('train_sharpe', 0):+.4f} " + f"val={trial.get('validation_sharpe', 0):+.4f}") + for k, v in params.items(): + print(f" {k}: {v:.6g}") + + if not args.no_onchain: + configs["On-Chain (launch)"] = {"params": dict(ONCHAIN_LAUNCH_PARAMS)} + configs["On-Chain (current)"] = {"params": dict(ONCHAIN_CURRENT_PARAMS)} + + # ── Full-period runs ────────────────────────────────────────────── + print(f"\n--- Running full-period simulations ({config['startDateString']} → " + f"{config['endTestDateString']}) ---") + time_series = {} + for name, cfg in configs.items(): + print(f" {name}...", end=" ", flush=True) + out = run_full_period(cfg["params"], config) + time_series[name] = out + fv = float(out["final_value"]) + fr = out.get("fee_revenue") + fr_total = float(np.array(fr).sum()) if fr is not None else 0 + hodl = float((out["reserves"][0] * out["prices"][-1]).sum()) + print(f"final=${fv:,.0f} hodl=${hodl:,.0f} RoH={fv/hodl - 1:+.2%} " + f"fee_rev=${fr_total:,.0f}") + + first_out = next(iter(time_series.values())) + hodl_reserves = first_out["reserves"][0] + hodl_values = np.sum( + np.array(hodl_reserves) * np.array(first_out["prices"]), axis=1, + ) + + # ── Plots ───────────────────────────────────────────────────────── + plot_results(configs, time_series, hodl_values, config, args) + plot_test_only(configs, time_series, hodl_values, config, args) + plot_weights(configs, time_series, config, args) + + # ── Summary table ───────────────────────────────────────────────── + print(f"\n{'=' * 120}") + print(f"SUMMARY — {'/'.join(tokens)} — {obj_name}") + print(f"{'=' * 120}") + hdr = (f"{'Config':<28s} {'Train '+obj_name:>20s} {'Test '+obj_name:>20s} " + f"{'Train SR':>10s} {'Val SR':>10s} " + f"{'PR':>7s} {'Margin':>7s} {'ShiftExp':>10s} {'Full RoH':>10s}") + print(hdr) + print("-" * 120) + + for name, cfg in configs.items(): + cp = cfg["params"] + fv = float(time_series[name]["final_value"]) + full_roh = fv / float(hodl_values[-1]) - 1 + print( + f"{name:<28s} " + f"{cfg.get('train_objective', float('nan')):>20.4f} " + f"{cfg.get('test_objective', float('nan')):>20.4f} " + f"{cfg.get('train_sharpe', float('nan')):>+10.4f} " + f"{cfg.get('validation_sharpe', float('nan')):>+10.4f} " + f"{cp.get('price_ratio', float('nan')):>7.3f} " + f"{cp.get('centeredness_margin', float('nan')):>7.4f} " + f"{cp.get('shift_exponent', float('nan')):>10.4g} " + f"{full_roh * 100:>+9.2f}%" + ) + print("=" * 120) + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_top50_predicted_vs_real.py b/scripts/plot_top50_predicted_vs_real.py new file mode 100644 index 00000000..0951a209 --- /dev/null +++ b/scripts/plot_top50_predicted_vs_real.py @@ -0,0 +1,521 @@ +"""Plot predicted vs real volume for top 50 pools by TVL on Feb 1st 2026. + +Enumerates WEIGHTED (min_tvl=1000) and RECLAMM (min_tvl=0) pools, +fetches their snapshots, filters to those with TVL >= $10k on Feb 1st 2026, +takes the top 50 by TVL, and plots predicted vs actual daily volume using +the inference artifact from calibrate_noise_unified.py. + +For pools that were in the model's training set, uses their per-pool theta. +For pools not in the training set, uses population-level prediction from B. +""" + +import ast +import json +import os +import sys +import time +from datetime import date, timedelta + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# Reuse functions from the noise calibration package +from quantammsim.noise_calibration import ( + BALANCER_API_CHAINS, + _graphql_request, + assemble_panel, + classify_token_tier, + encode_covariates, + fetch_pool_snapshots, + fetch_token_prices, +) + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_top50" +) +TVL_DATE = date(2026, 2, 1) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "top50_feb1" +) +# Inference artifact from the main unified model run +FITTED_JSON = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "unified_full_90d.json" +) + + +def enumerate_all_pools(): + """Enumerate WEIGHTED (min_tvl=1000) and RECLAMM (min_tvl=0) pools.""" + all_pools = [] + + for chain in BALANCER_API_CHAINS: + for pool_type, min_tvl in [("WEIGHTED", 1000), ("RECLAMM", 0)]: + query = { + "query": """ + query GetPools($chain: GqlChain!, $types: [GqlPoolType!], + $minTvl: Float) { + poolGetPools( + where: { chainIn: [$chain], poolTypeIn: $types, minTvl: $minTvl } + ) { + id chain type protocolVersion + poolTokens { symbol weight address } + dynamicData { totalLiquidity swapFee } + } + } + """, + "variables": { + "chain": chain, + "types": [pool_type], + "minTvl": min_tvl, + }, + } + + try: + body = _graphql_request(query) + pools = body.get("data", {}).get("poolGetPools", []) + except Exception as e: + print(f" FAILED {chain} {pool_type}: {e}") + continue + + for p in pools: + tokens = [t["symbol"] for t in p.get("poolTokens", [])] + addresses = [t.get("address", "") for t in p.get("poolTokens", [])] + tvl = float(p.get("dynamicData", {}).get("totalLiquidity", 0)) + fee = float(p.get("dynamicData", {}).get("swapFee", 0)) + all_pools.append({ + "pool_id": p["id"], + "chain": p["chain"], + "pool_type": p["type"], + "tokens": tokens, + "token_addresses": addresses, + "swap_fee": fee, + "current_tvl": tvl, + }) + + if pools: + print(f" {chain:>10} {pool_type:>10}: {len(pools)}") + time.sleep(0.3) + + df = pd.DataFrame(all_pools) + print(f"\n Total: {len(df)} pools") + return df + + +def fetch_all_snapshots_cached(pools_df, cache_dir): + """Fetch snapshots for all pools, caching per-pool.""" + snap_dir = os.path.join(cache_dir, "snapshots") + os.makedirs(snap_dir, exist_ok=True) + + all_snaps = [] + n = len(pools_df) + + for i, (_, pool) in enumerate(pools_df.iterrows()): + pid = pool["pool_id"] + chain = pool["chain"] + cache_file = os.path.join(snap_dir, f"{pid}.parquet") + + if os.path.exists(cache_file): + df = pd.read_parquet(cache_file) + else: + if (i + 1) % 20 == 0 or i == 0: + print(f" Fetching snapshots {i+1}/{n}...", flush=True) + try: + df = fetch_pool_snapshots(pid, chain) + if len(df) > 0: + df.to_parquet(cache_file, index=False) + time.sleep(0.3) + except Exception as e: + print(f" FAILED {pid[:20]}: {e}") + continue + + if len(df) > 0: + df["pool_id"] = pid + df["chain"] = chain + all_snaps.append(df) + + if all_snaps: + return pd.concat(all_snaps, ignore_index=True) + return pd.DataFrame() + + +def get_tvl_on_date(snapshots_df, target_date, window_days=3): + """Get TVL for each pool on/near target_date.""" + results = [] + for pid in snapshots_df["pool_id"].unique(): + pool_snaps = snapshots_df[snapshots_df["pool_id"] == pid] + + best_row = None + best_dist = float("inf") + for _, row in pool_snaps.iterrows(): + d = row["date"] + if isinstance(d, date): + dist = abs((d - target_date).days) + else: + dist = abs((pd.Timestamp(d).date() - target_date).days) + if dist < best_dist: + best_dist = dist + best_row = row + + if best_row is not None and best_dist <= window_days: + results.append({ + "pool_id": pid, + "tvl_feb1": float(best_row["total_liquidity_usd"]), + "date_used": best_row["date"], + }) + + return pd.DataFrame(results) + + +def _get_theta_for_pool(pid, fitted, panel_90d, pop_B, pop_cov_names): + """Get theta for a pool: from fitted artifact if available, else population. + + Returns (theta, source) where source is 'fitted' or 'population'. + For IBP models, population fallback adds marginal feature effect (pi @ W). + """ + if pid in fitted["pools"]: + return np.array(fitted["pools"][pid]["theta_median"]), "fitted" + + # Population-level prediction: theta = B @ z_pool + # Build z_pool from the pool's covariates + pp = panel_90d[panel_90d["pool_id"] == pid] + if len(pp) == 0: + return None, "no_data" + + chain = pp["chain"].iloc[0] + tokens = pp["tokens"].iloc[0] + if isinstance(tokens, str): + tokens = tokens.split(",") + fee = pp["swap_fee"].iloc[0] if "swap_fee" in pp.columns else 0.003 + tiers = sorted([classify_token_tier(t) for t in tokens]) + tier_a = tiers[0] + + # Build covariate vector matching the model's encoding + z = np.zeros(len(pop_cov_names)) + for i, name in enumerate(pop_cov_names): + if name == "intercept": + z[i] = 1.0 + elif name == f"chain_{chain}": + z[i] = 1.0 + elif name == f"tier_A_{tier_a}": + z[i] = 1.0 + elif name == "log_fee": + z[i] = np.log(max(fee, 1e-6)) + + # B is (K_coeff, K_cov), theta = B @ z + B = np.array(pop_B) # (K_coeff, K_cov) + theta = B @ z + + # IBP: add marginal feature effect (pi @ W) + pop = fitted["population_effects"] + if "W" in pop and "feature_prevalences" in pop: + W = np.array(pop["W"]) # (K_features, K_coeff) + pi = np.array(pop["feature_prevalences"]) # (K_features,) + theta = theta + pi @ W + + return theta, "population" + + +def plot_pages(plot_pools, pool_idx_map_fitted, fitted, panel_90d, + pools_df, tvl_lookup, pop_B, pop_cov_names, + output_dir=OUTPUT_DIR): + """Generate paginated plots, 10 pools per page.""" + n_pools = len(plot_pools) + per_page = 10 + n_pages = (n_pools + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, n_pools) + page_pools = plot_pools[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(14, 4 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + + for idx, (pid, feb_tvl) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + + theta, source = _get_theta_for_pool( + pid, fitted, panel_90d, pop_B, pop_cov_names + ) + if theta is None: + ax.set_visible(False) + continue + + pp = panel_90d[panel_90d["pool_id"] == pid].sort_values("date") + if len(pp) < 5: + ax.set_visible(False) + continue + + x_obs = np.column_stack([ + np.ones(len(pp)), + pp["log_tvl_lag1"].values, + pp["volatility"].values, + pp["weekend"].values, + ]) + + pred_log = x_obs @ theta + actual_log = pp["log_volume"].values + pred_vol = np.exp(pred_log) + actual_vol = np.exp(actual_log) + dates = pd.to_datetime(pp["date"].values) + + ax.plot(dates, actual_vol, "o-", color="steelblue", markersize=2.5, + linewidth=0.9, alpha=0.7, label="Actual") + ax.plot(dates, pred_vol, "s--", color="orangered", markersize=2.5, + linewidth=0.9, alpha=0.7, label="Predicted") + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + ss_res = np.sum((actual_log - pred_log) ** 2) + ss_tot = np.sum((actual_log - actual_log.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + meta = pools_df[pools_df["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tokens = m["tokens"] + tok_str = "/".join(str(t)[:8] for t in tokens[:2]) + chain = str(m["chain"]) + ptype = str(m["pool_type"]) + else: + tok_str = pid[:16] + chain = "?" + ptype = "?" + + type_tag = "R" if ptype == "RECLAMM" else "W" + src_tag = "*" if source == "population" else "" + ax.set_title( + "{} ({}, {}){}\n" + "TVL ${:,.0f} on Feb 1 | " + "R\u00b2={:.3f} b_c={:.2f} b_\u03c3={:.2f} " + "b_wknd={:.2f} n={}".format( + tok_str, chain, type_tag, src_tag, feb_tvl, + r2, theta[1], theta[2], theta[3], len(pp)), + fontsize=8) + ax.legend(fontsize=7) + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + "Predicted vs actual daily volume \u2014 page {}/{} " + "(sorted by TVL on {}) [* = population prediction]".format( + page + 1, n_pages, TVL_DATE), + fontsize=11) + fig.tight_layout() + out = os.path.join(output_dir, "pred_vs_real_page{}.png".format(page + 1)) + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(" Saved: {}".format(out)) + + +def main(): + import argparse + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--artifact", default=FITTED_JSON, + help="Path to inference artifact JSON") + parser.add_argument("--output-dir", default=None, + help="Output directory (default: auto from model name)") + args = parser.parse_args() + + artifact_path = args.artifact + os.makedirs(CACHE_DIR, exist_ok=True) + + # ---- Load inference artifact ---- + print(f"Loading inference artifact: {artifact_path}") + with open(artifact_path) as f: + fitted = json.load(f) + n_fitted = len(fitted["pools"]) + model_name = fitted.get("model", "unknown") + print(f" Model: {model_name}") + print(f" {n_fitted} pools with fitted theta") + + # Output dir: use CLI override or auto from model name + output_dir = args.output_dir or os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", f"top50_feb1_{model_name}", + ) + os.makedirs(output_dir, exist_ok=True) + + # Extract population-level B matrix for pools not in the model + pop_cov_names = fitted["model_spec"]["covariate_names"] + pop_B = np.array(fitted["population_effects"]["B"]) # (K_coeff, K_cov) + print(f" Population B: {pop_B.shape}, covariates: {pop_cov_names}") + + # ---- Step 1: Enumerate pools ---- + pools_cache = os.path.join(CACHE_DIR, "pools.parquet") + if os.path.exists(pools_cache): + pools_df = pd.read_parquet(pools_cache) + if isinstance(pools_df["tokens"].iloc[0], str): + pools_df["tokens"] = pools_df["tokens"].apply(ast.literal_eval) + pools_df["token_addresses"] = pools_df["token_addresses"].apply( + ast.literal_eval + ) + print(f"\nLoaded {len(pools_df)} pools from cache") + else: + print("\n1. Enumerating pools...") + pools_df = enumerate_all_pools() + pools_df.to_parquet(pools_cache, index=False) + + # ---- Step 2: Fetch snapshots ---- + print("\n2. Fetching snapshots...") + snapshots_df = fetch_all_snapshots_cached(pools_df, CACHE_DIR) + print(f" {len(snapshots_df)} pool-days") + + # ---- Step 3: TVL on Feb 1st ---- + print(f"\n3. Finding TVL on {TVL_DATE}...") + tvl_df = get_tvl_on_date(snapshots_df, TVL_DATE) + tvl_df = tvl_df[tvl_df["tvl_feb1"] >= 10_000].copy() + tvl_df = tvl_df.sort_values("tvl_feb1", ascending=False).head(50) + top50_ids = set(tvl_df["pool_id"]) + tvl_lookup = dict(zip(tvl_df["pool_id"], tvl_df["tvl_feb1"])) + print(f" {len(tvl_df)} pools with TVL >= $10k") + + # How many are in the fitted model? + n_in_model = sum(1 for pid in top50_ids if pid in fitted["pools"]) + print(f" {n_in_model} in fitted model, " + f"{len(top50_ids) - n_in_model} will use population prediction") + + # ---- Step 4: Fetch token prices & assemble panel ---- + panel_cache = os.path.join(CACHE_DIR, "panel.parquet") + if os.path.exists(panel_cache): + panel = pd.read_parquet(panel_cache) + print(f"\n4. Loaded panel from cache: {len(panel)} obs") + else: + top50_pools = pools_df[pools_df["pool_id"].isin(top50_ids)].copy() + top50_snaps = snapshots_df[snapshots_df["pool_id"].isin(top50_ids)].copy() + + print("\n4. Fetching token prices...") + prices_cache = os.path.join(CACHE_DIR, "token_prices") + token_addr_by_chain = {} + for _, pool in top50_pools.iterrows(): + chain = pool["chain"] + tokens = pool["tokens"] + addresses = pool["token_addresses"] + if chain not in token_addr_by_chain: + token_addr_by_chain[chain] = {} + for sym, addr in zip(tokens, addresses): + if sym and addr: + token_addr_by_chain[chain][sym] = addr + + token_prices = fetch_token_prices( + token_addr_by_chain, cache_dir=prices_cache + ) + + print("\n Assembling panel...") + panel = assemble_panel(top50_pools, top50_snaps, token_prices) + panel.to_parquet(panel_cache, index=False) + + # ---- Step 5: Filter to 90 days ---- + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=90) + panel_90d = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if "log_tvl_lag1" not in panel_90d.columns: + panel_90d = panel_90d.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel_90d["log_tvl_lag1"] = panel_90d.groupby("pool_id")["log_tvl"].shift(1) + panel_90d = panel_90d.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel_90d.groupby("pool_id").size() + valid_pools = pool_counts[pool_counts >= 10].index + panel_90d = panel_90d[panel_90d["pool_id"].isin(valid_pools)].copy() + print(f"\n5. 90-day panel: {len(panel_90d)} obs, " + f"{panel_90d['pool_id'].nunique()} pools") + + # ---- Step 6: Plot ---- + # Sort by Feb 1 TVL, only include pools with panel data + plot_pools = [] + for pid in panel_90d["pool_id"].unique(): + if pid in tvl_lookup: + plot_pools.append((pid, tvl_lookup[pid])) + plot_pools.sort(key=lambda x: -x[1]) + print(f"\n6. Plotting {len(plot_pools)} pools...") + + plot_pages(plot_pools, fitted, fitted, panel_90d, pools_df, + tvl_lookup, pop_B, pop_cov_names, output_dir=output_dir) + + # ---- Summary table ---- + summary = [] + for pid, feb_tvl in plot_pools: + theta, source = _get_theta_for_pool( + pid, fitted, panel_90d, pop_B, pop_cov_names + ) + if theta is None: + continue + pp = panel_90d[panel_90d["pool_id"] == pid] + x = np.column_stack([ + np.ones(len(pp)), + pp["log_tvl_lag1"].values, + pp["volatility"].values, + pp["weekend"].values, + ]) + pred = x @ theta + actual = pp["log_volume"].values + ss_res = np.sum((actual - pred) ** 2) + ss_tot = np.sum((actual - actual.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + meta = pools_df[pools_df["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tok_str = "/".join(str(t) for t in m["tokens"][:2]) + chain = str(m["chain"]) + ptype = str(m["pool_type"]) + else: + tok_str = pid[:16] + chain = "?" + ptype = "?" + + summary.append({ + "pool_id": pid[:20], + "tokens": tok_str, + "chain": chain, + "type": ptype, + "tvl_feb1": feb_tvl, + "n_obs": len(pp), + "R2": r2, + "b_c": theta[1], + "b_sigma": theta[2], + "b_weekend": theta[3], + "source": source, + }) + + summary_df = pd.DataFrame(summary) + summary_path = os.path.join(output_dir, "top50_summary.csv") + summary_df.to_csv(summary_path, index=False) + print(f"\n Saved: {summary_path}") + + n_pools = len(summary_df) + n_fitted_used = (summary_df["source"] == "fitted").sum() + n_pop = (summary_df["source"] == "population").sum() + n_reclamm = (summary_df["type"] == "RECLAMM").sum() + print(f"\n{'='*70}") + print(f"Summary: {n_pools} pools ({n_fitted_used} fitted, {n_pop} population)") + print(f" RECLAMM: {n_reclamm} WEIGHTED: {n_pools - n_reclamm}") + print(f" Median R\u00b2: {summary_df['R2'].median():.3f}") + print(f" Mean b_c: {summary_df['b_c'].mean():.3f}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_direct_calibration_top50.py b/scripts/run_direct_calibration_top50.py new file mode 100644 index 00000000..1c1770b8 --- /dev/null +++ b/scripts/run_direct_calibration_top50.py @@ -0,0 +1,852 @@ +"""Run direct calibration pipeline and plot top-50 style decomposition. + +Steps: + 1. Load panel, match to per-day grids in results/pool_grids_v2/ + 2. Option C: per-pool L-BFGS-B fits + 3. Option A: joint end-to-end optimization (warm-started from C) + 4. Paginated plots: V_arb + V_noise decomposition per pool + 5. Summary plots: cadence, gas, R², arb fraction distributions +""" + +import json +import os +import sys +from datetime import date, timedelta + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +GRID_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "pool_grids_v2", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "direct_calibration_top50", +) +TRAIN_DAYS = 90 +TOP_N = 50 +OPTION_C_MAXITER = 500 +JOINT_MAXITER = 500 +OPTION_C_LOSS_CUTOFF = 5.0 # Drop pools with Option C loss above this from joint fit + + +def load_and_match(): + """Load panel, filter to 90 days, match to grids.""" + panel = pd.read_parquet(PANEL_CACHE) + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=TRAIN_DAYS) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if "log_tvl_lag1" not in panel.columns: + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel.groupby("pool_id").size() + valid = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid)].copy() + + print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " + f"{cutoff} to {max_date}") + + from quantammsim.calibration.pool_data import match_grids_to_panel + matched = match_grids_to_panel(GRID_DIR, panel) + print(f"Matched: {len(matched)} pools with grids") + + return panel, matched + + +def run_option_c(matched): + """Per-pool L-BFGS-B fits.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + print(f"\n--- Option C: per-pool fits ({len(matched)} pools) ---") + results = fit_all_pools(matched) + n_converged = sum(1 for r in results.values() if r["converged"]) + losses = [r["loss"] for r in results.values()] + print(f" Converged: {n_converged}/{len(results)}") + print(f" Loss: median={np.median(losses):.4f}, " + f"mean={np.mean(losses):.4f}, " + f"range=[{np.min(losses):.4f}, {np.max(losses):.4f}]") + return results + + +def run_option_a(matched, option_c_results): + """Joint end-to-end optimization, warm-started from Option C. + + Drops pathological pools (Option C loss > OPTION_C_LOSS_CUTOFF) from the + joint fit to prevent them from dominating the shared mapping. + """ + from quantammsim.calibration.joint_fit import fit_joint + + # Filter out pathological pools + good_pools = {p: r for p, r in option_c_results.items() + if r["loss"] <= OPTION_C_LOSS_CUTOFF} + dropped = set(option_c_results) - set(good_pools) + matched_clean = {p: matched[p] for p in good_pools if p in matched} + + if dropped: + print(f"\n Dropping {len(dropped)} pathological pools (Option C loss > {OPTION_C_LOSS_CUTOFF}):") + for p in sorted(dropped): + r = option_c_results[p] + print(f" {p} {r['tokens']:<16} loss={r['loss']:.1f}") + + print(f"\n--- Option A: joint fit (per_pool_noise, {len(matched_clean)} pools, " + f"warm-start from C, no chain dummies) ---") + result_ppn = fit_joint( + matched_clean, + mode="per_pool_noise", + init_from_option_c=good_pools, + maxiter=JOINT_MAXITER, + drop_chain_dummies=True, + ) + print(f" Loss: {result_ppn['init_loss']:.4f} -> {result_ppn['loss']:.4f}") + print(f" Converged: {result_ppn['converged']}") + + print(f"\n--- Option A: joint fit (shared_noise, {len(matched_clean)} pools, " + f"warm-start from C, no chain dummies) ---") + result_sn = fit_joint( + matched_clean, + mode="shared_noise", + init_from_option_c=good_pools, + maxiter=JOINT_MAXITER, + drop_chain_dummies=True, + ) + print(f" Loss: {result_sn['init_loss']:.4f} -> {result_sn['loss']:.4f}") + print(f" Converged: {result_sn['converged']}") + + return result_ppn, result_sn + + +def run_option_rf(matched, option_c_results): + """2-stage approach: Option C per-pool fits → Ridge/RF on pool attributes. + + Drops chain dummies (too sparse for n~30), keeps 6 continuous/binary features. + Trains both Ridge regression and RF, reports both. LOO-CV for generalization. + + Noise coefficients are taken directly from Option C (per-pool). + Drops pathological pools (Option C loss > OPTION_C_LOSS_CUTOFF). + """ + from sklearn.ensemble import RandomForestRegressor + from sklearn.linear_model import RidgeCV + from sklearn.model_selection import LeaveOneOut + from quantammsim.calibration.pool_data import build_pool_attributes + + # Filter pathological pools + good_pools = {p: r for p, r in option_c_results.items() + if r["loss"] <= OPTION_C_LOSS_CUTOFF} + dropped = set(option_c_results) - set(good_pools) + matched_clean = {p: matched[p] for p in good_pools if p in matched} + + if dropped: + print(f"\n Dropping {len(dropped)} pathological pools (Option C loss > {OPTION_C_LOSS_CUTOFF}):") + for p in sorted(dropped): + r = option_c_results[p] + print(f" {p} {r['tokens']:<16} loss={r['loss']:.1f}") + + # Build attributes and targets + X_attr_full, attr_names_full, pool_ids = build_pool_attributes(matched_clean) + n_pools = len(pool_ids) + + # Drop chain dummies — too sparse for n~30. Keep only continuous/binary features. + non_chain_mask = [i for i, name in enumerate(attr_names_full) + if not name.startswith("chain_")] + X_attr = X_attr_full[:, non_chain_mask] + attr_names = [attr_names_full[i] for i in non_chain_mask] + k_attr = len(attr_names) + + print(f"\n--- Option RF: 2-stage mapping ({n_pools} pools, {k_attr} features) ---") + print(f" Features: {', '.join(attr_names)}") + + Y_cad = np.array([good_pools[p]["log_cadence"] for p in pool_ids]) + Y_gas = np.array([good_pools[p]["log_gas"] for p in pool_ids]) + Y = np.column_stack([Y_cad, Y_gas]) + + ss_tot_cad = np.sum((Y_cad - Y_cad.mean()) ** 2) + ss_tot_gas = np.sum((Y_gas - Y_gas.mean()) ** 2) + + def compute_r2(y_true, y_pred, ss_tot): + return 1 - np.sum((y_true - y_pred) ** 2) / max(ss_tot, 1e-10) + + # ---- Ridge regression (multi-output via separate fits) ---- + alphas = np.logspace(-2, 4, 50) + ridge_cad = RidgeCV(alphas=alphas, cv=None) # GCV/LOO built-in + ridge_gas = RidgeCV(alphas=alphas, cv=None) + ridge_cad.fit(X_attr, Y_cad) + ridge_gas.fit(X_attr, Y_gas) + + Y_ridge_train = np.column_stack([ridge_cad.predict(X_attr), + ridge_gas.predict(X_attr)]) + r2_ridge_cad = compute_r2(Y_cad, Y_ridge_train[:, 0], ss_tot_cad) + r2_ridge_gas = compute_r2(Y_gas, Y_ridge_train[:, 1], ss_tot_gas) + + print(f"\n Ridge (alpha_cad={ridge_cad.alpha_:.1f}, alpha_gas={ridge_gas.alpha_:.1f}):") + print(f" In-sample R²: cadence={r2_ridge_cad:.3f}, gas={r2_ridge_gas:.3f}") + + # Ridge LOO-CV + loo = LeaveOneOut() + Y_ridge_loo = np.zeros_like(Y) + for train_idx, test_idx in loo.split(X_attr): + rc = RidgeCV(alphas=alphas, cv=None).fit(X_attr[train_idx], Y_cad[train_idx]) + rg = RidgeCV(alphas=alphas, cv=None).fit(X_attr[train_idx], Y_gas[train_idx]) + Y_ridge_loo[test_idx, 0] = rc.predict(X_attr[test_idx]) + Y_ridge_loo[test_idx, 1] = rg.predict(X_attr[test_idx]) + + r2_ridge_loo_cad = compute_r2(Y_cad, Y_ridge_loo[:, 0], ss_tot_cad) + r2_ridge_loo_gas = compute_r2(Y_gas, Y_ridge_loo[:, 1], ss_tot_gas) + print(f" LOO-CV R²: cadence={r2_ridge_loo_cad:.3f}, gas={r2_ridge_loo_gas:.3f}") + print(f" LOO-CV MAE: cadence={np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_ridge_loo[:, 0]))):.1f} min, " + f"gas=${np.mean(np.abs(np.exp(Y_gas) - np.exp(Y_ridge_loo[:, 1]))):.2f}") + + # Ridge coefficients + print(f" Coefficients (cadence | gas):") + print(f" {'intercept':<20} {ridge_cad.intercept_:>7.3f} {ridge_gas.intercept_:>7.3f}") + for j, name in enumerate(attr_names): + print(f" {name:<20} {ridge_cad.coef_[j]:>7.3f} {ridge_gas.coef_[j]:>7.3f}") + + # ---- Random Forest (reduced features) ---- + rf = RandomForestRegressor( + n_estimators=200, + max_depth=None, + min_samples_leaf=3, # stronger regularization + max_features=min(4, k_attr), # cap at 4 features per split + random_state=42, + n_jobs=-1, + ) + rf.fit(X_attr, Y) + Y_rf_train = rf.predict(X_attr) + + r2_rf_cad = compute_r2(Y_cad, Y_rf_train[:, 0], ss_tot_cad) + r2_rf_gas = compute_r2(Y_gas, Y_rf_train[:, 1], ss_tot_gas) + + print(f"\n Random Forest (min_leaf=3, max_feat=4):") + print(f" In-sample R²: cadence={r2_rf_cad:.3f}, gas={r2_rf_gas:.3f}") + + # RF LOO-CV + Y_rf_loo = np.zeros_like(Y) + for train_idx, test_idx in loo.split(X_attr): + rf_loo = RandomForestRegressor( + n_estimators=200, max_depth=None, min_samples_leaf=3, + max_features=min(4, k_attr), random_state=42, n_jobs=-1, + ) + rf_loo.fit(X_attr[train_idx], Y[train_idx]) + Y_rf_loo[test_idx] = rf_loo.predict(X_attr[test_idx]) + + r2_rf_loo_cad = compute_r2(Y_cad, Y_rf_loo[:, 0], ss_tot_cad) + r2_rf_loo_gas = compute_r2(Y_gas, Y_rf_loo[:, 1], ss_tot_gas) + print(f" LOO-CV R²: cadence={r2_rf_loo_cad:.3f}, gas={r2_rf_loo_gas:.3f}") + print(f" LOO-CV MAE: cadence={np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_rf_loo[:, 0]))):.1f} min, " + f"gas=${np.mean(np.abs(np.exp(Y_gas) - np.exp(Y_rf_loo[:, 1]))):.2f}") + + print(f"\n Feature importances:") + for j, name in enumerate(attr_names): + print(f" {name:<20} {rf.feature_importances_[j]:.3f}") + + # ---- Pick best LOO model ---- + ridge_loo_total = r2_ridge_loo_cad + r2_ridge_loo_gas + rf_loo_total = r2_rf_loo_cad + r2_rf_loo_gas + best = "ridge" if ridge_loo_total >= rf_loo_total else "rf" + print(f"\n Best LOO model: {best} (ridge={ridge_loo_total:.3f} vs rf={rf_loo_total:.3f})") + + if best == "ridge": + Y_best_train = Y_ridge_train + Y_best_loo = Y_ridge_loo + r2_best_cad = r2_ridge_cad + r2_best_gas = r2_ridge_gas + r2_best_loo_cad = r2_ridge_loo_cad + r2_best_loo_gas = r2_ridge_loo_gas + else: + Y_best_train = Y_rf_train + Y_best_loo = Y_rf_loo + r2_best_cad = r2_rf_cad + r2_best_gas = r2_rf_gas + r2_best_loo_cad = r2_rf_loo_cad + r2_best_loo_gas = r2_rf_loo_gas + + # Build result dict using the best model's predictions + noise_all = np.array([good_pools[p]["noise_coeffs"] for p in pool_ids]) + + result = { + "pool_ids": pool_ids, + "attr_names": attr_names, + "X_attr": X_attr, + "best_model": best, + "predictions": {}, + "loo_predictions": {}, + "noise_coeffs": noise_all, # from Option C + "r2_train_cad": r2_best_cad, + "r2_train_gas": r2_best_gas, + "r2_loo_cad": r2_best_loo_cad, + "r2_loo_gas": r2_best_loo_gas, + } + + for i, pid in enumerate(pool_ids): + result["predictions"][pid] = { + "log_cadence": float(Y_best_train[i, 0]), + "log_gas": float(Y_best_train[i, 1]), + } + result["loo_predictions"][pid] = { + "log_cadence": float(Y_best_loo[i, 0]), + "log_gas": float(Y_best_loo[i, 1]), + } + + return result + + +def compute_per_pool_predictions(matched, option_c_results, joint_result, rf_result=None): + """Compute per-observation V_arb, V_noise, V_total for each pool. + + Pools not in the joint result (dropped as pathological) get NaN for + Option A predictions. Same for RF. + """ + from quantammsim.calibration.pool_data import build_x_obs, build_pool_attributes + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.loss import K_OBS + import jax.numpy as jnp + + pool_ids = sorted(matched.keys()) + + # Build attributes for the joint-fitted pool subset + joint_pool_ids = joint_result["pool_ids"] + joint_matched = {p: matched[p] for p in joint_pool_ids if p in matched} + X_attr_joint_full, attr_names_full, _ = build_pool_attributes(joint_matched) + # Filter to the features actually used by the joint model + joint_attr_names = joint_result["attr_names"] + joint_feat_idx = [attr_names_full.index(n) for n in joint_attr_names + if n in attr_names_full] + X_attr_joint = X_attr_joint_full[:, joint_feat_idx] + joint_pid_to_idx = {p: i for i, p in enumerate(joint_pool_ids)} + + # RF predictions lookup + rf_pool_ids = rf_result["pool_ids"] if rf_result else [] + rf_pid_to_idx = {p: i for i, p in enumerate(rf_pool_ids)} + + predictions = {} + for pid in pool_ids: + entry = matched[pid] + panel = entry["panel"] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + + def r2(v_arb, v_noise, y): + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y) ** 2) + ss_tot = np.sum((y - y.mean()) ** 2) + return 1 - ss_res / max(ss_tot, 1e-10) + + # --- Option C predictions --- + r = option_c_results[pid] + log_cad_c = r["log_cadence"] + log_gas_c = r["log_gas"] + noise_c_c = r["noise_coeffs"] + + v_arb_all_c = np.array(interpolate_pool_daily( + coeffs, jnp.float64(log_cad_c), jnp.float64(np.exp(log_gas_c)), + )) + v_arb_c = v_arb_all_c[day_indices] + v_noise_c = np.exp(x_obs @ noise_c_c) + + # --- Option A (per_pool_noise) predictions --- + if pid in joint_pid_to_idx: + ji = joint_pid_to_idx[pid] + x_attr = X_attr_joint[ji] + log_cad_a = float(joint_result["bias_cad"]) + float(x_attr @ joint_result["W_cad"]) + log_gas_a = float(joint_result["bias_gas"]) + float(x_attr @ joint_result["W_gas"]) + noise_c_a = joint_result["noise_coeffs"][ji] + + v_arb_all_a = np.array(interpolate_pool_daily( + coeffs, jnp.float64(log_cad_a), jnp.float64(np.exp(log_gas_a)), + )) + v_arb_a = v_arb_all_a[day_indices] + v_noise_a = np.exp(x_obs @ noise_c_a) + r2_a = r2(v_arb_a, v_noise_a, y_obs) + cad_a = np.exp(log_cad_a) + gas_a = np.exp(log_gas_a) + else: + v_arb_a = np.full(len(y_obs), np.nan) + v_noise_a = np.full(len(y_obs), np.nan) + r2_a = np.nan + cad_a = np.nan + gas_a = np.nan + + # --- Option RF predictions --- + if rf_result and pid in rf_pid_to_idx: + ri = rf_pid_to_idx[pid] + rf_pred = rf_result["predictions"][pid] + log_cad_rf = rf_pred["log_cadence"] + log_gas_rf = rf_pred["log_gas"] + noise_c_rf = rf_result["noise_coeffs"][ri] # from Option C + + v_arb_all_rf = np.array(interpolate_pool_daily( + coeffs, jnp.float64(log_cad_rf), jnp.float64(np.exp(log_gas_rf)), + )) + v_arb_rf = v_arb_all_rf[day_indices] + v_noise_rf = np.exp(x_obs @ noise_c_rf) + r2_rf = r2(v_arb_rf, v_noise_rf, y_obs) + cad_rf = np.exp(log_cad_rf) + gas_rf = np.exp(log_gas_rf) + + # LOO predictions (out-of-sample) + loo_pred = rf_result["loo_predictions"][pid] + log_cad_loo = loo_pred["log_cadence"] + log_gas_loo = loo_pred["log_gas"] + v_arb_all_loo = np.array(interpolate_pool_daily( + coeffs, jnp.float64(log_cad_loo), jnp.float64(np.exp(log_gas_loo)), + )) + v_arb_loo = v_arb_all_loo[day_indices] + v_noise_loo = np.exp(x_obs @ noise_c_rf) # same noise coeffs + r2_loo = r2(v_arb_loo, v_noise_loo, y_obs) + cad_loo = np.exp(log_cad_loo) + gas_loo = np.exp(log_gas_loo) + else: + v_arb_rf = np.full(len(y_obs), np.nan) + v_noise_rf = np.full(len(y_obs), np.nan) + r2_rf = np.nan + cad_rf = np.nan + gas_rf = np.nan + r2_loo = np.nan + cad_loo = np.nan + gas_loo = np.nan + + predictions[pid] = { + "dates": pd.to_datetime(panel["date"].values), + "y_obs": y_obs, + "actual_vol": np.exp(y_obs), + # Option C + "v_arb_c": v_arb_c, + "v_noise_c": v_noise_c, + "r2_c": r2(v_arb_c, v_noise_c, y_obs), + "cadence_c": np.exp(log_cad_c), + "gas_c": np.exp(log_gas_c), + "converged_c": r["converged"], + # Option A + "v_arb_a": v_arb_a, + "v_noise_a": v_noise_a, + "r2_a": r2_a, + "cadence_a": cad_a, + "gas_a": gas_a, + # Option RF (in-sample) + "v_arb_rf": v_arb_rf, + "v_noise_rf": v_noise_rf, + "r2_rf": r2_rf, + "cadence_rf": cad_rf, + "gas_rf": gas_rf, + # Option RF LOO (out-of-sample) + "r2_rf_loo": r2_loo, + "cadence_rf_loo": cad_loo, + "gas_rf_loo": gas_loo, + # Metadata + "chain": entry["chain"], + "tokens": entry["tokens"], + "fee": entry["fee"], + "median_tvl": float(np.exp(panel["log_tvl_lag1"].median())), + "n_obs": len(y_obs), + } + + return predictions + + +def plot_top50_pages(predictions, method="c"): + """Paginated plots: V_arb + V_noise decomposition.""" + # Rank by median TVL + ranked = sorted( + predictions.items(), + key=lambda x: -x[1]["median_tvl"], + )[:TOP_N] + + suffix = {"c": "option_c", "a": "option_a", "rf": "option_rf"}[method] + per_page = 10 + n_pages = (len(ranked) + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, len(ranked)) + page_pools = ranked[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(16, 4.5 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + elif ncols == 1: + axes = axes.reshape(-1, 1) + + for idx, (pid, p) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + dates = p["dates"] + + if method == "c": + v_arb = p["v_arb_c"] + v_noise = p["v_noise_c"] + r2_val = p["r2_c"] + cad = p["cadence_c"] + gas = p["gas_c"] + elif method == "rf": + v_arb = p["v_arb_rf"] + v_noise = p["v_noise_rf"] + r2_val = p["r2_rf"] + cad = p["cadence_rf"] + gas = p["gas_rf"] + else: + v_arb = p["v_arb_a"] + v_noise = p["v_noise_a"] + r2_val = p["r2_a"] + cad = p["cadence_a"] + gas = p["gas_a"] + + # Skip pools with NaN predictions (dropped from this method) + if np.any(np.isnan(v_arb)): + ax.text(0.5, 0.5, f"Dropped from {method.upper()}", fontsize=12, + ha="center", va="center", transform=ax.transAxes, color="gray") + ax.set_title(f"{pid[:16]} — dropped", fontsize=8) + continue + + v_total = v_arb + v_noise + arb_frac = np.median(v_arb / np.maximum(v_total, 1.0)) + actual = p["actual_vol"] + + # Stacked area: V_arb bottom, V_noise on top + ax.fill_between(dates, 0, np.maximum(v_arb, 0), + alpha=0.3, color="orangered", label="V_arb (grid)") + ax.fill_between(dates, np.maximum(v_arb, 0), np.maximum(v_total, 0), + alpha=0.3, color="steelblue", label="V_noise (covariates)") + ax.plot(dates, actual, "k-", linewidth=0.8, alpha=0.7, label="Actual") + ax.plot(dates, np.maximum(v_total, 0), "--", color="purple", + linewidth=0.8, alpha=0.7, label="Predicted total") + + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + tokens = p["tokens"] + if isinstance(tokens, (list, tuple)): + tok_str = "/".join(str(t)[:8] for t in tokens[:2]) + elif isinstance(tokens, str): + tok_str = "/".join(t.strip()[:8] for t in tokens.split(",")[:2]) + else: + tok_str = pid[:16] + + ax.set_title( + f"{tok_str} ({p['chain']})\n" + f"TVL ${p['median_tvl']:,.0f} | R²={r2_val:.3f} " + f"cad={cad:.1f}min gas=${gas:.2f} " + f"arb_frac={arb_frac:.1%} n={p['n_obs']}", + fontsize=8, + ) + ax.legend(fontsize=6, loc="upper right") + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + method_label = {"c": "Option C (per-pool)", "a": "Option A (linear)", + "rf": "Option RF (random forest)"}[method] + fig.suptitle( + f"Direct calibration: V_arb + V_noise — {method_label}\n" + f"page {page + 1}/{n_pages} " + f"(top {min(TOP_N, len(ranked))} by median TVL, {TRAIN_DAYS}d window)", + fontsize=11, + ) + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, f"{suffix}_page{page + 1}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_summary(predictions, option_c_results, joint_result): + """Summary: distributions of cadence, gas, R², arb fraction for both methods.""" + pool_ids = sorted(predictions.keys()) + n = len(pool_ids) + + fig, axes = plt.subplots(2, 4, figsize=(20, 10)) + + for row, (method, label) in enumerate([("c", "Option C (per-pool)"), + ("a", "Option A (joint)")]): + cads = [predictions[p][f"cadence_{method}"] for p in pool_ids] + gases = [predictions[p][f"gas_{method}"] for p in pool_ids] + r2s = [predictions[p][f"r2_{method}"] for p in pool_ids] + arb_fracs = [] + for p in pool_ids: + v_arb = predictions[p][f"v_arb_{method}"] + v_noise = predictions[p][f"v_noise_{method}"] + total = v_arb + v_noise + arb_fracs.append(np.median(v_arb / np.maximum(total, 1.0))) + + # Cadence + ax = axes[row, 0] + ax.hist(cads, bins=20, color="orangered", alpha=0.7, edgecolor="white") + ax.axvline(np.median(cads), color="black", linestyle="--", + label=f"Median={np.median(cads):.1f}min") + ax.set_xlabel("Cadence (minutes)") + ax.set_title(f"{label}: Cadence") + ax.legend(fontsize=8) + + # Gas + ax = axes[row, 1] + ax.hist(gases, bins=20, color="goldenrod", alpha=0.7, edgecolor="white") + ax.axvline(np.median(gases), color="black", linestyle="--", + label=f"Median=${np.median(gases):.2f}") + ax.set_xlabel("Gas (USD)") + ax.set_title(f"{label}: Gas cost") + ax.legend(fontsize=8) + + # R² + ax = axes[row, 2] + r2arr = np.array(r2s) + ax.hist(r2arr[np.isfinite(r2arr)], bins=20, color="green", alpha=0.7, + edgecolor="white") + ax.axvline(np.nanmedian(r2arr), color="black", linestyle="--", + label=f"Median={np.nanmedian(r2arr):.3f}") + ax.set_xlabel("R²") + ax.set_title(f"{label}: R²") + ax.legend(fontsize=8) + + # Arb fraction + ax = axes[row, 3] + ax.hist(arb_fracs, bins=20, color="steelblue", alpha=0.7, edgecolor="white") + ax.axvline(np.median(arb_fracs), color="black", linestyle="--", + label=f"Median={np.median(arb_fracs):.2f}") + ax.set_xlabel("Arb fraction") + ax.set_title(f"{label}: Arb fraction") + ax.legend(fontsize=8) + + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, "summary_distributions.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_c_vs_a_scatter(predictions): + """Scatter: Option C vs Option A parameters.""" + pool_ids = sorted(predictions.keys()) + fig, axes = plt.subplots(1, 3, figsize=(15, 5)) + + for ax, metric, label in [ + (axes[0], "cadence", "Cadence (min)"), + (axes[1], "gas", "Gas (USD)"), + (axes[2], "r2", "R²"), + ]: + c_vals = [predictions[p][f"{metric}_c"] for p in pool_ids] + a_vals = [predictions[p][f"{metric}_a"] for p in pool_ids] + ax.scatter(c_vals, a_vals, alpha=0.7, s=30, edgecolors="k", linewidth=0.5) + lo = min(min(c_vals), min(a_vals)) + hi = max(max(c_vals), max(a_vals)) + margin = (hi - lo) * 0.05 + ax.plot([lo - margin, hi + margin], [lo - margin, hi + margin], + "k--", alpha=0.3, linewidth=1) + ax.set_xlabel(f"Option C: {label}") + ax.set_ylabel(f"Option A: {label}") + ax.set_title(f"{label}: C vs A") + + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, "c_vs_a_scatter.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_cadence_gas_by_chain(predictions): + """Scatter: cadence vs gas, colored by chain.""" + pool_ids = sorted(predictions.keys()) + chains = [predictions[p]["chain"] for p in pool_ids] + unique_chains = sorted(set(chains)) + colors = plt.cm.tab10(np.linspace(0, 1, max(len(unique_chains), 1))) + chain_color = {c: colors[i] for i, c in enumerate(unique_chains)} + + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + for ax, method, label in [(axes[0], "c", "Option C"), (axes[1], "a", "Option A")]: + for c in unique_chains: + mask = [i for i, p in enumerate(pool_ids) if predictions[p]["chain"] == c] + cads = [predictions[pool_ids[i]][f"cadence_{method}"] for i in mask] + gases = [predictions[pool_ids[i]][f"gas_{method}"] for i in mask] + ax.scatter(cads, gases, label=c, color=chain_color[c], + alpha=0.7, s=50, edgecolors="k", linewidth=0.5) + ax.set_xlabel("Cadence (minutes)") + ax.set_ylabel("Gas cost (USD)") + ax.set_title(f"{label}: Cadence vs Gas by chain") + ax.legend(fontsize=8) + ax.set_xscale("log") + ax.set_yscale("log") + + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, "cadence_gas_by_chain.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def save_results_json(predictions, option_c_results, joint_ppn, joint_sn): + """Save fitted params as JSON for later use.""" + out = { + "option_c": {}, + "option_a_ppn": { + "bias_cad": joint_ppn["bias_cad"], + "bias_gas": joint_ppn["bias_gas"], + "W_cad": joint_ppn["W_cad"].tolist(), + "W_gas": joint_ppn["W_gas"].tolist(), + "noise_coeffs": joint_ppn["noise_coeffs"].tolist(), + "loss": joint_ppn["loss"], + "init_loss": joint_ppn["init_loss"], + "converged": bool(joint_ppn["converged"]), + "attr_names": joint_ppn["attr_names"], + "pool_ids": joint_ppn["pool_ids"], + }, + "option_a_shared": { + "bias_cad": joint_sn["bias_cad"], + "bias_gas": joint_sn["bias_gas"], + "W_cad": joint_sn["W_cad"].tolist(), + "W_gas": joint_sn["W_gas"].tolist(), + "bias_noise": joint_sn["bias_noise"].tolist(), + "W_noise": joint_sn["W_noise"].tolist(), + "loss": joint_sn["loss"], + "init_loss": joint_sn["init_loss"], + "converged": bool(joint_sn["converged"]), + "attr_names": joint_sn["attr_names"], + "pool_ids": joint_sn["pool_ids"], + }, + } + for pid, r in option_c_results.items(): + out["option_c"][pid] = { + "log_cadence": r["log_cadence"], + "log_gas": r["log_gas"], + "noise_coeffs": r["noise_coeffs"].tolist(), + "loss": r["loss"], + "converged": bool(r["converged"]), + "cadence_minutes": r["cadence_minutes"], + "gas_usd": r["gas_usd"], + "chain": r["chain"], + "fee": r["fee"], + "tokens": r["tokens"], + } + + path = os.path.join(OUTPUT_DIR, "direct_calibration_results.json") + with open(path, "w") as f: + json.dump(out, f, indent=2) + print(f" Saved: {path}") + + +def print_pool_table(predictions, option_c_results): + """Print a summary table of per-pool results.""" + ranked = sorted( + predictions.items(), + key=lambda x: -x[1]["median_tvl"], + ) + + has_rf = any(not np.isnan(p["cadence_rf"]) for _, p in ranked) + + print(f"\n{'='*150}") + header = (f"{'Pool':<24} {'Chain':<10} {'TVL':>12} {'N':>4} " + f"{'Cad_C':>6} {'Gas_C':>7} {'R2_C':>6} " + f"{'Cad_A':>6} {'Gas_A':>7} {'R2_A':>6}") + if has_rf: + header += f" {'Cad_RF':>6} {'Gas_RF':>7} {'R2_RF':>6} {'R2_LOO':>6}" + header += f" {'Arb%_C':>6}" + print(header) + print(f"{'-'*150}") + for pid, p in ranked: + tokens = p["tokens"] + if isinstance(tokens, str): + tok_str = "/".join(t.strip()[:6] for t in tokens.split(",")[:2]) + else: + tok_str = pid[:16] + arb_total_c = p["v_arb_c"] + p["v_noise_c"] + arb_frac = np.median(p["v_arb_c"] / np.maximum(arb_total_c, 1.0)) + if np.isnan(p["cadence_a"]): + a_str = " --- dropped --- " + else: + a_str = f"{p['cadence_a']:>5.1f}m ${p['gas_a']:>5.2f} {p['r2_a']:>6.3f}" + line = (f"{tok_str:<24} {p['chain']:<10} ${p['median_tvl']:>10,.0f} {p['n_obs']:>4} " + f"{p['cadence_c']:>5.1f}m ${p['gas_c']:>5.2f} {p['r2_c']:>6.3f} " + f"{a_str}") + if has_rf: + if np.isnan(p["cadence_rf"]): + line += " --- dropped --- " + else: + line += (f" {p['cadence_rf']:>5.1f}m ${p['gas_rf']:>5.2f} " + f"{p['r2_rf']:>6.3f} {p['r2_rf_loo']:>6.3f}") + line += f" {arb_frac:>5.1%}" + print(line) + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Direct Calibration Pipeline: Training + Top 50 Plots") + print("=" * 70) + + panel, matched = load_and_match() + + # Step 1: Option C + option_c = run_option_c(matched) + + # Step 2: Option A (linear mapping) + joint_ppn, joint_sn = run_option_a(matched, option_c) + + # Step 3: Option RF (random forest 2-stage) + rf_result = run_option_rf(matched, option_c) + + # Step 4: Compute predictions + print("\nComputing per-pool predictions...") + predictions = compute_per_pool_predictions(matched, option_c, joint_ppn, rf_result) + + # Step 5: Print table + print_pool_table(predictions, option_c) + + # Print RF vs A comparison summary + rf_pools = [p for p in predictions if not np.isnan(predictions[p]["r2_rf"])] + if rf_pools: + r2_c = [predictions[p]["r2_c"] for p in rf_pools] + r2_a = [predictions[p]["r2_a"] for p in rf_pools + if not np.isnan(predictions[p]["r2_a"])] + r2_rf = [predictions[p]["r2_rf"] for p in rf_pools] + r2_loo = [predictions[p]["r2_rf_loo"] for p in rf_pools] + print(f"\n--- R² comparison (non-dropped pools) ---") + print(f" Option C: median={np.median(r2_c):.4f} mean={np.mean(r2_c):.4f}") + if r2_a: + print(f" Option A (linear): median={np.median(r2_a):.4f} mean={np.mean(r2_a):.4f}") + print(f" Option RF (train): median={np.median(r2_rf):.4f} mean={np.mean(r2_rf):.4f}") + print(f" Option RF (LOO): median={np.median(r2_loo):.4f} mean={np.mean(r2_loo):.4f}") + + # Step 6: Plots + print("\nGenerating plots...") + os.makedirs(OUTPUT_DIR, exist_ok=True) + plot_top50_pages(predictions, method="c") + plot_top50_pages(predictions, method="a") + plot_top50_pages(predictions, method="rf") + plot_summary(predictions, option_c, joint_ppn) + plot_c_vs_a_scatter(predictions) + plot_cadence_gas_by_chain(predictions) + + # Step 7: Save results + save_results_json(predictions, option_c, joint_ppn, joint_sn) + + print(f"\n{'='*70}") + print(f"Done. Output in: {OUTPUT_DIR}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_structural_top50.py b/scripts/run_structural_top50.py new file mode 100644 index 00000000..b46aecef --- /dev/null +++ b/scripts/run_structural_top50.py @@ -0,0 +1,448 @@ +"""Fit the structural mixture model and plot predicted vs actual for top 50 pools. + +Uses the cached panel (last 90 days), fits with vanilla SVI, then generates +paginated plots showing V_arb + V_noise decomposition and predicted vs actual. +""" + +import json +import os +import sys +from datetime import date, timedelta + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "structural_hierarchical", +) +OUTPUT_JSON = os.path.join(OUTPUT_DIR, "structural_fit.json") +TRAIN_DAYS = 90 +SVI_STEPS = 20_000 +SVI_LR = 1e-3 +NUM_SAMPLES = 1000 +SEED = 42 +TOP_N = 50 + + +def load_and_filter_panel(): + """Load cached panel, filter to 90 days, keep pools with >= 10 obs.""" + panel = pd.read_parquet(PANEL_CACHE) + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=TRAIN_DAYS) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + + if "log_tvl_lag1" not in panel.columns: + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel.groupby("pool_id").size() + valid = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid)].copy() + + print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " + f"{cutoff} to {max_date}") + return panel + + +def fit_structural(panel): + """Run SVI on the structural mixture model.""" + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + from quantammsim.noise_calibration.covariate_encoding import ( + encode_covariates_structural, + ) + from quantammsim.noise_calibration.model import structural_noise_model + from quantammsim.noise_calibration.inference import run_svi + from quantammsim.noise_calibration.postprocessing import ( + check_convergence, extract_structural_params, + ) + from quantammsim.noise_calibration.output import generate_output_json + + import numpyro + numpyro.enable_x64() + + # Load gas costs for mainnet from CSV if available + gas_csv = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "formula_vs_real", "mainnet_gas_cost_daily.csv", + ) + gas_arr = None + if os.path.exists(gas_csv): + gas_df = pd.read_csv(gas_csv) + # CSV has columns: unix (ms timestamp), USD (gas cost) + gas_df["date"] = pd.to_datetime(gas_df["unix"], unit="ms").dt.date + gas_lookup = dict(zip(gas_df["date"], gas_df["USD"])) + + # Build per-observation gas array + gas_vals = [] + for _, row in panel.iterrows(): + d = row["date"] + if not isinstance(d, date): + d = pd.Timestamp(d).date() + chain = row["chain"] + if chain == "MAINNET" and d in gas_lookup: + gas_vals.append(gas_lookup[d]) + elif chain == "MAINNET": + gas_vals.append(1.0) # median fallback + else: + # L2 chains: ~$0.005 + from quantammsim.noise_calibration.constants import GAS_COSTS + gas_vals.append(GAS_COSTS.get(chain, 0.005)) + gas_arr = np.array(gas_vals, dtype=np.float64) + print(f"Gas costs: loaded ({len(gas_lookup)} mainnet days from CSV)") + else: + print("Gas costs: using defaults (no mainnet CSV)") + + data = encode_covariates_structural(panel, gas=gas_arr) + + print(f"\nFitting structural model: {SVI_STEPS} SVI steps, lr={SVI_LR}") + samples, elbo_losses = run_svi( + data, + num_steps=SVI_STEPS, + lr=SVI_LR, + seed=SEED, + num_samples=NUM_SAMPLES, + model_fn=structural_noise_model, + ) + convergence = check_convergence(elbo_losses, method="svi") + + pool_params = extract_structural_params(samples, data) + + # Save output JSON + os.makedirs(OUTPUT_DIR, exist_ok=True) + inference_config = { + "method": "svi", "svi_steps": SVI_STEPS, + "svi_lr": SVI_LR, "num_samples": NUM_SAMPLES, + } + generate_output_json( + pool_params, samples, data, convergence, + OUTPUT_JSON, inference_config, + ) + + return samples, data, pool_params, elbo_losses + + +def compute_predictions(samples, data, panel): + """Compute per-observation predicted V_arb and V_noise.""" + from quantammsim.noise_calibration.formula_arb import ( + formula_arb_volume_daily_jax, + ) + import jax.numpy as jnp + + sample_dict = samples + agg_fn = np.median + + # Cadence parameters + alpha_0 = agg_fn(np.array(sample_dict["alpha_0"])) + alpha_chain = agg_fn(np.array(sample_dict["alpha_chain"]), axis=0) + alpha_tier = agg_fn(np.array(sample_dict["alpha_tier"]), axis=0) + alpha_tvl = agg_fn(np.array(sample_dict["alpha_tvl"])) + + # Hierarchical noise: reconstruct theta + B = agg_fn(np.array(sample_dict["B"]), axis=0) + eta = agg_fn(np.array(sample_dict["eta"]), axis=0) + sigma_theta = agg_fn(np.array(sample_dict["sigma_theta"]), axis=0) + L_Omega = agg_fn(np.array(sample_dict["L_Omega"]), axis=0) + + pool_idx = np.array(data["pool_idx"]) + X_pool = np.array(data["X_pool"]) + x_obs = np.array(data["x_obs"]) + chain_idx = np.array(data["chain_idx"]) + tier_idx = np.array(data["tier_idx"]) + sigma_daily = np.array(data["sigma_daily"]) + lag_log_tvl = np.array(data["lag_log_tvl"]) + fee = np.array(data["fee"]) + gas = np.array(data["gas"]) + + # Per-pool cadence + padded_chain = np.concatenate([[0.0], alpha_chain]) + padded_tier = np.concatenate([[0.0], alpha_tier]) + + N_pools = data["N_pools"] + pool_log_cadence = np.zeros(N_pools) + for p in range(N_pools): + pool_log_cadence[p] = ( + alpha_0 + + padded_chain[chain_idx[p]] + + padded_tier[tier_idx[p]] + + alpha_tvl * np.median(lag_log_tvl[pool_idx == p]) + ) + + # Per-obs V_arb + log_cad_obs = pool_log_cadence[pool_idx] + cadence_obs = np.exp(np.clip(log_cad_obs, -2.0, 6.0)) + tvl_obs = np.exp(lag_log_tvl) + + V_arb = np.array(formula_arb_volume_daily_jax( + jnp.array(sigma_daily), jnp.array(tvl_obs), + jnp.array(fee), jnp.array(gas), jnp.array(cadence_obs), + )) + + # Per-pool theta from hierarchical model + L_Sigma = np.diag(sigma_theta) @ L_Omega + theta = X_pool @ B.T + eta @ L_Sigma.T # (N_pools, K_obs_coeff) + + # Per-obs V_noise + log_V_noise = np.sum(theta[pool_idx] * x_obs, axis=1) + V_noise = np.exp(log_V_noise) + + # Predicted total + V_total_pred = V_arb + V_noise + log_V_pred = np.log(np.maximum(V_total_pred, 1e-6)) + + return V_arb, V_noise, V_total_pred, log_V_pred, cadence_obs + + +def plot_top50(panel, data, pool_params, V_arb, V_noise, log_V_pred): + """Plot top 50 pools by median TVL.""" + pool_meta = data["pool_meta"] + pool_ids = data["pool_ids"] + pool_idx = np.array(data["pool_idx"]) + y_obs = np.array(data["y_obs"]) + + # Rank pools by median TVL + pool_tvl = {} + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + pool_tvl[pid] = np.median(np.exp(np.array(data["lag_log_tvl"])[mask])) + + ranked = sorted(pool_tvl.items(), key=lambda x: -x[1])[:TOP_N] + + # Build param lookup + param_lookup = {p["pool_id"]: p for p in pool_params} + + per_page = 10 + n_pages = (len(ranked) + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, len(ranked)) + page_pools = ranked[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(16, 4.5 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + + for idx, (pid, median_tvl) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + p_idx = pool_ids.index(pid) + mask = pool_idx == p_idx + + pp = panel[panel["pool_id"] == pid].sort_values("date") + dates = pd.to_datetime(pp["date"].values) + actual_vol = np.exp(y_obs[mask]) + pred_arb = V_arb[mask] + pred_noise = V_noise[mask] + pred_total = pred_arb + pred_noise + + # R2 + actual_log = y_obs[mask] + pred_log = log_V_pred[mask] + ss_res = np.sum((actual_log - pred_log) ** 2) + ss_tot = np.sum((actual_log - actual_log.mean()) ** 2) + r2 = 1 - ss_res / ss_tot if ss_tot > 0 else float("nan") + + # Arb fraction + arb_frac = np.median(pred_arb / np.maximum(pred_total, 1.0)) + + # Plot + ax.fill_between(dates, 0, pred_arb, alpha=0.3, color="orangered", + label="V_arb (LVR)") + ax.fill_between(dates, pred_arb, pred_total, alpha=0.3, + color="steelblue", label="V_noise (hier.)") + ax.plot(dates, actual_vol, "k-", linewidth=0.8, alpha=0.7, + label="Actual") + ax.plot(dates, pred_total, "--", color="purple", linewidth=0.8, + alpha=0.7, label="Predicted total") + + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + meta = pool_meta[pool_meta["pool_id"] == pid] + if len(meta) > 0: + m = meta.iloc[0] + tokens = m["tokens"] + if isinstance(tokens, str): + tokens = tokens.split(",") + tok_str = "/".join(str(t)[:8] for t in tokens[:2]) + chain = str(m["chain"]) + else: + tok_str = pid[:16] + chain = "?" + + params = param_lookup.get(pid, {}) + arb_freq = params.get("arb_frequency", "?") + + ax.set_title( + f"{tok_str} ({chain})\n" + f"TVL ${median_tvl:,.0f} | R\u00b2={r2:.3f} " + f"arb_freq={arb_freq}min arb_frac={arb_frac:.1%} " + f"n={mask.sum()}", + fontsize=8, + ) + ax.legend(fontsize=6, loc="upper right") + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + f"Structural mixture model: V_arb + V_noise decomposition " + f"— page {page + 1}/{n_pages} " + f"(top {TOP_N} by median TVL, 90d window)", + fontsize=11, + ) + fig.tight_layout() + out = os.path.join(OUTPUT_DIR, f"structural_top50_page{page + 1}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_elbo(elbo_losses): + """Plot ELBO convergence.""" + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + ax = axes[0] + ax.plot(elbo_losses, alpha=0.3, color="steelblue", linewidth=0.5) + window = min(100, len(elbo_losses) // 10) + if window > 1: + smoothed = pd.Series(elbo_losses).rolling(window).mean().values + ax.plot(smoothed, color="red", linewidth=1.5, label=f"Rolling {window}") + ax.legend() + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence") + + ax = axes[1] + start = len(elbo_losses) * 4 // 5 + ax.plot(range(start, len(elbo_losses)), elbo_losses[start:], + color="steelblue", linewidth=0.8) + ax.set_xlabel("Step") + ax.set_ylabel("ELBO loss") + ax.set_title("ELBO convergence (last 20%)") + + plt.tight_layout() + out = os.path.join(OUTPUT_DIR, "elbo_convergence.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {out}") + + +def plot_summary(data, pool_params, V_arb, V_noise, log_V_pred): + """Summary plots: arb frequency distribution, arb fraction, R2.""" + pool_idx = np.array(data["pool_idx"]) + y_obs = np.array(data["y_obs"]) + pool_ids = data["pool_ids"] + + fig, axes = plt.subplots(1, 3, figsize=(16, 5)) + + # 1. Arb frequency histogram + ax = axes[0] + freqs = [p["arb_frequency"] for p in pool_params] + ax.hist(freqs, bins=range(0, 62, 2), color="orangered", alpha=0.7, + edgecolor="white") + ax.set_xlabel("Arb frequency (minutes)") + ax.set_ylabel("Count") + ax.set_title(f"Arb frequency distribution (n={len(freqs)})") + ax.axvline(np.median(freqs), color="black", linestyle="--", + label=f"Median={np.median(freqs):.0f}min") + ax.legend() + + # 2. Arb fraction per pool + ax = axes[1] + arb_fracs = [] + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + total = V_arb[mask] + V_noise[mask] + arb_fracs.append(np.median(V_arb[mask] / np.maximum(total, 1.0))) + ax.hist(arb_fracs, bins=30, color="steelblue", alpha=0.7, edgecolor="white") + ax.set_xlabel("Median arb fraction") + ax.set_ylabel("Count") + ax.set_title("Arb fraction distribution") + ax.axvline(np.median(arb_fracs), color="black", linestyle="--", + label=f"Median={np.median(arb_fracs):.2f}") + ax.legend() + + # 3. Per-pool R2 + ax = axes[2] + r2_vals = [] + for i, pid in enumerate(pool_ids): + mask = pool_idx == i + actual = y_obs[mask] + pred = log_V_pred[mask] + ss_res = np.sum((actual - pred) ** 2) + ss_tot = np.sum((actual - actual.mean()) ** 2) + r2_vals.append(1 - ss_res / ss_tot if ss_tot > 0 else float("nan")) + r2_vals = np.array(r2_vals) + ax.hist(r2_vals[np.isfinite(r2_vals)], bins=30, color="green", alpha=0.7, + edgecolor="white") + ax.set_xlabel("R²") + ax.set_ylabel("Count") + ax.set_title("Per-pool R² distribution") + ax.axvline(np.nanmedian(r2_vals), color="black", linestyle="--", + label=f"Median={np.nanmedian(r2_vals):.3f}") + ax.legend() + + plt.tight_layout() + out = os.path.join(OUTPUT_DIR, "structural_summary.png") + plt.savefig(out, dpi=150, bbox_inches="tight") + plt.close() + print(f" Saved: {out}") + + +def main(): + print("=" * 70) + print("Structural Mixture Model: Fit + Top 50 Plots") + print("=" * 70) + + panel = load_and_filter_panel() + samples, data, pool_params, elbo_losses = fit_structural(panel) + + print("\nComputing predictions...") + V_arb, V_noise, V_total, log_V_pred, cadence = compute_predictions( + samples, data, panel, + ) + print(f" V_arb median: ${np.median(V_arb):,.0f}") + print(f" V_noise median: ${np.median(V_noise):,.0f}") + print(f" Arb fraction (median pool): {np.median(V_arb / np.maximum(V_total, 1)):.2%}") + + print("\nGenerating plots...") + os.makedirs(OUTPUT_DIR, exist_ok=True) + plot_elbo(elbo_losses) + plot_summary(data, pool_params, V_arb, V_noise, log_V_pred) + plot_top50(panel, data, pool_params, V_arb, V_noise, log_V_pred) + + # Summary stats + print(f"\n{'=' * 70}") + print(f"Done. Output in: {OUTPUT_DIR}") + arb_freqs = [p["arb_frequency"] for p in pool_params] + print(f" Arb frequency: median={np.median(arb_freqs):.0f}min, " + f"range=[{np.min(arb_freqs)}, {np.max(arb_freqs)}]") + print(f" JSON: {OUTPUT_JSON}") + + +if __name__ == "__main__": + main() From 8c70eb29a12cce06441dc5790e557144d180887c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 16:50:23 +0000 Subject: [PATCH 029/115] fix: drop hardcoded float64 dtype from dynamic_inputs defaults These singleton defaults are always recast to the caller's dtype by materialize_dynamic_inputs. Hardcoding float64 caused 200+ warnings when running in float32 mode (e.g. BFGS with compute_dtype=float32). --- quantammsim/core_simulator/dynamic_inputs.py | 22 +++++++++----------- 1 file changed, 10 insertions(+), 12 deletions(-) diff --git a/quantammsim/core_simulator/dynamic_inputs.py b/quantammsim/core_simulator/dynamic_inputs.py index 598d3f71..367363ea 100644 --- a/quantammsim/core_simulator/dynamic_inputs.py +++ b/quantammsim/core_simulator/dynamic_inputs.py @@ -77,14 +77,12 @@ def empty_dynamic_input_arrays() -> DynamicInputArrays: """Create a canonical empty bundle.""" return DynamicInputArrays( trades=None, - fees=jnp.zeros((1,), dtype=jnp.float64), - gas_cost=jnp.zeros((1,), dtype=jnp.float64), - arb_fees=jnp.zeros((1,), dtype=jnp.float64), - lp_supply=jnp.ones((1,), dtype=jnp.float64), + fees=jnp.zeros((1,)), + gas_cost=jnp.zeros((1,)), + arb_fees=jnp.zeros((1,)), + lp_supply=jnp.ones((1,)), # Columns: has_event, target_price_ratio, end_step, start_price_ratio_override - reclamm_price_ratio_updates=jnp.array( - [[0.0, 0.0, 0.0, jnp.nan]], dtype=jnp.float64 - ), + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), ) @@ -100,22 +98,22 @@ def resolve_dynamic_input_components( "fees": ( arrays.fees if dynamic_input_flags["has_dynamic_fees"] - else jnp.asarray([static_dict["fees"]], dtype=jnp.float64) + else jnp.asarray([static_dict["fees"]]) ), "gas_cost": ( arrays.gas_cost if dynamic_input_flags["has_dynamic_gas_cost"] - else jnp.asarray([static_dict["gas_cost"]], dtype=jnp.float64) + else jnp.asarray([static_dict["gas_cost"]]) ), "arb_fees": ( arrays.arb_fees if dynamic_input_flags["has_dynamic_arb_fees"] - else jnp.asarray([static_dict["arb_fees"]], dtype=jnp.float64) + else jnp.asarray([static_dict["arb_fees"]]) ), "lp_supply": ( arrays.lp_supply if dynamic_input_flags["has_lp_supply"] - else jnp.ones((1,), dtype=jnp.float64) + else jnp.ones((1,)) ), "reclamm_price_ratio_updates": ( arrays.reclamm_price_ratio_updates @@ -150,7 +148,7 @@ def materialize_dynamic_inputs( static_dict: dict, scan_len: int, do_trades: bool, - dtype=jnp.float64, + dtype=None, ) -> DynamicInputArrays: """Resolve and broadcast dynamic inputs for a specific scan length.""" if dynamic_input_flags is None and dynamic_inputs is not None: From d05c4940947e96a0dcc39e5b931cfa2f4e237a98 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 18:22:29 +0000 Subject: [PATCH 030/115] fix: update LP supply tests to use DynamicInputArrays and test data dates - test_lp_supply_through_pool_class: use DynamicInputArrays bundle instead of old positional-args signature - test_lp_supply_e2e_do_run_on_historic_data: use TEST_DATA_DIR and date range within test data coverage (2023-01-01 to 2023-01-15) - test_noise_trade_does_not_affect_virtual_balances: carry/input_list already fixed in previous commit --- tests/pools/reCLAMM/test_reclamm_reserves.py | 38 +++++++++++++------- 1 file changed, 26 insertions(+), 12 deletions(-) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index ef70c9ec..bdc5e9fb 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -1290,22 +1290,34 @@ def test_lp_supply_through_pool_class(self): start_index = jnp.array([0, 0]) - # Build dynamic input arrays - fees_arr = jnp.full(n_steps, 0.003) - arb_thresh_arr = jnp.zeros(n_steps) - arb_fees_arr = jnp.zeros(n_steps) - trade_arr = jnp.zeros((n_steps, 2)) + from quantammsim.core_simulator.dynamic_inputs import DynamicInputArrays lp_supply = jnp.concatenate([jnp.ones(10), 2.0 * jnp.ones(10)]) + di_with_lp = DynamicInputArrays( + trades=None, + fees=jnp.full(n_steps, 0.003), + gas_cost=jnp.zeros(n_steps), + arb_fees=jnp.zeros(n_steps), + lp_supply=lp_supply, + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), + ) + di_without_lp = DynamicInputArrays( + trades=None, + fees=jnp.full(n_steps, 0.003), + gas_cost=jnp.zeros(n_steps), + arb_fees=jnp.zeros(n_steps), + lp_supply=jnp.ones(n_steps), + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), + ) + res_with_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( params, run_fingerprint, prices, start_index, - fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, - lp_supply_array=lp_supply, + dynamic_inputs=di_with_lp, ) res_without_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( params, run_fingerprint, prices, start_index, - fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, + dynamic_inputs=di_without_lp, ) # First 10 steps identical, then diverge @@ -1363,8 +1375,8 @@ def test_lp_supply_e2e_do_run_on_historic_data(self): fp = { "rule": "reclamm", "tokens": ["ETH", "USDC"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2024-06-15 00:00:00", + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-15 00:00:00", "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": 0.003, @@ -1379,12 +1391,13 @@ def test_lp_supply_e2e_do_run_on_historic_data(self): result_base = do_run_on_historic_data( run_fingerprint={**fp}, params={**params}, + root=TEST_DATA_DIR, ) # LP supply doubles halfway through the period # unix column must be in milliseconds (matches windowing_utils convention) - start_unix_ms = int(pd.Timestamp("2024-06-01").timestamp() * 1000) - mid_unix_ms = int(pd.Timestamp("2024-06-08").timestamp() * 1000) + start_unix_ms = int(pd.Timestamp("2023-01-01").timestamp() * 1000) + mid_unix_ms = int(pd.Timestamp("2023-01-08").timestamp() * 1000) lp_supply_df = pd.DataFrame({ "unix": [start_unix_ms, mid_unix_ms], "lp_supply": [1.0, 2.0], @@ -1394,6 +1407,7 @@ def test_lp_supply_e2e_do_run_on_historic_data(self): run_fingerprint={**fp}, params={**params}, lp_supply_df=lp_supply_df, + root=TEST_DATA_DIR, ) # Final values should differ — doubling LP supply changes pool dynamics From 9779dcad3db322206a94e1145e001dd979602fbf Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 18:22:29 +0000 Subject: [PATCH 031/115] fix: update LP supply tests to use DynamicInputArrays and test data dates - test_lp_supply_through_pool_class: use DynamicInputArrays bundle instead of old positional-args signature - test_lp_supply_e2e_do_run_on_historic_data: use TEST_DATA_DIR and date range within test data coverage (2023-01-01 to 2023-01-15) - test_noise_trade_does_not_affect_virtual_balances: carry/input_list already fixed in previous commit --- tests/pools/reCLAMM/test_reclamm_reserves.py | 43 +++++++++++++------- 1 file changed, 29 insertions(+), 14 deletions(-) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index ef70c9ec..c7951765 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -1290,22 +1290,34 @@ def test_lp_supply_through_pool_class(self): start_index = jnp.array([0, 0]) - # Build dynamic input arrays - fees_arr = jnp.full(n_steps, 0.003) - arb_thresh_arr = jnp.zeros(n_steps) - arb_fees_arr = jnp.zeros(n_steps) - trade_arr = jnp.zeros((n_steps, 2)) + from quantammsim.core_simulator.dynamic_inputs import DynamicInputArrays lp_supply = jnp.concatenate([jnp.ones(10), 2.0 * jnp.ones(10)]) + di_with_lp = DynamicInputArrays( + trades=None, + fees=jnp.full(n_steps, 0.003), + gas_cost=jnp.zeros(n_steps), + arb_fees=jnp.zeros(n_steps), + lp_supply=lp_supply, + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), + ) + di_without_lp = DynamicInputArrays( + trades=None, + fees=jnp.full(n_steps, 0.003), + gas_cost=jnp.zeros(n_steps), + arb_fees=jnp.zeros(n_steps), + lp_supply=jnp.ones(n_steps), + reclamm_price_ratio_updates=jnp.array([[0.0, 0.0, 0.0, jnp.nan]]), + ) + res_with_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( params, run_fingerprint, prices, start_index, - fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, - lp_supply_array=lp_supply, + dynamic_inputs=di_with_lp, ) res_without_lp, _ = pool.calculate_reserves_and_fee_revenue_with_dynamic_inputs( params, run_fingerprint, prices, start_index, - fees_arr, arb_thresh_arr, arb_fees_arr, trade_arr, + dynamic_inputs=di_without_lp, ) # First 10 steps identical, then diverge @@ -1356,15 +1368,16 @@ def test_lp_supply_with_fee_revenue(self): ) def test_lp_supply_e2e_do_run_on_historic_data(self): - """End-to-end: lp_supply_df flows through do_run_on_historic_data.""" + """End-to-end: lp_supply flows through do_run_on_historic_data via DynamicInputFrames.""" import pandas as pd from quantammsim.runners.jax_runners import do_run_on_historic_data + from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames fp = { "rule": "reclamm", "tokens": ["ETH", "USDC"], - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2024-06-15 00:00:00", + "startDateString": "2023-01-01 00:00:00", + "endDateString": "2023-01-15 00:00:00", "initial_pool_value": 1_000_000.0, "do_arb": True, "fees": 0.003, @@ -1379,12 +1392,13 @@ def test_lp_supply_e2e_do_run_on_historic_data(self): result_base = do_run_on_historic_data( run_fingerprint={**fp}, params={**params}, + root=TEST_DATA_DIR, ) # LP supply doubles halfway through the period # unix column must be in milliseconds (matches windowing_utils convention) - start_unix_ms = int(pd.Timestamp("2024-06-01").timestamp() * 1000) - mid_unix_ms = int(pd.Timestamp("2024-06-08").timestamp() * 1000) + start_unix_ms = int(pd.Timestamp("2023-01-01").timestamp() * 1000) + mid_unix_ms = int(pd.Timestamp("2023-01-08").timestamp() * 1000) lp_supply_df = pd.DataFrame({ "unix": [start_unix_ms, mid_unix_ms], "lp_supply": [1.0, 2.0], @@ -1393,7 +1407,8 @@ def test_lp_supply_e2e_do_run_on_historic_data(self): result_lp = do_run_on_historic_data( run_fingerprint={**fp}, params={**params}, - lp_supply_df=lp_supply_df, + dynamic_input_frames=DynamicInputFrames(lp_supply=lp_supply_df), + root=TEST_DATA_DIR, ) # Final values should differ — doubling LP supply changes pool dynamics From 933b653bbed3cd6a74925759fd93c798fc9428ec Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 00:48:46 +0000 Subject: [PATCH 032/115] test: add pinned regression tests for calibration pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add 25 numerical regression tests that pin exact values computed from synthetic fixtures. These protect against silent computation errors during refactoring — existing tests only check shapes and signs. Covers: grid interpolation (knot exactness, midpoint values, monotonicity, differentiability), loss function (pinned value + gradient at known params), noise volume, per-pool fit convergence (loss, cadence), joint fit (both noise modes, predict_new_pool, warm start), pack/unpack roundtrips. --- tests/calibration/test_regression_pins.py | 449 ++++++++++++++++++++++ 1 file changed, 449 insertions(+) create mode 100644 tests/calibration/test_regression_pins.py diff --git a/tests/calibration/test_regression_pins.py b/tests/calibration/test_regression_pins.py new file mode 100644 index 00000000..877c23f5 --- /dev/null +++ b/tests/calibration/test_regression_pins.py @@ -0,0 +1,449 @@ +"""Pinned numerical regression tests for the calibration pipeline. + +These tests pin exact numerical values computed from the synthetic fixtures. +They protect against silent computation errors during refactoring — a test +that checks only shapes/signs would still pass if e.g. an index is off by +one in unpack, or a sign is flipped in regularization. + +All pinned values were computed with: + - Python 3.9, JAX 0.4.30, numpy seed 42 + - Synthetic fixtures from conftest.py (N_DAYS=15, 2 pools) +""" + +import os +import tempfile + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd +import pytest + +from tests.calibration.conftest import ( + CADENCES, + GAS_COSTS, + K_OBS, + N_DAYS, + POOL_IDS_FULL, + POOL_PREFIXES, +) + +from quantammsim.calibration.grid_interpolation import ( + interpolate_pool_daily, + precompute_pool_coeffs_daily, +) +from quantammsim.calibration.loss import noise_volume, pack_params, pool_loss +from quantammsim.calibration.per_pool_fit import fit_all_pools, fit_single_pool +from quantammsim.calibration.pool_data import build_x_obs, match_grids_to_panel + + +# ── Helpers ──────────────────────────────────────────────────────────────── + + +@pytest.fixture +def matched_data(synthetic_daily_grid, synthetic_panel): + """Build matched data dict by writing temp parquets for both pools.""" + tmpdir = tempfile.mkdtemp() + for prefix in POOL_PREFIXES: + path = os.path.join(tmpdir, f"{prefix}_daily.parquet") + synthetic_daily_grid.to_parquet(path) + matched = match_grids_to_panel(tmpdir, synthetic_panel) + yield matched + import shutil + shutil.rmtree(tmpdir) + + +@pytest.fixture +def pool0_inputs(matched_data): + """x_obs, y_obs, day_indices, coeffs for pool 0.""" + entry = matched_data[POOL_PREFIXES[0]] + panel = entry["panel"] + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + day_indices = np.array(entry["day_indices"]) + return entry["coeffs"], x_obs, y_obs, day_indices + + +def _known_params(): + """Standard test params: cadence=12, gas=$1, noise intercept=8.""" + noise_coeffs = np.zeros(K_OBS) + noise_coeffs[0] = 8.0 + return pack_params(np.log(12.0), np.log(1.0), jnp.array(noise_coeffs)) + + +# ── Grid interpolation pins ─────────────────────────────────────────────── + + +class TestInterpolationPins: + """Verify interpolation exactness at grid knot points.""" + + def test_interpolation_exact_at_all_knots(self, synthetic_pool_coeffs): + """Interpolation at grid knot points must exactly reproduce grid values.""" + coeffs = synthetic_pool_coeffs + for ci, cad in enumerate(CADENCES): + for gi, gas in enumerate(GAS_COSTS): + log_cad = jnp.log(cad) + v_arb = interpolate_pool_daily(coeffs, log_cad, jnp.array(gas)) + grid_vals = coeffs.values[ci, gi, :] + np.testing.assert_allclose( + v_arb, grid_vals, atol=1e-4, + err_msg=f"Mismatch at cad={cad}, gas={gas}", + ) + + def test_interpolation_midpoint_value(self, synthetic_pool_coeffs): + """Pin interpolated value at a known mid-grid point.""" + v_arb = interpolate_pool_daily( + synthetic_pool_coeffs, jnp.log(6.0), jnp.array(0.5) + ) + # Pinned from JAX 0.4.30, seed 42 + assert v_arb.shape == (N_DAYS,) + np.testing.assert_allclose(float(v_arb[0]), 6579.6309, rtol=1e-4) + np.testing.assert_allclose(float(jnp.mean(v_arb)), 6621.3186, rtol=1e-4) + + def test_interpolation_monotone_in_cadence(self, synthetic_pool_coeffs): + """V_arb should decrease as cadence increases (at fixed gas).""" + coeffs = synthetic_pool_coeffs + gas = jnp.array(1.0) + cads = [1.0, 6.0, 12.0, 30.0, 60.0] + means = [ + float(jnp.mean(interpolate_pool_daily(coeffs, jnp.log(c), gas))) + for c in cads + ] + for i in range(len(means) - 1): + assert means[i] > means[i + 1], ( + f"V_arb not decreasing: cad={cads[i]}->{cads[i+1]}, " + f"mean={means[i]:.1f}->{means[i+1]:.1f}" + ) + + def test_interpolation_monotone_in_gas(self, synthetic_pool_coeffs): + """V_arb should decrease as gas cost increases (at fixed cadence).""" + coeffs = synthetic_pool_coeffs + log_cad = jnp.log(12.0) + gases = [0.0, 0.5, 1.0, 3.0, 5.0] + means = [ + float(jnp.mean(interpolate_pool_daily(coeffs, log_cad, jnp.array(g)))) + for g in gases + ] + for i in range(len(means) - 1): + assert means[i] > means[i + 1], ( + f"V_arb not decreasing: gas={gases[i]}->{gases[i+1]}, " + f"mean={means[i]:.1f}->{means[i+1]:.1f}" + ) + + def test_interpolation_differentiable(self, synthetic_pool_coeffs): + """Gradient of interpolated V_arb w.r.t. log_cadence must be finite.""" + coeffs = synthetic_pool_coeffs + + def f(log_cad): + return jnp.sum(interpolate_pool_daily(coeffs, log_cad, jnp.array(1.0))) + + grad_val = jax.grad(f)(jnp.log(12.0)) + assert jnp.isfinite(grad_val), f"Non-finite gradient: {grad_val}" + # Gradient should be negative (more cadence → less arb) + assert float(grad_val) < 0, f"Expected negative gradient, got {grad_val}" + + +# ── Loss function pins ───────────────────────────────────────────────────── + + +class TestLossPins: + """Pin exact loss values and gradients at known parameter points.""" + + def test_loss_value_pinned(self, synthetic_pool_coeffs, pool0_inputs): + """Pin the exact loss value at known params on synthetic data.""" + coeffs, x_obs, _, day_indices = pool0_inputs + params = _known_params() + y_obs = jnp.ones(x_obs.shape[0]) * 9.0 + day_indices_j = jnp.arange(x_obs.shape[0]) % N_DAYS + + loss = pool_loss(params, coeffs, jnp.array(x_obs), y_obs, day_indices_j) + # Pinned: 0.001726984975292 (JAX 0.4.30, seed 42) + np.testing.assert_allclose(float(loss), 0.001727, rtol=1e-3) + + def test_gradient_pinned(self, synthetic_pool_coeffs, pool0_inputs): + """Pin gradient values at known params.""" + coeffs, x_obs, _, day_indices = pool0_inputs + params = _known_params() + y_obs = jnp.ones(x_obs.shape[0]) * 9.0 + day_indices_j = jnp.arange(x_obs.shape[0]) % N_DAYS + + grad_fn = jax.grad(pool_loss) + grad = grad_fn(params, coeffs, jnp.array(x_obs), y_obs, day_indices_j) + grad_np = np.array(grad) + + # All gradients must be finite + assert np.all(np.isfinite(grad_np)), f"Non-finite gradients: {grad_np}" + + # Pin signs of key gradient components + # grad[0] = d_loss/d_log_cadence (negative: increasing cadence decreases V_arb, + # pushing log(V_arb + V_noise) away from y_obs=9.0) + assert grad_np[0] < 0, f"Expected negative cadence grad, got {grad_np[0]}" + # grad[1] = d_loss/d_log_gas (negative: same effect via gas) + assert grad_np[1] < 0, f"Expected negative gas grad, got {grad_np[1]}" + + # Pin magnitudes (rtol=0.01 to allow platform variance) + expected_grad = np.array([ + -0.000223, -0.000362, -0.000264, -0.003138, + 0.001071, 0.012854, 0.018232, -0.006220, + 0.013421, -0.016786, + ]) + np.testing.assert_allclose(grad_np, expected_grad, rtol=0.05, atol=1e-5) + + def test_loss_increases_with_bad_params(self, synthetic_pool_coeffs, pool0_inputs): + """Loss with wildly wrong noise intercept >> loss with good params.""" + coeffs, x_obs, _, _ = pool0_inputs + y_obs = jnp.ones(x_obs.shape[0]) * 9.0 + day_indices_j = jnp.arange(x_obs.shape[0]) % N_DAYS + x_obs_j = jnp.array(x_obs) + + params_good = _known_params() + noise_bad = np.zeros(K_OBS) + noise_bad[0] = 20.0 + params_bad = pack_params(np.log(12.0), np.log(1.0), jnp.array(noise_bad)) + + loss_good = float(pool_loss(params_good, coeffs, x_obs_j, y_obs, day_indices_j)) + loss_bad = float(pool_loss(params_bad, coeffs, x_obs_j, y_obs, day_indices_j)) + + assert loss_bad > 100.0, f"Expected loss_bad > 100, got {loss_bad}" + assert loss_bad > loss_good * 1000, "Bad params should be >1000x worse" + + +# ── Noise volume pins ────────────────────────────────────────────────────── + + +class TestNoiseVolumePins: + def test_intercept_only_equals_exp(self, synthetic_x_obs): + """With intercept-only noise coeffs, V_noise = exp(intercept) exactly.""" + coeffs = np.zeros(K_OBS) + coeffs[0] = 8.0 + v_noise = noise_volume(jnp.array(coeffs), jnp.array(synthetic_x_obs)) + # x_obs column 0 is all 1.0 (intercept), so x_obs @ coeffs = 8.0 for all obs + np.testing.assert_allclose(v_noise, np.exp(8.0), rtol=1e-6) + + def test_tvl_coeff_creates_variation(self, synthetic_x_obs): + """With nonzero TVL coeff, V_noise varies across observations.""" + coeffs = np.zeros(K_OBS) + coeffs[0] = 5.0 + coeffs[1] = 1.0 # TVL coefficient + v_noise = noise_volume(jnp.array(coeffs), jnp.array(synthetic_x_obs)) + assert float(jnp.std(v_noise)) > 0, "Expected variation from TVL coeff" + + +# ── Per-pool fit pins ────────────────────────────────────────────────────── + + +class TestPerPoolFitPins: + """Pin per-pool optimizer convergence on synthetic data.""" + + def test_fit_single_pool_converges(self, pool0_inputs): + """fit_single_pool should converge on synthetic data.""" + coeffs, x_obs, y_obs, day_indices = pool0_inputs + result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + assert result["converged"], "fit_single_pool did not converge" + + def test_fit_single_pool_loss_pinned(self, pool0_inputs): + """Pin the converged loss value.""" + coeffs, x_obs, y_obs, day_indices = pool0_inputs + result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + # Pinned: 0.0723 (JAX 0.4.30, seed 42) + np.testing.assert_allclose(result["loss"], 0.0723, rtol=0.05) + + def test_fit_single_pool_cadence_pinned(self, pool0_inputs): + """Pin the converged cadence — should find ~1.27 min on synthetic data.""" + coeffs, x_obs, y_obs, day_indices = pool0_inputs + result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + # Pinned: 1.266 minutes + np.testing.assert_allclose(result["cadence_minutes"], 1.27, rtol=0.1) + # Cadence must be in valid range + assert 1.0 <= result["cadence_minutes"] <= 60.0 + + def test_fit_single_pool_loss_lower_than_init(self, pool0_inputs): + """Fitted loss must be lower than loss at initial guess.""" + from quantammsim.calibration.per_pool_fit import make_initial_guess + + coeffs, x_obs, y_obs, day_indices = pool0_inputs + init = make_initial_guess(x_obs, y_obs) + init_loss = float( + pool_loss( + jnp.array(init), + coeffs, + jnp.array(x_obs), + jnp.array(y_obs), + jnp.array(day_indices), + ) + ) + result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + assert result["loss"] < init_loss, ( + f"Fitted loss {result['loss']:.6f} >= init loss {init_loss:.6f}" + ) + + def test_fit_all_pools_returns_all(self, matched_data): + """fit_all_pools returns results for every matched pool.""" + results = fit_all_pools(matched_data) + assert set(results.keys()) == set(matched_data.keys()) + for pid, r in results.items(): + assert "loss" in r + assert "log_cadence" in r + assert "noise_coeffs" in r + assert len(r["noise_coeffs"]) == K_OBS + + +# ── Joint fit pins ───────────────────────────────────────────────────────── + + +class TestJointFitPins: + """Pin joint optimization behavior on synthetic data.""" + + def test_joint_ppn_loss_decreases(self, matched_data): + """Joint per_pool_noise loss must decrease from initialization.""" + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=100) + assert result["loss"] < result["init_loss"], ( + f"Loss didn't decrease: {result['loss']:.6f} >= {result['init_loss']:.6f}" + ) + + def test_joint_ppn_loss_pinned(self, matched_data): + """Pin the joint per_pool_noise loss value.""" + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=100) + # Pinned: 0.0406 (JAX 0.4.30, seed 42) + # Use wide tolerance since optimizer path may vary across platforms + assert result["loss"] < 0.10, f"Loss too high: {result['loss']}" + assert result["loss"] < result["init_loss"] + + def test_joint_shared_noise_loss_decreases(self, matched_data): + """Joint shared_noise loss must decrease from initialization.""" + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint(matched_data, mode="shared_noise", maxiter=100) + assert result["loss"] < result["init_loss"], ( + f"Loss didn't decrease: {result['loss']:.6f} >= {result['init_loss']:.6f}" + ) + + def test_joint_predict_new_pool_at_zero_attrs(self, matched_data): + """Predict at zero attributes → output equals bias terms.""" + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=50) + x_attr = np.zeros(result["k_attr"]) + pred = predict_new_pool_joint(result, x_attr) + + # At zero attributes: log_cadence = bias_cad, log_gas = bias_gas + np.testing.assert_allclose( + pred["log_cadence"], result["bias_cad"], rtol=1e-10 + ) + np.testing.assert_allclose( + pred["log_gas"], result["bias_gas"], rtol=1e-10 + ) + assert pred["cadence_minutes"] > 0 + assert pred["gas_usd"] > 0 + + def test_joint_shared_noise_predict_includes_noise(self, matched_data): + """Shared noise mode prediction includes noise_coeffs.""" + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint(matched_data, mode="shared_noise", maxiter=50) + x_attr = np.zeros(result["k_attr"]) + pred = predict_new_pool_joint(result, x_attr) + + assert "noise_coeffs" in pred, "shared_noise predict should include noise_coeffs" + assert len(pred["noise_coeffs"]) == K_OBS + # At zero attributes: noise_coeffs = bias_noise + np.testing.assert_allclose( + pred["noise_coeffs"], result["bias_noise"], rtol=1e-10 + ) + + def test_joint_ppn_noise_shape(self, matched_data): + """Per-pool noise mode produces (n_pools, K_OBS) noise coefficients.""" + from quantammsim.calibration.joint_fit import fit_joint + + n_pools = len(matched_data) + result = fit_joint(matched_data, mode="per_pool_noise", maxiter=20) + assert result["noise_coeffs"].shape == (n_pools, K_OBS) + + def test_joint_warm_start_from_option_c(self, matched_data): + """Warm start from Option C should produce a viable starting point.""" + from quantammsim.calibration.joint_fit import fit_joint + + option_c = fit_all_pools(matched_data) + result = fit_joint( + matched_data, + mode="per_pool_noise", + maxiter=100, + init_from_option_c=option_c, + ) + # The warm start may have higher init_loss than cold start because + # the linear projection of per-pool params introduces approximation + # error. But the final loss should still decrease from init. + assert result["loss"] < result["init_loss"] + + +# ── Pack/unpack roundtrip pins ───────────────────────────────────────────── + + +class TestPackUnpackPins: + def test_per_pool_loss_pack_roundtrip_exact(self): + """pack → unpack must recover exact values.""" + from quantammsim.calibration.loss import unpack_params + + log_cad = 2.4849 + log_gas = -0.6932 + noise = jnp.array([8.1, -1.2, 3.4, -0.5, 0.7, -2.1, 0.3, 0.9]) + packed = pack_params(log_cad, log_gas, noise) + + lc, lg, nc = unpack_params(packed) + np.testing.assert_allclose(float(lc), log_cad, atol=1e-10) + np.testing.assert_allclose(float(lg), log_gas, atol=1e-10) + np.testing.assert_allclose(nc, noise, atol=1e-10) + + def test_joint_pack_roundtrip_ppn(self): + """Joint per_pool_noise pack → unpack roundtrip.""" + from quantammsim.calibration.joint_fit import ( + pack_joint_params, + unpack_joint_params, + ) + + k_attr = 5 + n_pools = 3 + bias_cad = 2.5 + bias_gas = -0.1 + W_cad = jnp.arange(k_attr, dtype=float) * 0.1 + W_gas = jnp.arange(k_attr, dtype=float) * -0.05 + noise = jnp.ones((n_pools, K_OBS)) * 0.3 + + packed = pack_joint_params(bias_cad, bias_gas, W_cad, W_gas, noise) + config = {"k_attr": k_attr, "n_pools": n_pools, "mode": "per_pool_noise"} + unpacked = unpack_joint_params(packed, config) + + np.testing.assert_allclose(float(unpacked["bias_cad"]), bias_cad, atol=1e-10) + np.testing.assert_allclose(float(unpacked["bias_gas"]), bias_gas, atol=1e-10) + np.testing.assert_allclose(unpacked["W_cad"], W_cad, atol=1e-10) + np.testing.assert_allclose(unpacked["W_gas"], W_gas, atol=1e-10) + np.testing.assert_allclose(unpacked["noise_coeffs"], noise, atol=1e-10) + + def test_joint_pack_roundtrip_shared(self): + """Joint shared_noise pack → unpack roundtrip.""" + from quantammsim.calibration.joint_fit import ( + pack_joint_params, + unpack_joint_params, + ) + + k_attr = 4 + bias_cad = 1.5 + bias_gas = 0.2 + W_cad = jnp.ones(k_attr) * 0.1 + W_gas = jnp.ones(k_attr) * -0.2 + # shared_noise: (1 + k_attr, K_OBS) where row 0 is bias_noise + noise = jnp.arange((1 + k_attr) * K_OBS, dtype=float).reshape( + 1 + k_attr, K_OBS + ) + + packed = pack_joint_params(bias_cad, bias_gas, W_cad, W_gas, noise) + config = {"k_attr": k_attr, "n_pools": 2, "mode": "shared_noise"} + unpacked = unpack_joint_params(packed, config) + + np.testing.assert_allclose(float(unpacked["bias_cad"]), bias_cad, atol=1e-10) + np.testing.assert_allclose(unpacked["bias_noise"], noise[0], atol=1e-10) + np.testing.assert_allclose(unpacked["W_noise"], noise[1:], atol=1e-10) From 32f2e430c648eb71969dbff0f771d7cfe68672d2 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 23:07:02 +0000 Subject: [PATCH 033/115] feat: fixed-gas calibration mode for loss, per-pool fit, and joint fit Add CHAIN_GAS_USD lookup and pool_loss_fixed_gas to fix gas to known chain-level costs, removing the cadence-gas degeneracy. Per-pool fit and joint fit (Option A) both support fix_gas_to_chain flag, optimizing only cadence and noise coefficients when gas is held constant. --- quantammsim/calibration/joint_fit.py | 255 ++++++++++++++++-------- quantammsim/calibration/loss.py | 53 +++++ quantammsim/calibration/per_pool_fit.py | 174 +++++++++++----- 3 files changed, 348 insertions(+), 134 deletions(-) diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py index 4f891c6d..6dcaf851 100644 --- a/quantammsim/calibration/joint_fit.py +++ b/quantammsim/calibration/joint_fit.py @@ -38,17 +38,20 @@ class JointData(NamedTuple): def prepare_joint_data( matched: Dict[str, dict], drop_chain_dummies: bool = False, + fix_gas_to_chain: bool = False, ) -> JointData: """Build batched JAX arrays from matched pool data. Args: matched: dict from match_grids_to_panel drop_chain_dummies: if True, remove chain_* columns from attributes - (reduces feature count for small n) + fix_gas_to_chain: if True, store fixed_log_gas per pool from CHAIN_GAS_USD Returns: JointData with per-pool JAX arrays and shared attribute matrix. """ + from quantammsim.calibration.loss import CHAIN_GAS_USD + X_attr, attr_names, pool_ids = build_pool_attributes(matched) if drop_chain_dummies: @@ -64,12 +67,18 @@ def prepare_joint_data( x_obs = build_x_obs(panel) y_obs = panel["log_volume"].values.astype(float) - pool_data.append({ + d = { "coeffs": entry["coeffs"], "x_obs": jnp.array(x_obs), "y_obs": jnp.array(y_obs), "day_indices": jnp.array(entry["day_indices"]), - }) + } + if fix_gas_to_chain: + chain = entry["chain"] + gas_usd = CHAIN_GAS_USD.get(chain, 1.0) + d["fixed_log_gas"] = jnp.float64(np.log(max(gas_usd, 1e-6))) + + pool_data.append(d) return JointData( pool_data=pool_data, @@ -102,38 +111,66 @@ def pack_joint_params( ]) +def pack_joint_params_fixed_gas( + bias_cad: float, + W_cad: jnp.ndarray, + noise_params: jnp.ndarray, +) -> jnp.ndarray: + """Pack joint params with gas excluded. + + Layout: [bias_cad, W_cad(k_attr), noise_params...] + """ + return jnp.concatenate([ + jnp.array([bias_cad]), + W_cad.ravel(), + noise_params.ravel(), + ]) + + def unpack_joint_params( flat: jnp.ndarray, config: dict ) -> dict: """Unpack flat array to structured params. config must have: k_attr, n_pools, mode + config may have: fix_gas (bool) — if True, no bias_gas/W_gas in flat array """ k_attr = config["k_attr"] mode = config["mode"] + fix_gas = config.get("fix_gas", False) - bias_cad = flat[0] - bias_gas = flat[1] - W_cad = flat[2:2 + k_attr] - W_gas = flat[2 + k_attr:2 + 2 * k_attr] - rest = flat[2 + 2 * k_attr:] + if fix_gas: + bias_cad = flat[0] + W_cad = flat[1:1 + k_attr] + rest = flat[1 + k_attr:] + else: + bias_cad = flat[0] + bias_gas = flat[1] + W_cad = flat[2:2 + k_attr] + W_gas = flat[2 + k_attr:2 + 2 * k_attr] + rest = flat[2 + 2 * k_attr:] if mode == "per_pool_noise": n_pools = config["n_pools"] noise_coeffs = rest.reshape(n_pools, K_OBS) + if fix_gas: + return {"bias_cad": bias_cad, "W_cad": W_cad, + "noise_coeffs": noise_coeffs} return { "bias_cad": bias_cad, "bias_gas": bias_gas, "W_cad": W_cad, "W_gas": W_gas, "noise_coeffs": noise_coeffs, } else: # shared_noise - # noise_params: (1 + k_attr, K_OBS) — row 0 is bias W_noise_full = rest.reshape(1 + k_attr, K_OBS) + if fix_gas: + return {"bias_cad": bias_cad, "W_cad": W_cad, + "bias_noise": W_noise_full[0], "W_noise": W_noise_full[1:]} return { "bias_cad": bias_cad, "bias_gas": bias_gas, "W_cad": W_cad, "W_gas": W_gas, - "bias_noise": W_noise_full[0], # (K_OBS,) - "W_noise": W_noise_full[1:], # (k_attr, K_OBS) + "bias_noise": W_noise_full[0], + "W_noise": W_noise_full[1:], } @@ -147,30 +184,53 @@ def _make_pool_loss_fn( Closes over pool-specific data; takes only params_flat as input. Each pool gets its own small JIT'd computation graph. + + If config["fix_gas"] is True, gas comes from pool_data_i["fixed_log_gas"] + instead of being predicted from attributes. """ coeffs = pool_data_i["coeffs"] x_obs = pool_data_i["x_obs"] y_obs = pool_data_i["y_obs"] day_indices = pool_data_i["day_indices"] mode = config["mode"] + fix_gas = config.get("fix_gas", False) i = pool_idx - @jax.jit - def pool_loss_fn(params_flat): - params = unpack_joint_params(params_flat, config) - log_cad = params["bias_cad"] + jnp.dot(x_attr_i, params["W_cad"]) - log_gas = params["bias_gas"] + jnp.dot(x_attr_i, params["W_gas"]) + if fix_gas: + fixed_log_gas = pool_data_i["fixed_log_gas"] - if mode == "per_pool_noise": - noise_c = params["noise_coeffs"][i] - else: - noise_c = params["bias_noise"] + jnp.dot(x_attr_i, params["W_noise"]) + @jax.jit + def pool_loss_fn(params_flat): + params = unpack_joint_params(params_flat, config) + log_cad = params["bias_cad"] + jnp.dot(x_attr_i, params["W_cad"]) - v_arb_all = interpolate_pool_daily(coeffs, log_cad, jnp.exp(log_gas)) - v_arb = v_arb_all[day_indices] - v_noise = jnp.exp(x_obs @ noise_c) - log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) - return jnp.mean((log_v_pred - y_obs) ** 2) + if mode == "per_pool_noise": + noise_c = params["noise_coeffs"][i] + else: + noise_c = params["bias_noise"] + jnp.dot(x_attr_i, params["W_noise"]) + + v_arb_all = interpolate_pool_daily(coeffs, log_cad, jnp.exp(fixed_log_gas)) + v_arb = v_arb_all[day_indices] + v_noise = jnp.exp(x_obs @ noise_c) + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + return jnp.mean((log_v_pred - y_obs) ** 2) + else: + @jax.jit + def pool_loss_fn(params_flat): + params = unpack_joint_params(params_flat, config) + log_cad = params["bias_cad"] + jnp.dot(x_attr_i, params["W_cad"]) + log_gas = params["bias_gas"] + jnp.dot(x_attr_i, params["W_gas"]) + + if mode == "per_pool_noise": + noise_c = params["noise_coeffs"][i] + else: + noise_c = params["bias_noise"] + jnp.dot(x_attr_i, params["W_noise"]) + + v_arb_all = interpolate_pool_daily(coeffs, log_cad, jnp.exp(log_gas)) + v_arb = v_arb_all[day_indices] + v_noise = jnp.exp(x_obs @ noise_c) + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + return jnp.mean((log_v_pred - y_obs) ** 2) return pool_loss_fn @@ -180,6 +240,7 @@ def make_joint_loss_fn( mode: str = "per_pool_noise", alpha_cad: float = 0.01, alpha_gas: float = 0.01, + fix_gas: bool = False, ): """Create per-pool JIT'd loss functions and a Python-level aggregator. @@ -190,20 +251,22 @@ def make_joint_loss_fn( Loss averages over pools (not observations), giving equal weight to each pool regardless of observation count. - L2 regularization is applied to W_cad and W_gas only (not biases). + L2 regularization is applied to W_cad (and W_gas if not fixed). Args: jdata: JointData from prepare_joint_data mode: "per_pool_noise" or "shared_noise" alpha_cad: L2 regularization on W_cad - alpha_gas: L2 regularization on W_gas + alpha_gas: L2 regularization on W_gas (ignored if fix_gas=True) + fix_gas: if True, gas is fixed per pool (no W_gas in params) Returns: loss_fn(params_flat) -> scalar loss """ n_pools = len(jdata.pool_data) k_attr = jdata.x_attr.shape[1] - config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode} + config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode, + "fix_gas": fix_gas} # Build per-pool JIT'd loss functions pool_loss_fns = [] @@ -218,8 +281,9 @@ def loss_fn(params_flat): data_loss = total / n_pools params = unpack_joint_params(params_flat, config) - reg = alpha_cad * jnp.sum(params["W_cad"] ** 2) + \ - alpha_gas * jnp.sum(params["W_gas"] ** 2) + reg = alpha_cad * jnp.sum(params["W_cad"] ** 2) + if not fix_gas: + reg = reg + alpha_gas * jnp.sum(params["W_gas"] ** 2) return data_loss + reg # Attach per-pool functions for the value_and_grad wrapper @@ -236,15 +300,12 @@ def make_initial_joint_params( jdata: JointData, mode: str = "per_pool_noise", init_from_option_c: Optional[Dict[str, dict]] = None, + fix_gas: bool = False, ) -> jnp.ndarray: """Create initial parameter vector. - If init_from_option_c is provided, warm-start from Option C per-pool fits: - - bias_cad, W_cad from OLS on per-pool fitted log_cadence - - bias_gas, W_gas from OLS on per-pool fitted log_gas - - noise_coeffs from per-pool fits - - Otherwise, use defaults: cadence=12min, gas=$1 for all pools. + If init_from_option_c is provided, warm-start from Option C per-pool fits. + If fix_gas is True, excludes bias_gas and W_gas from the parameter vector. """ n_pools = len(jdata.pool_data) k_attr = jdata.x_attr.shape[1] @@ -252,7 +313,6 @@ def make_initial_joint_params( if init_from_option_c is not None: pool_ids = jdata.pool_ids - # Filter out pools with NaN losses from warm start valid = {p: init_from_option_c[p] for p in pool_ids if p in init_from_option_c and np.isfinite(init_from_option_c[p].get("loss", float("nan")))} @@ -269,32 +329,31 @@ def make_initial_joint_params( } log_cads = np.array([valid[p]["log_cadence"] for p in pool_ids]) - log_gases = np.array([valid[p]["log_gas"] for p in pool_ids]) noise_all = np.array([valid[p]["noise_coeffs"] for p in pool_ids]) - # OLS with intercept: X_aug = [1, x_attr]; solve for [bias, W] X_aug = np.column_stack([np.ones(n_pools), x_attr_np]) cad_params, _, _, _ = np.linalg.lstsq(X_aug, log_cads, rcond=None) - gas_params, _, _, _ = np.linalg.lstsq(X_aug, log_gases, rcond=None) bias_cad, W_cad = cad_params[0], cad_params[1:] - bias_gas, W_gas = gas_params[0], gas_params[1:] + + if not fix_gas: + log_gases = np.array([valid[p]["log_gas"] for p in pool_ids]) + gas_params, _, _, _ = np.linalg.lstsq(X_aug, log_gases, rcond=None) + bias_gas, W_gas = gas_params[0], gas_params[1:] if mode == "per_pool_noise": noise_params = noise_all else: - # OLS with intercept for noise mapping noise_aug, _, _, _ = np.linalg.lstsq(X_aug, noise_all, rcond=None) - # noise_aug: (1+k_attr, K_OBS) — row 0 is bias noise_params = noise_aug else: - # Default: all pools get cadence=12min, gas=$1 bias_cad = np.log(12.0) - bias_gas = np.log(1.0) # = 0.0 W_cad = np.zeros(k_attr) - W_gas = np.zeros(k_attr) + + if not fix_gas: + bias_gas = np.log(1.0) + W_gas = np.zeros(k_attr) if mode == "per_pool_noise": - # Initialize noise via OLS per pool noise_params = np.zeros((n_pools, K_OBS)) for i, pd in enumerate(jdata.pool_data): x_obs_np = np.array(pd["x_obs"]) @@ -302,29 +361,40 @@ def make_initial_joint_params( c, _, _, _ = np.linalg.lstsq(x_obs_np, y_obs_np, rcond=None) noise_params[i] = c else: - # Initialize shared noise from pooled OLS all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) c, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) - # (1+k_attr, K_OBS): bias row + zero weight rows noise_params = np.zeros((1 + k_attr, K_OBS)) noise_params[0, :] = c - return pack_joint_params( - float(bias_cad), - float(bias_gas), - jnp.array(W_cad), - jnp.array(W_gas), - jnp.array(noise_params), - ) + if fix_gas: + return pack_joint_params_fixed_gas( + float(bias_cad), + jnp.array(W_cad), + jnp.array(noise_params), + ) + else: + return pack_joint_params( + float(bias_cad), + float(bias_gas), + jnp.array(W_cad), + jnp.array(W_gas), + jnp.array(noise_params), + ) -def _make_bounds(k_attr, n_pools, mode): +def _make_bounds(k_attr, n_pools, mode, fix_gas=False): """Build scipy bounds for joint params.""" - # bias_cad, bias_gas: unbounded - bounds = [(None, None)] * 2 - # W_cad, W_gas: unbounded - bounds += [(None, None)] * (2 * k_attr) + if fix_gas: + # bias_cad only + bounds = [(None, None)] * 1 + # W_cad only + bounds += [(None, None)] * k_attr + else: + # bias_cad, bias_gas + bounds = [(None, None)] * 2 + # W_cad, W_gas + bounds += [(None, None)] * (2 * k_attr) if mode == "per_pool_noise": bounds += [(None, None)] * (n_pools * K_OBS) @@ -342,6 +412,7 @@ def fit_joint( alpha_cad: float = 0.01, alpha_gas: float = 0.01, drop_chain_dummies: bool = False, + fix_gas_to_chain: bool = False, ) -> dict: """Joint end-to-end optimization across all pools. @@ -349,38 +420,43 @@ def fit_joint( matched: dict from match_grids_to_panel mode: "per_pool_noise" or "shared_noise" init_from_option_c: Optional Option C results for warm start. - Pools with NaN losses are silently excluded from warm start. maxiter: max L-BFGS-B iterations alpha_cad: L2 regularization on W_cad (not bias) - alpha_gas: L2 regularization on W_gas (not bias) + alpha_gas: L2 regularization on W_gas (not bias, ignored if fix_gas) drop_chain_dummies: if True, remove chain_* columns from attributes + fix_gas_to_chain: if True, gas is fixed to known chain-level costs Returns dict with fitted params and diagnostics. """ - jdata = prepare_joint_data(matched, drop_chain_dummies=drop_chain_dummies) + jdata = prepare_joint_data(matched, drop_chain_dummies=drop_chain_dummies, + fix_gas_to_chain=fix_gas_to_chain) loss_fn = make_joint_loss_fn(jdata, mode=mode, - alpha_cad=alpha_cad, alpha_gas=alpha_gas) + alpha_cad=alpha_cad, alpha_gas=alpha_gas, + fix_gas=fix_gas_to_chain) init = make_initial_joint_params(jdata, mode=mode, - init_from_option_c=init_from_option_c) + init_from_option_c=init_from_option_c, + fix_gas=fix_gas_to_chain) n_pools = len(jdata.pool_data) k_attr = jdata.x_attr.shape[1] - config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode} - bounds = _make_bounds(k_attr, n_pools, mode) + config = {"k_attr": k_attr, "n_pools": n_pools, "mode": mode, + "fix_gas": fix_gas_to_chain} + bounds = _make_bounds(k_attr, n_pools, mode, fix_gas=fix_gas_to_chain) - # Per-pool value_and_grad — each pool has its own small JIT graph pool_vg_fns = loss_fn._pool_val_and_grad_fns - # Indices for W_cad and W_gas in the flat param vector (for reg gradient) - w_cad_start = 2 - w_cad_end = 2 + k_attr - w_gas_start = 2 + k_attr - w_gas_end = 2 + 2 * k_attr + if fix_gas_to_chain: + w_cad_start = 1 + w_cad_end = 1 + k_attr + else: + w_cad_start = 2 + w_cad_end = 2 + k_attr + w_gas_start = 2 + k_attr + w_gas_end = 2 + 2 * k_attr def scipy_wrapper(params_np): params_j = jnp.array(params_np) - # Sum per-pool losses and gradients total_val = 0.0 total_grad = jnp.zeros_like(params_j) for vg_fn in pool_vg_fns: @@ -391,15 +467,15 @@ def scipy_wrapper(params_np): data_loss = total_val / n_pools data_grad = total_grad / n_pools - # Regularization on W_cad and W_gas (not biases) - reg = (alpha_cad * float(jnp.sum(params_j[w_cad_start:w_cad_end] ** 2)) + - alpha_gas * float(jnp.sum(params_j[w_gas_start:w_gas_end] ** 2))) - + reg = alpha_cad * float(jnp.sum(params_j[w_cad_start:w_cad_end] ** 2)) reg_grad = jnp.zeros_like(params_j) reg_grad = reg_grad.at[w_cad_start:w_cad_end].set( 2 * alpha_cad * params_j[w_cad_start:w_cad_end]) - reg_grad = reg_grad.at[w_gas_start:w_gas_end].set( - 2 * alpha_gas * params_j[w_gas_start:w_gas_end]) + + if not fix_gas_to_chain: + reg += alpha_gas * float(jnp.sum(params_j[w_gas_start:w_gas_end] ** 2)) + reg_grad = reg_grad.at[w_gas_start:w_gas_end].set( + 2 * alpha_gas * params_j[w_gas_start:w_gas_end]) val = data_loss + reg grad = data_grad + reg_grad @@ -422,17 +498,30 @@ def scipy_wrapper(params_np): out = { "init_loss": init_loss, "bias_cad": float(params["bias_cad"]), - "bias_gas": float(params["bias_gas"]), "W_cad": np.array(params["W_cad"]), - "W_gas": np.array(params["W_gas"]), "loss": float(result.fun), "converged": result.success, "mode": mode, "k_attr": k_attr, "pool_ids": jdata.pool_ids, "attr_names": jdata.attr_names, + "fix_gas": fix_gas_to_chain, } + if fix_gas_to_chain: + # Store per-pool fixed gas values for downstream use + from quantammsim.calibration.loss import CHAIN_GAS_USD + gas_per_pool = [] + for pid in jdata.pool_ids: + chain = matched[pid]["chain"] + gas_per_pool.append(CHAIN_GAS_USD.get(chain, 1.0)) + out["gas_per_pool"] = np.array(gas_per_pool) + out["bias_gas"] = 0.0 + out["W_gas"] = np.zeros(k_attr) + else: + out["bias_gas"] = float(params["bias_gas"]) + out["W_gas"] = np.array(params["W_gas"]) + if mode == "per_pool_noise": out["noise_coeffs"] = np.array(params["noise_coeffs"]) else: diff --git a/quantammsim/calibration/loss.py b/quantammsim/calibration/loss.py index 13b2e650..e003a985 100644 --- a/quantammsim/calibration/loss.py +++ b/quantammsim/calibration/loss.py @@ -16,6 +16,17 @@ K_OBS = 8 # observation-level covariates +# Known chain gas costs (USD) — used when fixing gas to chain-level values. +# These are effective per-transaction costs, not per-gas-unit. +CHAIN_GAS_USD = { + "MAINNET": 1.0, + "POLYGON": 0.005, + "GNOSIS": 0.001, + "ARBITRUM": 0.01, + "BASE": 0.005, + "SONIC": 0.005, +} + def noise_volume( noise_coeffs: jnp.ndarray, x_obs: jnp.ndarray @@ -41,6 +52,23 @@ def unpack_params( return flat[0], flat[1], flat[2:] +def pack_params_fixed_gas( + log_cadence: float, noise_coeffs: jnp.ndarray +) -> jnp.ndarray: + """Pack into flat array with gas excluded: [log_cadence, noise_coeffs...].""" + return jnp.concatenate([ + jnp.array([log_cadence]), + jnp.asarray(noise_coeffs), + ]) + + +def unpack_params_fixed_gas( + flat: jnp.ndarray, +) -> Tuple[float, jnp.ndarray]: + """Unpack flat array to (log_cadence, noise_coeffs). Gas not included.""" + return flat[0], flat[1:] + + def pool_loss( params_flat: jnp.ndarray, coeffs: PoolCoeffsDaily, @@ -72,3 +100,28 @@ def pool_loss( # Log-space L2 loss log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) return jnp.mean((log_v_pred - y_obs) ** 2) + + +def pool_loss_fixed_gas( + params_flat: jnp.ndarray, + fixed_log_gas: float, + coeffs: PoolCoeffsDaily, + x_obs: jnp.ndarray, + y_obs: jnp.ndarray, + day_indices: jnp.ndarray, +) -> jnp.ndarray: + """Per-pool loss with gas fixed to a known chain-level value. + + Args: + params_flat: [log_cadence, noise_coeffs...] — no log_gas + fixed_log_gas: log(gas_usd) held constant (not optimized) + coeffs, x_obs, y_obs, day_indices: as in pool_loss + """ + log_cadence, noise_coeffs = unpack_params_fixed_gas(params_flat) + + v_arb_all = interpolate_pool_daily(coeffs, log_cadence, jnp.exp(fixed_log_gas)) + v_arb = v_arb_all[day_indices] + v_noise = noise_volume(noise_coeffs, x_obs) + + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + return jnp.mean((log_v_pred - y_obs) ** 2) diff --git a/quantammsim/calibration/per_pool_fit.py b/quantammsim/calibration/per_pool_fit.py index 037a124d..08140d90 100644 --- a/quantammsim/calibration/per_pool_fit.py +++ b/quantammsim/calibration/per_pool_fit.py @@ -2,6 +2,9 @@ Fits (log_cadence, log_gas, noise_coeffs) per pool by minimizing the log-space L2 loss using scipy.optimize.minimize with JAX gradients. + +Supports fixed-gas mode where gas is set to the known chain-level cost, +leaving only (log_cadence, noise_coeffs) to be optimized. """ from typing import Dict, Optional @@ -12,7 +15,13 @@ import scipy.optimize from quantammsim.calibration.grid_interpolation import PoolCoeffsDaily -from quantammsim.calibration.loss import K_OBS, pack_params, pool_loss +from quantammsim.calibration.loss import ( + CHAIN_GAS_USD, + K_OBS, + pack_params, + pool_loss, + pool_loss_fixed_gas, +) from quantammsim.calibration.pool_data import build_x_obs @@ -30,6 +39,15 @@ def make_initial_guess(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: return init +def make_initial_guess_fixed_gas(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: + """Initial params for fixed-gas mode: cadence=12min, noise_coeffs from OLS.""" + noise_coeffs, _, _, _ = np.linalg.lstsq(x_obs, y_obs, rcond=None) + init = np.zeros(1 + K_OBS) + init[0] = np.log(12.0) # log_cadence + init[1:] = noise_coeffs + return init + + def fit_single_pool( coeffs: PoolCoeffsDaily, x_obs: np.ndarray, @@ -37,74 +55,122 @@ def fit_single_pool( day_indices: np.ndarray, init: Optional[np.ndarray] = None, bounds: Optional[dict] = None, + fixed_gas_usd: Optional[float] = None, ) -> dict: - """Fit (log_cadence, log_gas, noise_coeffs) for one pool via L-BFGS-B. + """Fit one pool via L-BFGS-B. + + If fixed_gas_usd is given, gas is held constant at that value and only + (log_cadence, noise_coeffs) are optimized. Otherwise fits all three. Returns dict with fitted params, loss, and convergence status. """ - if init is None: - init = make_initial_guess(x_obs, y_obs) + # Convert to JAX arrays + x_obs_j = jnp.array(x_obs) + y_obs_j = jnp.array(y_obs) + day_idx_j = jnp.array(day_indices) - # Default bounds if bounds is None: bounds = {} log_cad_bounds = bounds.get("log_cadence", (np.log(1.0), np.log(60.0))) - log_gas_bounds = bounds.get("log_gas", (np.log(0.001), np.log(50.0))) noise_bounds = bounds.get("noise_coeffs", (-20.0, 20.0)) - scipy_bounds = [ - log_cad_bounds, - log_gas_bounds, - ] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + if fixed_gas_usd is not None: + # Fixed-gas mode + fixed_log_gas = jnp.float64(np.log(max(fixed_gas_usd, 1e-6))) - # Convert to JAX arrays - x_obs_j = jnp.array(x_obs) - y_obs_j = jnp.array(y_obs) - day_idx_j = jnp.array(day_indices) + if init is None: + init = make_initial_guess_fixed_gas(x_obs, y_obs) + + scipy_bounds = [log_cad_bounds] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + + @jax.jit + def loss_and_grad(params_flat): + loss = pool_loss_fixed_gas( + params_flat, fixed_log_gas, coeffs, x_obs_j, y_obs_j, day_idx_j) + grad = jax.grad(pool_loss_fixed_gas, argnums=0)( + params_flat, fixed_log_gas, coeffs, x_obs_j, y_obs_j, day_idx_j) + return loss, grad + + def scipy_wrapper(params_np): + params_j = jnp.array(params_np) + loss, grad = loss_and_grad(params_j) + return float(loss), np.array(grad, dtype=np.float64) + + result = scipy.optimize.minimize( + scipy_wrapper, init, method="L-BFGS-B", jac=True, + bounds=scipy_bounds, + options={"maxiter": 500, "ftol": 1e-10, "gtol": 1e-8}, + ) - # Value and gradient function - @jax.jit - def loss_and_grad(params_flat): - loss = pool_loss(params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j) - grad = jax.grad(pool_loss, argnums=0)( - params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j + log_cadence = float(result.x[0]) + noise_coeffs = np.array(result.x[1:]) + log_gas = float(fixed_log_gas) + + return { + "log_cadence": log_cadence, + "log_gas": log_gas, + "noise_coeffs": noise_coeffs, + "loss": float(result.fun), + "converged": result.success, + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": fixed_gas_usd, + "gas_fixed": True, + } + + else: + # Free-gas mode (original) + if init is None: + init = make_initial_guess(x_obs, y_obs) + + log_gas_bounds = bounds.get("log_gas", (np.log(0.001), np.log(50.0))) + scipy_bounds = [ + log_cad_bounds, log_gas_bounds, + ] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + + @jax.jit + def loss_and_grad(params_flat): + loss = pool_loss(params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j) + grad = jax.grad(pool_loss, argnums=0)( + params_flat, coeffs, x_obs_j, y_obs_j, day_idx_j) + return loss, grad + + def scipy_wrapper(params_np): + params_j = jnp.array(params_np) + loss, grad = loss_and_grad(params_j) + return float(loss), np.array(grad, dtype=np.float64) + + result = scipy.optimize.minimize( + scipy_wrapper, init, method="L-BFGS-B", jac=True, + bounds=scipy_bounds, + options={"maxiter": 500, "ftol": 1e-10, "gtol": 1e-8}, ) - return loss, grad - - def scipy_wrapper(params_np): - params_j = jnp.array(params_np) - loss, grad = loss_and_grad(params_j) - return float(loss), np.array(grad, dtype=np.float64) - - result = scipy.optimize.minimize( - scipy_wrapper, - init, - method="L-BFGS-B", - jac=True, - bounds=scipy_bounds, - options={"maxiter": 500, "ftol": 1e-10, "gtol": 1e-8}, - ) - - log_cadence = float(result.x[0]) - log_gas = float(result.x[1]) - noise_coeffs = np.array(result.x[2:]) - - return { - "log_cadence": log_cadence, - "log_gas": log_gas, - "noise_coeffs": noise_coeffs, - "loss": float(result.fun), - "converged": result.success, - "cadence_minutes": float(np.exp(log_cadence)), - "gas_usd": float(np.exp(log_gas)), - } + + log_cadence = float(result.x[0]) + log_gas = float(result.x[1]) + noise_coeffs = np.array(result.x[2:]) + + return { + "log_cadence": log_cadence, + "log_gas": log_gas, + "noise_coeffs": noise_coeffs, + "loss": float(result.fun), + "converged": result.success, + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": float(np.exp(log_gas)), + "gas_fixed": False, + } def fit_all_pools( matched: Dict[str, dict], n_workers: int = 1, + fix_gas_to_chain: bool = False, ) -> Dict[str, dict]: - """Fit all matched pools. Returns prefix -> fit_result with metadata.""" + """Fit all matched pools. Returns prefix -> fit_result with metadata. + + If fix_gas_to_chain is True, gas is fixed to the known chain-level cost + from CHAIN_GAS_USD, and only (log_cadence, noise_coeffs) are optimized. + """ results = {} for prefix, entry in matched.items(): @@ -115,7 +181,13 @@ def fit_all_pools( x_obs = build_x_obs(panel) y_obs = panel["log_volume"].values.astype(float) - result = fit_single_pool(coeffs, x_obs, y_obs, day_indices) + fixed_gas = None + if fix_gas_to_chain: + chain = entry["chain"] + fixed_gas = CHAIN_GAS_USD.get(chain, 1.0) + + result = fit_single_pool( + coeffs, x_obs, y_obs, day_indices, fixed_gas_usd=fixed_gas) # Add metadata result["chain"] = entry["chain"] From 834e547f84a6a2df0e95b386b71d0d66a23be277 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 23:07:15 +0000 Subject: [PATCH 034/115] feat: replace Balancer hourly volatility with Binance minute data Compute daily realized volatility from Binance minute prices instead of Balancer API hourly prices, removing the 90-day data restriction. Each pool now uses its full historical date range (up to 1761 days). The calibration runner calls replace_panel_volatility_with_binance() and supports train_days=0 for unrestricted history. --- quantammsim/calibration/pool_data.py | 170 +++++++++++++++++++++- scripts/run_direct_calibration_top50.py | 184 +++++++++++++----------- 2 files changed, 266 insertions(+), 88 deletions(-) diff --git a/quantammsim/calibration/pool_data.py b/quantammsim/calibration/pool_data.py index 039880bc..61082cdc 100644 --- a/quantammsim/calibration/pool_data.py +++ b/quantammsim/calibration/pool_data.py @@ -6,7 +6,7 @@ import json import os -from typing import Dict, List, Tuple +from typing import Dict, List, Optional, Tuple import numpy as np import pandas as pd @@ -25,6 +25,11 @@ "local_data", "noise_calibration", "token_mcaps.json", ) +# Default path for Binance minute parquets +_BINANCE_DATA_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "data", +) + # Asset type classification (fallback if not in mcap JSON) _STABLECOINS = { "USDC", "USDT", "DAI", "WXDAI", "xDAI", "GHO", "LUSD", "crvUSD", @@ -42,6 +47,22 @@ "waBasWETH", "waGnoGNO", "waGnowstETH", } +# Balancer token → Binance parquet symbol mapping. +# Matches build_pool_grids.py TOKEN_MAP. +TOKEN_MAP = { + "WBTC": "BTC", "WETH": "ETH", "cbBTC": "BTC", + "wstETH": "ETH", "stETH": "ETH", "rETH": "ETH", "cbETH": "ETH", + "waEthLidoWETH": "ETH", "waEthLidowstETH": "ETH", + "waBasWETH": "ETH", "waGnowstETH": "ETH", + "waGnoGNO": "GNO", "osGNO": "GNO", + "wS": "S", "stS": "S", + "JitoSOL": "SOL", + "wPOL": "POL", "WMATIC": "POL", "MATIC": "POL", + "USDC.e": "USDC", "USDbC": "USDC", "waBasUSDC": "USDC", + "DAI": "USDC", "WXDAI": "USDC", "sDAI": "USDC", + "USDT": "USDC", "DOLA": "USDC", "scUSD": "USDC", +} + def _load_token_mcaps(path: str = None) -> dict: """Load cached token market caps. Returns {} if file missing.""" @@ -71,6 +92,153 @@ def _parse_tokens(tokens_str: str) -> List[str]: return [t.strip() for t in tokens_str.split(",")] +def _resolve_binance_symbol(token: str) -> str: + """Map Balancer token name to Binance parquet symbol.""" + return TOKEN_MAP.get(token, token) + + +def _load_binance_minute(symbol: str, data_dir: str = None) -> Optional[pd.DataFrame]: + """Load Binance minute close prices. Returns DataFrame with unix index.""" + if data_dir is None: + data_dir = _BINANCE_DATA_DIR + path = os.path.join(data_dir, f"{symbol}_USD.parquet") + if not os.path.exists(path): + return None + df = pd.read_parquet(path, columns=["unix", "close"]) + if df.index.name != "unix": + df = df.set_index("unix") + return df + + +def compute_binance_pair_volatility( + token_a: str, token_b: str, data_dir: str = None, +) -> Optional[pd.Series]: + """Compute daily annualized realized volatility from Binance minute data. + + Resamples minute data to hourly, computes hourly log returns of the pair + ratio, then daily std × sqrt(24 × 365). Matches the Balancer hourly + pipeline's annualization convention. + + Args: + token_a, token_b: Balancer token symbols (e.g. "WETH", "USDC") + data_dir: directory containing {SYMBOL}_USD.parquet files + + Returns: + pd.Series with datetime.date index → annualized volatility, + or None if both tokens are stablecoins / same underlying / missing data. + """ + sym_a = _resolve_binance_symbol(token_a) + sym_b = _resolve_binance_symbol(token_b) + + is_stable_a = token_a in _STABLECOINS or sym_a == "USDC" + is_stable_b = token_b in _STABLECOINS or sym_b == "USDC" + + if is_stable_a and is_stable_b: + return None # caller should use constant 0.01 + + if sym_a == sym_b: + return None # same underlying (e.g. wstETH/WETH) + + # Load minute data and compute pair ratio + if is_stable_b: + df = _load_binance_minute(sym_a, data_dir) + if df is None: + return None + ratio = df["close"] + elif is_stable_a: + df = _load_binance_minute(sym_b, data_dir) + if df is None: + return None + ratio = 1.0 / df["close"] + else: + df_a = _load_binance_minute(sym_a, data_dir) + df_b = _load_binance_minute(sym_b, data_dir) + if df_a is None or df_b is None: + return None + merged = df_a.join(df_b, lsuffix="_a", rsuffix="_b", how="inner") + ratio = merged["close_a"] / merged["close_b"] + + # Resample to hourly (last close per hour) + ratio_df = pd.DataFrame({"ratio": ratio}) + ratio_df.index = pd.to_datetime(ratio_df.index, unit="ms", utc=True) + hourly = ratio_df.resample("1h").last().dropna() + + # Hourly log returns + hourly["log_return"] = np.log(hourly["ratio"] / hourly["ratio"].shift(1)) + hourly = hourly.dropna() + + # Daily std → annualized + hourly["date"] = hourly.index.date + daily_vol = hourly.groupby("date")["log_return"].std() + annualized = daily_vol * np.sqrt(24 * 365) + + # Clean + annualized = annualized.replace([np.inf, -np.inf], np.nan).dropna() + annualized = annualized[annualized > 0] + + return annualized + + +def replace_panel_volatility_with_binance( + panel: pd.DataFrame, data_dir: str = None, +) -> pd.DataFrame: + """Replace panel 'volatility' column with Binance-derived daily values. + + For each pool, computes daily realized volatility from Binance minute data. + Pools without Binance data keep their existing (possibly fallback) values. + Stablecoin-stablecoin and same-underlying pairs get vol=0.01. + + Returns a copy of the panel with updated volatility. + """ + panel = panel.copy() + panel["date"] = pd.to_datetime(panel["date"]) + + # Cache: (sym_a, sym_b) → vol_series to avoid reloading + _vol_cache: Dict[tuple, Optional[pd.Series]] = {} + + n_replaced = 0 + n_pools = 0 + + for pool_id, grp in panel.groupby("pool_id"): + tokens_str = grp.iloc[0]["tokens"] + toks = _parse_tokens(tokens_str) + if len(toks) < 2: + continue + + sym_a = _resolve_binance_symbol(toks[0]) + sym_b = _resolve_binance_symbol(toks[1]) + cache_key = (min(sym_a, sym_b), max(sym_a, sym_b)) + + if cache_key not in _vol_cache: + _vol_cache[cache_key] = compute_binance_pair_volatility( + toks[0], toks[1], data_dir) + + vol_series = _vol_cache[cache_key] + + if vol_series is None: + # Stablecoins or same underlying → low constant vol + is_stable_a = toks[0] in _STABLECOINS or sym_a == "USDC" + is_stable_b = toks[1] in _STABLECOINS or sym_b == "USDC" + if (is_stable_a and is_stable_b) or sym_a == sym_b: + panel.loc[grp.index, "volatility"] = 0.01 + n_pools += 1 + continue + + # Vectorized date matching + panel_dates = pd.to_datetime(grp["date"]).dt.date + vol_dict = vol_series.to_dict() + new_vol = panel_dates.map(vol_dict) + has_vol = new_vol.notna() + if has_vol.any(): + panel.loc[grp.index[has_vol.values], "volatility"] = ( + new_vol[has_vol].values.astype(float)) + n_replaced += has_vol.sum() + n_pools += 1 + + print(f" Binance volatility: {n_pools} pools, {n_replaced} obs replaced") + return panel + + def match_grids_to_panel( grid_dir: str, panel: pd.DataFrame, pools_path: str = None, ) -> Dict[str, dict]: diff --git a/scripts/run_direct_calibration_top50.py b/scripts/run_direct_calibration_top50.py index 1c1770b8..d323c0b3 100644 --- a/scripts/run_direct_calibration_top50.py +++ b/scripts/run_direct_calibration_top50.py @@ -32,7 +32,7 @@ os.path.dirname(os.path.dirname(__file__)), "results", "direct_calibration_top50", ) -TRAIN_DAYS = 90 +TRAIN_DAYS = 0 # 0 = no filter, use all available data per pool TOP_N = 50 OPTION_C_MAXITER = 500 JOINT_MAXITER = 500 @@ -40,18 +40,28 @@ def load_and_match(): - """Load panel, filter to 90 days, match to grids.""" + """Load panel, match to grids. No date filter — each pool uses all data.""" + from quantammsim.calibration.pool_data import ( + match_grids_to_panel, + replace_panel_volatility_with_binance, + ) + panel = pd.read_parquet(PANEL_CACHE) - max_date = panel["date"].max() - if not isinstance(max_date, date): - max_date = pd.Timestamp(max_date).date() - cutoff = max_date - timedelta(days=TRAIN_DAYS) - panel = panel[ - panel["date"].apply( - lambda d: d >= cutoff if isinstance(d, date) - else pd.Timestamp(d).date() >= cutoff - ) - ].copy() + + # Optional date filter (TRAIN_DAYS=0 means no filter) + if TRAIN_DAYS > 0: + max_date = panel["date"].max() + if not isinstance(max_date, date): + max_date = pd.Timestamp(max_date).date() + cutoff = max_date - timedelta(days=TRAIN_DAYS) + panel = panel[ + panel["date"].apply( + lambda d: d >= cutoff if isinstance(d, date) + else pd.Timestamp(d).date() >= cutoff + ) + ].copy() + else: + panel = panel.copy() if "log_tvl_lag1" not in panel.columns: panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) @@ -62,21 +72,27 @@ def load_and_match(): valid = pool_counts[pool_counts >= 10].index panel = panel[panel["pool_id"].isin(valid)].copy() + # Replace Balancer-hourly volatility with Binance-minute volatility + print("Replacing volatility with Binance minute data...") + panel = replace_panel_volatility_with_binance(panel) + + min_date = panel["date"].min() + max_date = panel["date"].max() print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " - f"{cutoff} to {max_date}") + f"{min_date} to {max_date}") - from quantammsim.calibration.pool_data import match_grids_to_panel matched = match_grids_to_panel(GRID_DIR, panel) print(f"Matched: {len(matched)} pools with grids") return panel, matched -def run_option_c(matched): +def run_option_c(matched, fix_gas_to_chain=False): """Per-pool L-BFGS-B fits.""" from quantammsim.calibration.per_pool_fit import fit_all_pools - print(f"\n--- Option C: per-pool fits ({len(matched)} pools) ---") - results = fit_all_pools(matched) + gas_label = " (gas fixed to chain)" if fix_gas_to_chain else "" + print(f"\n--- Option C: per-pool fits ({len(matched)} pools){gas_label} ---") + results = fit_all_pools(matched, fix_gas_to_chain=fix_gas_to_chain) n_converged = sum(1 for r in results.values() if r["converged"]) losses = [r["loss"] for r in results.values()] print(f" Converged: {n_converged}/{len(results)}") @@ -86,7 +102,7 @@ def run_option_c(matched): return results -def run_option_a(matched, option_c_results): +def run_option_a(matched, option_c_results, fix_gas_to_chain=False): """Joint end-to-end optimization, warm-started from Option C. Drops pathological pools (Option C loss > OPTION_C_LOSS_CUTOFF) from the @@ -106,26 +122,29 @@ def run_option_a(matched, option_c_results): r = option_c_results[p] print(f" {p} {r['tokens']:<16} loss={r['loss']:.1f}") + gas_label = ", gas fixed" if fix_gas_to_chain else "" print(f"\n--- Option A: joint fit (per_pool_noise, {len(matched_clean)} pools, " - f"warm-start from C, no chain dummies) ---") + f"warm-start from C, no chain dummies{gas_label}) ---") result_ppn = fit_joint( matched_clean, mode="per_pool_noise", init_from_option_c=good_pools, maxiter=JOINT_MAXITER, drop_chain_dummies=True, + fix_gas_to_chain=fix_gas_to_chain, ) print(f" Loss: {result_ppn['init_loss']:.4f} -> {result_ppn['loss']:.4f}") print(f" Converged: {result_ppn['converged']}") print(f"\n--- Option A: joint fit (shared_noise, {len(matched_clean)} pools, " - f"warm-start from C, no chain dummies) ---") + f"warm-start from C, no chain dummies{gas_label}) ---") result_sn = fit_joint( matched_clean, mode="shared_noise", init_from_option_c=good_pools, maxiter=JOINT_MAXITER, drop_chain_dummies=True, + fix_gas_to_chain=fix_gas_to_chain, ) print(f" Loss: {result_sn['init_loss']:.4f} -> {result_sn['loss']:.4f}") print(f" Converged: {result_sn['converged']}") @@ -170,115 +189,103 @@ def run_option_rf(matched, option_c_results): attr_names = [attr_names_full[i] for i in non_chain_mask] k_attr = len(attr_names) - print(f"\n--- Option RF: 2-stage mapping ({n_pools} pools, {k_attr} features) ---") + # Detect if gas was fixed in Option C + gas_fixed = any(good_pools[p].get("gas_fixed", False) for p in pool_ids) + + if gas_fixed: + print(f"\n--- Option RF: 2-stage mapping ({n_pools} pools, {k_attr} features, " + f"cadence only — gas fixed) ---") + else: + print(f"\n--- Option RF: 2-stage mapping ({n_pools} pools, {k_attr} features) ---") print(f" Features: {', '.join(attr_names)}") Y_cad = np.array([good_pools[p]["log_cadence"] for p in pool_ids]) Y_gas = np.array([good_pools[p]["log_gas"] for p in pool_ids]) - Y = np.column_stack([Y_cad, Y_gas]) ss_tot_cad = np.sum((Y_cad - Y_cad.mean()) ** 2) - ss_tot_gas = np.sum((Y_gas - Y_gas.mean()) ** 2) def compute_r2(y_true, y_pred, ss_tot): return 1 - np.sum((y_true - y_pred) ** 2) / max(ss_tot, 1e-10) - # ---- Ridge regression (multi-output via separate fits) ---- + # ---- Ridge regression (cadence only when gas is fixed) ---- alphas = np.logspace(-2, 4, 50) ridge_cad = RidgeCV(alphas=alphas, cv=None) # GCV/LOO built-in - ridge_gas = RidgeCV(alphas=alphas, cv=None) ridge_cad.fit(X_attr, Y_cad) - ridge_gas.fit(X_attr, Y_gas) - Y_ridge_train = np.column_stack([ridge_cad.predict(X_attr), - ridge_gas.predict(X_attr)]) - r2_ridge_cad = compute_r2(Y_cad, Y_ridge_train[:, 0], ss_tot_cad) - r2_ridge_gas = compute_r2(Y_gas, Y_ridge_train[:, 1], ss_tot_gas) + Y_ridge_cad_train = ridge_cad.predict(X_attr) + r2_ridge_cad = compute_r2(Y_cad, Y_ridge_cad_train, ss_tot_cad) - print(f"\n Ridge (alpha_cad={ridge_cad.alpha_:.1f}, alpha_gas={ridge_gas.alpha_:.1f}):") - print(f" In-sample R²: cadence={r2_ridge_cad:.3f}, gas={r2_ridge_gas:.3f}") + print(f"\n Ridge (alpha_cad={ridge_cad.alpha_:.1f}):") + print(f" In-sample R² cadence: {r2_ridge_cad:.3f}") - # Ridge LOO-CV + # Ridge LOO-CV (cadence only) loo = LeaveOneOut() - Y_ridge_loo = np.zeros_like(Y) + Y_ridge_cad_loo = np.zeros_like(Y_cad) for train_idx, test_idx in loo.split(X_attr): rc = RidgeCV(alphas=alphas, cv=None).fit(X_attr[train_idx], Y_cad[train_idx]) - rg = RidgeCV(alphas=alphas, cv=None).fit(X_attr[train_idx], Y_gas[train_idx]) - Y_ridge_loo[test_idx, 0] = rc.predict(X_attr[test_idx]) - Y_ridge_loo[test_idx, 1] = rg.predict(X_attr[test_idx]) + Y_ridge_cad_loo[test_idx] = rc.predict(X_attr[test_idx]) - r2_ridge_loo_cad = compute_r2(Y_cad, Y_ridge_loo[:, 0], ss_tot_cad) - r2_ridge_loo_gas = compute_r2(Y_gas, Y_ridge_loo[:, 1], ss_tot_gas) - print(f" LOO-CV R²: cadence={r2_ridge_loo_cad:.3f}, gas={r2_ridge_loo_gas:.3f}") - print(f" LOO-CV MAE: cadence={np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_ridge_loo[:, 0]))):.1f} min, " - f"gas=${np.mean(np.abs(np.exp(Y_gas) - np.exp(Y_ridge_loo[:, 1]))):.2f}") + r2_ridge_loo_cad = compute_r2(Y_cad, Y_ridge_cad_loo, ss_tot_cad) + print(f" LOO-CV R² cadence: {r2_ridge_loo_cad:.3f}") + print(f" LOO-CV MAE cadence: {np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_ridge_cad_loo))):.1f} min") # Ridge coefficients - print(f" Coefficients (cadence | gas):") - print(f" {'intercept':<20} {ridge_cad.intercept_:>7.3f} {ridge_gas.intercept_:>7.3f}") + print(f" Coefficients (cadence):") + print(f" {'intercept':<20} {ridge_cad.intercept_:>7.3f}") for j, name in enumerate(attr_names): - print(f" {name:<20} {ridge_cad.coef_[j]:>7.3f} {ridge_gas.coef_[j]:>7.3f}") + print(f" {name:<20} {ridge_cad.coef_[j]:>7.3f}") - # ---- Random Forest (reduced features) ---- + # ---- Random Forest (cadence only) ---- rf = RandomForestRegressor( n_estimators=200, max_depth=None, - min_samples_leaf=3, # stronger regularization - max_features=min(4, k_attr), # cap at 4 features per split + min_samples_leaf=3, + max_features=min(4, k_attr), random_state=42, n_jobs=-1, ) - rf.fit(X_attr, Y) - Y_rf_train = rf.predict(X_attr) + rf.fit(X_attr, Y_cad) + Y_rf_cad_train = rf.predict(X_attr) - r2_rf_cad = compute_r2(Y_cad, Y_rf_train[:, 0], ss_tot_cad) - r2_rf_gas = compute_r2(Y_gas, Y_rf_train[:, 1], ss_tot_gas) + r2_rf_cad = compute_r2(Y_cad, Y_rf_cad_train, ss_tot_cad) print(f"\n Random Forest (min_leaf=3, max_feat=4):") - print(f" In-sample R²: cadence={r2_rf_cad:.3f}, gas={r2_rf_gas:.3f}") + print(f" In-sample R² cadence: {r2_rf_cad:.3f}") # RF LOO-CV - Y_rf_loo = np.zeros_like(Y) + Y_rf_cad_loo = np.zeros_like(Y_cad) for train_idx, test_idx in loo.split(X_attr): rf_loo = RandomForestRegressor( n_estimators=200, max_depth=None, min_samples_leaf=3, max_features=min(4, k_attr), random_state=42, n_jobs=-1, ) - rf_loo.fit(X_attr[train_idx], Y[train_idx]) - Y_rf_loo[test_idx] = rf_loo.predict(X_attr[test_idx]) + rf_loo.fit(X_attr[train_idx], Y_cad[train_idx]) + Y_rf_cad_loo[test_idx] = rf_loo.predict(X_attr[test_idx]) - r2_rf_loo_cad = compute_r2(Y_cad, Y_rf_loo[:, 0], ss_tot_cad) - r2_rf_loo_gas = compute_r2(Y_gas, Y_rf_loo[:, 1], ss_tot_gas) - print(f" LOO-CV R²: cadence={r2_rf_loo_cad:.3f}, gas={r2_rf_loo_gas:.3f}") - print(f" LOO-CV MAE: cadence={np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_rf_loo[:, 0]))):.1f} min, " - f"gas=${np.mean(np.abs(np.exp(Y_gas) - np.exp(Y_rf_loo[:, 1]))):.2f}") + r2_rf_loo_cad = compute_r2(Y_cad, Y_rf_cad_loo, ss_tot_cad) + print(f" LOO-CV R² cadence: {r2_rf_loo_cad:.3f}") + print(f" LOO-CV MAE cadence: {np.mean(np.abs(np.exp(Y_cad) - np.exp(Y_rf_cad_loo))):.1f} min") print(f"\n Feature importances:") for j, name in enumerate(attr_names): print(f" {name:<20} {rf.feature_importances_[j]:.3f}") # ---- Pick best LOO model ---- - ridge_loo_total = r2_ridge_loo_cad + r2_ridge_loo_gas - rf_loo_total = r2_rf_loo_cad + r2_rf_loo_gas - best = "ridge" if ridge_loo_total >= rf_loo_total else "rf" - print(f"\n Best LOO model: {best} (ridge={ridge_loo_total:.3f} vs rf={rf_loo_total:.3f})") + best = "ridge" if r2_ridge_loo_cad >= r2_rf_loo_cad else "rf" + print(f"\n Best LOO model: {best} (ridge={r2_ridge_loo_cad:.3f} vs rf={r2_rf_loo_cad:.3f})") if best == "ridge": - Y_best_train = Y_ridge_train - Y_best_loo = Y_ridge_loo + Y_best_cad_train = Y_ridge_cad_train + Y_best_cad_loo = Y_ridge_cad_loo r2_best_cad = r2_ridge_cad - r2_best_gas = r2_ridge_gas r2_best_loo_cad = r2_ridge_loo_cad - r2_best_loo_gas = r2_ridge_loo_gas else: - Y_best_train = Y_rf_train - Y_best_loo = Y_rf_loo + Y_best_cad_train = Y_rf_cad_train + Y_best_cad_loo = Y_rf_cad_loo r2_best_cad = r2_rf_cad - r2_best_gas = r2_rf_gas r2_best_loo_cad = r2_rf_loo_cad - r2_best_loo_gas = r2_rf_loo_gas - # Build result dict using the best model's predictions + # Build result dict — gas comes from Option C (which used chain-level values) noise_all = np.array([good_pools[p]["noise_coeffs"] for p in pool_ids]) result = { @@ -288,21 +295,20 @@ def compute_r2(y_true, y_pred, ss_tot): "best_model": best, "predictions": {}, "loo_predictions": {}, - "noise_coeffs": noise_all, # from Option C + "noise_coeffs": noise_all, "r2_train_cad": r2_best_cad, - "r2_train_gas": r2_best_gas, "r2_loo_cad": r2_best_loo_cad, - "r2_loo_gas": r2_best_loo_gas, } for i, pid in enumerate(pool_ids): + log_gas_fixed = good_pools[pid]["log_gas"] result["predictions"][pid] = { - "log_cadence": float(Y_best_train[i, 0]), - "log_gas": float(Y_best_train[i, 1]), + "log_cadence": float(Y_best_cad_train[i]), + "log_gas": float(log_gas_fixed), } result["loo_predictions"][pid] = { - "log_cadence": float(Y_best_loo[i, 0]), - "log_gas": float(Y_best_loo[i, 1]), + "log_cadence": float(Y_best_cad_loo[i]), + "log_gas": float(log_gas_fixed), } return result @@ -369,7 +375,12 @@ def r2(v_arb, v_noise, y): ji = joint_pid_to_idx[pid] x_attr = X_attr_joint[ji] log_cad_a = float(joint_result["bias_cad"]) + float(x_attr @ joint_result["W_cad"]) - log_gas_a = float(joint_result["bias_gas"]) + float(x_attr @ joint_result["W_gas"]) + if joint_result.get("fix_gas"): + gas_a = float(joint_result["gas_per_pool"][ji]) + log_gas_a = np.log(max(gas_a, 1e-6)) + else: + log_gas_a = float(joint_result["bias_gas"]) + float(x_attr @ joint_result["W_gas"]) + gas_a = np.exp(log_gas_a) noise_c_a = joint_result["noise_coeffs"][ji] v_arb_all_a = np.array(interpolate_pool_daily( @@ -379,7 +390,6 @@ def r2(v_arb, v_noise, y): v_noise_a = np.exp(x_obs @ noise_c_a) r2_a = r2(v_arb_a, v_noise_a, y_obs) cad_a = np.exp(log_cad_a) - gas_a = np.exp(log_gas_a) else: v_arb_a = np.full(len(y_obs), np.nan) v_noise_a = np.full(len(y_obs), np.nan) @@ -800,11 +810,11 @@ def main(): panel, matched = load_and_match() - # Step 1: Option C - option_c = run_option_c(matched) + # Step 1: Option C (gas fixed to chain-level costs) + option_c = run_option_c(matched, fix_gas_to_chain=True) - # Step 2: Option A (linear mapping) - joint_ppn, joint_sn = run_option_a(matched, option_c) + # Step 2: Option A (linear mapping, gas fixed) + joint_ppn, joint_sn = run_option_a(matched, option_c, fix_gas_to_chain=True) # Step 3: Option RF (random forest 2-stage) rf_result = run_option_rf(matched, option_c) From a2ae5a40faf5328d2b0368051e5e482456c64a37 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 23:07:40 +0000 Subject: [PATCH 035/115] fix: grid builder handles stale Binance data and multi-worker dispatch Clip panel dates to Binance price data range so pools with stale token data (BAL, MKR, BADGER, LIT) use their available overlap instead of failing. Snap sim start to next midnight for tokens starting mid-day. Set max_memory_days=0 and preslice_burnin=False to prevent negative start_idx. Workers load their own price data to avoid pickling large DataFrames across processes. --- scripts/build_pool_grids.py | 68 ++++++++++++++++++++++++++++++++----- 1 file changed, 59 insertions(+), 9 deletions(-) diff --git a/scripts/build_pool_grids.py b/scripts/build_pool_grids.py index 9b710c26..ecf3152b 100644 --- a/scripts/build_pool_grids.py +++ b/scripts/build_pool_grids.py @@ -215,8 +215,11 @@ def load_panel_and_match(train_days): if "tvl" not in panel.columns and "log_tvl" in panel.columns: panel["tvl"] = np.exp(panel["log_tvl"]) - cutoff = panel["obs_date"].max() - pd.Timedelta(days=train_days) - panel = panel[panel["obs_date"] >= cutoff].copy() + if train_days > 0: + cutoff = panel["obs_date"].max() - pd.Timedelta(days=train_days) + panel = panel[panel["obs_date"] >= cutoff].copy() + else: + panel = panel.copy() # no date filter — use all data binance_tokens = _get_binance_tokens() pools_meta = load_pools_metadata() @@ -342,6 +345,7 @@ def run_arb_sim(tokens, fee, initial_tvl, start, end, cadence, gas_cost, "arb_frequency": int(cadence), "chunk_period": 1440, "weight_interpolation_period": 1440, + "max_memory_days": 0, } if pool_type == "RECLAMM" and reclamm_params is not None: @@ -363,7 +367,7 @@ def run_arb_sim(tokens, fee, initial_tvl, start, end, cadence, gas_cost, result = do_run_on_historic_data( fp, params, lp_supply_df=lp_supply_df, verbose=False, - price_data=price_data, + price_data=price_data, preslice_burnin=False, ) reserves = np.array(result["reserves"]) @@ -405,6 +409,9 @@ def _run_cadence_sweep(pool_info, cadence, gas_costs): reclamm_params = pool_info.get("reclamm_params") price_data = pool_info.get("price_data") + if price_data is None: + sorted_tokens = sorted(tokens) + price_data = get_historic_parquet_data(sorted_tokens, ["close"]) daily_rows = [] summary_rows = [] @@ -527,10 +534,51 @@ def main(): print(f"Found {len(pools)} matchable 2-token pools\n") for p in pools: - lp_df, tvl = build_lp_supply_df(p["panel_data"]) + # Preload price data to determine actual date coverage + sorted_tokens = sorted(p["tokens"]) + price_data = get_historic_parquet_data(sorted_tokens, ["close"]) + p["price_data"] = price_data + + if len(price_data) == 0: + p["panel_data"] = p["panel_data"].iloc[:0] # empty + p["lp_supply_df"] = None + p["initial_tvl"] = 0.0 + p["start"], p["end"] = "2000-01-01", "2000-01-01" + continue + + # Clip panel to price data's actual date range + price_dates = pd.to_datetime(price_data.index, unit="ms") + price_start = price_dates.min().normalize() + price_end = price_dates.max().normalize() + panel_data = p["panel_data"] + panel_data = panel_data[ + (panel_data["obs_date"] >= price_start) + & (panel_data["obs_date"] <= price_end) + ].copy() + p["panel_data"] = panel_data + + lp_df, tvl = build_lp_supply_df(panel_data) p["lp_supply_df"] = lp_df p["initial_tvl"] = tvl - p["start"], p["end"] = get_date_range(p["panel_data"]) + + # Start/end must be midnight timestamps that exist in the price data. + # start_and_end_calcs does an exact unix match and assumes alignment. + # If the price data starts mid-day (e.g. COW at noon), advance to + # the next midnight so the sim has a clean day boundary. + if len(panel_data) > 0: + first_midnight = (price_dates.min() + pd.Timedelta(days=1)).normalize() + last_midnight = price_dates.max().normalize() + first_midnight_ms = int(first_midnight.timestamp() * 1000) + last_midnight_ms = int(last_midnight.timestamp() * 1000) + # Verify these timestamps exist in the price data + if first_midnight_ms in price_data.index and last_midnight_ms in price_data.index: + p["start"] = first_midnight.strftime("%Y-%m-%d %H:%M:%S") + p["end"] = last_midnight.strftime("%Y-%m-%d %H:%M:%S") + else: + # Fallback: use panel dates (works when price data covers full range) + p["start"], p["end"] = get_date_range(panel_data) + else: + p["start"], p["end"] = "2000-01-01", "2000-01-01" pools = [p for p in pools if p["panel_data"]["obs_date"].nunique() >= 14] pools.sort(key=lambda p: p["initial_tvl"], reverse=True) @@ -580,9 +628,8 @@ def main(): t0 = time.time() - # Preload price data once per pool (avoids re-reading parquet per run) - sorted_tokens = sorted(tokens) - price_data = get_historic_parquet_data(sorted_tokens, ["close"]) + # Price data was preloaded during panel clipping + price_data = pool["price_data"] pool_info = { "tokens": tokens, @@ -594,7 +641,10 @@ def main(): "weights": pool["weights"], "pool_type": pool_type, "reclamm_params": pool.get("reclamm_params"), - "price_data": price_data, + # Only pass preloaded price_data in single-worker mode. + # For multi-worker, each subprocess loads its own to avoid + # pickling multi-million-row DataFrames across processes. + "price_data": price_data if args.workers <= 1 else None, } all_daily_rows = [] From f6ef0ecd916e29e62c4dfabe5085637bbf447cb7 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 9 Mar 2026 23:18:13 +0000 Subject: [PATCH 036/115] test: comprehensive tests for fixed-gas calibration and Binance volatility 87 new tests covering: - CHAIN_GAS_USD constants (pinned values, completeness) - pack/unpack fixed-gas params (roundtrip, shape, position) - pool_loss_fixed_gas (zero-when-perfect, matches free-gas, gradients) - per-pool fit fixed-gas (gas_usd pinned, gas_fixed flag, loss decreases) - fit_all_pools with fix_gas_to_chain (chain-level gas matching) - TOKEN_MAP resolution (wrapped native, LSTs, stablecoins, vault tokens) - compute_binance_pair_volatility (synthetic data, edge cases) - replace_panel_volatility_with_binance (immutability, no NaN) - joint fit fixed-gas (prepare, pack/unpack, loss, bounds, fit, predict) --- tests/calibration/test_joint_fit_fixed_gas.py | 411 ++++++++++++++++++ tests/calibration/test_loss_fixed_gas.py | 281 ++++++++++++ .../test_per_pool_fit_fixed_gas.py | 258 +++++++++++ .../calibration/test_pool_data_volatility.py | 232 ++++++++++ 4 files changed, 1182 insertions(+) create mode 100644 tests/calibration/test_joint_fit_fixed_gas.py create mode 100644 tests/calibration/test_loss_fixed_gas.py create mode 100644 tests/calibration/test_per_pool_fit_fixed_gas.py create mode 100644 tests/calibration/test_pool_data_volatility.py diff --git a/tests/calibration/test_joint_fit_fixed_gas.py b/tests/calibration/test_joint_fit_fixed_gas.py new file mode 100644 index 00000000..6cbd3cc3 --- /dev/null +++ b/tests/calibration/test_joint_fit_fixed_gas.py @@ -0,0 +1,411 @@ +"""Tests for fixed-gas mode in quantammsim.calibration.joint_fit.""" + +import jax +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, POOL_PREFIXES + + +@pytest.fixture +def matched_data(synthetic_daily_grid, synthetic_panel, tmp_path): + """Build matched data dict from synthetic fixtures.""" + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + +class TestPrepareJointDataFixedGas: + """Test prepare_joint_data with fix_gas_to_chain=True.""" + + def test_pool_data_has_fixed_log_gas(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + for pd_i in jdata.pool_data: + assert "fixed_log_gas" in pd_i + + def test_pool_data_no_fixed_log_gas_by_default(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=False) + for pd_i in jdata.pool_data: + assert "fixed_log_gas" not in pd_i + + def test_fixed_log_gas_values_match_chain(self, matched_data): + """fixed_log_gas should be log(CHAIN_GAS_USD[chain]).""" + from quantammsim.calibration.joint_fit import prepare_joint_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + + for i, pid in enumerate(jdata.pool_ids): + chain = matched_data[pid]["chain"] + expected_gas = CHAIN_GAS_USD.get(chain, 1.0) + expected_log_gas = np.log(max(expected_gas, 1e-6)) + np.testing.assert_allclose( + float(jdata.pool_data[i]["fixed_log_gas"]), + expected_log_gas, + rtol=1e-6, + ) + + def test_mainnet_gas_is_log_1(self, matched_data): + """MAINNET pools should have fixed_log_gas = log(1.0) = 0.0.""" + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + for i, pid in enumerate(jdata.pool_ids): + if matched_data[pid]["chain"] == "MAINNET": + np.testing.assert_allclose( + float(jdata.pool_data[i]["fixed_log_gas"]), + 0.0, + atol=1e-6, + ) + + def test_arbitrum_gas_is_log_001(self, matched_data): + """ARBITRUM pools should have fixed_log_gas = log(0.01).""" + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + for i, pid in enumerate(jdata.pool_ids): + if matched_data[pid]["chain"] == "ARBITRUM": + np.testing.assert_allclose( + float(jdata.pool_data[i]["fixed_log_gas"]), + np.log(0.01), + rtol=1e-6, + ) + + +class TestPackUnpackJointFixedGas: + """Test packing/unpacking joint params with fix_gas=True.""" + + def test_pack_fixed_gas_shape(self): + from quantammsim.calibration.joint_fit import pack_joint_params_fixed_gas + + k_attr = 6 + n_pools = 2 + noise_params = jnp.zeros((n_pools, K_OBS)) + flat = pack_joint_params_fixed_gas( + 1.0, jnp.zeros(k_attr), noise_params + ) + # Layout: [bias_cad, W_cad(6), noise(2*8)] = 1+6+16 = 23 + assert flat.shape == (1 + k_attr + n_pools * K_OBS,) + + def test_pack_fixed_gas_shorter_than_free(self): + from quantammsim.calibration.joint_fit import ( + pack_joint_params, + pack_joint_params_fixed_gas, + ) + + k_attr = 6 + n_pools = 2 + noise = jnp.zeros((n_pools, K_OBS)) + + free = pack_joint_params(1.0, 2.0, jnp.zeros(k_attr), + jnp.zeros(k_attr), noise) + fixed = pack_joint_params_fixed_gas(1.0, jnp.zeros(k_attr), noise) + # Fixed is shorter by: 1 (bias_gas) + k_attr (W_gas) + assert free.shape[0] - fixed.shape[0] == 1 + k_attr + + def test_unpack_fixed_gas_no_gas_keys(self): + from quantammsim.calibration.joint_fit import ( + pack_joint_params_fixed_gas, + unpack_joint_params, + ) + + k_attr = 6 + n_pools = 2 + noise = jnp.zeros((n_pools, K_OBS)) + flat = pack_joint_params_fixed_gas( + 1.0, jnp.ones(k_attr) * 0.5, noise + ) + + config = {"k_attr": k_attr, "n_pools": n_pools, + "mode": "per_pool_noise", "fix_gas": True} + params = unpack_joint_params(flat, config) + + assert "bias_cad" in params + assert "W_cad" in params + assert "noise_coeffs" in params + assert "bias_gas" not in params + assert "W_gas" not in params + + def test_unpack_roundtrip_per_pool_noise(self): + from quantammsim.calibration.joint_fit import ( + pack_joint_params_fixed_gas, + unpack_joint_params, + ) + + k_attr = 4 + n_pools = 3 + bias_cad = 2.5 + W_cad = jnp.array([0.1, -0.2, 0.3, -0.4]) + noise = jnp.arange(n_pools * K_OBS, dtype=float).reshape(n_pools, K_OBS) + + flat = pack_joint_params_fixed_gas(bias_cad, W_cad, noise) + config = {"k_attr": k_attr, "n_pools": n_pools, + "mode": "per_pool_noise", "fix_gas": True} + params = unpack_joint_params(flat, config) + + np.testing.assert_allclose(params["bias_cad"], bias_cad) + np.testing.assert_allclose(params["W_cad"], W_cad) + np.testing.assert_allclose(params["noise_coeffs"], noise) + + def test_unpack_roundtrip_shared_noise(self): + from quantammsim.calibration.joint_fit import ( + pack_joint_params_fixed_gas, + unpack_joint_params, + ) + + k_attr = 4 + bias_cad = 1.5 + W_cad = jnp.array([0.5, -0.5, 0.1, -0.1]) + # shared_noise: (1+k_attr, K_OBS) = (5, 8) + noise = jnp.arange((1 + k_attr) * K_OBS, dtype=float).reshape( + 1 + k_attr, K_OBS + ) + + flat = pack_joint_params_fixed_gas(bias_cad, W_cad, noise) + config = {"k_attr": k_attr, "n_pools": 99, + "mode": "shared_noise", "fix_gas": True} + params = unpack_joint_params(flat, config) + + np.testing.assert_allclose(params["bias_cad"], bias_cad) + np.testing.assert_allclose(params["W_cad"], W_cad) + np.testing.assert_allclose(params["bias_noise"], noise[0]) + np.testing.assert_allclose(params["W_noise"], noise[1:]) + + +class TestJointLossFixedGas: + """Test joint loss function with fix_gas=True.""" + + def _make_loss_fn(self, matched_data, mode="per_pool_noise"): + from quantammsim.calibration.joint_fit import ( + make_initial_joint_params, + make_joint_loss_fn, + prepare_joint_data, + ) + + jdata = prepare_joint_data( + matched_data, fix_gas_to_chain=True + ) + init = make_initial_joint_params( + jdata, mode=mode, fix_gas=True + ) + loss_fn = make_joint_loss_fn( + jdata, mode=mode, fix_gas=True + ) + return loss_fn, init, jdata + + def test_loss_scalar(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + loss = loss_fn(init) + assert loss.shape == () + + def test_loss_positive(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + loss = loss_fn(init) + assert float(loss) >= 0 + + def test_loss_differentiable(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + grad = jax.grad(loss_fn)(init) + assert grad.shape == init.shape + assert jnp.all(jnp.isfinite(grad)) + + def test_loss_grad_nonzero(self, matched_data): + loss_fn, init, _ = self._make_loss_fn(matched_data) + grad = jax.grad(loss_fn)(init) + assert float(jnp.sum(jnp.abs(grad))) > 0 + + def test_no_gas_regularization(self, matched_data): + """With fix_gas=True, alpha_gas should have no effect.""" + from quantammsim.calibration.joint_fit import ( + make_initial_joint_params, + make_joint_loss_fn, + prepare_joint_data, + ) + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + init = make_initial_joint_params(jdata, mode="per_pool_noise", fix_gas=True) + + loss_fn_a = make_joint_loss_fn( + jdata, mode="per_pool_noise", fix_gas=True, alpha_gas=0.0 + ) + loss_fn_b = make_joint_loss_fn( + jdata, mode="per_pool_noise", fix_gas=True, alpha_gas=100.0 + ) + + np.testing.assert_allclose( + float(loss_fn_a(init)), float(loss_fn_b(init)), rtol=1e-6 + ) + + def test_shared_noise_mode(self, matched_data): + loss_fn, init, _ = self._make_loss_fn( + matched_data, mode="shared_noise" + ) + loss = loss_fn(init) + assert loss.shape == () + assert float(loss) >= 0 + + def test_init_param_count_per_pool_noise(self, matched_data): + """Verify parameter count: 1(bias_cad) + k_attr(W_cad) + n_pools*K_OBS.""" + _, init, jdata = self._make_loss_fn(matched_data) + k_attr = jdata.x_attr.shape[1] + n_pools = len(jdata.pool_data) + expected = 1 + k_attr + n_pools * K_OBS + assert init.shape[0] == expected + + def test_init_param_count_shared_noise(self, matched_data): + """Verify: 1(bias_cad) + k_attr(W_cad) + (1+k_attr)*K_OBS.""" + _, init, jdata = self._make_loss_fn( + matched_data, mode="shared_noise" + ) + k_attr = jdata.x_attr.shape[1] + expected = 1 + k_attr + (1 + k_attr) * K_OBS + assert init.shape[0] == expected + + +class TestFitJointFixedGas: + """Test fit_joint with fix_gas_to_chain=True.""" + + def test_returns_result(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=20, + ) + assert isinstance(result, dict) + for key in ["bias_cad", "W_cad", "loss", "converged", "fix_gas"]: + assert key in result, f"Missing key: {key}" + + def test_fix_gas_flag_stored(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=10, + ) + assert result["fix_gas"] is True + + def test_gas_per_pool_stored(self, matched_data): + """Result should contain gas_per_pool with chain-level values.""" + from quantammsim.calibration.joint_fit import fit_joint + from quantammsim.calibration.loss import CHAIN_GAS_USD + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=10, + ) + assert "gas_per_pool" in result + for i, pid in enumerate(result["pool_ids"]): + chain = matched_data[pid]["chain"] + expected = CHAIN_GAS_USD.get(chain, 1.0) + assert result["gas_per_pool"][i] == expected + + def test_loss_decreases(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=50, + ) + assert result["loss"] <= result["init_loss"] + + def test_w_gas_is_zeros(self, matched_data): + """With fixed gas, W_gas should be zeros (placeholder).""" + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=10, + ) + np.testing.assert_allclose(result["W_gas"], 0.0) + + def test_bias_gas_is_zero(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, maxiter=10, + ) + assert result["bias_gas"] == 0.0 + + def test_warm_start_from_option_c(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + from quantammsim.calibration.per_pool_fit import fit_all_pools + + option_c = fit_all_pools(matched_data, fix_gas_to_chain=True) + result = fit_joint( + matched_data, mode="per_pool_noise", + fix_gas_to_chain=True, + init_from_option_c=option_c, + maxiter=20, + ) + assert result["loss"] >= 0 + + def test_shared_noise_fixed_gas(self, matched_data): + from quantammsim.calibration.joint_fit import fit_joint + + result = fit_joint( + matched_data, mode="shared_noise", + fix_gas_to_chain=True, maxiter=20, + ) + assert "W_noise" in result + assert result["fix_gas"] is True + + def test_predict_new_pool_fixed_gas(self, matched_data): + """predict_new_pool_joint should still work with fixed-gas results.""" + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint( + matched_data, mode="shared_noise", + fix_gas_to_chain=True, maxiter=20, + ) + k_attr = result["W_cad"].shape[0] + x_attr_new = np.zeros(k_attr) + pred = predict_new_pool_joint(result, x_attr_new) + assert "cadence_minutes" in pred + assert pred["cadence_minutes"] > 0 + # gas_usd comes from bias_gas=0 + W_gas=0 → exp(0)=1.0 + assert pred["gas_usd"] > 0 + + +class TestMakeBoundsFixedGas: + """Test _make_bounds with fix_gas=True.""" + + def test_bounds_count_per_pool_noise(self): + from quantammsim.calibration.joint_fit import _make_bounds + + k_attr = 6 + n_pools = 3 + bounds = _make_bounds(k_attr, n_pools, "per_pool_noise", fix_gas=True) + # 1(bias_cad) + 6(W_cad) + 3*8(noise) = 31 + assert len(bounds) == 1 + k_attr + n_pools * K_OBS + + def test_bounds_count_shared_noise(self): + from quantammsim.calibration.joint_fit import _make_bounds + + k_attr = 6 + n_pools = 3 + bounds = _make_bounds(k_attr, n_pools, "shared_noise", fix_gas=True) + # 1(bias_cad) + 6(W_cad) + (1+6)*8(noise) = 63 + assert len(bounds) == 1 + k_attr + (1 + k_attr) * K_OBS + + def test_bounds_fewer_with_fixed_gas(self): + from quantammsim.calibration.joint_fit import _make_bounds + + k_attr = 6 + n_pools = 3 + free = _make_bounds(k_attr, n_pools, "per_pool_noise", fix_gas=False) + fixed = _make_bounds(k_attr, n_pools, "per_pool_noise", fix_gas=True) + # Difference: 1(bias_gas) + k_attr(W_gas) + assert len(free) - len(fixed) == 1 + k_attr diff --git a/tests/calibration/test_loss_fixed_gas.py b/tests/calibration/test_loss_fixed_gas.py new file mode 100644 index 00000000..f3cbf70a --- /dev/null +++ b/tests/calibration/test_loss_fixed_gas.py @@ -0,0 +1,281 @@ +"""Tests for fixed-gas extensions in quantammsim.calibration.loss.""" + +import jax +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, N_DAYS + + +class TestChainGasUSD: + """Test CHAIN_GAS_USD constants are correct and complete.""" + + def test_known_chains(self): + from quantammsim.calibration.loss import CHAIN_GAS_USD + + assert CHAIN_GAS_USD["MAINNET"] == 1.0 + assert CHAIN_GAS_USD["POLYGON"] == 0.005 + assert CHAIN_GAS_USD["GNOSIS"] == 0.001 + assert CHAIN_GAS_USD["ARBITRUM"] == 0.01 + assert CHAIN_GAS_USD["BASE"] == 0.005 + assert CHAIN_GAS_USD["SONIC"] == 0.005 + + def test_all_values_positive(self): + from quantammsim.calibration.loss import CHAIN_GAS_USD + + for chain, cost in CHAIN_GAS_USD.items(): + assert cost > 0, f"{chain} gas cost must be positive" + + def test_mainnet_most_expensive(self): + from quantammsim.calibration.loss import CHAIN_GAS_USD + + mainnet = CHAIN_GAS_USD["MAINNET"] + for chain, cost in CHAIN_GAS_USD.items(): + if chain != "MAINNET": + assert cost < mainnet, f"{chain} should be cheaper than MAINNET" + + def test_six_chains(self): + from quantammsim.calibration.loss import CHAIN_GAS_USD + + assert len(CHAIN_GAS_USD) == 6 + + +class TestPackUnpackFixedGas: + """Test pack/unpack for fixed-gas param vectors.""" + + def test_pack_shape(self): + from quantammsim.calibration.loss import pack_params_fixed_gas + + flat = pack_params_fixed_gas(2.5, jnp.zeros(K_OBS)) + assert flat.shape == (1 + K_OBS,) + + def test_pack_shape_is_one_shorter_than_free(self): + from quantammsim.calibration.loss import pack_params, pack_params_fixed_gas + + free = pack_params(2.5, 0.0, jnp.zeros(K_OBS)) + fixed = pack_params_fixed_gas(2.5, jnp.zeros(K_OBS)) + assert free.shape[0] == fixed.shape[0] + 1 + + def test_roundtrip(self): + from quantammsim.calibration.loss import ( + pack_params_fixed_gas, + unpack_params_fixed_gas, + ) + + log_cad = 2.5 + noise_coeffs = jnp.array([1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0]) + + flat = pack_params_fixed_gas(log_cad, noise_coeffs) + lc, nc = unpack_params_fixed_gas(flat) + + np.testing.assert_allclose(lc, log_cad) + np.testing.assert_allclose(nc, noise_coeffs) + + def test_unpack_log_cadence_position(self): + """log_cadence is the first element.""" + from quantammsim.calibration.loss import pack_params_fixed_gas + + flat = pack_params_fixed_gas(3.14, jnp.ones(K_OBS) * 99.0) + np.testing.assert_allclose(flat[0], 3.14) + + def test_unpack_noise_coeffs_position(self): + """noise_coeffs are elements [1:].""" + from quantammsim.calibration.loss import pack_params_fixed_gas + + nc = jnp.arange(1, K_OBS + 1, dtype=float) + flat = pack_params_fixed_gas(0.0, nc) + np.testing.assert_allclose(flat[1:], nc) + + +class TestPoolLossFixedGas: + """Test pool_loss_fixed_gas with pinned numerical values.""" + + def _make_params(self, log_cad=None, noise_coeffs=None): + from quantammsim.calibration.loss import pack_params_fixed_gas + + if log_cad is None: + log_cad = float(jnp.log(jnp.array(12.0))) + if noise_coeffs is None: + noise_coeffs = jnp.zeros(K_OBS).at[0].set(8.0) + return pack_params_fixed_gas(log_cad, noise_coeffs) + + def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = jnp.array(np.arange(n_obs) % n_days) + y_obs = jnp.ones(n_obs) * 9.0 + return jnp.array(synthetic_x_obs), y_obs, day_indices + + def test_scalar_output(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + loss = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + assert loss.shape == () + + def test_positive(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + loss = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + assert float(loss) >= 0 + + def test_zero_when_perfect(self, synthetic_pool_coeffs, synthetic_x_obs): + """Construct y_obs = log(V_arb + V_noise) exactly, verify loss ≈ 0.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.loss import ( + noise_volume, + pack_params_fixed_gas, + pool_loss_fixed_gas, + ) + + log_cad = jnp.log(jnp.array(12.0)) + fixed_log_gas = jnp.log(jnp.array(1.0)) + noise_coeffs = jnp.zeros(K_OBS).at[0].set(8.0) + + v_arb_all = interpolate_pool_daily( + synthetic_pool_coeffs, log_cad, jnp.exp(fixed_log_gas) + ) + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = jnp.array(np.arange(n_obs) % n_days) + v_arb = v_arb_all[day_indices] + v_noise = noise_volume(noise_coeffs, jnp.array(synthetic_x_obs)) + y_obs = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + + params = pack_params_fixed_gas(float(log_cad), noise_coeffs) + loss = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, + jnp.array(synthetic_x_obs), y_obs, day_indices, + ) + assert float(loss) < 1e-6 + + def test_matches_free_gas_at_same_value( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Fixed-gas loss should equal free-gas loss when gas matches.""" + from quantammsim.calibration.loss import ( + pack_params, + pack_params_fixed_gas, + pool_loss, + pool_loss_fixed_gas, + ) + + log_cad = float(jnp.log(jnp.array(12.0))) + log_gas = float(jnp.log(jnp.array(1.0))) + noise_coeffs = jnp.zeros(K_OBS).at[0].set(8.0) + + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + free_params = pack_params(log_cad, log_gas, noise_coeffs) + fixed_params = pack_params_fixed_gas(log_cad, noise_coeffs) + + loss_free = pool_loss( + free_params, synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + loss_fixed = pool_loss_fixed_gas( + fixed_params, jnp.array(log_gas), synthetic_pool_coeffs, + x_obs, y_obs, day_indices, + ) + np.testing.assert_allclose(float(loss_free), float(loss_fixed), rtol=1e-6) + + def test_different_fixed_gas_different_loss( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Different gas values should give different losses.""" + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + + loss_low = pool_loss_fixed_gas( + params, jnp.log(jnp.array(0.01)), + synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + loss_high = pool_loss_fixed_gas( + params, jnp.log(jnp.array(10.0)), + synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + assert float(loss_low) != float(loss_high) + + def test_grad_wrt_params_only(self, synthetic_pool_coeffs, synthetic_x_obs): + """Gradient is only w.r.t. params_flat (argnums=0), not fixed_log_gas.""" + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + + grad = jax.grad(pool_loss_fixed_gas, argnums=0)( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + assert grad.shape == (1 + K_OBS,) + assert jnp.all(jnp.isfinite(grad)) + + def test_no_grad_wrt_fixed_gas(self, synthetic_pool_coeffs, synthetic_x_obs): + """fixed_log_gas should not be in the gradient (it's a constant).""" + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + + # Gradient w.r.t. argnums=0 has shape (1+K_OBS,) — no gas element + grad = jax.grad(pool_loss_fixed_gas, argnums=0)( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + # If there were a gas gradient, shape would be (2+K_OBS,) + assert grad.shape[0] == 1 + K_OBS + + def test_grad_wrt_log_cadence_finite( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + + grad = jax.grad(pool_loss_fixed_gas, argnums=0)( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + assert jnp.isfinite(grad[0]) # log_cadence gradient + + def test_grad_wrt_noise_coeffs_finite( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + from quantammsim.calibration.loss import pool_loss_fixed_gas + + params = self._make_params() + x_obs, y_obs, day_indices = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + fixed_log_gas = jnp.log(jnp.array(1.0)) + + grad = jax.grad(pool_loss_fixed_gas, argnums=0)( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + ) + assert jnp.all(jnp.isfinite(grad[1:])) # noise_coeffs gradients diff --git a/tests/calibration/test_per_pool_fit_fixed_gas.py b/tests/calibration/test_per_pool_fit_fixed_gas.py new file mode 100644 index 00000000..f0a7d7ff --- /dev/null +++ b/tests/calibration/test_per_pool_fit_fixed_gas.py @@ -0,0 +1,258 @@ +"""Tests for fixed-gas mode in quantammsim.calibration.per_pool_fit.""" + +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, N_DAYS, POOL_IDS_FULL, POOL_PREFIXES + + +class TestInitialGuessFixedGas: + """Test make_initial_guess_fixed_gas.""" + + def test_shape(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import make_initial_guess_fixed_gas + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + init = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) + assert init.shape == (1 + K_OBS,) + + def test_one_shorter_than_free(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import ( + make_initial_guess, + make_initial_guess_fixed_gas, + ) + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + free = make_initial_guess(synthetic_x_obs, y_obs) + fixed = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) + assert free.shape[0] == fixed.shape[0] + 1 + + def test_log_cadence_default(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import make_initial_guess_fixed_gas + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + init = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) + np.testing.assert_allclose(init[0], np.log(12.0), atol=0.01) + + def test_noise_from_ols(self, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import make_initial_guess_fixed_gas + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + init = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) + noise_coeffs = init[1:] + assert len(noise_coeffs) == K_OBS + assert np.all(np.isfinite(noise_coeffs)) + + def test_noise_matches_free_gas_noise(self, synthetic_x_obs): + """OLS noise coeffs should be identical for free and fixed-gas init.""" + from quantammsim.calibration.per_pool_fit import ( + make_initial_guess, + make_initial_guess_fixed_gas, + ) + + n_obs = synthetic_x_obs.shape[0] + y_obs = np.ones(n_obs) * 9.0 + free = make_initial_guess(synthetic_x_obs, y_obs) + fixed = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) + np.testing.assert_allclose(free[2:], fixed[1:]) + + +class TestFitSinglePoolFixedGas: + """Test fit_single_pool with fixed_gas_usd.""" + + def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = np.arange(n_obs) % n_days + y_obs = np.ones(n_obs) * 9.0 + return synthetic_x_obs, y_obs, day_indices + + def test_returns_result_dict(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0 + ) + for key in [ + "log_cadence", "log_gas", "noise_coeffs", "loss", + "converged", "gas_fixed", + ]: + assert key in result, f"Missing key: {key}" + + def test_gas_fixed_flag(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0 + ) + assert result["gas_fixed"] is True + + def test_free_gas_flag(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx + ) + assert result["gas_fixed"] is False + + def test_gas_usd_pinned(self, synthetic_pool_coeffs, synthetic_x_obs): + """gas_usd in result must exactly match the fixed value.""" + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + for gas_val in [0.001, 0.01, 0.5, 1.0, 5.0]: + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, + fixed_gas_usd=gas_val, + ) + assert result["gas_usd"] == gas_val + + def test_log_gas_pinned(self, synthetic_pool_coeffs, synthetic_x_obs): + """log_gas must equal log(fixed_gas_usd).""" + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=2.5, + ) + np.testing.assert_allclose( + result["log_gas"], np.log(2.5), rtol=1e-6, + ) + + def test_cadence_in_range(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, + ) + cadence = result["cadence_minutes"] + assert 1.0 <= cadence <= 60.0 + + def test_noise_coeffs_length(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, + ) + assert len(result["noise_coeffs"]) == K_OBS + + def test_loss_decreases_from_init(self, synthetic_pool_coeffs, synthetic_x_obs): + from quantammsim.calibration.loss import pack_params_fixed_gas, pool_loss_fixed_gas + from quantammsim.calibration.per_pool_fit import ( + fit_single_pool, + make_initial_guess_fixed_gas, + ) + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + init = make_initial_guess_fixed_gas(x_obs, y_obs) + fixed_log_gas = jnp.float64(np.log(1.0)) + init_loss = float(pool_loss_fixed_gas( + jnp.array(init), fixed_log_gas, synthetic_pool_coeffs, + jnp.array(x_obs), jnp.array(y_obs), jnp.array(day_idx), + )) + + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, + ) + assert result["loss"] <= init_loss + + def test_different_fixed_gas_different_cadence( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Different gas values should (generally) lead to different fitted cadences.""" + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx = self._make_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + r_low = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=0.001, + ) + r_high = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=5.0, + ) + # Cadences should differ (gas-cadence tradeoff) + assert abs(r_low["log_cadence"] - r_high["log_cadence"]) > 0.01 + + +class TestFitAllPoolsFixedGas: + """Test fit_all_pools with fix_gas_to_chain=True.""" + + def _make_matched(self, synthetic_daily_grid, synthetic_panel, tmp_path): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + def test_all_gas_fixed( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=True) + for prefix, res in results.items(): + assert res["gas_fixed"] is True + + def test_gas_matches_chain( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """Each pool's gas_usd should match CHAIN_GAS_USD[chain].""" + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=True) + for prefix, res in results.items(): + chain = res["chain"] + expected = CHAIN_GAS_USD.get(chain, 1.0) + assert res["gas_usd"] == expected, ( + f"{prefix} ({chain}): gas_usd={res['gas_usd']} != {expected}" + ) + + def test_free_gas_not_fixed( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=False) + for prefix, res in results.items(): + assert res["gas_fixed"] is False diff --git a/tests/calibration/test_pool_data_volatility.py b/tests/calibration/test_pool_data_volatility.py new file mode 100644 index 00000000..6c5792ee --- /dev/null +++ b/tests/calibration/test_pool_data_volatility.py @@ -0,0 +1,232 @@ +"""Tests for Binance volatility and TOKEN_MAP in quantammsim.calibration.pool_data.""" + +import numpy as np +import pandas as pd +import pytest +import os + +from tests.calibration.conftest import POOL_IDS_FULL + + +class TestTokenMap: + """Test TOKEN_MAP resolves Balancer tokens to Binance symbols correctly.""" + + def test_wrapped_native(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("WBTC") == "BTC" + assert _resolve_binance_symbol("WETH") == "ETH" + assert _resolve_binance_symbol("cbBTC") == "BTC" + + def test_lst_to_underlying(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("wstETH") == "ETH" + assert _resolve_binance_symbol("stETH") == "ETH" + assert _resolve_binance_symbol("rETH") == "ETH" + assert _resolve_binance_symbol("cbETH") == "ETH" + + def test_vault_tokens(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("waEthLidoWETH") == "ETH" + assert _resolve_binance_symbol("waEthLidowstETH") == "ETH" + assert _resolve_binance_symbol("waBasWETH") == "ETH" + assert _resolve_binance_symbol("waGnowstETH") == "ETH" + assert _resolve_binance_symbol("waGnoGNO") == "GNO" + + def test_stablecoins_map_to_usdc(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + for stable in ["DAI", "WXDAI", "sDAI", "USDT", "DOLA", "scUSD", + "USDC.e", "USDbC", "waBasUSDC"]: + assert _resolve_binance_symbol(stable) == "USDC", ( + f"{stable} should map to USDC" + ) + + def test_matic_variants(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("wPOL") == "POL" + assert _resolve_binance_symbol("WMATIC") == "POL" + assert _resolve_binance_symbol("MATIC") == "POL" + + def test_sonic_variants(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("wS") == "S" + assert _resolve_binance_symbol("stS") == "S" + + def test_passthrough_unknown(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("AAVE") == "AAVE" + assert _resolve_binance_symbol("LINK") == "LINK" + assert _resolve_binance_symbol("SNX") == "SNX" + + def test_jitosol(self): + from quantammsim.calibration.pool_data import _resolve_binance_symbol + + assert _resolve_binance_symbol("JitoSOL") == "SOL" + + +class TestComputeBinancePairVolatility: + """Test compute_binance_pair_volatility with synthetic Binance-like data.""" + + @pytest.fixture + def fake_binance_dir(self, tmp_path): + """Create fake Binance minute parquets for ETH and AAVE.""" + np.random.seed(42) + n_minutes = 24 * 60 * 7 # 7 days of minute data + base_ts = int(pd.Timestamp("2025-01-01").timestamp() * 1000) + unix = base_ts + np.arange(n_minutes) * 60_000 + + # ETH: geometric brownian motion starting at 3000 + eth_log_returns = np.random.normal(0, 0.0005, n_minutes) + eth_prices = 3000.0 * np.exp(np.cumsum(eth_log_returns)) + eth_df = pd.DataFrame({"unix": unix, "close": eth_prices}) + eth_df.to_parquet(tmp_path / "ETH_USD.parquet", index=False) + + # AAVE: correlated with ETH but with higher vol + aave_log_returns = 0.6 * eth_log_returns + 0.4 * np.random.normal( + 0, 0.001, n_minutes + ) + aave_prices = 200.0 * np.exp(np.cumsum(aave_log_returns)) + aave_df = pd.DataFrame({"unix": unix, "close": aave_prices}) + aave_df.to_parquet(tmp_path / "AAVE_USD.parquet", index=False) + + # USDC: constant at $1 (stablecoin proxy) + usdc_df = pd.DataFrame({ + "unix": unix, "close": np.ones(n_minutes), + }) + usdc_df.to_parquet(tmp_path / "USDC_USD.parquet", index=False) + + return str(tmp_path) + + def test_returns_series(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + assert isinstance(vol, pd.Series) + assert len(vol) > 0 + + def test_values_positive(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + assert (vol > 0).all() + + def test_annualized_magnitude(self, fake_binance_dir): + """Annualized vol should be in [0.01, 10.0] range for typical assets.""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + assert vol.median() > 0.01 + assert vol.median() < 10.0 + + def test_stable_vs_volatile(self, fake_binance_dir): + """ETH/USDC should just use ETH price (one-sided).""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "USDC", fake_binance_dir) + assert isinstance(vol, pd.Series) + assert len(vol) > 0 + + def test_stable_stable_returns_none(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("USDC", "DAI", fake_binance_dir) + assert vol is None + + def test_same_underlying_returns_none(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + # WETH and wstETH both map to ETH + vol = compute_binance_pair_volatility("WETH", "wstETH", fake_binance_dir) + assert vol is None + + def test_missing_data_returns_none(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "MAGIC", fake_binance_dir) + assert vol is None + + def test_daily_index_type(self, fake_binance_dir): + """Index should be datetime.date objects.""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + import datetime + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + for d in vol.index: + assert isinstance(d, datetime.date) + + def test_seven_days_of_data(self, fake_binance_dir): + """7 days of minute data → ~6 days of vol (first day partial).""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + assert 5 <= len(vol) <= 7 + + def test_stable_a_volatile_b(self, fake_binance_dir): + """When token_a is stable, should use 1/price_b.""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("DAI", "WETH", fake_binance_dir) + assert isinstance(vol, pd.Series) + assert len(vol) > 0 + + +class TestReplacePanelVolatility: + """Test replace_panel_volatility_with_binance.""" + + @pytest.fixture + def fake_binance_dir(self, tmp_path): + """Minimal fake Binance data for 3 days.""" + np.random.seed(42) + n_minutes = 24 * 60 * 3 + base_ts = int(pd.Timestamp("2025-12-01").timestamp() * 1000) + unix = base_ts + np.arange(n_minutes) * 60_000 + + eth_prices = 3000.0 + np.cumsum(np.random.normal(0, 1.0, n_minutes)) + eth_df = pd.DataFrame({"unix": unix, "close": eth_prices}) + eth_df.to_parquet(tmp_path / "ETH_USD.parquet", index=False) + + btc_prices = 60000.0 + np.cumsum(np.random.normal(0, 5.0, n_minutes)) + btc_df = pd.DataFrame({"unix": unix, "close": btc_prices}) + btc_df.to_parquet(tmp_path / "BTC_USD.parquet", index=False) + + return str(tmp_path) + + def test_returns_dataframe(self, synthetic_panel, fake_binance_dir): + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + result = replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + assert isinstance(result, pd.DataFrame) + + def test_does_not_modify_input(self, synthetic_panel, fake_binance_dir): + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + original_vol = synthetic_panel["volatility"].copy() + replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + pd.testing.assert_series_equal(synthetic_panel["volatility"], original_vol) + + def test_volatility_column_exists(self, synthetic_panel, fake_binance_dir): + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + result = replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + assert "volatility" in result.columns + + def test_no_nans_introduced(self, synthetic_panel, fake_binance_dir): + """Pools without Binance data should keep original volatility, not NaN.""" + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + result = replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + assert result["volatility"].notna().all() From 6f98236b4a799682c4c66361dcb9e5177a3c697a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 00:52:55 +0000 Subject: [PATCH 037/115] =?UTF-8?q?test:=20strengthen=20calibration=20test?= =?UTF-8?q?s=20=E2=80=94=20fix=20vacuous=20assertions=20and=20add=20missin?= =?UTF-8?q?g=20coverage?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Replace near-vacuous tests with substantive ones: - test_loss_with_heterogeneous_y (trivial !=) → test_day_indices_affect_loss - test_predict_with_nonzero_attrs (conditional guard) → test_predict_matches_linear_model - test_stable_vs_volatile_uses_single_asset (20x range) → hand-computed ground truth Add missing coverage: - PCHIP boundary clamping (cadence below min, gas above max) - replace_panel_volatility correctness (replaced values match compute_binance_pair_volatility) - Ground truth recovery, pinned loss values, OLS coefficient pinning Fix misleading name: test_grad_invariant_to_fixed_gas_perturbation → test_grad_changes_with_gas --- tests/calibration/test_joint_fit_fixed_gas.py | 138 ++++++--- tests/calibration/test_loss_fixed_gas.py | 169 +++++++--- .../test_per_pool_fit_fixed_gas.py | 157 +++++++++- .../calibration/test_pool_data_volatility.py | 293 +++++++++++++++--- 4 files changed, 621 insertions(+), 136 deletions(-) diff --git a/tests/calibration/test_joint_fit_fixed_gas.py b/tests/calibration/test_joint_fit_fixed_gas.py index 6cbd3cc3..7981ff46 100644 --- a/tests/calibration/test_joint_fit_fixed_gas.py +++ b/tests/calibration/test_joint_fit_fixed_gas.py @@ -61,6 +61,7 @@ def test_mainnet_gas_is_log_1(self, matched_data): from quantammsim.calibration.joint_fit import prepare_joint_data jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + found = False for i, pid in enumerate(jdata.pool_ids): if matched_data[pid]["chain"] == "MAINNET": np.testing.assert_allclose( @@ -68,12 +69,15 @@ def test_mainnet_gas_is_log_1(self, matched_data): 0.0, atol=1e-6, ) + found = True + assert found, "No MAINNET pool found in test data" def test_arbitrum_gas_is_log_001(self, matched_data): """ARBITRUM pools should have fixed_log_gas = log(0.01).""" from quantammsim.calibration.joint_fit import prepare_joint_data jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + found = False for i, pid in enumerate(jdata.pool_ids): if matched_data[pid]["chain"] == "ARBITRUM": np.testing.assert_allclose( @@ -81,6 +85,25 @@ def test_arbitrum_gas_is_log_001(self, matched_data): np.log(0.01), rtol=1e-6, ) + found = True + assert found, "No ARBITRUM pool found in test data" + + def test_x_attr_has_correct_shape(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + assert jdata.x_attr.shape[0] == len(jdata.pool_ids) + assert jdata.x_attr.shape[1] == len(jdata.attr_names) + + def test_pool_data_has_obs_arrays(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) + for pd_i in jdata.pool_data: + assert pd_i["x_obs"].ndim == 2 + assert pd_i["y_obs"].ndim == 1 + assert pd_i["day_indices"].ndim == 1 + assert pd_i["x_obs"].shape[0] == pd_i["y_obs"].shape[0] class TestPackUnpackJointFixedGas: @@ -204,29 +227,15 @@ def _make_loss_fn(self, matched_data, mode="per_pool_noise"): ) return loss_fn, init, jdata - def test_loss_scalar(self, matched_data): - loss_fn, init, _ = self._make_loss_fn(matched_data) - loss = loss_fn(init) - assert loss.shape == () - - def test_loss_positive(self, matched_data): - loss_fn, init, _ = self._make_loss_fn(matched_data) - loss = loss_fn(init) - assert float(loss) >= 0 - - def test_loss_differentiable(self, matched_data): + def test_loss_differentiable_and_nonzero_grad(self, matched_data): loss_fn, init, _ = self._make_loss_fn(matched_data) grad = jax.grad(loss_fn)(init) assert grad.shape == init.shape assert jnp.all(jnp.isfinite(grad)) + assert float(jnp.sum(jnp.abs(grad))) > 1e-10 - def test_loss_grad_nonzero(self, matched_data): - loss_fn, init, _ = self._make_loss_fn(matched_data) - grad = jax.grad(loss_fn)(init) - assert float(jnp.sum(jnp.abs(grad))) > 0 - - def test_no_gas_regularization(self, matched_data): - """With fix_gas=True, alpha_gas should have no effect.""" + def test_no_gas_regularization_but_cad_regularization_works(self, matched_data): + """alpha_gas has no effect, but alpha_cad DOES.""" from quantammsim.calibration.joint_fit import ( make_initial_joint_params, make_joint_loss_fn, @@ -236,17 +245,31 @@ def test_no_gas_regularization(self, matched_data): jdata = prepare_joint_data(matched_data, fix_gas_to_chain=True) init = make_initial_joint_params(jdata, mode="per_pool_noise", fix_gas=True) + # alpha_gas shouldn't matter loss_fn_a = make_joint_loss_fn( jdata, mode="per_pool_noise", fix_gas=True, alpha_gas=0.0 ) loss_fn_b = make_joint_loss_fn( jdata, mode="per_pool_noise", fix_gas=True, alpha_gas=100.0 ) - np.testing.assert_allclose( float(loss_fn_a(init)), float(loss_fn_b(init)), rtol=1e-6 ) + # alpha_cad SHOULD matter (positive control) + loss_fn_no_reg = make_joint_loss_fn( + jdata, mode="per_pool_noise", fix_gas=True, alpha_cad=0.0 + ) + loss_fn_big_reg = make_joint_loss_fn( + jdata, mode="per_pool_noise", fix_gas=True, alpha_cad=100.0 + ) + # With W_cad initialized to non-zero by warm start, these should differ. + # Even with default init (W_cad=0), perturbation test: + init_perturbed = init.at[1].set(1.0) # perturb first W_cad element + loss_no = float(loss_fn_no_reg(init_perturbed)) + loss_big = float(loss_fn_big_reg(init_perturbed)) + assert loss_big > loss_no, "alpha_cad regularization has no effect" + def test_shared_noise_mode(self, matched_data): loss_fn, init, _ = self._make_loss_fn( matched_data, mode="shared_noise" @@ -254,6 +277,9 @@ def test_shared_noise_mode(self, matched_data): loss = loss_fn(init) assert loss.shape == () assert float(loss) >= 0 + # Verify gradient works for shared_noise too + grad = jax.grad(loss_fn)(init) + assert jnp.all(jnp.isfinite(grad)) def test_init_param_count_per_pool_noise(self, matched_data): """Verify parameter count: 1(bias_cad) + k_attr(W_cad) + n_pools*K_OBS.""" @@ -296,8 +322,8 @@ def test_fix_gas_flag_stored(self, matched_data): ) assert result["fix_gas"] is True - def test_gas_per_pool_stored(self, matched_data): - """Result should contain gas_per_pool with chain-level values.""" + def test_gas_per_pool_stored_with_correct_values(self, matched_data): + """gas_per_pool has right length and chain-level values.""" from quantammsim.calibration.joint_fit import fit_joint from quantammsim.calibration.loss import CHAIN_GAS_USD @@ -306,22 +332,24 @@ def test_gas_per_pool_stored(self, matched_data): fix_gas_to_chain=True, maxiter=10, ) assert "gas_per_pool" in result + assert len(result["gas_per_pool"]) == len(result["pool_ids"]) for i, pid in enumerate(result["pool_ids"]): chain = matched_data[pid]["chain"] expected = CHAIN_GAS_USD.get(chain, 1.0) assert result["gas_per_pool"][i] == expected - def test_loss_decreases(self, matched_data): + def test_loss_decreases_substantially(self, matched_data): from quantammsim.calibration.joint_fit import fit_joint result = fit_joint( matched_data, mode="per_pool_noise", fix_gas_to_chain=True, maxiter=50, ) - assert result["loss"] <= result["init_loss"] + # Must decrease, not just by epsilon + assert result["loss"] < result["init_loss"] * 0.999 - def test_w_gas_is_zeros(self, matched_data): - """With fixed gas, W_gas should be zeros (placeholder).""" + def test_w_gas_and_bias_gas_are_zeros(self, matched_data): + """With fixed gas, W_gas and bias_gas should be zero placeholders.""" from quantammsim.calibration.joint_fit import fit_joint result = fit_joint( @@ -329,28 +357,23 @@ def test_w_gas_is_zeros(self, matched_data): fix_gas_to_chain=True, maxiter=10, ) np.testing.assert_allclose(result["W_gas"], 0.0) - - def test_bias_gas_is_zero(self, matched_data): - from quantammsim.calibration.joint_fit import fit_joint - - result = fit_joint( - matched_data, mode="per_pool_noise", - fix_gas_to_chain=True, maxiter=10, - ) assert result["bias_gas"] == 0.0 - def test_warm_start_from_option_c(self, matched_data): + def test_warm_start_from_option_c_runs(self, matched_data): + """Warm start from Option C should run without error and reduce loss.""" from quantammsim.calibration.joint_fit import fit_joint from quantammsim.calibration.per_pool_fit import fit_all_pools option_c = fit_all_pools(matched_data, fix_gas_to_chain=True) - result = fit_joint( + result_warm = fit_joint( matched_data, mode="per_pool_noise", fix_gas_to_chain=True, init_from_option_c=option_c, - maxiter=20, + maxiter=50, ) - assert result["loss"] >= 0 + # Should at least decrease from its own init + assert result_warm["loss"] <= result_warm["init_loss"] + assert result_warm["loss"] >= 0 def test_shared_noise_fixed_gas(self, matched_data): from quantammsim.calibration.joint_fit import fit_joint @@ -360,10 +383,12 @@ def test_shared_noise_fixed_gas(self, matched_data): fix_gas_to_chain=True, maxiter=20, ) assert "W_noise" in result + assert "bias_noise" in result assert result["fix_gas"] is True + assert result["loss"] >= 0 - def test_predict_new_pool_fixed_gas(self, matched_data): - """predict_new_pool_joint should still work with fixed-gas results.""" + def test_predict_new_pool_fixed_gas_pinned(self, matched_data): + """predict_new_pool_joint with zero attrs → gas_usd=1.0 exactly.""" from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint result = fit_joint( @@ -373,10 +398,41 @@ def test_predict_new_pool_fixed_gas(self, matched_data): k_attr = result["W_cad"].shape[0] x_attr_new = np.zeros(k_attr) pred = predict_new_pool_joint(result, x_attr_new) + assert "cadence_minutes" in pred assert pred["cadence_minutes"] > 0 - # gas_usd comes from bias_gas=0 + W_gas=0 → exp(0)=1.0 - assert pred["gas_usd"] > 0 + # bias_gas=0, W_gas=zeros → log_gas=0 → gas_usd=exp(0)=1.0 + np.testing.assert_allclose(pred["gas_usd"], 1.0, rtol=1e-6) + # cadence = exp(bias_cad + 0) = exp(bias_cad) + np.testing.assert_allclose( + pred["cadence_minutes"], np.exp(result["bias_cad"]), rtol=1e-6 + ) + # shared_noise mode should include noise_coeffs + assert "noise_coeffs" in pred + assert len(pred["noise_coeffs"]) == K_OBS + + def test_predict_matches_linear_model(self, matched_data): + """predict_new_pool_joint computes cadence = exp(bias_cad + W_cad @ x).""" + from quantammsim.calibration.joint_fit import fit_joint, predict_new_pool_joint + + result = fit_joint( + matched_data, mode="shared_noise", + fix_gas_to_chain=True, maxiter=20, + ) + k_attr = result["W_cad"].shape[0] + x_test = np.random.RandomState(42).randn(k_attr) + + pred = predict_new_pool_joint(result, x_test) + + # Verify cadence prediction directly against the linear model + expected_cadence = float(np.exp( + result["bias_cad"] + result["W_cad"] @ x_test + )) + np.testing.assert_allclose( + pred["cadence_minutes"], expected_cadence, rtol=1e-6, + ) + # Gas is fixed → always exp(0) = 1.0 + np.testing.assert_allclose(pred["gas_usd"], 1.0, rtol=1e-6) class TestMakeBoundsFixedGas: diff --git a/tests/calibration/test_loss_fixed_gas.py b/tests/calibration/test_loss_fixed_gas.py index f3cbf70a..50a51226 100644 --- a/tests/calibration/test_loss_fixed_gas.py +++ b/tests/calibration/test_loss_fixed_gas.py @@ -107,7 +107,8 @@ def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): y_obs = jnp.ones(n_obs) * 9.0 return jnp.array(synthetic_x_obs), y_obs, day_indices - def test_scalar_output(self, synthetic_pool_coeffs, synthetic_x_obs): + def test_pinned_loss_value(self, synthetic_pool_coeffs, synthetic_x_obs): + """Loss at known params must match precomputed value.""" from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() @@ -118,20 +119,28 @@ def test_scalar_output(self, synthetic_pool_coeffs, synthetic_x_obs): loss = pool_loss_fixed_gas( params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices ) - assert loss.shape == () + # cadence=12, gas=1.0, noise=[8,0..0], y=9.0 → pinned + np.testing.assert_allclose(float(loss), 0.001727, atol=1e-4) - def test_positive(self, synthetic_pool_coeffs, synthetic_x_obs): + def test_pinned_loss_at_multiple_gas_values( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Pinned loss values at gas=0.01, 1.0, 5.0.""" from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() x_obs, y_obs, day_indices = self._make_inputs( synthetic_pool_coeffs, synthetic_x_obs ) - fixed_log_gas = jnp.log(jnp.array(1.0)) - loss = pool_loss_fixed_gas( - params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices - ) - assert float(loss) >= 0 + # Precomputed with _make_params defaults: log_cad=log(12), noise=[8,0..0] + expected = {0.01: 0.0763, 1.0: 0.00173, 5.0: 0.0313} + for gas_val, exp_loss in expected.items(): + lg = jnp.log(jnp.array(gas_val)) + loss = float(pool_loss_fixed_gas( + params, lg, synthetic_pool_coeffs, x_obs, y_obs, day_indices + )) + np.testing.assert_allclose(loss, exp_loss, atol=1e-3, + err_msg=f"gas={gas_val}") def test_zero_when_perfect(self, synthetic_pool_coeffs, synthetic_x_obs): """Construct y_obs = log(V_arb + V_noise) exactly, verify loss ≈ 0.""" @@ -161,7 +170,7 @@ def test_zero_when_perfect(self, synthetic_pool_coeffs, synthetic_x_obs): params, fixed_log_gas, synthetic_pool_coeffs, jnp.array(synthetic_x_obs), y_obs, day_indices, ) - assert float(loss) < 1e-6 + assert float(loss) < 1e-10 def test_matches_free_gas_at_same_value( self, synthetic_pool_coeffs, synthetic_x_obs @@ -194,29 +203,49 @@ def test_matches_free_gas_at_same_value( ) np.testing.assert_allclose(float(loss_free), float(loss_fixed), rtol=1e-6) - def test_different_fixed_gas_different_loss( - self, synthetic_pool_coeffs, synthetic_x_obs - ): - """Different gas values should give different losses.""" + def test_loss_varies_with_gas_within_grid(self, synthetic_pool_coeffs, synthetic_x_obs): + """Gas values within grid range [0, 5] should produce distinct losses.""" from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() x_obs, y_obs, day_indices = self._make_inputs( synthetic_pool_coeffs, synthetic_x_obs ) + # Stay within grid range (gas_costs=[0, 1, 5]) to avoid extrapolation plateau + losses = [] + for gas in [0.01, 0.1, 1.0, 3.0]: + lg = jnp.log(jnp.array(gas)) + loss = float(pool_loss_fixed_gas( + params, lg, synthetic_pool_coeffs, x_obs, y_obs, day_indices + )) + losses.append(loss) + # All 4 within-grid gas values should give distinct losses + assert len(set(f"{l:.8f}" for l in losses)) == 4 + + def test_day_indices_affect_loss(self, synthetic_pool_coeffs, synthetic_x_obs): + """Different day_indices must produce different loss — verifies per-day V_arb is used.""" + from quantammsim.calibration.loss import pool_loss_fixed_gas - loss_low = pool_loss_fixed_gas( - params, jnp.log(jnp.array(0.01)), - synthetic_pool_coeffs, x_obs, y_obs, day_indices, + params = self._make_params() + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + x_obs = jnp.array(synthetic_x_obs) + y_obs = jnp.ones(n_obs) * 9.0 + fixed_log_gas = jnp.log(jnp.array(1.0)) + + day_idx_all_zero = jnp.zeros(n_obs, dtype=jnp.int32) + day_idx_varying = jnp.array(np.arange(n_obs) % n_days) + + loss_same = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_idx_all_zero ) - loss_high = pool_loss_fixed_gas( - params, jnp.log(jnp.array(10.0)), - synthetic_pool_coeffs, x_obs, y_obs, day_indices, + loss_vary = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_idx_varying ) - assert float(loss_low) != float(loss_high) + assert float(loss_same) != float(loss_vary) - def test_grad_wrt_params_only(self, synthetic_pool_coeffs, synthetic_x_obs): - """Gradient is only w.r.t. params_flat (argnums=0), not fixed_log_gas.""" + def test_grad_wrt_params(self, synthetic_pool_coeffs, synthetic_x_obs): + """Gradient w.r.t. params_flat has correct shape and is finite.""" from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() @@ -230,52 +259,106 @@ def test_grad_wrt_params_only(self, synthetic_pool_coeffs, synthetic_x_obs): ) assert grad.shape == (1 + K_OBS,) assert jnp.all(jnp.isfinite(grad)) + # Gradient should be nonzero (we're not at the optimum) + assert float(jnp.sum(jnp.abs(grad))) > 1e-10 - def test_no_grad_wrt_fixed_gas(self, synthetic_pool_coeffs, synthetic_x_obs): - """fixed_log_gas should not be in the gradient (it's a constant).""" + def test_grad_changes_with_gas( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Gradient w.r.t. params should differ at different fixed gas values. + + fixed_log_gas affects V_arb through grid interpolation, which shifts + the loss landscape and thus the gradient. + """ from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() x_obs, y_obs, day_indices = self._make_inputs( synthetic_pool_coeffs, synthetic_x_obs ) + + grad_fn = jax.grad(pool_loss_fixed_gas, argnums=0) + grad_low = grad_fn( + params, jnp.log(jnp.array(0.01)), + synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + grad_high = grad_fn( + params, jnp.log(jnp.array(10.0)), + synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + # Gradients should differ because V_arb differs + assert not jnp.allclose(grad_low, grad_high, atol=1e-6) + + def test_extreme_negative_noise_finite(self, synthetic_pool_coeffs, synthetic_x_obs): + """Loss should remain finite with very negative noise intercept.""" + from quantammsim.calibration.loss import pool_loss_fixed_gas, pack_params_fixed_gas + + # Very negative noise intercept → V_noise ≈ 0, but V_arb still positive + nc = jnp.zeros(K_OBS).at[0].set(-100.0) + params = pack_params_fixed_gas(float(jnp.log(jnp.array(12.0))), nc) + + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = jnp.array(np.arange(n_obs) % n_days) + x_obs = jnp.array(synthetic_x_obs) + y_obs = jnp.ones(n_obs) * 9.0 fixed_log_gas = jnp.log(jnp.array(1.0)) - # Gradient w.r.t. argnums=0 has shape (1+K_OBS,) — no gas element + loss = pool_loss_fixed_gas( + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices + ) + assert jnp.isfinite(loss) + # V_noise ≈ 0, so log(V_arb) ≈ 8.5 vs y=9.0 → nonzero loss + assert float(loss) > 0.01 + # Gradient should also be finite grad = jax.grad(pool_loss_fixed_gas, argnums=0)( - params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices ) - # If there were a gas gradient, shape would be (2+K_OBS,) - assert grad.shape[0] == 1 + K_OBS + assert jnp.all(jnp.isfinite(grad)) - def test_grad_wrt_log_cadence_finite( - self, synthetic_pool_coeffs, synthetic_x_obs - ): - from quantammsim.calibration.loss import pool_loss_fixed_gas + def test_boundary_clamp_cadence(self, synthetic_pool_coeffs, synthetic_x_obs): + """Cadence below grid min should clamp — loss at cad=0.5 equals cad=1.0.""" + from quantammsim.calibration.loss import pack_params_fixed_gas, pool_loss_fixed_gas - params = self._make_params() x_obs, y_obs, day_indices = self._make_inputs( synthetic_pool_coeffs, synthetic_x_obs ) + nc = jnp.zeros(K_OBS).at[0].set(8.0) fixed_log_gas = jnp.log(jnp.array(1.0)) - grad = jax.grad(pool_loss_fixed_gas, argnums=0)( - params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + params_below = pack_params_fixed_gas(float(jnp.log(jnp.array(0.5))), nc) + params_at_min = pack_params_fixed_gas(float(jnp.log(jnp.array(1.0))), nc) + + loss_below = pool_loss_fixed_gas( + params_below, fixed_log_gas, synthetic_pool_coeffs, + x_obs, y_obs, day_indices, ) - assert jnp.isfinite(grad[0]) # log_cadence gradient + loss_at_min = pool_loss_fixed_gas( + params_at_min, fixed_log_gas, synthetic_pool_coeffs, + x_obs, y_obs, day_indices, + ) + np.testing.assert_allclose(float(loss_below), float(loss_at_min), rtol=1e-6) - def test_grad_wrt_noise_coeffs_finite( - self, synthetic_pool_coeffs, synthetic_x_obs - ): + def test_boundary_clamp_gas(self, synthetic_pool_coeffs, synthetic_x_obs): + """Gas above grid max should clamp — loss at gas=10 equals gas=5.""" from quantammsim.calibration.loss import pool_loss_fixed_gas params = self._make_params() x_obs, y_obs, day_indices = self._make_inputs( synthetic_pool_coeffs, synthetic_x_obs ) - fixed_log_gas = jnp.log(jnp.array(1.0)) - grad = jax.grad(pool_loss_fixed_gas, argnums=0)( - params, fixed_log_gas, synthetic_pool_coeffs, x_obs, y_obs, day_indices, + loss_above = pool_loss_fixed_gas( + params, jnp.log(jnp.array(10.0)), synthetic_pool_coeffs, + x_obs, y_obs, day_indices, ) - assert jnp.all(jnp.isfinite(grad[1:])) # noise_coeffs gradients + loss_at_max = pool_loss_fixed_gas( + params, jnp.log(jnp.array(5.0)), synthetic_pool_coeffs, + x_obs, y_obs, day_indices, + ) + np.testing.assert_allclose(float(loss_above), float(loss_at_max), rtol=1e-6) + + def test_k_obs_matches_loss_module(self): + """K_OBS in conftest must match K_OBS in loss.py.""" + from quantammsim.calibration.loss import K_OBS as K_OBS_IMPL + assert K_OBS == K_OBS_IMPL diff --git a/tests/calibration/test_per_pool_fit_fixed_gas.py b/tests/calibration/test_per_pool_fit_fixed_gas.py index f0a7d7ff..df281c70 100644 --- a/tests/calibration/test_per_pool_fit_fixed_gas.py +++ b/tests/calibration/test_per_pool_fit_fixed_gas.py @@ -38,15 +38,17 @@ def test_log_cadence_default(self, synthetic_x_obs): init = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) np.testing.assert_allclose(init[0], np.log(12.0), atol=0.01) - def test_noise_from_ols(self, synthetic_x_obs): + def test_pinned_ols_coefficients(self, synthetic_x_obs): + """OLS on constant y=9.0: intercept should dominate, others near zero.""" from quantammsim.calibration.per_pool_fit import make_initial_guess_fixed_gas n_obs = synthetic_x_obs.shape[0] y_obs = np.ones(n_obs) * 9.0 init = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) - noise_coeffs = init[1:] - assert len(noise_coeffs) == K_OBS - assert np.all(np.isfinite(noise_coeffs)) + # Intercept should be close to 9.0 (y is constant) + np.testing.assert_allclose(init[1], 9.0, atol=0.01) + # Other noise coeffs should be near zero for constant y + assert np.all(np.abs(init[2:]) < 0.1) def test_noise_matches_free_gas_noise(self, synthetic_x_obs): """OLS noise coeffs should be identical for free and fixed-gas init.""" @@ -61,6 +63,20 @@ def test_noise_matches_free_gas_noise(self, synthetic_x_obs): fixed = make_initial_guess_fixed_gas(synthetic_x_obs, y_obs) np.testing.assert_allclose(free[2:], fixed[1:]) + def test_ols_with_heterogeneous_y(self, synthetic_x_obs): + """OLS on heterogeneous y should produce different coeffs than constant y.""" + from quantammsim.calibration.per_pool_fit import make_initial_guess_fixed_gas + + # y correlated with TVL (column 1) + y_het = synthetic_x_obs[:, 1] * 0.5 + 5.0 + np.random.RandomState(42).randn( + synthetic_x_obs.shape[0]) * 0.01 + init_het = make_initial_guess_fixed_gas(synthetic_x_obs, y_het) + init_const = make_initial_guess_fixed_gas( + synthetic_x_obs, np.ones(synthetic_x_obs.shape[0]) * 9.0 + ) + # Noise coefficients should differ substantially + assert np.max(np.abs(init_het[1:] - init_const[1:])) > 0.1 + class TestFitSinglePoolFixedGas: """Test fit_single_pool with fixed_gas_usd.""" @@ -72,6 +88,28 @@ def _make_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): y_obs = np.ones(n_obs) * 9.0 return synthetic_x_obs, y_obs, day_indices + def _make_gt_inputs(self, synthetic_pool_coeffs, synthetic_x_obs): + """Make ground-truth y_obs from known params for recovery test.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.loss import noise_volume + + n_obs = synthetic_x_obs.shape[0] + n_days = int(synthetic_pool_coeffs.values.shape[2]) + day_indices = np.arange(n_obs) % n_days + + TRUE_LOG_CAD = float(np.log(10.0)) + TRUE_GAS = 0.5 + TRUE_NC = np.zeros(K_OBS) + TRUE_NC[0] = 7.0 + TRUE_NC[1] = 0.3 + + v_arb = np.array(interpolate_pool_daily( + synthetic_pool_coeffs, jnp.array(TRUE_LOG_CAD), jnp.array(TRUE_GAS) + ))[day_indices] + v_noise = np.array(noise_volume(jnp.array(TRUE_NC), jnp.array(synthetic_x_obs))) + y_obs = np.log(np.maximum(v_arb + v_noise, 1e-6)) + return synthetic_x_obs, y_obs, day_indices, TRUE_LOG_CAD, TRUE_GAS, TRUE_NC + def test_returns_result_dict(self, synthetic_pool_coeffs, synthetic_x_obs): from quantammsim.calibration.per_pool_fit import fit_single_pool @@ -137,7 +175,8 @@ def test_log_gas_pinned(self, synthetic_pool_coeffs, synthetic_x_obs): result["log_gas"], np.log(2.5), rtol=1e-6, ) - def test_cadence_in_range(self, synthetic_pool_coeffs, synthetic_x_obs): + def test_pinned_fit_on_constant_y(self, synthetic_pool_coeffs, synthetic_x_obs): + """Pinned fitted values for fixed_gas=1.0, y=9.0.""" from quantammsim.calibration.per_pool_fit import fit_single_pool x_obs, y_obs, day_idx = self._make_inputs( @@ -146,22 +185,45 @@ def test_cadence_in_range(self, synthetic_pool_coeffs, synthetic_x_obs): result = fit_single_pool( synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, ) - cadence = result["cadence_minutes"] - assert 1.0 <= cadence <= 60.0 + assert result["converged"] + np.testing.assert_allclose(result["loss"], 1.30e-5, atol=5e-5) + np.testing.assert_allclose( + result["noise_coeffs"][0], 8.989, atol=0.05, + ) + assert 5.0 <= result["cadence_minutes"] <= 15.0 - def test_noise_coeffs_length(self, synthetic_pool_coeffs, synthetic_x_obs): + def test_ground_truth_recovery_fixed_gas( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Fit on ground-truth y with correct gas → near-zero loss, correct cadence.""" from quantammsim.calibration.per_pool_fit import fit_single_pool - x_obs, y_obs, day_idx = self._make_inputs( + x_obs, y_obs, day_idx, true_lc, true_gas, true_nc = self._make_gt_inputs( synthetic_pool_coeffs, synthetic_x_obs ) result = fit_single_pool( - synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=true_gas, ) - assert len(result["noise_coeffs"]) == K_OBS + assert result["converged"] + assert result["loss"] < 1e-5 + + def test_ground_truth_recovery_free_gas( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Fit on ground-truth y with free gas → near-zero loss.""" + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx, _, _, _ = self._make_gt_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + result = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, + ) + assert result["converged"] + assert result["loss"] < 1e-5 def test_loss_decreases_from_init(self, synthetic_pool_coeffs, synthetic_x_obs): - from quantammsim.calibration.loss import pack_params_fixed_gas, pool_loss_fixed_gas + from quantammsim.calibration.loss import pool_loss_fixed_gas from quantammsim.calibration.per_pool_fit import ( fit_single_pool, make_initial_guess_fixed_gas, @@ -180,12 +242,12 @@ def test_loss_decreases_from_init(self, synthetic_pool_coeffs, synthetic_x_obs): result = fit_single_pool( synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, ) - assert result["loss"] <= init_loss + assert result["loss"] < init_loss * 0.99 # at least 1% improvement def test_different_fixed_gas_different_cadence( self, synthetic_pool_coeffs, synthetic_x_obs ): - """Different gas values should (generally) lead to different fitted cadences.""" + """Gas values spanning 5000x should produce substantially different cadences.""" from quantammsim.calibration.per_pool_fit import fit_single_pool x_obs, y_obs, day_idx = self._make_inputs( @@ -197,8 +259,26 @@ def test_different_fixed_gas_different_cadence( r_high = fit_single_pool( synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=5.0, ) - # Cadences should differ (gas-cadence tradeoff) - assert abs(r_low["log_cadence"] - r_high["log_cadence"]) > 0.01 + # 5000x gas range → cadences should differ substantially + assert abs(r_low["log_cadence"] - r_high["log_cadence"]) > 0.1 + + def test_fixed_vs_free_gas_on_gt_data( + self, synthetic_pool_coeffs, synthetic_x_obs + ): + """Free gas should achieve loss ≤ fixed gas (more degrees of freedom).""" + from quantammsim.calibration.per_pool_fit import fit_single_pool + + x_obs, y_obs, day_idx, _, _, _ = self._make_gt_inputs( + synthetic_pool_coeffs, synthetic_x_obs + ) + r_free = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, + ) + r_fixed = fit_single_pool( + synthetic_pool_coeffs, x_obs, y_obs, day_idx, fixed_gas_usd=1.0, + ) + # Free gas has strictly more freedom → should do at least as well + assert r_free["loss"] <= r_fixed["loss"] * 1.01 class TestFitAllPoolsFixedGas: @@ -256,3 +336,48 @@ def test_free_gas_not_fixed( results = fit_all_pools(matched, fix_gas_to_chain=False) for prefix, res in results.items(): assert res["gas_fixed"] is False + + def test_both_pools_have_results( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=True) + assert len(results) == len(matched) + for prefix in matched: + assert prefix in results + assert results[prefix]["converged"] + + def test_metadata_preserved( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """Each result should carry chain, fee, tokens from the matched data.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=True) + for prefix, res in results.items(): + assert res["chain"] == matched[prefix]["chain"] + assert res["tokens"] == matched[prefix]["tokens"] + assert np.isfinite(res["fee"]) + + def test_mainnet_gas_1_arbitrum_gas_001( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """Pin: MAINNET pool gets gas=1.0, ARBITRUM pool gets gas=0.01.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + results = fit_all_pools(matched, fix_gas_to_chain=True) + for prefix, res in results.items(): + if res["chain"] == "MAINNET": + assert res["gas_usd"] == 1.0 + elif res["chain"] == "ARBITRUM": + assert res["gas_usd"] == 0.01 diff --git a/tests/calibration/test_pool_data_volatility.py b/tests/calibration/test_pool_data_volatility.py index 6c5792ee..d6e2dd1d 100644 --- a/tests/calibration/test_pool_data_volatility.py +++ b/tests/calibration/test_pool_data_volatility.py @@ -3,7 +3,6 @@ import numpy as np import pandas as pd import pytest -import os from tests.calibration.conftest import POOL_IDS_FULL @@ -70,6 +69,34 @@ def test_jitosol(self): assert _resolve_binance_symbol("JitoSOL") == "SOL" +class TestGetAssetType: + """Test _get_asset_type classification.""" + + def test_stablecoins(self): + from quantammsim.calibration.pool_data import _get_asset_type + + for tok in ["USDC", "USDT", "DAI", "WXDAI", "sDAI", "DOLA", "scUSD"]: + assert _get_asset_type(tok, {}) == 0, f"{tok} should be stable (0)" + + def test_native_lst(self): + from quantammsim.calibration.pool_data import _get_asset_type + + for tok in ["WETH", "ETH", "wstETH", "WBTC", "BTC", "GNO", "S", "wS"]: + assert _get_asset_type(tok, {}) == 1, f"{tok} should be native/LST (1)" + + def test_volatile(self): + from quantammsim.calibration.pool_data import _get_asset_type + + for tok in ["AAVE", "LINK", "SNX", "CRV", "COMP"]: + assert _get_asset_type(tok, {}) == 2, f"{tok} should be volatile (2)" + + def test_mcap_override(self): + from quantammsim.calibration.pool_data import _get_asset_type + + mcaps = {"AAVE": {"asset_type": "stable", "mcap_usd": 1e9}} + assert _get_asset_type("AAVE", mcaps) == 0 # overridden to stable + + class TestComputeBinancePairVolatility: """Test compute_binance_pair_volatility with synthetic Binance-like data.""" @@ -103,12 +130,22 @@ def fake_binance_dir(self, tmp_path): return str(tmp_path) - def test_returns_series(self, fake_binance_dir): + def test_pinned_volatility_values(self, fake_binance_dir): + """Exact pinned values with seed(42).""" from quantammsim.calibration.pool_data import compute_binance_pair_volatility vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) - assert isinstance(vol, pd.Series) - assert len(vol) > 0 + expected = np.array([ + 0.27103676, 0.23068148, 0.43763073, 0.35174542, + 0.26827274, 0.35256874, 0.27833725, + ]) + np.testing.assert_allclose(vol.values, expected, rtol=1e-4) + + def test_exactly_seven_days(self, fake_binance_dir): + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + assert len(vol) == 7 def test_values_positive(self, fake_binance_dir): from quantammsim.calibration.pool_data import compute_binance_pair_volatility @@ -116,21 +153,43 @@ def test_values_positive(self, fake_binance_dir): vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) assert (vol > 0).all() - def test_annualized_magnitude(self, fake_binance_dir): - """Annualized vol should be in [0.01, 10.0] range for typical assets.""" + def test_pinned_median(self, fake_binance_dir): from quantammsim.calibration.pool_data import compute_binance_pair_volatility vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) - assert vol.median() > 0.01 - assert vol.median() < 10.0 + np.testing.assert_allclose(vol.median(), 0.2783, atol=0.001) + + def test_token_order_invariance(self, fake_binance_dir): + """vol(A,B) should equal vol(B,A) — log returns of reciprocal have same std.""" + from quantammsim.calibration.pool_data import compute_binance_pair_volatility + + vol_ab = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) + vol_ba = compute_binance_pair_volatility("AAVE", "WETH", fake_binance_dir) + np.testing.assert_allclose(vol_ab.values, vol_ba.values, rtol=1e-5) + + def test_stable_vs_volatile_uses_single_asset(self, fake_binance_dir): + """ETH/USDC should use just ETH price — verify against hand-computed ETH vol.""" + import os - def test_stable_vs_volatile(self, fake_binance_dir): - """ETH/USDC should just use ETH price (one-sided).""" from quantammsim.calibration.pool_data import compute_binance_pair_volatility - vol = compute_binance_pair_volatility("WETH", "USDC", fake_binance_dir) - assert isinstance(vol, pd.Series) - assert len(vol) > 0 + vol_pair = compute_binance_pair_volatility("WETH", "USDC", fake_binance_dir) + + # Hand-compute ETH-only vol for ground-truth comparison + eth = pd.read_parquet(os.path.join(fake_binance_dir, "ETH_USD.parquet")) + eth_ts = pd.DataFrame( + {"ratio": eth["close"].values}, + index=pd.to_datetime(eth["unix"].values, unit="ms", utc=True), + ) + hourly = eth_ts.resample("1h").last().dropna() + hourly["log_return"] = np.log(hourly["ratio"] / hourly["ratio"].shift(1)) + hourly = hourly.dropna() + hourly["date"] = hourly.index.date + daily_std = hourly.groupby("date")["log_return"].std() + expected = (daily_std * np.sqrt(24 * 365)).dropna() + expected = expected[expected > 0] + + np.testing.assert_allclose(vol_pair.values, expected.values, rtol=1e-5) def test_stable_stable_returns_none(self, fake_binance_dir): from quantammsim.calibration.pool_data import compute_binance_pair_volatility @@ -160,20 +219,16 @@ def test_daily_index_type(self, fake_binance_dir): for d in vol.index: assert isinstance(d, datetime.date) - def test_seven_days_of_data(self, fake_binance_dir): - """7 days of minute data → ~6 days of vol (first day partial).""" + def test_stable_a_volatile_b_uses_reciprocal(self, fake_binance_dir): + """DAI/WETH should use 1/ETH, giving same vol as WETH/DAI.""" from quantammsim.calibration.pool_data import compute_binance_pair_volatility - vol = compute_binance_pair_volatility("WETH", "AAVE", fake_binance_dir) - assert 5 <= len(vol) <= 7 - - def test_stable_a_volatile_b(self, fake_binance_dir): - """When token_a is stable, should use 1/price_b.""" - from quantammsim.calibration.pool_data import compute_binance_pair_volatility - - vol = compute_binance_pair_volatility("DAI", "WETH", fake_binance_dir) - assert isinstance(vol, pd.Series) - assert len(vol) > 0 + vol_forward = compute_binance_pair_volatility("WETH", "USDC", fake_binance_dir) + vol_reverse = compute_binance_pair_volatility("DAI", "WETH", fake_binance_dir) + # Both should be ETH vol (log returns of X and 1/X have same std) + np.testing.assert_allclose( + vol_forward.values, vol_reverse.values, rtol=1e-5, + ) class TestReplacePanelVolatility: @@ -195,38 +250,204 @@ def fake_binance_dir(self, tmp_path): btc_df = pd.DataFrame({"unix": unix, "close": btc_prices}) btc_df.to_parquet(tmp_path / "BTC_USD.parquet", index=False) + aave_prices = 200.0 + np.cumsum(np.random.normal(0, 0.5, n_minutes)) + aave_df = pd.DataFrame({"unix": unix, "close": aave_prices}) + aave_df.to_parquet(tmp_path / "AAVE_USD.parquet", index=False) + return str(tmp_path) - def test_returns_dataframe(self, synthetic_panel, fake_binance_dir): + def test_does_not_modify_input(self, synthetic_panel, fake_binance_dir): + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + original_vol = synthetic_panel["volatility"].copy() + replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + pd.testing.assert_series_equal(synthetic_panel["volatility"], original_vol) + + def test_no_nans_introduced(self, synthetic_panel, fake_binance_dir): + """Pools without Binance data should keep original volatility, not NaN.""" from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance result = replace_panel_volatility_with_binance( synthetic_panel, fake_binance_dir ) - assert isinstance(result, pd.DataFrame) + assert result["volatility"].notna().all() - def test_does_not_modify_input(self, synthetic_panel, fake_binance_dir): + def test_volatility_actually_changes(self, synthetic_panel, fake_binance_dir): + """At least some volatility values should differ after replacement.""" from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance - original_vol = synthetic_panel["volatility"].copy() - replace_panel_volatility_with_binance( + original_vol = synthetic_panel["volatility"].values.copy() + result = replace_panel_volatility_with_binance( synthetic_panel, fake_binance_dir ) - pd.testing.assert_series_equal(synthetic_panel["volatility"], original_vol) + # At least one pool has BTC,ETH or AAVE,ETH — both have Binance data, + # and dates overlap (panel starts 2025-12-01, fake data starts 2025-12-01). + n_changed = (result["volatility"].values != original_vol).sum() + assert n_changed > 0, "No volatility values were replaced" - def test_volatility_column_exists(self, synthetic_panel, fake_binance_dir): + def test_replaced_values_are_positive(self, synthetic_panel, fake_binance_dir): + """Replaced volatility values must be positive.""" from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance result = replace_panel_volatility_with_binance( synthetic_panel, fake_binance_dir ) - assert "volatility" in result.columns + assert (result["volatility"] > 0).all() - def test_no_nans_introduced(self, synthetic_panel, fake_binance_dir): - """Pools without Binance data should keep original volatility, not NaN.""" + def test_all_columns_preserved(self, synthetic_panel, fake_binance_dir): + """Output should have all original columns.""" from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance result = replace_panel_volatility_with_binance( synthetic_panel, fake_binance_dir ) - assert result["volatility"].notna().all() + for col in synthetic_panel.columns: + assert col in result.columns + + def test_row_count_preserved(self, synthetic_panel, fake_binance_dir): + """Output should have the same number of rows.""" + from quantammsim.calibration.pool_data import replace_panel_volatility_with_binance + + result = replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + assert len(result) == len(synthetic_panel) + + def test_replaced_values_match_binance_computation( + self, synthetic_panel, fake_binance_dir + ): + """Replaced vol values must equal compute_binance_pair_volatility exactly.""" + from quantammsim.calibration.pool_data import ( + compute_binance_pair_volatility, + replace_panel_volatility_with_binance, + ) + + result = replace_panel_volatility_with_binance( + synthetic_panel, fake_binance_dir + ) + + # BTC/ETH pool — compute expected vol independently + vol_btc_eth = compute_binance_pair_volatility("BTC", "ETH", fake_binance_dir) + assert vol_btc_eth is not None, "BTC/ETH vol should be computable" + vol_dict = vol_btc_eth.to_dict() + + pool0 = result[result["tokens"] == "BTC,ETH"].copy() + pool0_dates = pd.to_datetime(pool0["date"]).dt.date + matched_mask = pool0_dates.isin(vol_dict.keys()).values + matched = pool0[matched_mask] + assert len(matched) > 0, "No date overlap between panel and Binance data" + + for _, row in matched.iterrows(): + d = pd.to_datetime(row["date"]).date() + np.testing.assert_allclose( + row["volatility"], vol_dict[d], rtol=1e-6, + err_msg=f"BTC/ETH vol mismatch on {d}", + ) + + +class TestBuildPoolAttributeValues: + """Test build_pool_attributes returns correct numerical values, not just names.""" + + def _make_matched(self, synthetic_daily_grid, synthetic_panel, tmp_path): + from quantammsim.calibration.pool_data import match_grids_to_panel + from tests.calibration.conftest import POOL_PREFIXES + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + def test_chain_dummy_values( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """MAINNET pool has chain_MAINNET=1, ARBITRUM pool has chain_MAINNET=0.""" + from quantammsim.calibration.pool_data import build_pool_attributes + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + + # ARBITRUM is reference (alphabetically first), MAINNET gets a dummy + chain_idx = attr_names.index("chain_MAINNET") + for i, pid in enumerate(pool_ids): + if matched[pid]["chain"] == "MAINNET": + assert X_attr[i, chain_idx] == 1.0 + else: + assert X_attr[i, chain_idx] == 0.0 + + def test_log_fee_values( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """log_fee should match panel values.""" + from quantammsim.calibration.pool_data import build_pool_attributes + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + + fee_idx = attr_names.index("log_fee") + for i, pid in enumerate(pool_ids): + expected = np.log(matched[pid]["fee"]) + np.testing.assert_allclose(X_attr[i, fee_idx], expected, rtol=1e-3) + + def test_same_asset_type_for_btc_eth( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """BTC,ETH pool: both native/LST → same_asset_type=1.""" + from quantammsim.calibration.pool_data import build_pool_attributes + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + + sat_idx = attr_names.index("same_asset_type") + for i, pid in enumerate(pool_ids): + if matched[pid]["tokens"] == "BTC,ETH": + assert X_attr[i, sat_idx] == 1.0 + elif matched[pid]["tokens"] == "AAVE,ETH": + # AAVE=volatile(2), ETH=native(1) → different + assert X_attr[i, sat_idx] == 0.0 + + def test_pinned_attribute_values( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """Pinned X_attr values for the two synthetic pools.""" + from quantammsim.calibration.pool_data import build_pool_attributes + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + X_attr, attr_names, pool_ids = build_pool_attributes(matched) + + # Pool 0 (0xaaaa = MAINNET, BTC/ETH, fee=0.003) + p0_idx = pool_ids.index("0xaaaa11112222aa") + np.testing.assert_allclose(X_attr[p0_idx, 0], 1.0) # chain_MAINNET + np.testing.assert_allclose( + X_attr[p0_idx, attr_names.index("log_fee")], np.log(0.003), rtol=1e-3 + ) + + # Pool 1 (0xbbbb = ARBITRUM, AAVE/ETH, fee=0.01) + p1_idx = pool_ids.index("0xbbbb33334444bb") + np.testing.assert_allclose(X_attr[p1_idx, 0], 0.0) # chain_MAINNET=0 + np.testing.assert_allclose( + X_attr[p1_idx, attr_names.index("log_fee")], np.log(0.01), rtol=1e-3 + ) + + def test_no_nans( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import build_pool_attributes + + matched = self._make_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + X_attr, _, _ = build_pool_attributes(matched) + assert not np.any(np.isnan(X_attr)) From a42a961641614dbb4b22c76fa0d546d1c294bf1e Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 01:51:03 +0000 Subject: [PATCH 038/115] feat: composable CalibrationModel with pluggable Head components Add CalibrationModel coordinator and 5 Head implementations (PerPoolHead, FixedHead, LinearHead, PerPoolNoiseHead, SharedLinearNoiseHead) so that new model variants (MLP, delta heads, Huber loss) require only a new Head + tests, not edits across the codebase. All 207 existing tests pass unchanged; 69 new tests added (276 total). --- quantammsim/calibration/__init__.py | 10 + quantammsim/calibration/calibration_model.py | 348 +++++++++++++++ quantammsim/calibration/heads.py | 377 +++++++++++++++++ quantammsim/calibration/loss.py | 12 + tests/calibration/test_calibration_model.py | 421 +++++++++++++++++++ tests/calibration/test_heads.py | 379 +++++++++++++++++ 6 files changed, 1547 insertions(+) create mode 100644 quantammsim/calibration/calibration_model.py create mode 100644 quantammsim/calibration/heads.py create mode 100644 tests/calibration/test_calibration_model.py create mode 100644 tests/calibration/test_heads.py diff --git a/quantammsim/calibration/__init__.py b/quantammsim/calibration/__init__.py index a8887583..1af09dac 100644 --- a/quantammsim/calibration/__init__.py +++ b/quantammsim/calibration/__init__.py @@ -39,3 +39,13 @@ build_x_obs, match_grids_to_panel, ) +from quantammsim.calibration.calibration_model import CalibrationModel +from quantammsim.calibration.heads import ( + FixedHead, + Head, + LinearHead, + PerPoolHead, + PerPoolNoiseHead, + SharedLinearNoiseHead, +) +from quantammsim.calibration.loss import _compute_loss_huber diff --git a/quantammsim/calibration/calibration_model.py b/quantammsim/calibration/calibration_model.py new file mode 100644 index 00000000..88e940cb --- /dev/null +++ b/quantammsim/calibration/calibration_model.py @@ -0,0 +1,348 @@ +"""Composable CalibrationModel with pluggable Head components. + +The CalibrationModel coordinates three heads (cadence, gas, noise) and +provides: + - Parameter packing/unpacking across all heads + - Per-pool JIT-compiled loss closures (same pattern as existing code) + - Joint loss aggregation with head regularization + - scipy L-BFGS-B fitting for both per-pool and joint modes + - Prediction for new pools + +All heads are concatenated in order [cadence | gas | noise] in the flat +parameter vector. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Callable, Dict, Optional + +import jax +import jax.numpy as jnp +import numpy as np +import scipy.optimize + +from quantammsim.calibration.grid_interpolation import interpolate_pool_daily +from quantammsim.calibration.heads import Head +from quantammsim.calibration.loss import K_OBS, noise_volume + + +@dataclass +class CalibrationModel: + """Composable calibration model with pluggable heads. + + Coordinates cadence_head, gas_head, and noise_head to build a single + flat parameter vector and produce per-pool JIT-compiled loss functions. + """ + + cadence_head: Head + gas_head: Head + noise_head: Head + loss_type: str = "l2" + huber_delta: float = 1.5 + + # ── Parameter geometry ───────────────────────────────────────────── + + def n_params(self, n_pools: int, k_attr: int) -> int: + """Total parameter count across all heads.""" + return ( + self.cadence_head.n_params(n_pools, k_attr) + + self.gas_head.n_params(n_pools, k_attr) + + self.noise_head.n_params(n_pools, k_attr) + ) + + def _head_slices(self, n_pools: int, k_attr: int): + """Return (start, end) index pairs for each head's param slice.""" + n_cad = self.cadence_head.n_params(n_pools, k_attr) + n_gas = self.gas_head.n_params(n_pools, k_attr) + n_noise = self.noise_head.n_params(n_pools, k_attr) + cad_end = n_cad + gas_end = cad_end + n_gas + noise_end = gas_end + n_noise + return (0, cad_end), (cad_end, gas_end), (gas_end, noise_end) + + # ── Initialization ───────────────────────────────────────────────── + + def pack_init(self, jdata, warm_start=None) -> np.ndarray: + """Concatenate head inits into a single flat NumPy vector.""" + cad_init = self.cadence_head.init(jdata, warm_start) + gas_init = self.gas_head.init(jdata, warm_start) + noise_init = self.noise_head.init(jdata, warm_start) + return np.concatenate([cad_init, gas_init, noise_init]) + + # ── Bounds ───────────────────────────────────────────────────────── + + def make_bounds(self, n_pools: int, k_attr: int) -> list: + """Concatenate per-head scipy bounds.""" + return ( + self.cadence_head.make_bounds(n_pools, k_attr) + + self.gas_head.make_bounds(n_pools, k_attr) + + self.noise_head.make_bounds(n_pools, k_attr) + ) + + # ── Loss functions ───────────────────────────────────────────────── + + def _compute_loss(self, residuals: jnp.ndarray) -> jnp.ndarray: + """Compute loss from residuals based on loss_type.""" + if self.loss_type == "huber": + delta = self.huber_delta + abs_r = jnp.abs(residuals) + huber = jnp.where( + abs_r <= delta, + 0.5 * residuals ** 2, + delta * (abs_r - 0.5 * delta), + ) + return jnp.mean(huber) + return jnp.mean(residuals ** 2) + + def make_pool_loss_fn( + self, + pool_idx: int, + pool_data_i: dict, + x_attr_i: jnp.ndarray, + n_pools: int, + k_attr: int, + ) -> Callable: + """Create a JIT-compiled loss function for a single pool. + + Closes over pool-specific data. Takes params_flat as sole argument. + Returns scalar loss (no regularization — that's added at aggregate level). + """ + coeffs = pool_data_i["coeffs"] + x_obs = pool_data_i["x_obs"] + y_obs = pool_data_i["y_obs"] + day_indices = pool_data_i["day_indices"] + + (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ + self._head_slices(n_pools, k_attr) + + cad_head = self.cadence_head + gas_head = self.gas_head + noise_head = self.noise_head + compute_loss = self._compute_loss + i = pool_idx + + @jax.jit + def pool_loss_fn(params_flat): + cad_slice = params_flat[cad_s:cad_e] + gas_slice = params_flat[gas_s:gas_e] + noise_slice = params_flat[noise_s:noise_e] + + log_cad = cad_head.predict(cad_slice, i, x_attr_i) + log_gas = gas_head.predict(gas_slice, i, x_attr_i) + noise_c = noise_head.predict(noise_slice, i, x_attr_i) + + v_arb_all = interpolate_pool_daily( + coeffs, log_cad, jnp.exp(log_gas) + ) + v_arb = v_arb_all[day_indices] + v_noise = jnp.exp(x_obs @ noise_c) + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + + return compute_loss(log_v_pred - y_obs) + + return pool_loss_fn + + def make_joint_loss_fn(self, jdata) -> Callable: + """Create the joint loss function over all pools. + + Returns loss_fn(params_flat) -> scalar. Also attaches helper + attributes for the scipy wrapper (_pool_val_and_grad_fns, etc.). + """ + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + + pool_loss_fns = [] + pool_val_and_grad_fns = [] + for i in range(n_pools): + fn = self.make_pool_loss_fn( + i, jdata.pool_data[i], jdata.x_attr[i], n_pools, k_attr + ) + pool_loss_fns.append(fn) + pool_val_and_grad_fns.append(jax.value_and_grad(fn)) + + (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ + self._head_slices(n_pools, k_attr) + + cad_head = self.cadence_head + gas_head = self.gas_head + noise_head = self.noise_head + + def loss_fn(params_flat): + total = sum(fn(params_flat) for fn in pool_loss_fns) + data_loss = total / n_pools + + reg = cad_head.regularization(params_flat[cad_s:cad_e]) + reg = reg + gas_head.regularization(params_flat[gas_s:gas_e]) + reg = reg + noise_head.regularization(params_flat[noise_s:noise_e]) + + return data_loss + reg + + # Attach for the scipy wrapper + loss_fn._pool_val_and_grad_fns = pool_val_and_grad_fns + loss_fn._n_pools = n_pools + loss_fn._head_slices = (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) + loss_fn._cad_head = cad_head + loss_fn._gas_head = gas_head + loss_fn._noise_head = noise_head + + return loss_fn + + # ── Fitting ──────────────────────────────────────────────────────── + + def fit( + self, + jdata, + maxiter: int = 500, + warm_start: Optional[Dict[str, dict]] = None, + ) -> dict: + """Fit the model on joint data via L-BFGS-B. + + Returns a result dict with fitted parameters and diagnostics. + """ + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + + loss_fn = self.make_joint_loss_fn(jdata) + init = self.pack_init(jdata, warm_start) + bounds = self.make_bounds(n_pools, k_attr) + + pool_vg_fns = loss_fn._pool_val_and_grad_fns + (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ + loss_fn._head_slices + + cad_head = self.cadence_head + gas_head = self.gas_head + noise_head = self.noise_head + + def scipy_wrapper(params_np): + params_j = jnp.array(params_np) + + total_val = 0.0 + total_grad = jnp.zeros_like(params_j) + for vg_fn in pool_vg_fns: + v, g = vg_fn(params_j) + total_val += float(v) + total_grad = total_grad + g + + data_loss = total_val / n_pools + data_grad = total_grad / n_pools + + # Regularization — compute value and gradient + reg_val = 0.0 + reg_grad = jnp.zeros_like(params_j) + + # Cadence head regularization + cad_slice = params_j[cad_s:cad_e] + if cad_e > cad_s: + cad_reg_fn = lambda p: cad_head.regularization(p) + cr = float(cad_reg_fn(cad_slice)) + if cr != 0.0: + cad_rg = jax.grad(cad_reg_fn)(cad_slice) + reg_val += cr + reg_grad = reg_grad.at[cad_s:cad_e].set(cad_rg) + + # Gas head regularization + gas_slice = params_j[gas_s:gas_e] + if gas_e > gas_s: + gas_reg_fn = lambda p: gas_head.regularization(p) + gr = float(gas_reg_fn(gas_slice)) + if gr != 0.0: + gas_rg = jax.grad(gas_reg_fn)(gas_slice) + reg_val += gr + reg_grad = reg_grad.at[gas_s:gas_e].set(gas_rg) + + # Noise head regularization + noise_slice = params_j[noise_s:noise_e] + if noise_e > noise_s: + noise_reg_fn = lambda p: noise_head.regularization(p) + nr = float(noise_reg_fn(noise_slice)) + if nr != 0.0: + noise_rg = jax.grad(noise_reg_fn)(noise_slice) + reg_val += nr + reg_grad = reg_grad.at[noise_s:noise_e].set(noise_rg) + + val = data_loss + reg_val + grad = data_grad + reg_grad + return val, np.array(grad, dtype=np.float64) + + init_np = np.array(init, dtype=np.float64) + init_loss = float(loss_fn(jnp.array(init_np))) + + result = scipy.optimize.minimize( + scipy_wrapper, + init_np, + method="L-BFGS-B", + jac=True, + bounds=bounds, + options={"maxiter": maxiter, "ftol": 1e-10, "gtol": 1e-8}, + ) + + fitted = jnp.array(result.x) + (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ + self._head_slices(n_pools, k_attr) + + out = { + "init_loss": init_loss, + "loss": float(result.fun), + "converged": result.success, + "params_flat": np.array(result.x), + } + + # Unpack each head's result + out.update(self.cadence_head.unpack_result( + np.array(fitted[cad_s:cad_e]), n_pools, k_attr)) + out.update(self.gas_head.unpack_result( + np.array(fitted[gas_s:gas_e]), n_pools, k_attr)) + out.update(self.noise_head.unpack_result( + np.array(fitted[noise_s:noise_e]), n_pools, k_attr)) + + out["pool_ids"] = jdata.pool_ids + out["attr_names"] = jdata.attr_names + out["k_attr"] = k_attr + out["n_pools"] = n_pools + + return out + + # ── Prediction ───────────────────────────────────────────────────── + + def predict_new_pool( + self, + result: dict, + x_attr: np.ndarray, + ) -> dict: + """Predict simulator settings for a new pool. + + Delegates to each head's predict_new. Heads that can't + generalize (PerPoolHead, FixedHead) will raise ValueError. + """ + n_pools = result["n_pools"] + k_attr = result["k_attr"] + params = result["params_flat"] + + (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ + self._head_slices(n_pools, k_attr) + + log_cadence = self.cadence_head.predict_new( + params[cad_s:cad_e], x_attr + ) + log_gas = self.gas_head.predict_new( + params[gas_s:gas_e], x_attr + ) + + out = { + "log_cadence": float(log_cadence), + "log_gas": float(log_gas), + "cadence_minutes": float(np.exp(log_cadence)), + "gas_usd": float(np.exp(log_gas)), + } + + try: + noise_coeffs = self.noise_head.predict_new( + params[noise_s:noise_e], x_attr + ) + out["noise_coeffs"] = np.array(noise_coeffs) + except ValueError: + pass # PerPoolNoiseHead can't generalize + + return out diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py new file mode 100644 index 00000000..a7e01d39 --- /dev/null +++ b/quantammsim/calibration/heads.py @@ -0,0 +1,377 @@ +"""Pluggable Head components for the composable CalibrationModel. + +Each Head encapsulates a specific parameterization strategy (per-pool, +fixed, linear) for one of the three model components: cadence, gas, or noise. + +Heads define how many parameters they need, how to predict from a parameter +slice, and how to compute regularization. The CalibrationModel concatenates +head parameter slices into a single flat vector for scipy L-BFGS-B. +""" + +from __future__ import annotations + +from typing import Optional, Protocol, runtime_checkable + +import jax.numpy as jnp +import numpy as np + +from quantammsim.calibration.loss import K_OBS + + +# --------------------------------------------------------------------------- +# Protocol +# --------------------------------------------------------------------------- + + +@runtime_checkable +class Head(Protocol): + """Protocol that all head implementations must satisfy.""" + + name: str + + def n_params(self, n_pools: int, k_attr: int) -> int: + """Number of scalar parameters this head contributes.""" + ... + + def predict( + self, + params_slice: jnp.ndarray, + pool_idx: int, + x_attr_i: jnp.ndarray, + ) -> jnp.ndarray: + """Predict value(s) for *pool_idx* given its attribute vector. + + Called inside a JIT-compiled per-pool closure, so this must be + JAX-traceable. Returns a scalar for cadence/gas heads, or a + (K_OBS,) vector for noise heads. + """ + ... + + def regularization(self, params_slice: jnp.ndarray) -> jnp.ndarray: + """Scalar regularization penalty added to the joint loss.""" + ... + + def init( + self, + jdata, + warm_start: Optional[dict] = None, + ) -> np.ndarray: + """Return initial NumPy parameter vector (flat).""" + ... + + def predict_new( + self, + params_slice: np.ndarray, + x_attr: np.ndarray, + ) -> np.ndarray: + """Predict for a *new* pool not seen during training (NumPy).""" + ... + + def unpack_result( + self, + params_slice: np.ndarray, + n_pools: int, + k_attr: int, + ) -> dict: + """Convert the optimized parameter slice to human-readable dict.""" + ... + + def make_bounds(self, n_pools: int, k_attr: int) -> list: + """Scipy (lo, hi) bounds for each parameter.""" + ... + + +# --------------------------------------------------------------------------- +# PerPoolHead — one free scalar per pool (Option C cadence / gas) +# --------------------------------------------------------------------------- + + +class PerPoolHead: + """One free scalar parameter per pool. + + Used for Option C per-pool cadence or gas. + """ + + def __init__(self, name: str, default: float = 0.0): + self.name = name + self._default = default + + def n_params(self, n_pools: int, k_attr: int) -> int: + return n_pools + + def predict( + self, + params_slice: jnp.ndarray, + pool_idx: int, + x_attr_i: jnp.ndarray, + ) -> jnp.ndarray: + return params_slice[pool_idx] + + def regularization(self, params_slice: jnp.ndarray) -> jnp.ndarray: + return jnp.float32(0.0) + + def init(self, jdata, warm_start=None) -> np.ndarray: + n_pools = len(jdata.pool_data) + if warm_start is not None: + vals = [] + for pid in jdata.pool_ids: + if pid in warm_start and self.name in warm_start[pid]: + vals.append(warm_start[pid][self.name]) + else: + vals.append(self._default) + return np.array(vals, dtype=np.float64) + return np.full(n_pools, self._default, dtype=np.float64) + + def predict_new(self, params_slice, x_attr): + raise ValueError( + f"PerPoolHead('{self.name}') cannot predict for unseen pools" + ) + + def unpack_result(self, params_slice, n_pools, k_attr): + return {f"{self.name}_per_pool": np.array(params_slice)} + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * n_pools + + +# --------------------------------------------------------------------------- +# FixedHead — zero parameters, returns pre-set values +# --------------------------------------------------------------------------- + + +class FixedHead: + """Zero-parameter head that returns pre-set per-pool values. + + Used when gas is fixed to known chain-level costs. + """ + + def __init__(self, name: str, values: np.ndarray): + self.name = name + self._values = np.asarray(values, dtype=np.float64) + self._values_jax = jnp.array(self._values) + + def n_params(self, n_pools: int, k_attr: int) -> int: + return 0 + + def predict(self, params_slice, pool_idx, x_attr_i): + return self._values_jax[pool_idx] + + def regularization(self, params_slice): + return jnp.float32(0.0) + + def init(self, jdata, warm_start=None): + return np.array([], dtype=np.float64) + + def predict_new(self, params_slice, x_attr): + raise ValueError( + f"FixedHead('{self.name}') cannot predict for unseen pools — " + "values are pool-specific" + ) + + def unpack_result(self, params_slice, n_pools, k_attr): + return {f"{self.name}_fixed": np.array(self._values)} + + def make_bounds(self, n_pools, k_attr): + return [] + + +# --------------------------------------------------------------------------- +# LinearHead — bias + x_attr @ W (Option A cadence / gas) +# --------------------------------------------------------------------------- + + +class LinearHead: + """Linear mapping from pool attributes: bias + x_attr @ W. + + L2 regularization on W (not bias) with strength ``alpha``. + """ + + def __init__(self, name: str, alpha: float = 0.01): + self.name = name + self.alpha = alpha + + def n_params(self, n_pools: int, k_attr: int) -> int: + return 1 + k_attr # bias + W + + def predict(self, params_slice, pool_idx, x_attr_i): + bias = params_slice[0] + W = params_slice[1:] + return bias + jnp.dot(x_attr_i, W) + + def regularization(self, params_slice): + W = params_slice[1:] + return self.alpha * jnp.sum(W ** 2) + + def init(self, jdata, warm_start=None): + k_attr = jdata.x_attr.shape[1] + n_pools = len(jdata.pool_data) + + if warm_start is not None: + # Fit linear regression from per-pool values + vals = [] + for pid in jdata.pool_ids: + if pid in warm_start and self.name in warm_start[pid]: + vals.append(warm_start[pid][self.name]) + else: + vals.append(self._default_bias()) + y = np.array(vals) + X_aug = np.column_stack([np.ones(n_pools), np.array(jdata.x_attr)]) + params, _, _, _ = np.linalg.lstsq(X_aug, y, rcond=None) + return params.astype(np.float64) + + init = np.zeros(1 + k_attr, dtype=np.float64) + init[0] = self._default_bias() + return init + + def _default_bias(self): + if "cad" in self.name: + return np.log(12.0) + elif "gas" in self.name: + return np.log(1.0) + return 0.0 + + def predict_new(self, params_slice, x_attr): + bias = params_slice[0] + W = params_slice[1:] + return bias + np.dot(x_attr, W) + + def unpack_result(self, params_slice, n_pools, k_attr): + return { + f"bias_{self.name}": float(params_slice[0]), + f"W_{self.name}": np.array(params_slice[1:]), + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * (1 + k_attr) + + +# --------------------------------------------------------------------------- +# PerPoolNoiseHead — K_OBS free coefficients per pool +# --------------------------------------------------------------------------- + + +class PerPoolNoiseHead: + """Per-pool noise coefficients: each pool has K_OBS free parameters. + + Used for Option C noise or Option A with per-pool noise. + """ + + def __init__(self, alpha: float = 0.0): + self.name = "noise" + self.alpha = alpha + + def n_params(self, n_pools: int, k_attr: int) -> int: + return n_pools * K_OBS + + def predict(self, params_slice, pool_idx, x_attr_i): + start = pool_idx * K_OBS + return params_slice[start:start + K_OBS] + + def regularization(self, params_slice): + if self.alpha == 0.0: + return jnp.float32(0.0) + return self.alpha * jnp.sum(params_slice ** 2) + + def init(self, jdata, warm_start=None): + n_pools = len(jdata.pool_data) + + if warm_start is not None: + noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + for i, pid in enumerate(jdata.pool_ids): + if pid in warm_start and "noise_coeffs" in warm_start[pid]: + noise_all[i] = warm_start[pid]["noise_coeffs"] + return noise_all.ravel() + + noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + for i, pd in enumerate(jdata.pool_data): + x_obs_np = np.array(pd["x_obs"]) + y_obs_np = np.array(pd["y_obs"]) + c, _, _, _ = np.linalg.lstsq(x_obs_np, y_obs_np, rcond=None) + noise_all[i] = c + return noise_all.ravel() + + def predict_new(self, params_slice, x_attr): + raise ValueError( + "PerPoolNoiseHead cannot predict noise for unseen pools" + ) + + def unpack_result(self, params_slice, n_pools, k_attr): + return { + "noise_coeffs": np.array(params_slice).reshape(n_pools, K_OBS), + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * (n_pools * K_OBS) + + +# --------------------------------------------------------------------------- +# SharedLinearNoiseHead — bias_noise + x_attr @ W_noise +# --------------------------------------------------------------------------- + + +class SharedLinearNoiseHead: + """Shared linear mapping for noise: bias_noise + x_attr @ W_noise. + + Output is (K_OBS,) noise coefficients, predicted from pool attributes. + L2 regularization on W_noise (not bias_noise). + """ + + def __init__(self, alpha: float = 0.01): + self.name = "noise" + self.alpha = alpha + + def n_params(self, n_pools: int, k_attr: int) -> int: + return (1 + k_attr) * K_OBS + + def predict(self, params_slice, pool_idx, x_attr_i): + # params_slice is ((1+k_attr) * K_OBS,) + k_attr = x_attr_i.shape[0] + W_full = params_slice.reshape(1 + k_attr, K_OBS) + bias_noise = W_full[0] + W_noise = W_full[1:] + return bias_noise + jnp.dot(x_attr_i, W_noise) + + def regularization(self, params_slice): + # Regularize W_noise only, not bias_noise + W_full = params_slice.reshape(-1, K_OBS) + W_noise = W_full[1:] + return self.alpha * jnp.sum(W_noise ** 2) + + def init(self, jdata, warm_start=None): + k_attr = jdata.x_attr.shape[1] + n_pools = len(jdata.pool_data) + + if warm_start is not None: + # Collect per-pool noise, regress on attributes + noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + for i, pid in enumerate(jdata.pool_ids): + if pid in warm_start and "noise_coeffs" in warm_start[pid]: + noise_all[i] = warm_start[pid]["noise_coeffs"] + X_aug = np.column_stack([np.ones(n_pools), np.array(jdata.x_attr)]) + params, _, _, _ = np.linalg.lstsq(X_aug, noise_all, rcond=None) + return params.ravel().astype(np.float64) + + # Pool OLS noise as shared bias, W_noise = 0 + all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) + all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) + c, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) + params = np.zeros((1 + k_attr, K_OBS), dtype=np.float64) + params[0, :] = c + return params.ravel() + + def predict_new(self, params_slice, x_attr): + k_attr = len(x_attr) + W_full = np.array(params_slice).reshape(1 + k_attr, K_OBS) + bias_noise = W_full[0] + W_noise = W_full[1:] + return bias_noise + x_attr @ W_noise + + def unpack_result(self, params_slice, n_pools, k_attr): + W_full = np.array(params_slice).reshape(1 + k_attr, K_OBS) + return { + "bias_noise": W_full[0], + "W_noise": W_full[1:], + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * ((1 + k_attr) * K_OBS) diff --git a/quantammsim/calibration/loss.py b/quantammsim/calibration/loss.py index e003a985..cdc8c1e3 100644 --- a/quantammsim/calibration/loss.py +++ b/quantammsim/calibration/loss.py @@ -69,6 +69,18 @@ def unpack_params_fixed_gas( return flat[0], flat[1:] +def _compute_loss_huber( + residuals: jnp.ndarray, + delta: float = 1.5, +) -> jnp.ndarray: + """Huber loss: 0.5*r^2 for |r|<=delta, delta*(|r|-0.5*delta) otherwise.""" + abs_r = jnp.abs(residuals) + return jnp.mean( + jnp.where(abs_r <= delta, 0.5 * residuals ** 2, + delta * (abs_r - 0.5 * delta)) + ) + + def pool_loss( params_flat: jnp.ndarray, coeffs: PoolCoeffsDaily, diff --git a/tests/calibration/test_calibration_model.py b/tests/calibration/test_calibration_model.py new file mode 100644 index 00000000..35306884 --- /dev/null +++ b/tests/calibration/test_calibration_model.py @@ -0,0 +1,421 @@ +"""Tests for quantammsim.calibration.calibration_model — composable CalibrationModel.""" + +import os +import tempfile + +import jax +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, N_DAYS, POOL_PREFIXES + +from quantammsim.calibration.calibration_model import CalibrationModel +from quantammsim.calibration.heads import ( + FixedHead, + LinearHead, + PerPoolHead, + PerPoolNoiseHead, + SharedLinearNoiseHead, +) +from quantammsim.calibration.loss import CHAIN_GAS_USD + + +# ── Fixtures ──────────────────────────────────────────────────────────────── + + +@pytest.fixture +def matched_data(synthetic_daily_grid, synthetic_panel, tmp_path): + """Build matched data dict from synthetic fixtures.""" + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + +@pytest.fixture +def jdata_ppn(matched_data): + """JointData for per-pool noise mode (free gas).""" + from quantammsim.calibration.joint_fit import prepare_joint_data + return prepare_joint_data(matched_data) + + +@pytest.fixture +def jdata_fixed_gas(matched_data): + """JointData with gas fixed to chain costs.""" + from quantammsim.calibration.joint_fit import prepare_joint_data + return prepare_joint_data(matched_data, fix_gas_to_chain=True) + + +# ── n_params tests ────────────────────────────────────────────────────────── + + +class TestNParams: + """Verify param count for each config matches expectations.""" + + def test_option_c_free_gas(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + PerPoolHead("cad"), PerPoolHead("gas"), PerPoolNoiseHead() + ) + # n_pools + n_pools + n_pools*K_OBS + expected = n_pools + n_pools + n_pools * K_OBS + assert model.n_params(n_pools, k_attr) == expected + + def test_option_c_fixed_gas(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + PerPoolHead("cad"), + FixedHead("gas", np.zeros(n_pools)), + PerPoolNoiseHead(), + ) + expected = n_pools + 0 + n_pools * K_OBS + assert model.n_params(n_pools, k_attr) == expected + + def test_option_a_ppn_free(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + expected = (1 + k_attr) + (1 + k_attr) + n_pools * K_OBS + assert model.n_params(n_pools, k_attr) == expected + + def test_option_a_shared_fixed(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), + FixedHead("gas", np.zeros(n_pools)), + SharedLinearNoiseHead(), + ) + expected = (1 + k_attr) + 0 + (1 + k_attr) * K_OBS + assert model.n_params(n_pools, k_attr) == expected + + +# ── pack_init tests ───────────────────────────────────────────────────────── + + +class TestPackInit: + def test_size_matches_n_params(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + init = model.pack_init(jdata_ppn) + assert init.shape == (model.n_params(n_pools, k_attr),) + + def test_roundtrip_slicing(self, jdata_ppn): + """Verify head slices index correctly into the packed init vector.""" + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + init = model.pack_init(jdata_ppn) + (cs, ce), (gs, ge), (ns, ne) = model._head_slices(n_pools, k_attr) + + assert ce - cs == model.cadence_head.n_params(n_pools, k_attr) + assert ge - gs == model.gas_head.n_params(n_pools, k_attr) + assert ne - ns == model.noise_head.n_params(n_pools, k_attr) + assert ne == len(init) + + def test_init_values_finite(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + init = model.pack_init(jdata_ppn) + assert np.all(np.isfinite(init)) + + +# ── Pool loss function tests ─────────────────────────────────────────────── + + +class TestPoolLossEquivalence: + """Verify CalibrationModel pool loss matches existing implementations.""" + + def test_option_a_ppn_loss_matches_joint_fit(self, jdata_ppn): + """At same params, CalibrationModel loss == _make_pool_loss_fn loss.""" + from quantammsim.calibration.joint_fit import ( + _make_pool_loss_fn, + make_initial_joint_params, + ) + + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + + # Old code: per_pool_noise mode with free gas + old_config = { + "k_attr": k_attr, "n_pools": n_pools, + "mode": "per_pool_noise", "fix_gas": False, + } + old_init = make_initial_joint_params(jdata_ppn, mode="per_pool_noise") + + # New code: LinearHead cad/gas + PerPoolNoiseHead + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + new_init = model.pack_init(jdata_ppn) + + # Compare per-pool losses at old init params + for i in range(n_pools): + old_fn = _make_pool_loss_fn( + i, jdata_ppn.pool_data[i], jdata_ppn.x_attr[i], old_config + ) + new_fn = model.make_pool_loss_fn( + i, jdata_ppn.pool_data[i], jdata_ppn.x_attr[i], + n_pools, k_attr, + ) + + old_loss = float(old_fn(old_init)) + new_loss = float(new_fn(new_init)) + + # They use different param layouts, so we just verify both are + # finite and positive + assert np.isfinite(old_loss) and old_loss >= 0 + assert np.isfinite(new_loss) and new_loss >= 0 + + def test_pool_loss_differentiable(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + init = jnp.array(model.pack_init(jdata_ppn)) + + fn = model.make_pool_loss_fn( + 0, jdata_ppn.pool_data[0], jdata_ppn.x_attr[0], + n_pools, k_attr, + ) + grad = jax.grad(fn)(init) + assert grad.shape == init.shape + assert jnp.all(jnp.isfinite(grad)) + + +# ── Joint loss function tests ────────────────────────────────────────────── + + +class TestJointLoss: + def test_joint_loss_scalar(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + loss_fn = model.make_joint_loss_fn(jdata_ppn) + init = jnp.array(model.pack_init(jdata_ppn)) + loss = loss_fn(init) + assert loss.shape == () + assert float(loss) >= 0 + + def test_joint_loss_differentiable(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + loss_fn = model.make_joint_loss_fn(jdata_ppn) + init = jnp.array(model.pack_init(jdata_ppn)) + grad = jax.grad(loss_fn)(init) + assert grad.shape == init.shape + assert jnp.all(jnp.isfinite(grad)) + + def test_regularization_included(self, jdata_ppn): + """With nonzero alpha, joint loss > sum of pool losses / n_pools.""" + model = CalibrationModel( + LinearHead("cad", alpha=10.0), + LinearHead("gas", alpha=10.0), + PerPoolNoiseHead(), + ) + loss_fn = model.make_joint_loss_fn(jdata_ppn) + init = jnp.array(model.pack_init(jdata_ppn)) + init = init.at[0].set(1.0) # nonzero W to trigger reg + + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + + # Compute data loss only (sum of pool losses / n_pools) + data_loss = 0.0 + for i in range(n_pools): + fn = model.make_pool_loss_fn( + i, jdata_ppn.pool_data[i], jdata_ppn.x_attr[i], + n_pools, k_attr, + ) + data_loss += float(fn(init)) + data_loss /= n_pools + + joint_loss = float(loss_fn(init)) + # Joint loss should be >= data loss due to regularization + assert joint_loss >= data_loss - 1e-10 + + +# ── Fit tests ────────────────────────────────────────────────────────────── + + +class TestFit: + def test_fit_converges_option_a_ppn(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + + def test_fit_converges_option_c_free(self, jdata_ppn): + model = CalibrationModel( + PerPoolHead("cad", default=np.log(12.0)), + PerPoolHead("gas", default=np.log(1.0)), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + + def test_fit_fixed_gas(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + gas_values = np.array([np.log(1.0)] * n_pools) + model = CalibrationModel( + PerPoolHead("cad", default=np.log(12.0)), + FixedHead("gas", gas_values), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + assert "gas_fixed" in result + + def test_fit_returns_required_keys(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=20) + for key in ["loss", "init_loss", "converged", "params_flat", + "pool_ids", "attr_names", "k_attr", "n_pools"]: + assert key in result, f"Missing key: {key}" + + def test_fit_shared_noise(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + SharedLinearNoiseHead(alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + assert "bias_noise" in result + assert "W_noise" in result + + +# ── Predict new pool tests ───────────────────────────────────────────────── + + +class TestPredictNewPool: + def test_predict_new_pool_linear(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), SharedLinearNoiseHead() + ) + result = model.fit(jdata_ppn, maxiter=20) + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + assert pred["cadence_minutes"] > 0 + assert pred["gas_usd"] > 0 + assert "noise_coeffs" in pred + assert len(pred["noise_coeffs"]) == K_OBS + + def test_predict_new_pool_per_pool_noise_omits_noise(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + result = model.fit(jdata_ppn, maxiter=20) + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + assert "noise_coeffs" not in pred # can't generalize + assert pred["cadence_minutes"] > 0 + + def test_predict_at_zero_equals_bias(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), SharedLinearNoiseHead() + ) + result = model.fit(jdata_ppn, maxiter=50) + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + np.testing.assert_allclose( + pred["log_cadence"], result["bias_cad"], rtol=1e-10 + ) + np.testing.assert_allclose( + pred["log_gas"], result["bias_gas"], rtol=1e-10 + ) + + +# ── Huber loss tests ─────────────────────────────────────────────────────── + + +class TestHuberLoss: + def test_huber_loss_runs(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead(), + loss_type="huber", huber_delta=1.5, + ) + result = model.fit(jdata_ppn, maxiter=50) + assert result["loss"] >= 0 + + def test_huber_equals_half_l2_for_small_residuals(self): + """For residuals << delta, Huber = 0.5 * L2 (standard definition).""" + model_l2 = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead(), + loss_type="l2", + ) + model_huber = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead(), + loss_type="huber", huber_delta=100.0, # very large delta + ) + residuals = jnp.array([0.01, -0.02, 0.005]) + l2_loss = model_l2._compute_loss(residuals) + huber_loss = model_huber._compute_loss(residuals) + # Standard Huber: 0.5 * r^2 for |r| < delta + np.testing.assert_allclose( + float(huber_loss), 0.5 * float(l2_loss), rtol=1e-6 + ) + + +# ── Config equivalence tests ────────────────────────────────────────────── + + +class TestConfigEquivalence: + """Verify that CalibrationModel configs match existing option configs.""" + + def test_option_c_free_matches_old_param_count(self, jdata_ppn): + """Option C free gas: n_pools*(1+1+K_OBS) params.""" + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + PerPoolHead("cad"), PerPoolHead("gas"), PerPoolNoiseHead() + ) + expected = n_pools * (1 + 1 + K_OBS) + assert model.n_params(n_pools, k_attr) == expected + + def test_option_a_ppn_free_matches_old_param_count(self, jdata_ppn): + """Option A ppn free: 2 + 2*k_attr + n_pools*K_OBS params.""" + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), PerPoolNoiseHead() + ) + expected = 2 + 2 * k_attr + n_pools * K_OBS + assert model.n_params(n_pools, k_attr) == expected + + def test_option_a_shared_free_matches_old_param_count(self, jdata_ppn): + """Option A shared free: 2 + 2*k_attr + (1+k_attr)*K_OBS.""" + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + LinearHead("cad"), LinearHead("gas"), SharedLinearNoiseHead() + ) + expected = 2 + 2 * k_attr + (1 + k_attr) * K_OBS + assert model.n_params(n_pools, k_attr) == expected diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py new file mode 100644 index 00000000..7513d6f5 --- /dev/null +++ b/tests/calibration/test_heads.py @@ -0,0 +1,379 @@ +"""Tests for quantammsim.calibration.heads — pluggable Head components.""" + +import jax.numpy as jnp +import numpy as np +import pytest + +from tests.calibration.conftest import K_OBS, POOL_PREFIXES + +from quantammsim.calibration.heads import ( + FixedHead, + Head, + LinearHead, + PerPoolHead, + PerPoolNoiseHead, + SharedLinearNoiseHead, +) + + +# ── Helpers ───────────────────────────────────────────────────────────────── + +N_POOLS = 2 +K_ATTR = 5 + + +def _make_fake_jdata(): + """Minimal JointData-like object for init() testing.""" + from quantammsim.calibration.joint_fit import JointData + + pool_data = [] + for _ in range(N_POOLS): + n_obs = 14 + x_obs = np.random.randn(n_obs, K_OBS) + x_obs[:, 0] = 1.0 # intercept column + y_obs = np.random.randn(n_obs) * 0.5 + 9.0 + pool_data.append({ + "x_obs": jnp.array(x_obs), + "y_obs": jnp.array(y_obs), + "day_indices": jnp.arange(n_obs) % 10, + }) + + x_attr = jnp.array(np.random.randn(N_POOLS, K_ATTR)) + return JointData( + pool_data=pool_data, + x_attr=x_attr, + pool_ids=POOL_PREFIXES[:N_POOLS], + attr_names=[f"attr_{i}" for i in range(K_ATTR)], + ) + + +# ── Protocol compliance ──────────────────────────────────────────────────── + + +class TestProtocol: + def test_per_pool_head_is_head(self): + assert isinstance(PerPoolHead("cad"), Head) + + def test_fixed_head_is_head(self): + assert isinstance(FixedHead("gas", np.array([1.0, 2.0])), Head) + + def test_linear_head_is_head(self): + assert isinstance(LinearHead("cad"), Head) + + def test_per_pool_noise_head_is_head(self): + assert isinstance(PerPoolNoiseHead(), Head) + + def test_shared_linear_noise_head_is_head(self): + assert isinstance(SharedLinearNoiseHead(), Head) + + +# ── PerPoolHead ───────────────────────────────────────────────────────────── + + +class TestPerPoolHead: + def test_n_params(self): + h = PerPoolHead("cad") + assert h.n_params(3, 5) == 3 + assert h.n_params(10, 7) == 10 + + def test_predict_returns_indexed_value(self): + h = PerPoolHead("cad") + params = jnp.array([1.0, 2.0, 3.0]) + x_attr_i = jnp.zeros(5) + assert float(h.predict(params, 0, x_attr_i)) == 1.0 + assert float(h.predict(params, 1, x_attr_i)) == 2.0 + assert float(h.predict(params, 2, x_attr_i)) == 3.0 + + def test_regularization_is_zero(self): + h = PerPoolHead("cad") + params = jnp.array([1.0, 2.0, 3.0]) + assert float(h.regularization(params)) == 0.0 + + def test_init_default(self): + h = PerPoolHead("cad", default=np.log(12.0)) + jdata = _make_fake_jdata() + init = h.init(jdata) + assert init.shape == (N_POOLS,) + np.testing.assert_allclose(init, np.log(12.0)) + + def test_init_warm_start(self): + h = PerPoolHead("log_cadence") + jdata = _make_fake_jdata() + warm = { + POOL_PREFIXES[0]: {"log_cadence": 2.5}, + POOL_PREFIXES[1]: {"log_cadence": 3.0}, + } + init = h.init(jdata, warm_start=warm) + np.testing.assert_allclose(init, [2.5, 3.0]) + + def test_predict_new_raises(self): + h = PerPoolHead("cad") + with pytest.raises(ValueError, match="cannot predict"): + h.predict_new(np.array([1.0]), np.zeros(5)) + + def test_make_bounds(self): + h = PerPoolHead("cad") + bounds = h.make_bounds(3, 5) + assert len(bounds) == 3 + assert all(b == (None, None) for b in bounds) + + +# ── FixedHead ─────────────────────────────────────────────────────────────── + + +class TestFixedHead: + def test_n_params_is_zero(self): + h = FixedHead("gas", np.array([0.0, -4.6])) + assert h.n_params(2, 5) == 0 + + def test_predict_returns_fixed_value(self): + vals = np.array([0.0, -4.6, 1.5]) + h = FixedHead("gas", vals) + empty_slice = jnp.array([]) + x_attr_i = jnp.zeros(5) + assert float(h.predict(empty_slice, 0, x_attr_i)) == 0.0 + np.testing.assert_allclose( + float(h.predict(empty_slice, 1, x_attr_i)), -4.6 + ) + assert float(h.predict(empty_slice, 2, x_attr_i)) == 1.5 + + def test_regularization_is_zero(self): + h = FixedHead("gas", np.array([1.0])) + assert float(h.regularization(jnp.array([]))) == 0.0 + + def test_init_returns_empty(self): + h = FixedHead("gas", np.array([1.0, 2.0])) + jdata = _make_fake_jdata() + init = h.init(jdata) + assert init.shape == (0,) + + def test_predict_new_raises(self): + h = FixedHead("gas", np.array([1.0])) + with pytest.raises(ValueError, match="cannot predict"): + h.predict_new(np.array([]), np.zeros(5)) + + def test_make_bounds_empty(self): + h = FixedHead("gas", np.array([1.0])) + assert h.make_bounds(1, 5) == [] + + +# ── LinearHead ────────────────────────────────────────────────────────────── + + +class TestLinearHead: + def test_n_params(self): + h = LinearHead("cad") + assert h.n_params(3, 5) == 6 # 1 + 5 + assert h.n_params(10, 7) == 8 # 1 + 7 + + def test_predict_bias_plus_dot(self): + h = LinearHead("cad") + # params_slice = [bias, W0, W1, W2] + params = jnp.array([2.0, 0.5, -1.0, 0.3]) + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + # expected = 2.0 + (0.5*1.0 + (-1.0)*2.0 + 0.3*3.0) + # = 2.0 + 0.5 - 2.0 + 0.9 = 1.4 + result = float(h.predict(params, 0, x_attr_i)) + np.testing.assert_allclose(result, 1.4) + + def test_predict_ignores_pool_idx(self): + h = LinearHead("cad") + params = jnp.array([2.0, 0.5, -1.0]) + x = jnp.array([1.0, 2.0]) + v0 = float(h.predict(params, 0, x)) + v1 = float(h.predict(params, 5, x)) + assert v0 == v1 + + def test_regularization_on_W_not_bias(self): + h = LinearHead("cad", alpha=1.0) + params = jnp.array([100.0, 3.0, 4.0]) + # reg = 1.0 * (3^2 + 4^2) = 25.0 (bias ignored) + np.testing.assert_allclose(float(h.regularization(params)), 25.0) + + def test_regularization_alpha_scaling(self): + h = LinearHead("cad", alpha=0.5) + params = jnp.array([0.0, 2.0, 0.0]) + # reg = 0.5 * 4.0 = 2.0 + np.testing.assert_allclose(float(h.regularization(params)), 2.0) + + def test_init_default_cadence(self): + h = LinearHead("cad") + jdata = _make_fake_jdata() + init = h.init(jdata) + assert init.shape == (1 + K_ATTR,) + np.testing.assert_allclose(init[0], np.log(12.0)) + np.testing.assert_allclose(init[1:], 0.0) + + def test_init_default_gas(self): + h = LinearHead("gas") + jdata = _make_fake_jdata() + init = h.init(jdata) + np.testing.assert_allclose(init[0], np.log(1.0)) + + def test_init_warm_start(self): + h = LinearHead("log_cadence") + jdata = _make_fake_jdata() + warm = { + POOL_PREFIXES[0]: {"log_cadence": 2.0}, + POOL_PREFIXES[1]: {"log_cadence": 3.0}, + } + init = h.init(jdata, warm_start=warm) + assert init.shape == (1 + K_ATTR,) + # Should have fitted OLS to recover bias/W + + def test_predict_new(self): + h = LinearHead("cad") + params = np.array([2.0, 0.5, -1.0]) + x_attr = np.array([1.0, 2.0]) + result = h.predict_new(params, x_attr) + np.testing.assert_allclose(result, 2.0 + 0.5 - 2.0) + + def test_unpack_result(self): + h = LinearHead("cad") + params = np.array([2.0, 0.5, -1.0]) + result = h.unpack_result(params, 3, 2) + assert "bias_cad" in result + assert "W_cad" in result + np.testing.assert_allclose(result["bias_cad"], 2.0) + np.testing.assert_allclose(result["W_cad"], [0.5, -1.0]) + + def test_make_bounds(self): + h = LinearHead("cad") + bounds = h.make_bounds(3, 5) + assert len(bounds) == 6 # 1 + 5 + + +# ── PerPoolNoiseHead ──────────────────────────────────────────────────────── + + +class TestPerPoolNoiseHead: + def test_n_params(self): + h = PerPoolNoiseHead() + assert h.n_params(3, 5) == 3 * K_OBS + assert h.n_params(2, 7) == 2 * K_OBS + + def test_predict_correct_slice(self): + h = PerPoolNoiseHead() + n_pools = 3 + params = jnp.arange(n_pools * K_OBS, dtype=float) + x_attr_i = jnp.zeros(5) + + for i in range(n_pools): + result = h.predict(params, i, x_attr_i) + expected = params[i * K_OBS:(i + 1) * K_OBS] + np.testing.assert_allclose(result, expected) + + def test_regularization_zero_by_default(self): + h = PerPoolNoiseHead() + params = jnp.ones(16) + assert float(h.regularization(params)) == 0.0 + + def test_regularization_with_alpha(self): + h = PerPoolNoiseHead(alpha=1.0) + params = jnp.array([3.0, 4.0]) + np.testing.assert_allclose(float(h.regularization(params)), 25.0) + + def test_init_from_ols(self): + np.random.seed(42) + h = PerPoolNoiseHead() + jdata = _make_fake_jdata() + init = h.init(jdata) + assert init.shape == (N_POOLS * K_OBS,) + assert np.all(np.isfinite(init)) + + def test_init_warm_start(self): + h = PerPoolNoiseHead() + jdata = _make_fake_jdata() + warm = { + POOL_PREFIXES[0]: {"noise_coeffs": np.ones(K_OBS) * 5.0}, + POOL_PREFIXES[1]: {"noise_coeffs": np.ones(K_OBS) * 7.0}, + } + init = h.init(jdata, warm_start=warm) + assert init.shape == (N_POOLS * K_OBS,) + np.testing.assert_allclose(init[:K_OBS], 5.0) + np.testing.assert_allclose(init[K_OBS:], 7.0) + + def test_predict_new_raises(self): + h = PerPoolNoiseHead() + with pytest.raises(ValueError, match="cannot predict"): + h.predict_new(np.zeros(K_OBS * 2), np.zeros(5)) + + def test_unpack_result(self): + h = PerPoolNoiseHead() + params = np.arange(N_POOLS * K_OBS, dtype=float) + result = h.unpack_result(params, N_POOLS, K_ATTR) + assert result["noise_coeffs"].shape == (N_POOLS, K_OBS) + + +# ── SharedLinearNoiseHead ─────────────────────────────────────────────────── + + +class TestSharedLinearNoiseHead: + def test_n_params(self): + h = SharedLinearNoiseHead() + assert h.n_params(3, 5) == (1 + 5) * K_OBS + assert h.n_params(10, 7) == (1 + 7) * K_OBS + + def test_predict_bias_plus_dot(self): + k_attr = 3 + h = SharedLinearNoiseHead() + W_full = np.zeros((1 + k_attr, K_OBS)) + W_full[0, :] = 1.0 # bias_noise = [1, 1, ..., 1] + W_full[1, 0] = 2.0 # first feature maps to first noise coeff + params = jnp.array(W_full.ravel()) + x_attr_i = jnp.array([1.0, 0.0, 0.0]) + result = h.predict(params, 0, x_attr_i) + assert result.shape == (K_OBS,) + np.testing.assert_allclose(float(result[0]), 3.0) # 1 + 2*1 + np.testing.assert_allclose(float(result[1]), 1.0) # 1 + 0 + + def test_predict_ignores_pool_idx(self): + k_attr = 2 + h = SharedLinearNoiseHead() + params = jnp.ones((1 + k_attr) * K_OBS) + x = jnp.array([1.0, 2.0]) + r0 = h.predict(params, 0, x) + r5 = h.predict(params, 5, x) + np.testing.assert_allclose(r0, r5) + + def test_regularization_on_W_not_bias(self): + k_attr = 2 + h = SharedLinearNoiseHead(alpha=1.0) + W_full = np.zeros((1 + k_attr, K_OBS)) + W_full[0, :] = 100.0 # bias — not regularized + W_full[1, 0] = 3.0 + W_full[2, 0] = 4.0 + params = jnp.array(W_full.ravel()) + # reg = 1.0 * (9 + 16) = 25.0 + np.testing.assert_allclose(float(h.regularization(params)), 25.0) + + def test_init_default(self): + np.random.seed(42) + h = SharedLinearNoiseHead() + jdata = _make_fake_jdata() + init = h.init(jdata) + assert init.shape == ((1 + K_ATTR) * K_OBS,) + assert np.all(np.isfinite(init)) + + def test_predict_new(self): + k_attr = 2 + h = SharedLinearNoiseHead() + W_full = np.zeros((1 + k_attr, K_OBS)) + W_full[0, :] = 5.0 + W_full[1, 0] = 1.0 + params = W_full.ravel() + x_attr = np.array([2.0, 0.0]) + result = h.predict_new(params, x_attr) + assert result.shape == (K_OBS,) + np.testing.assert_allclose(result[0], 7.0) + np.testing.assert_allclose(result[1], 5.0) + + def test_unpack_result(self): + h = SharedLinearNoiseHead() + k_attr = 3 + W_full = np.arange((1 + k_attr) * K_OBS, dtype=float) + result = h.unpack_result(W_full, 2, k_attr) + assert "bias_noise" in result + assert "W_noise" in result + assert result["bias_noise"].shape == (K_OBS,) + assert result["W_noise"].shape == (k_attr, K_OBS) From c7ee40da61b20ff60f339d152144261fe31179b8 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 10:35:18 +0000 Subject: [PATCH 039/115] feat: add MLPHead for nonlinear pool-attribute-to-cadence mapping MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two-layer MLP (x_attr → Dense(hidden, ReLU) → Dense(1)) with He initialization, L2 regularization on weights, and warm-start from per-pool fits. 16 unit tests + 6 integration tests with CalibrationModel. --- quantammsim/calibration/__init__.py | 1 + quantammsim/calibration/heads.py | 132 ++++++++++++++ tests/calibration/test_calibration_model.py | 79 ++++++++ tests/calibration/test_heads.py | 189 ++++++++++++++++++++ 4 files changed, 401 insertions(+) diff --git a/quantammsim/calibration/__init__.py b/quantammsim/calibration/__init__.py index 1af09dac..c229fb3c 100644 --- a/quantammsim/calibration/__init__.py +++ b/quantammsim/calibration/__init__.py @@ -44,6 +44,7 @@ FixedHead, Head, LinearHead, + MLPHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index a7e01d39..53d65ad2 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -375,3 +375,135 @@ def unpack_result(self, params_slice, n_pools, k_attr): def make_bounds(self, n_pools, k_attr): return [(None, None)] * ((1 + k_attr) * K_OBS) + + +# --------------------------------------------------------------------------- +# MLPHead — x_attr → Dense(hidden, relu) → Dense(1) +# --------------------------------------------------------------------------- + + +class MLPHead: + """Two-layer MLP mapping from pool attributes to a scalar. + + Architecture: x_attr → Dense(hidden, ReLU) → Dense(1) → scalar + + Parameter layout (flat): + [W1(k_attr * hidden), b1(hidden), W2(hidden), b2(1)] + + L2 regularization on W1 and W2 (not biases). + + Initialization: + - W1: He (scaled normal), b1: zeros + - W2: zeros (so initial output ≈ b2 = default bias) + - b2: sensible default (log(12) for cadence, log(1) for gas) + """ + + def __init__( + self, + name: str, + hidden: int = 16, + alpha: float = 0.01, + seed: int = 0, + ): + self.name = name + self.hidden = hidden + self.alpha = alpha + self._seed = seed + + def n_params(self, n_pools: int, k_attr: int) -> int: + h = self.hidden + return k_attr * h + h + h + 1 # W1 + b1 + W2 + b2 + + def _unpack_weights(self, params_slice, k_attr): + """Unpack flat slice → (W1, b1, W2, b2) as JAX arrays.""" + h = self.hidden + idx = 0 + W1 = params_slice[idx:idx + k_attr * h].reshape(k_attr, h) + idx += k_attr * h + b1 = params_slice[idx:idx + h] + idx += h + W2 = params_slice[idx:idx + h] + idx += h + b2 = params_slice[idx] + return W1, b1, W2, b2 + + def predict(self, params_slice, pool_idx, x_attr_i): + k_attr = x_attr_i.shape[0] + W1, b1, W2, b2 = self._unpack_weights(params_slice, k_attr) + hidden = jnp.maximum(x_attr_i @ W1 + b1, 0.0) # ReLU + return hidden @ W2 + b2 + + def regularization(self, params_slice): + # Regularize W1 and W2, not biases + # We can't call _unpack_weights without k_attr, so compute + # the total weight norm from the full slice minus biases. + # Layout: [W1(k*h), b1(h), W2(h), b2(1)] + # But we don't know k_attr here. Use a simpler approach: + # regularize the entire slice — biases are small relative to + # weights and the approximation error is negligible. + # Actually, let's extract properly by computing h from params. + h = self.hidden + total = params_slice.shape[0] + k_attr = (total - 2 * h - 1) // h + W1 = params_slice[:k_attr * h] + # b1 = params_slice[k_attr*h : k_attr*h + h] # skip + W2 = params_slice[k_attr * h + h:k_attr * h + 2 * h] + # b2 = params_slice[-1] # skip + return self.alpha * (jnp.sum(W1 ** 2) + jnp.sum(W2 ** 2)) + + def init(self, jdata, warm_start=None): + k_attr = jdata.x_attr.shape[1] + n_pools = len(jdata.pool_data) + h = self.hidden + rng = np.random.RandomState(self._seed) + + # He initialization for W1 + std = np.sqrt(2.0 / k_attr) + W1 = rng.randn(k_attr, h).astype(np.float64) * std + b1 = np.zeros(h, dtype=np.float64) + + # W2 = 0 so initial output = b2 (warm-start friendly) + W2 = np.zeros(h, dtype=np.float64) + b2 = np.array([self._default_bias()], dtype=np.float64) + + if warm_start is not None: + # Fit linear mapping from per-pool values, use as last-layer init + vals = [] + for pid in jdata.pool_ids: + if pid in warm_start and self.name in warm_start[pid]: + vals.append(warm_start[pid][self.name]) + else: + vals.append(self._default_bias()) + y = np.array(vals) + # Use mean as b2 (since W2=0, output = b2) + b2 = np.array([np.mean(y)], dtype=np.float64) + + return np.concatenate([W1.ravel(), b1, W2, b2]) + + def _default_bias(self): + if "cad" in self.name: + return np.log(12.0) + elif "gas" in self.name: + return np.log(1.0) + return 0.0 + + def predict_new(self, params_slice, x_attr): + k_attr = len(x_attr) + W1, b1, W2, b2 = self._unpack_weights( + np.asarray(params_slice), k_attr + ) + hidden = np.maximum(x_attr @ W1 + b1, 0.0) + return float(hidden @ W2 + b2) + + def unpack_result(self, params_slice, n_pools, k_attr): + params_np = np.array(params_slice) + W1, b1, W2, b2 = self._unpack_weights(params_np, k_attr) + return { + f"mlp_{self.name}_W1": np.array(W1), + f"mlp_{self.name}_b1": np.array(b1), + f"mlp_{self.name}_W2": np.array(W2), + f"mlp_{self.name}_b2": float(b2), + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * self.n_params(n_pools, k_attr) diff --git a/tests/calibration/test_calibration_model.py b/tests/calibration/test_calibration_model.py index 35306884..3f38f955 100644 --- a/tests/calibration/test_calibration_model.py +++ b/tests/calibration/test_calibration_model.py @@ -14,6 +14,7 @@ from quantammsim.calibration.heads import ( FixedHead, LinearHead, + MLPHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, @@ -419,3 +420,81 @@ def test_option_a_shared_free_matches_old_param_count(self, jdata_ppn): ) expected = 2 + 2 * k_attr + (1 + k_attr) * K_OBS assert model.n_params(n_pools, k_attr) == expected + + +# ── MLP integration tests ───────────────────────────────────────────────── + + +class TestMLPIntegration: + """Test CalibrationModel with MLPHead for cadence.""" + + def test_mlp_cadence_n_params(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + mlp_params = k_attr * 8 + 8 + 8 + 1 + linear_params = 1 + k_attr + noise_params = n_pools * K_OBS + assert model.n_params(n_pools, k_attr) == mlp_params + linear_params + noise_params + + def test_mlp_cadence_fit_converges(self, jdata_ppn): + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + assert np.isfinite(result["loss"]) + + def test_mlp_cadence_and_gas_fit(self, jdata_ppn): + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + MLPHead("gas", hidden=8, alpha=0.01), + PerPoolNoiseHead(), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + + def test_mlp_predict_new_pool(self, jdata_ppn): + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + MLPHead("gas", hidden=8, alpha=0.01), + SharedLinearNoiseHead(alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=50) + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + assert pred["cadence_minutes"] > 0 + assert pred["gas_usd"] > 0 + assert "noise_coeffs" in pred + + def test_mlp_loss_differentiable(self, jdata_ppn): + import jax + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + ) + loss_fn = model.make_joint_loss_fn(jdata_ppn) + init = jnp.array(model.pack_init(jdata_ppn)) + grad = jax.grad(loss_fn)(init) + assert jnp.all(jnp.isfinite(grad)) + assert float(jnp.sum(jnp.abs(grad))) > 0 + + def test_mlp_with_huber_loss(self, jdata_ppn): + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + LinearHead("gas", alpha=0.01), + PerPoolNoiseHead(), + loss_type="huber", huber_delta=1.5, + ) + result = model.fit(jdata_ppn, maxiter=50) + assert result["loss"] >= 0 + assert np.isfinite(result["loss"]) diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index 7513d6f5..9627c7ea 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -10,6 +10,7 @@ FixedHead, Head, LinearHead, + MLPHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, @@ -66,6 +67,9 @@ def test_per_pool_noise_head_is_head(self): def test_shared_linear_noise_head_is_head(self): assert isinstance(SharedLinearNoiseHead(), Head) + def test_mlp_head_is_head(self): + assert isinstance(MLPHead("cad"), Head) + # ── PerPoolHead ───────────────────────────────────────────────────────────── @@ -377,3 +381,188 @@ def test_unpack_result(self): assert "W_noise" in result assert result["bias_noise"].shape == (K_OBS,) assert result["W_noise"].shape == (k_attr, K_OBS) + + +# ── MLPHead ───────────────────────────────────────────────────────────────── + + +class TestMLPHead: + def test_n_params(self): + h = MLPHead("cad", hidden=16) + # k_attr=5: 5*16 + 16 + 16 + 1 = 113 + assert h.n_params(3, 5) == 113 + # k_attr=7: 7*16 + 16 + 16 + 1 = 145 + assert h.n_params(3, 7) == 145 + + def test_n_params_custom_hidden(self): + h = MLPHead("cad", hidden=8) + # k_attr=5: 5*8 + 8 + 8 + 1 = 57 + assert h.n_params(3, 5) == 57 + + def test_predict_with_zero_W2_equals_b2(self): + """With W2=0, output should be b2 regardless of input.""" + k_attr = 3 + h = MLPHead("cad", hidden=4) + n_p = h.n_params(1, k_attr) + params = np.zeros(n_p) + params[-1] = 2.5 # b2 + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + result = float(h.predict(jnp.array(params), 0, x_attr_i)) + np.testing.assert_allclose(result, 2.5) + + def test_predict_nonlinear(self): + """MLP should produce different outputs for different inputs.""" + k_attr = 3 + h = MLPHead("cad", hidden=4, seed=42) + jdata = _make_fake_jdata() + # Override x_attr to have k_attr=3 + from quantammsim.calibration.joint_fit import JointData + jdata = JointData( + pool_data=jdata.pool_data, + x_attr=jnp.array(np.random.randn(N_POOLS, k_attr)), + pool_ids=jdata.pool_ids, + attr_names=[f"a{i}" for i in range(k_attr)], + ) + init = jnp.array(h.init(jdata)) + # Set W1 to nonzero so ReLU activations vary + np.random.seed(42) + W1 = np.random.randn(k_attr * 4) * 0.5 + init = init.at[:k_attr * 4].set(jnp.array(W1)) + # Set W2 to nonzero so output varies + init = init.at[k_attr * 4 + 4:k_attr * 4 + 8].set(jnp.ones(4) * 0.1) + + x1 = jnp.array([1.0, 0.0, 0.0]) + x2 = jnp.array([0.0, 1.0, 0.0]) + v1 = float(h.predict(init, 0, x1)) + v2 = float(h.predict(init, 0, x2)) + assert v1 != v2, "MLP should produce different outputs for different inputs" + + def test_predict_ignores_pool_idx(self): + """MLP output depends only on x_attr, not pool_idx.""" + k_attr = 3 + h = MLPHead("cad", hidden=4) + params = jnp.ones(h.n_params(5, k_attr)) * 0.1 + x = jnp.array([1.0, 2.0, 3.0]) + v0 = float(h.predict(params, 0, x)) + v3 = float(h.predict(params, 3, x)) + assert v0 == v3 + + def test_regularization_on_weights_not_biases(self): + k_attr = 2 + h_alpha1 = MLPHead("cad", hidden=2, alpha=1.0) + # Layout: W1(2*2=4), b1(2), W2(2), b2(1) = 9 params + params = np.zeros(9) + params[0] = 3.0 # W1[0,0] + params[1] = 4.0 # W1[0,1] + # b1 = 0 (indices 4,5) + params[6] = 1.0 # W2[0] + params[7] = 2.0 # W2[1] + params[8] = 999.0 # b2 — should not be regularized + # reg = 1.0 * (9 + 16 + 1 + 4) = 30.0 + result = float(h_alpha1.regularization(jnp.array(params))) + np.testing.assert_allclose(result, 30.0) + + def test_regularization_alpha_scaling(self): + k_attr = 2 + h = MLPHead("cad", hidden=2, alpha=0.5) + params = np.zeros(9) + params[0] = 2.0 # W1 weight + # reg = 0.5 * 4.0 = 2.0 + np.testing.assert_allclose(float(h.regularization(jnp.array(params))), 2.0) + + def test_init_default_cadence(self): + h = MLPHead("cad", hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + n_p = h.n_params(N_POOLS, K_ATTR) + assert init.shape == (n_p,) + assert np.all(np.isfinite(init)) + # b2 should be log(12) + np.testing.assert_allclose(init[-1], np.log(12.0)) + + def test_init_default_gas(self): + h = MLPHead("gas", hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + # b2 should be log(1) = 0 + np.testing.assert_allclose(init[-1], 0.0) + + def test_init_W2_is_zero(self): + """W2 should be zero at init so output = b2.""" + h = MLPHead("cad", hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + # W2 is at [k_attr*hidden + hidden : k_attr*hidden + 2*hidden] + w2_start = K_ATTR * 4 + 4 + w2_end = w2_start + 4 + np.testing.assert_allclose(init[w2_start:w2_end], 0.0) + + def test_init_warm_start(self): + h = MLPHead("log_cadence", hidden=4) + jdata = _make_fake_jdata() + warm = { + POOL_PREFIXES[0]: {"log_cadence": 2.0}, + POOL_PREFIXES[1]: {"log_cadence": 3.0}, + } + init = h.init(jdata, warm_start=warm) + # b2 should be mean of warm-start values + np.testing.assert_allclose(init[-1], 2.5) + + def test_predict_new(self): + k_attr = 3 + h = MLPHead("cad", hidden=4) + n_p = h.n_params(1, k_attr) + params = np.zeros(n_p) + params[-1] = 2.5 # b2 + x_attr = np.array([1.0, 2.0, 3.0]) + result = h.predict_new(params, x_attr) + np.testing.assert_allclose(result, 2.5) + + def test_predict_new_matches_predict(self): + """predict_new should give same result as predict for same input.""" + k_attr = 3 + h = MLPHead("cad", hidden=4, seed=42) + n_p = h.n_params(1, k_attr) + np.random.seed(99) + params = np.random.randn(n_p) * 0.1 + x_attr = np.array([0.5, -1.0, 2.0]) + + jax_result = float(h.predict(jnp.array(params), 0, jnp.array(x_attr))) + np_result = h.predict_new(params, x_attr) + np.testing.assert_allclose(jax_result, np_result, rtol=1e-6) + + def test_unpack_result(self): + k_attr = 3 + h = MLPHead("cad", hidden=4) + n_p = h.n_params(1, k_attr) + params = np.arange(n_p, dtype=float) + result = h.unpack_result(params, 2, k_attr) + assert f"mlp_cad_W1" in result + assert f"mlp_cad_b1" in result + assert f"mlp_cad_W2" in result + assert f"mlp_cad_b2" in result + assert result["mlp_cad_W1"].shape == (k_attr, 4) + assert result["mlp_cad_b1"].shape == (4,) + assert result["mlp_cad_W2"].shape == (4,) + + def test_make_bounds(self): + h = MLPHead("cad", hidden=4) + bounds = h.make_bounds(3, 5) + assert len(bounds) == h.n_params(3, 5) + + def test_jax_differentiable(self): + """MLP predict should be JAX-differentiable.""" + import jax + k_attr = 3 + h = MLPHead("cad", hidden=4) + n_p = h.n_params(1, k_attr) + np.random.seed(42) + params = jnp.array(np.random.randn(n_p) * 0.1) + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + + def loss(p): + return h.predict(p, 0, x_attr_i) ** 2 + + grad = jax.grad(loss)(params) + assert grad.shape == params.shape + assert jnp.all(jnp.isfinite(grad)) From ef7d24ad291c5e43b5d3b64d7abdc7c89c1b3675 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 11:05:55 +0000 Subject: [PATCH 040/115] feat: add MLPNoiseHead for nonlinear pool-attribute-to-noise mapping Two-layer MLP (x_attr, Dense(hidden, ReLU), Dense(K_OBS)) that replaces the linear SharedLinearNoiseHead for the noise coefficient mapping. Initialized with W2=0 so output starts at pooled OLS noise coefficients. 16 unit tests + 6 CalibrationModel integration tests. --- quantammsim/calibration/__init__.py | 1 + quantammsim/calibration/heads.py | 115 +++++++++++++ tests/calibration/test_calibration_model.py | 84 +++++++++ tests/calibration/test_heads.py | 182 ++++++++++++++++++++ 4 files changed, 382 insertions(+) diff --git a/quantammsim/calibration/__init__.py b/quantammsim/calibration/__init__.py index c229fb3c..48a556ab 100644 --- a/quantammsim/calibration/__init__.py +++ b/quantammsim/calibration/__init__.py @@ -45,6 +45,7 @@ Head, LinearHead, MLPHead, + MLPNoiseHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 53d65ad2..2edd3430 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -507,3 +507,118 @@ def unpack_result(self, params_slice, n_pools, k_attr): def make_bounds(self, n_pools, k_attr): return [(None, None)] * self.n_params(n_pools, k_attr) + + +# --------------------------------------------------------------------------- +# MLPNoiseHead — x_attr → Dense(hidden, relu) → Dense(K_OBS) +# --------------------------------------------------------------------------- + + +class MLPNoiseHead: + """Two-layer MLP mapping from pool attributes to noise coefficients. + + Architecture: x_attr → Dense(hidden, ReLU) → Dense(K_OBS) + + Parameter layout (flat): + [W1(k_attr * hidden), b1(hidden), W2(hidden * K_OBS), b2(K_OBS)] + + L2 regularization on W1 and W2 (not biases). + + Initialization: + - W1: He (scaled normal), b1: zeros + - W2: zeros (so initial output = b2 = pooled OLS noise coefficients) + - b2: pooled OLS noise from training data + """ + + def __init__( + self, + hidden: int = 16, + alpha: float = 0.01, + seed: int = 0, + ): + self.name = "noise" + self.hidden = hidden + self.alpha = alpha + self._seed = seed + + def n_params(self, n_pools: int, k_attr: int) -> int: + h = self.hidden + # W1(k_attr*h) + b1(h) + W2(h*K_OBS) + b2(K_OBS) + return k_attr * h + h + h * K_OBS + K_OBS + + def _unpack_weights(self, params_slice, k_attr): + """Unpack flat slice → (W1, b1, W2, b2).""" + h = self.hidden + idx = 0 + W1 = params_slice[idx:idx + k_attr * h].reshape(k_attr, h) + idx += k_attr * h + b1 = params_slice[idx:idx + h] + idx += h + W2 = params_slice[idx:idx + h * K_OBS].reshape(h, K_OBS) + idx += h * K_OBS + b2 = params_slice[idx:idx + K_OBS] + return W1, b1, W2, b2 + + def predict(self, params_slice, pool_idx, x_attr_i): + k_attr = x_attr_i.shape[0] + W1, b1, W2, b2 = self._unpack_weights(params_slice, k_attr) + hidden = jnp.maximum(x_attr_i @ W1 + b1, 0.0) # ReLU + return hidden @ W2 + b2 # (K_OBS,) + + def regularization(self, params_slice): + h = self.hidden + total = params_slice.shape[0] + # Solve for k_attr: total = k*h + h + h*K_OBS + K_OBS + # k*h = total - h - h*K_OBS - K_OBS + k_attr = (total - h - h * K_OBS - K_OBS) // h + W1 = params_slice[:k_attr * h] + W2 = params_slice[k_attr * h + h:k_attr * h + h + h * K_OBS] + return self.alpha * (jnp.sum(W1 ** 2) + jnp.sum(W2 ** 2)) + + def init(self, jdata, warm_start=None): + k_attr = jdata.x_attr.shape[1] + n_pools = len(jdata.pool_data) + h = self.hidden + rng = np.random.RandomState(self._seed) + + # He initialization for W1 + std = np.sqrt(2.0 / k_attr) + W1 = rng.randn(k_attr, h).astype(np.float64) * std + b1 = np.zeros(h, dtype=np.float64) + + # W2 = 0 so initial output = b2 + W2 = np.zeros((h, K_OBS), dtype=np.float64) + + if warm_start is not None: + # Use mean of per-pool noise as b2 + noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + for i, pid in enumerate(jdata.pool_ids): + if pid in warm_start and "noise_coeffs" in warm_start[pid]: + noise_all[i] = warm_start[pid]["noise_coeffs"] + b2 = np.mean(noise_all, axis=0) + else: + # Pooled OLS noise as b2 + all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) + all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) + b2, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) + + return np.concatenate([W1.ravel(), b1, W2.ravel(), b2]) + + def predict_new(self, params_slice, x_attr): + k_attr = len(x_attr) + W1, b1, W2, b2 = self._unpack_weights(np.asarray(params_slice), k_attr) + hidden = np.maximum(x_attr @ W1 + b1, 0.0) + return hidden @ W2 + b2 # (K_OBS,) + + def unpack_result(self, params_slice, n_pools, k_attr): + params_np = np.array(params_slice) + W1, b1, W2, b2 = self._unpack_weights(params_np, k_attr) + return { + "mlp_noise_W1": np.array(W1), + "mlp_noise_b1": np.array(b1), + "mlp_noise_W2": np.array(W2), + "mlp_noise_b2": np.array(b2), + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * self.n_params(n_pools, k_attr) diff --git a/tests/calibration/test_calibration_model.py b/tests/calibration/test_calibration_model.py index 3f38f955..95973f47 100644 --- a/tests/calibration/test_calibration_model.py +++ b/tests/calibration/test_calibration_model.py @@ -15,6 +15,7 @@ FixedHead, LinearHead, MLPHead, + MLPNoiseHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, @@ -498,3 +499,86 @@ def test_mlp_with_huber_loss(self, jdata_ppn): result = model.fit(jdata_ppn, maxiter=50) assert result["loss"] >= 0 assert np.isfinite(result["loss"]) + + +# ── MLP noise integration tests ─────────────────────────────────────────── + + +class TestMLPNoiseIntegration: + """Test CalibrationModel with MLPNoiseHead — the key use case.""" + + def test_mlp_noise_fit_converges(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + MLPNoiseHead(hidden=8, alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + assert np.isfinite(result["loss"]) + + def test_mlp_noise_predict_new_pool(self, jdata_ppn): + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + MLPNoiseHead(hidden=8, alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=50) + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + assert pred["cadence_minutes"] > 0 + assert pred["gas_usd"] > 0 + assert "noise_coeffs" in pred + assert len(pred["noise_coeffs"]) == K_OBS + + def test_mlp_noise_loss_differentiable(self, jdata_ppn): + import jax + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + LinearHead("gas", alpha=0.01), + MLPNoiseHead(hidden=8, alpha=0.01), + ) + loss_fn = model.make_joint_loss_fn(jdata_ppn) + init = jnp.array(model.pack_init(jdata_ppn)) + grad = jax.grad(loss_fn)(init) + assert jnp.all(jnp.isfinite(grad)) + assert float(jnp.sum(jnp.abs(grad))) > 0 + + def test_full_mlp_model(self, jdata_ppn): + """MLP for all three heads — most expressive config.""" + model = CalibrationModel( + MLPHead("cad", hidden=8, alpha=0.01), + MLPHead("gas", hidden=8, alpha=0.01), + MLPNoiseHead(hidden=8, alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + x_attr = np.zeros(result["k_attr"]) + pred = model.predict_new_pool(result, x_attr) + assert pred["cadence_minutes"] > 0 + assert "noise_coeffs" in pred + + def test_mlp_noise_with_fixed_gas(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + gas_values = np.array([np.log(1.0)] * n_pools) + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + FixedHead("gas", gas_values), + MLPNoiseHead(hidden=8, alpha=0.01), + ) + result = model.fit(jdata_ppn, maxiter=100) + assert result["loss"] <= result["init_loss"] + + def test_mlp_noise_param_count(self, jdata_ppn): + n_pools = len(jdata_ppn.pool_data) + k_attr = jdata_ppn.x_attr.shape[1] + h = 8 + model = CalibrationModel( + LinearHead("cad"), + LinearHead("gas"), + MLPNoiseHead(hidden=h), + ) + # Linear cad: 1+k, Linear gas: 1+k, + # MLP noise: k*h + h + h*K_OBS + K_OBS + expected = (1 + k_attr) * 2 + k_attr * h + h + h * K_OBS + K_OBS + assert model.n_params(n_pools, k_attr) == expected diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index 9627c7ea..5eae07c3 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -11,6 +11,7 @@ Head, LinearHead, MLPHead, + MLPNoiseHead, PerPoolHead, PerPoolNoiseHead, SharedLinearNoiseHead, @@ -70,6 +71,9 @@ def test_shared_linear_noise_head_is_head(self): def test_mlp_head_is_head(self): assert isinstance(MLPHead("cad"), Head) + def test_mlp_noise_head_is_head(self): + assert isinstance(MLPNoiseHead(), Head) + # ── PerPoolHead ───────────────────────────────────────────────────────────── @@ -566,3 +570,181 @@ def loss(p): grad = jax.grad(loss)(params) assert grad.shape == params.shape assert jnp.all(jnp.isfinite(grad)) + + +# ── MLPNoiseHead ──────────────────────────────────────────────────────────── + + +class TestMLPNoiseHead: + def test_n_params(self): + h = MLPNoiseHead(hidden=16) + # k_attr=5: 5*16 + 16 + 16*8 + 8 = 80+16+128+8 = 232 + assert h.n_params(3, 5) == 232 + # k_attr=7: 7*16 + 16 + 16*8 + 8 = 112+16+128+8 = 264 + assert h.n_params(3, 7) == 264 + + def test_n_params_custom_hidden(self): + h = MLPNoiseHead(hidden=8) + # k_attr=5: 5*8 + 8 + 8*8 + 8 = 40+8+64+8 = 120 + assert h.n_params(3, 5) == 120 + + def test_predict_output_shape(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4) + n_p = h.n_params(1, k_attr) + params = jnp.zeros(n_p) + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + result = h.predict(params, 0, x_attr_i) + assert result.shape == (K_OBS,) + + def test_predict_with_zero_W2_equals_b2(self): + """With W2=0, output should be b2 regardless of input.""" + k_attr = 3 + h = MLPNoiseHead(hidden=4) + n_p = h.n_params(1, k_attr) + params = np.zeros(n_p) + # b2 is the last K_OBS elements + params[-K_OBS:] = np.arange(K_OBS) + 1.0 + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + result = h.predict(jnp.array(params), 0, x_attr_i) + np.testing.assert_allclose(result, np.arange(K_OBS) + 1.0) + + def test_predict_nonlinear(self): + """MLP should produce different outputs for different inputs.""" + k_attr = 3 + h = MLPNoiseHead(hidden=4, seed=42) + n_p = h.n_params(1, k_attr) + np.random.seed(42) + params = jnp.array(np.random.randn(n_p) * 0.1) + x1 = jnp.array([1.0, 0.0, 0.0]) + x2 = jnp.array([0.0, 1.0, 0.0]) + v1 = h.predict(params, 0, x1) + v2 = h.predict(params, 0, x2) + assert not jnp.allclose(v1, v2), "Should produce different outputs" + + def test_predict_ignores_pool_idx(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4) + params = jnp.ones(h.n_params(5, k_attr)) * 0.1 + x = jnp.array([1.0, 2.0, 3.0]) + v0 = h.predict(params, 0, x) + v3 = h.predict(params, 3, x) + np.testing.assert_allclose(v0, v3) + + def test_regularization_on_weights_not_biases(self): + k_attr = 2 + h = MLPNoiseHead(hidden=2, alpha=1.0) + # Layout: W1(2*2=4), b1(2), W2(2*8=16), b2(8) = 30 params + n_p = h.n_params(1, k_attr) + assert n_p == 30 + params = np.zeros(n_p) + params[0] = 3.0 # W1[0,0] + params[1] = 4.0 # W1[0,1] + # b1 at indices 4,5 — not regularized + params[6] = 1.0 # W2[0,0] + params[7] = 2.0 # W2[0,1] + params[-1] = 999.0 # b2[-1] — not regularized + # reg = 1.0 * (9 + 16 + 1 + 4) = 30.0 + result = float(h.regularization(jnp.array(params))) + np.testing.assert_allclose(result, 30.0) + + def test_init_default(self): + np.random.seed(42) + h = MLPNoiseHead(hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + n_p = h.n_params(N_POOLS, K_ATTR) + assert init.shape == (n_p,) + assert np.all(np.isfinite(init)) + + def test_init_W2_is_zero(self): + """W2 should be zero at init so output = b2.""" + h = MLPNoiseHead(hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + # W2 starts at k_attr*hidden + hidden + w2_start = K_ATTR * 4 + 4 + w2_end = w2_start + 4 * K_OBS + np.testing.assert_allclose(init[w2_start:w2_end], 0.0) + + def test_init_b2_from_ols(self): + """b2 should be pooled OLS noise coefficients.""" + np.random.seed(42) + h = MLPNoiseHead(hidden=4) + jdata = _make_fake_jdata() + init = h.init(jdata) + b2 = init[-K_OBS:] + assert np.all(np.isfinite(b2)) + # Should be nonzero (OLS on random data) + assert np.any(b2 != 0.0) + + def test_init_warm_start(self): + h = MLPNoiseHead(hidden=4) + jdata = _make_fake_jdata() + warm = { + POOL_PREFIXES[0]: {"noise_coeffs": np.ones(K_OBS) * 5.0}, + POOL_PREFIXES[1]: {"noise_coeffs": np.ones(K_OBS) * 7.0}, + } + init = h.init(jdata, warm_start=warm) + b2 = init[-K_OBS:] + # b2 should be mean of warm-start noise: (5+7)/2 = 6 + np.testing.assert_allclose(b2, 6.0) + + def test_predict_new(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4) + n_p = h.n_params(1, k_attr) + params = np.zeros(n_p) + params[-K_OBS:] = np.arange(K_OBS) + 1.0 # b2 + x_attr = np.array([1.0, 2.0, 3.0]) + result = h.predict_new(params, x_attr) + assert result.shape == (K_OBS,) + np.testing.assert_allclose(result, np.arange(K_OBS) + 1.0) + + def test_predict_new_matches_predict(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4, seed=42) + n_p = h.n_params(1, k_attr) + np.random.seed(99) + params = np.random.randn(n_p) * 0.1 + x_attr = np.array([0.5, -1.0, 2.0]) + + jax_result = np.array(h.predict(jnp.array(params), 0, jnp.array(x_attr))) + np_result = h.predict_new(params, x_attr) + np.testing.assert_allclose(jax_result, np_result, rtol=1e-6) + + def test_unpack_result(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4) + n_p = h.n_params(1, k_attr) + params = np.arange(n_p, dtype=float) + result = h.unpack_result(params, 2, k_attr) + assert "mlp_noise_W1" in result + assert "mlp_noise_b1" in result + assert "mlp_noise_W2" in result + assert "mlp_noise_b2" in result + assert result["mlp_noise_W1"].shape == (k_attr, 4) + assert result["mlp_noise_b1"].shape == (4,) + assert result["mlp_noise_W2"].shape == (4, K_OBS) + assert result["mlp_noise_b2"].shape == (K_OBS,) + + def test_make_bounds(self): + h = MLPNoiseHead(hidden=4) + bounds = h.make_bounds(3, 5) + assert len(bounds) == h.n_params(3, 5) + + def test_jax_differentiable(self): + import jax + k_attr = 3 + h = MLPNoiseHead(hidden=4) + n_p = h.n_params(1, k_attr) + np.random.seed(42) + params = jnp.array(np.random.randn(n_p) * 0.1) + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + + def loss(p): + return jnp.sum(h.predict(p, 0, x_attr_i) ** 2) + + grad = jax.grad(loss)(params) + assert grad.shape == params.shape + assert jnp.all(jnp.isfinite(grad)) From ccc3c6cf915a9db4913122a89b2c293be92a3108 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 12:39:26 +0000 Subject: [PATCH 041/115] fix: use small random W2 init for MLP heads to avoid degenerate L-BFGS Hessian Add MLP calibration and sweep scripts. --- quantammsim/calibration/heads.py | 10 +- scripts/run_mlp_calibration.py | 753 +++++++++++++++++++++++++++++++ scripts/run_mlp_sweep.py | 429 ++++++++++++++++++ tests/calibration/test_heads.py | 16 +- 4 files changed, 1197 insertions(+), 11 deletions(-) create mode 100644 scripts/run_mlp_calibration.py create mode 100644 scripts/run_mlp_sweep.py diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 2edd3430..1191d0a0 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -462,8 +462,8 @@ def init(self, jdata, warm_start=None): W1 = rng.randn(k_attr, h).astype(np.float64) * std b1 = np.zeros(h, dtype=np.float64) - # W2 = 0 so initial output = b2 (warm-start friendly) - W2 = np.zeros(h, dtype=np.float64) + # Small random W2 so L-BFGS gets a non-degenerate initial Hessian + W2 = rng.randn(h).astype(np.float64) * 0.01 b2 = np.array([self._default_bias()], dtype=np.float64) if warm_start is not None: @@ -475,7 +475,7 @@ def init(self, jdata, warm_start=None): else: vals.append(self._default_bias()) y = np.array(vals) - # Use mean as b2 (since W2=0, output = b2) + # Use mean as b2 b2 = np.array([np.mean(y)], dtype=np.float64) return np.concatenate([W1.ravel(), b1, W2, b2]) @@ -586,8 +586,8 @@ def init(self, jdata, warm_start=None): W1 = rng.randn(k_attr, h).astype(np.float64) * std b1 = np.zeros(h, dtype=np.float64) - # W2 = 0 so initial output = b2 - W2 = np.zeros((h, K_OBS), dtype=np.float64) + # Small random W2 so L-BFGS gets a non-degenerate initial Hessian + W2 = rng.randn(h, K_OBS).astype(np.float64) * 0.01 if warm_start is not None: # Use mean of per-pool noise as b2 diff --git a/scripts/run_mlp_calibration.py b/scripts/run_mlp_calibration.py new file mode 100644 index 00000000..39189c49 --- /dev/null +++ b/scripts/run_mlp_calibration.py @@ -0,0 +1,753 @@ +"""Run calibration with MLPNoiseHead and compare against linear baselines. + +Steps: + 1. Load panel, match to per-day grids + 2. Option C: per-pool L-BFGS-B fits (baseline, gas fixed to chain) + 3. Linear joint: SharedLinearNoiseHead baseline + 4. MLP noise joint: MLPNoiseHead (new) + 5. Full MLP joint: MLPHead cadence + MLPNoiseHead (new) + 6. Per-pool prediction, R², decomposition for each method + 7. Paginated plots, summary distributions, comparison scatter + 8. JSON export +""" + +import json +import os + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +GRID_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "pool_grids_v2", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "mlp_calibration", +) +OPTION_C_LOSS_CUTOFF = 5.0 +OPTION_C_MAXITER = 500 +JOINT_MAXITER = 500 +MLP_HIDDEN = 16 +TOP_N = 50 + + +# ---- Data loading ---- + + +def load_and_match(): + """Load panel, match to grids.""" + from quantammsim.calibration.pool_data import ( + match_grids_to_panel, + replace_panel_volatility_with_binance, + ) + + panel = pd.read_parquet(PANEL_CACHE) + + if "log_tvl_lag1" not in panel.columns: + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel.groupby("pool_id").size() + valid = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid)].copy() + + print("Replacing volatility with Binance minute data...") + panel = replace_panel_volatility_with_binance(panel) + + print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " + f"{panel['date'].min()} to {panel['date'].max()}") + + matched = match_grids_to_panel(GRID_DIR, panel) + print(f"Matched: {len(matched)} pools with grids") + return panel, matched + + +def filter_pathological(matched, option_c): + """Drop pools with high Option C loss.""" + good = {p: r for p, r in option_c.items() if r["loss"] <= OPTION_C_LOSS_CUTOFF} + dropped = set(option_c) - set(good) + matched_clean = {p: matched[p] for p in good if p in matched} + if dropped: + print(f" Dropping {len(dropped)} pools (loss > {OPTION_C_LOSS_CUTOFF}):") + for p in sorted(dropped): + print(f" {p} loss={option_c[p]['loss']:.1f}") + return matched_clean, good + + +# ---- Fitting ---- + + +def run_option_c(matched): + """Per-pool fits with gas fixed to chain costs.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + + print(f"\n--- Option C: per-pool fits ({len(matched)} pools, gas fixed) ---") + results = fit_all_pools(matched, fix_gas_to_chain=True) + + losses = [r["loss"] for r in results.values()] + n_conv = sum(1 for r in results.values() if r["converged"]) + print(f" Converged: {n_conv}/{len(results)}") + print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") + return results + + +def _build_gas_values(jdata, matched_clean): + """Build fixed gas values (log-space) from chain data.""" + from quantammsim.calibration.loss import CHAIN_GAS_USD + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_clean[pid]["chain"] + gas_usd = CHAIN_GAS_USD.get(chain, 1.0) + gas_values.append(np.log(max(gas_usd, 1e-6))) + return np.array(gas_values) + + +def run_linear_joint(matched_clean, option_c_clean): + """Joint fit with LinearHead + SharedLinearNoiseHead (baseline).""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import FixedHead, LinearHead, SharedLinearNoiseHead + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, fix_gas_to_chain=True) + gas_values = _build_gas_values(jdata, matched_clean) + + model = CalibrationModel( + cadence_head=LinearHead("cad", alpha=0.01), + gas_head=FixedHead("gas", gas_values), + noise_head=SharedLinearNoiseHead(alpha=0.01), + ) + + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + n_p = model.n_params(n_pools, k_attr) + print(f"\n--- Linear baseline: SharedLinearNoiseHead ({n_pools} pools, {n_p} params) ---") + + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + print(f" Converged: {result['converged']}") + return result, model, jdata + + +def run_mlp_noise_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): + """Joint fit with LinearHead cadence + FixedHead gas + MLPNoiseHead.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import FixedHead, LinearHead, MLPNoiseHead + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, fix_gas_to_chain=True) + gas_values = _build_gas_values(jdata, matched_clean) + + model = CalibrationModel( + cadence_head=LinearHead("cad", alpha=0.01), + gas_head=FixedHead("gas", gas_values), + noise_head=MLPNoiseHead(hidden=hidden, alpha=0.01), + ) + + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + n_p = model.n_params(n_pools, k_attr) + print(f"\n--- MLP noise: MLPNoiseHead(hidden={hidden}) ({n_pools} pools, {n_p} params) ---") + + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + print(f" Converged: {result['converged']}") + return result, model, jdata + + +def run_mlp_full_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): + """Joint fit with MLPHead cadence + FixedHead gas + MLPNoiseHead.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import FixedHead, MLPHead, MLPNoiseHead + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, fix_gas_to_chain=True) + gas_values = _build_gas_values(jdata, matched_clean) + + model = CalibrationModel( + cadence_head=MLPHead("cad", hidden=hidden, alpha=0.01), + gas_head=FixedHead("gas", gas_values), + noise_head=MLPNoiseHead(hidden=hidden, alpha=0.01), + ) + + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + n_p = model.n_params(n_pools, k_attr) + print(f"\n--- Full MLP: MLPHead(cad) + MLPNoiseHead ({n_pools} pools, {n_p} params) ---") + + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + print(f" Converged: {result['converged']}") + return result, model, jdata + + +# ---- Per-pool predictions ---- + + +def _extract_per_pool_params(model, result, jdata): + """Extract per-pool (log_cadence, log_gas, noise_coeffs) from a CalibrationModel result.""" + import jax.numpy as jnp + + params = jnp.array(result["params_flat"]) + n_pools = result["n_pools"] + k_attr = result["k_attr"] + (cs, ce), (gs, ge), (ns, ne) = model._head_slices(n_pools, k_attr) + + cad_slice = params[cs:ce] + gas_slice = params[gs:ge] + noise_slice = params[ns:ne] + + per_pool = [] + for i in range(n_pools): + x_attr_i = jdata.x_attr[i] + log_cad = float(model.cadence_head.predict(cad_slice, i, x_attr_i)) + log_gas = float(model.gas_head.predict(gas_slice, i, x_attr_i)) + noise_c = np.array(model.noise_head.predict(noise_slice, i, x_attr_i)) + per_pool.append({ + "log_cadence": log_cad, + "log_gas": log_gas, + "noise_coeffs": noise_c, + "cadence_minutes": float(np.exp(log_cad)), + "gas_usd": float(np.exp(log_gas)), + }) + return per_pool + + +def compute_per_pool_predictions(matched, option_c_results, + model_results): + """Compute V_arb, V_noise, R² per pool for Option C and each joint model. + + model_results: list of (label, per_pool_params, pool_ids) tuples, + where per_pool_params[i] is a dict with log_cadence, log_gas, noise_coeffs. + """ + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import build_x_obs + import jax.numpy as jnp + + pool_ids = sorted(matched.keys()) + + # Build lookup for each model's per-pool params + model_lookups = [] + for label, per_pool_params, m_pool_ids in model_results: + lookup = {pid: per_pool_params[i] for i, pid in enumerate(m_pool_ids)} + model_lookups.append((label, lookup)) + + def r2(v_arb, v_noise, y): + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y) ** 2) + ss_tot = np.sum((y - y.mean()) ** 2) + return 1 - ss_res / max(ss_tot, 1e-10) + + predictions = {} + for pid in pool_ids: + entry = matched[pid] + panel = entry["panel"] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + + p = { + "dates": pd.to_datetime(panel["date"].values), + "y_obs": y_obs, + "actual_vol": np.exp(y_obs), + "chain": entry["chain"], + "tokens": entry["tokens"], + "fee": entry["fee"], + "median_tvl": float(np.exp(panel["log_tvl_lag1"].median())), + "n_obs": len(y_obs), + } + + # Option C + rc = option_c_results[pid] + v_arb_all = np.array(interpolate_pool_daily( + coeffs, jnp.float64(rc["log_cadence"]), + jnp.float64(np.exp(rc["log_gas"])))) + v_arb_c = v_arb_all[day_indices] + v_noise_c = np.exp(x_obs @ rc["noise_coeffs"]) + p["v_arb_c"] = v_arb_c + p["v_noise_c"] = v_noise_c + p["r2_c"] = r2(v_arb_c, v_noise_c, y_obs) + p["cadence_c"] = rc["cadence_minutes"] + p["gas_c"] = rc["gas_usd"] + + # Each joint model + for label, lookup in model_lookups: + if pid in lookup: + mp = lookup[pid] + v_arb_all = np.array(interpolate_pool_daily( + coeffs, jnp.float64(mp["log_cadence"]), + jnp.float64(np.exp(mp["log_gas"])))) + v_arb = v_arb_all[day_indices] + v_noise = np.exp(x_obs @ mp["noise_coeffs"]) + p[f"v_arb_{label}"] = v_arb + p[f"v_noise_{label}"] = v_noise + p[f"r2_{label}"] = r2(v_arb, v_noise, y_obs) + p[f"cadence_{label}"] = mp["cadence_minutes"] + p[f"gas_{label}"] = mp["gas_usd"] + else: + n = len(y_obs) + p[f"v_arb_{label}"] = np.full(n, np.nan) + p[f"v_noise_{label}"] = np.full(n, np.nan) + p[f"r2_{label}"] = np.nan + p[f"cadence_{label}"] = np.nan + p[f"gas_{label}"] = np.nan + + predictions[pid] = p + + return predictions + + +# ---- Tables ---- + + +def print_pool_table(predictions, method_labels): + """Print per-pool results ranked by TVL.""" + ranked = sorted(predictions.items(), key=lambda x: -x[1]["median_tvl"]) + + header = f"{'Pool':<24} {'Chain':<10} {'TVL':>12} {'N':>4}" + header += f" {'Cad_C':>6} {'R2_C':>6}" + for label in method_labels: + short = label[:8] + header += f" {'Cad_'+short:>10} {'R2_'+short:>8}" + header += f" {'Arb%_C':>6}" + + print(f"\n{'='*len(header)}") + print(header) + print(f"{'-'*len(header)}") + for pid, p in ranked: + tokens = p["tokens"] + if isinstance(tokens, str): + tok_str = "/".join(t.strip()[:6] for t in tokens.split(",")[:2]) + else: + tok_str = pid[:16] + arb_total = p["v_arb_c"] + p["v_noise_c"] + arb_frac = np.median(p["v_arb_c"] / np.maximum(arb_total, 1.0)) + + line = (f"{tok_str:<24} {p['chain']:<10} ${p['median_tvl']:>10,.0f} " + f"{p['n_obs']:>4}") + line += f" {p['cadence_c']:>5.1f}m {p['r2_c']:>6.3f}" + for label in method_labels: + cad = p[f"cadence_{label}"] + r2v = p[f"r2_{label}"] + if np.isnan(cad): + line += f" {'---':>10} {'---':>8}" + else: + line += f" {cad:>9.1f}m {r2v:>8.3f}" + line += f" {arb_frac:>5.1%}" + print(line) + + +def print_r2_comparison(predictions, method_labels): + """Print aggregate R² comparison.""" + pool_ids = sorted(predictions.keys()) + + print(f"\n{'='*70}") + print("R² comparison (per-pool, in-sample)") + print(f"{'='*70}") + + r2_c = [predictions[p]["r2_c"] for p in pool_ids] + print(f" Option C (per-pool): median={np.median(r2_c):.4f} mean={np.mean(r2_c):.4f}") + + for label in method_labels: + r2_vals = [predictions[p][f"r2_{label}"] for p in pool_ids + if np.isfinite(predictions[p][f"r2_{label}"])] + if r2_vals: + print(f" {label:<22} median={np.median(r2_vals):.4f} mean={np.mean(r2_vals):.4f}") + + +def print_loss_comparison(option_c, joint_results): + """Print joint loss comparison.""" + print(f"\n{'='*70}") + print("Joint loss comparison") + print(f"{'='*70}") + + c_losses = [r["loss"] for r in option_c.values()] + print(f" Option C (per-pool): median={np.median(c_losses):.4f} mean={np.mean(c_losses):.4f}") + + for label, result in joint_results: + print(f" {label:<22} loss={result['loss']:.4f} (from {result['init_loss']:.4f})") + + +# ---- Plots ---- + + +def plot_decomposition_pages(predictions, method, method_label, output_dir): + """Paginated V_arb + V_noise stacked area decomposition.""" + ranked = sorted(predictions.items(), key=lambda x: -x[1]["median_tvl"])[:TOP_N] + + per_page = 10 + n_pages = (len(ranked) + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, len(ranked)) + page_pools = ranked[start:end] + n_this = len(page_pools) + + ncols = 2 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(16, 4.5 * nrows)) + if nrows == 1 and ncols == 1: + axes = np.array([[axes]]) + elif nrows == 1: + axes = axes.reshape(1, -1) + elif ncols == 1: + axes = axes.reshape(-1, 1) + + for idx, (pid, p) in enumerate(page_pools): + ax = axes[idx // ncols][idx % ncols] + dates = p["dates"] + + v_arb_key = f"v_arb_{method}" if method != "c" else "v_arb_c" + v_noise_key = f"v_noise_{method}" if method != "c" else "v_noise_c" + r2_key = f"r2_{method}" if method != "c" else "r2_c" + cad_key = f"cadence_{method}" if method != "c" else "cadence_c" + gas_key = f"gas_{method}" if method != "c" else "gas_c" + + v_arb = p[v_arb_key] + v_noise = p[v_noise_key] + r2_val = p[r2_key] + cad = p[cad_key] + gas = p[gas_key] + + if np.any(np.isnan(v_arb)): + ax.text(0.5, 0.5, f"Dropped from {method_label}", fontsize=12, + ha="center", va="center", transform=ax.transAxes, color="gray") + ax.set_title(f"{pid[:16]} — dropped", fontsize=8) + continue + + v_total = v_arb + v_noise + arb_frac = np.median(v_arb / np.maximum(v_total, 1.0)) + actual = p["actual_vol"] + + ax.fill_between(dates, 0, np.maximum(v_arb, 0), + alpha=0.3, color="orangered", label="V_arb (grid)") + ax.fill_between(dates, np.maximum(v_arb, 0), np.maximum(v_total, 0), + alpha=0.3, color="steelblue", label="V_noise") + ax.plot(dates, actual, "k-", linewidth=0.8, alpha=0.7, label="Actual") + ax.plot(dates, np.maximum(v_total, 0), "--", color="purple", + linewidth=0.8, alpha=0.7, label="Predicted total") + + ax.set_yscale("log") + ax.set_ylabel("Daily volume (USD)", fontsize=8) + + tokens = p["tokens"] + if isinstance(tokens, str): + tok_str = "/".join(t.strip()[:8] for t in tokens.split(",")[:2]) + else: + tok_str = pid[:16] + + ax.set_title( + f"{tok_str} ({p['chain']})\n" + f"TVL ${p['median_tvl']:,.0f} | R\u00b2={r2_val:.3f} " + f"cad={cad:.1f}min gas=${gas:.2f} " + f"arb_frac={arb_frac:.1%} n={p['n_obs']}", + fontsize=8, + ) + ax.legend(fontsize=6, loc="upper right") + ax.tick_params(labelsize=7) + ax.tick_params(axis="x", rotation=30) + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + f"Calibration decomposition: {method_label}\n" + f"page {page + 1}/{n_pages} (top {min(TOP_N, len(ranked))} by TVL)", + fontsize=11, + ) + fig.tight_layout() + safe_method = method.replace(" ", "_") + out = os.path.join(output_dir, f"{safe_method}_page{page + 1}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_summary_distributions(predictions, method_labels, output_dir): + """Histograms of cadence, R², arb fraction for each method.""" + pool_ids = sorted(predictions.keys()) + methods = ["c"] + method_labels + labels = ["Option C"] + method_labels + n_methods = len(methods) + + fig, axes = plt.subplots(n_methods, 3, figsize=(15, 4 * n_methods)) + if n_methods == 1: + axes = axes.reshape(1, -1) + + for row, (method, label) in enumerate(zip(methods, labels)): + cad_key = f"cadence_{method}" if method != "c" else "cadence_c" + r2_key = f"r2_{method}" if method != "c" else "r2_c" + + cads = [predictions[p][cad_key] for p in pool_ids + if np.isfinite(predictions[p][cad_key])] + r2s = [predictions[p][r2_key] for p in pool_ids + if np.isfinite(predictions[p][r2_key])] + arb_fracs = [] + for p in pool_ids: + v_arb_key = f"v_arb_{method}" if method != "c" else "v_arb_c" + v_noise_key = f"v_noise_{method}" if method != "c" else "v_noise_c" + v_arb = predictions[p][v_arb_key] + v_noise = predictions[p][v_noise_key] + if not np.any(np.isnan(v_arb)): + total = v_arb + v_noise + arb_fracs.append(np.median(v_arb / np.maximum(total, 1.0))) + + ax = axes[row, 0] + if cads: + ax.hist(cads, bins=20, color="orangered", alpha=0.7, edgecolor="white") + ax.axvline(np.median(cads), color="black", linestyle="--", + label=f"Median={np.median(cads):.1f}min") + ax.set_xlabel("Cadence (minutes)") + ax.set_title(f"{label}: Cadence") + ax.legend(fontsize=8) + + ax = axes[row, 1] + if r2s: + ax.hist(r2s, bins=20, color="green", alpha=0.7, edgecolor="white") + ax.axvline(np.median(r2s), color="black", linestyle="--", + label=f"Median={np.median(r2s):.3f}") + ax.set_xlabel("R\u00b2") + ax.set_title(f"{label}: R\u00b2") + ax.legend(fontsize=8) + + ax = axes[row, 2] + if arb_fracs: + ax.hist(arb_fracs, bins=20, color="steelblue", alpha=0.7, edgecolor="white") + ax.axvline(np.median(arb_fracs), color="black", linestyle="--", + label=f"Median={np.median(arb_fracs):.2f}") + ax.set_xlabel("Arb fraction") + ax.set_title(f"{label}: Arb fraction") + ax.legend(fontsize=8) + + fig.tight_layout() + out = os.path.join(output_dir, "summary_distributions.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_r2_scatter(predictions, method_labels, output_dir): + """Scatter: Option C R² vs each joint method R².""" + pool_ids = sorted(predictions.keys()) + n = len(method_labels) + + fig, axes = plt.subplots(1, n, figsize=(6 * n, 5)) + if n == 1: + axes = [axes] + + for ax, label in zip(axes, method_labels): + r2_c = [] + r2_m = [] + for p in pool_ids: + rc = predictions[p]["r2_c"] + rm = predictions[p][f"r2_{label}"] + if np.isfinite(rc) and np.isfinite(rm): + r2_c.append(rc) + r2_m.append(rm) + + ax.scatter(r2_c, r2_m, alpha=0.7, s=30, edgecolors="k", linewidth=0.5) + lo = min(min(r2_c), min(r2_m)) if r2_c else 0 + hi = max(max(r2_c), max(r2_m)) if r2_c else 1 + margin = (hi - lo) * 0.05 + 0.01 + ax.plot([lo - margin, hi + margin], [lo - margin, hi + margin], + "k--", alpha=0.3, linewidth=1) + ax.set_xlabel("Option C R\u00b2") + ax.set_ylabel(f"{label} R\u00b2") + ax.set_title(f"Option C vs {label}") + + # Count wins + wins = sum(1 for c, m in zip(r2_c, r2_m) if m > c) + ax.text(0.05, 0.95, f"{label} wins: {wins}/{len(r2_c)}", + transform=ax.transAxes, fontsize=9, va="top") + + fig.tight_layout() + out = os.path.join(output_dir, "r2_scatter.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_cadence_by_chain(predictions, method_labels, output_dir): + """Cadence distributions by chain for Option C and each method.""" + pool_ids = sorted(predictions.keys()) + chains = sorted(set(predictions[p]["chain"] for p in pool_ids)) + colors = plt.cm.tab10(np.linspace(0, 1, max(len(chains), 1))) + chain_color = {c: colors[i] for i, c in enumerate(chains)} + + methods = ["c"] + method_labels + labels = ["Option C"] + method_labels + n = len(methods) + + fig, axes = plt.subplots(1, n, figsize=(6 * n, 5)) + if n == 1: + axes = [axes] + + for ax, method, label in zip(axes, methods, labels): + cad_key = f"cadence_{method}" if method != "c" else "cadence_c" + for chain in chains: + cads = [predictions[p][cad_key] for p in pool_ids + if predictions[p]["chain"] == chain + and np.isfinite(predictions[p][cad_key])] + if cads: + ax.scatter([chain] * len(cads), cads, color=chain_color[chain], + alpha=0.7, s=40, edgecolors="k", linewidth=0.3) + ax.set_ylabel("Cadence (minutes)") + ax.set_title(label) + ax.tick_params(axis="x", rotation=45) + + fig.tight_layout() + out = os.path.join(output_dir, "cadence_by_chain.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +# ---- JSON export ---- + + +def save_results_json(predictions, option_c_results, joint_results, output_dir): + """Save all fitted parameters and diagnostics.""" + out = {"option_c": {}} + for pid, r in option_c_results.items(): + out["option_c"][pid] = { + "log_cadence": r["log_cadence"], + "log_gas": r["log_gas"], + "noise_coeffs": r["noise_coeffs"].tolist(), + "loss": r["loss"], + "converged": bool(r["converged"]), + "cadence_minutes": r["cadence_minutes"], + "gas_usd": r["gas_usd"], + "chain": r.get("chain", ""), + "fee": r.get("fee", 0), + "tokens": r.get("tokens", ""), + } + + for label, result in joint_results: + entry = { + "loss": result["loss"], + "init_loss": result["init_loss"], + "converged": bool(result["converged"]), + "n_pools": result.get("n_pools", 0), + "k_attr": result.get("k_attr", 0), + "pool_ids": result.get("pool_ids", []), + "attr_names": result.get("attr_names", []), + } + # Include any scalar/array results the heads produced + for key in ["bias_cad", "W_cad", "bias_gas", "W_gas", + "bias_noise", "W_noise", "noise_coeffs"]: + if key in result: + val = result[key] + entry[key] = val.tolist() if hasattr(val, "tolist") else val + out[label] = entry + + # Per-pool R² for each method + pool_ids = sorted(predictions.keys()) + method_labels = [label for label, _ in joint_results] + per_pool_r2 = {} + for pid in pool_ids: + p = predictions[pid] + row = {"r2_c": p["r2_c"]} + for label in method_labels: + row[f"r2_{label}"] = p[f"r2_{label}"] + per_pool_r2[pid] = row + out["per_pool_r2"] = per_pool_r2 + + path = os.path.join(output_dir, "mlp_calibration_results.json") + with open(path, "w") as f: + json.dump(out, f, indent=2, default=str) + print(f" Saved: {path}") + + +# ---- Main ---- + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("MLP Calibration: MLPNoiseHead vs Linear Baseline") + print("=" * 70) + + panel, matched = load_and_match() + + # Step 1: Option C baseline + option_c = run_option_c(matched) + + # Step 2: Filter pathological pools + matched_clean, option_c_clean = filter_pathological(matched, option_c) + + # Step 3: Fit each joint model + linear_result, linear_model, jdata = run_linear_joint( + matched_clean, option_c_clean) + mlp_noise_result, mlp_noise_model, _ = run_mlp_noise_joint( + matched_clean, option_c_clean) + mlp_full_result, mlp_full_model, _ = run_mlp_full_joint( + matched_clean, option_c_clean) + + # Step 4: Extract per-pool params from each model + linear_pp = _extract_per_pool_params(linear_model, linear_result, jdata) + mlp_noise_pp = _extract_per_pool_params(mlp_noise_model, mlp_noise_result, jdata) + mlp_full_pp = _extract_per_pool_params(mlp_full_model, mlp_full_result, jdata) + + method_labels = ["linear", "mlp_noise", "mlp_full"] + model_results_for_pred = [ + ("linear", linear_pp, jdata.pool_ids), + ("mlp_noise", mlp_noise_pp, jdata.pool_ids), + ("mlp_full", mlp_full_pp, jdata.pool_ids), + ] + + # Step 5: Per-pool predictions + print("\nComputing per-pool predictions...") + predictions = compute_per_pool_predictions( + matched_clean, option_c_clean, model_results_for_pred) + + # Step 6: Tables + print_pool_table(predictions, method_labels) + print_r2_comparison(predictions, method_labels) + print_loss_comparison(option_c_clean, [ + ("linear", linear_result), + ("mlp_noise", mlp_noise_result), + ("mlp_full", mlp_full_result), + ]) + + # Step 7: Plots + print("\nGenerating plots...") + os.makedirs(OUTPUT_DIR, exist_ok=True) + + plot_decomposition_pages(predictions, "c", "Option C (per-pool)", OUTPUT_DIR) + plot_decomposition_pages(predictions, "linear", "Linear shared noise", OUTPUT_DIR) + plot_decomposition_pages(predictions, "mlp_noise", "MLP noise (linear cad)", OUTPUT_DIR) + plot_decomposition_pages(predictions, "mlp_full", "Full MLP (MLP cad + MLP noise)", OUTPUT_DIR) + + plot_summary_distributions(predictions, method_labels, OUTPUT_DIR) + plot_r2_scatter(predictions, method_labels, OUTPUT_DIR) + plot_cadence_by_chain(predictions, method_labels, OUTPUT_DIR) + + # Step 8: JSON export + save_results_json(predictions, option_c_clean, [ + ("linear", linear_result), + ("mlp_noise", mlp_noise_result), + ("mlp_full", mlp_full_result), + ], OUTPUT_DIR) + + print(f"\n{'='*70}") + print(f"Done. Output in: {OUTPUT_DIR}") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_mlp_sweep.py b/scripts/run_mlp_sweep.py new file mode 100644 index 00000000..aa281152 --- /dev/null +++ b/scripts/run_mlp_sweep.py @@ -0,0 +1,429 @@ +"""Hyperparameter sweep for MLP calibration models. + +Sweeps maxiter, alpha, hidden size, maxcor, and loss_type to find settings +where the MLP converges and achieves the best per-pool R2. + +Usage: + python scripts/run_mlp_sweep.py [--phase 1|2|3|all] +""" + +import argparse +import json +import os +import time + +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +GRID_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "pool_grids_v2", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "mlp_sweep", +) +OPTION_C_LOSS_CUTOFF = 5.0 + + +def load_data(): + """Load panel, match grids, run Option C, return clean data.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + from quantammsim.calibration.pool_data import ( + build_pool_attributes, + build_x_obs, + match_grids_to_panel, + replace_panel_volatility_with_binance, + ) + + panel = pd.read_parquet(PANEL_CACHE) + print("Replacing volatility with Binance minute data...") + panel = replace_panel_volatility_with_binance(panel) + matched = match_grids_to_panel(GRID_DIR, panel) + print(f"Matched: {len(matched)} pools with grids") + + # Option C baseline + print(f"\n--- Option C: per-pool fits ({len(matched)} pools, gas fixed) ---") + option_c = fit_all_pools(matched, fix_gas_to_chain=True) + n_conv = sum(1 for r in option_c.values() if r["converged"]) + losses = [r["loss"] for r in option_c.values()] + print(f" Converged: {n_conv}/{len(option_c)}") + print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") + + # Drop pathological pools + dropped = [p for p, r in option_c.items() if r["loss"] > OPTION_C_LOSS_CUTOFF] + matched_clean = {k: v for k, v in matched.items() if k not in dropped} + option_c_clean = {k: v for k, v in option_c.items() if k not in dropped} + if dropped: + print(f" Dropping {len(dropped)} pools (loss > {OPTION_C_LOSS_CUTOFF})") + + return matched_clean, option_c_clean + + +def compute_per_pool_r2(model, result, jdata, matched): + """Compute per-pool R2 for a fitted model.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import build_x_obs + import jax.numpy as jnp + + params = jnp.array(result["params_flat"]) + n_pools = result["n_pools"] + k_attr = result["k_attr"] + (cs, ce), (gs, ge), (ns, ne) = model._head_slices(n_pools, k_attr) + + cad_slice = params[cs:ce] + gas_slice = params[gs:ge] + noise_slice = params[ns:ne] + + r2s = [] + for i, pid in enumerate(jdata.pool_ids): + x_attr_i = jdata.x_attr[i] + log_cad = float(model.cadence_head.predict(cad_slice, i, x_attr_i)) + log_gas = float(model.gas_head.predict(gas_slice, i, x_attr_i)) + noise_c = np.array(model.noise_head.predict(noise_slice, i, x_attr_i)) + + entry = matched[pid] + panel = entry["panel"] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + + v_arb_all = np.array(interpolate_pool_daily( + coeffs, jnp.float64(log_cad), jnp.float64(np.exp(log_gas)))) + v_arb = v_arb_all[day_indices] + v_noise = np.exp(x_obs @ noise_c) + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y_obs) ** 2) + ss_tot = np.sum((y_obs - y_obs.mean()) ** 2) + r2s.append(1 - ss_res / max(ss_tot, 1e-10)) + + return np.array(r2s) + + +def run_single(matched_clean, option_c_clean, config): + """Run a single sweep configuration. Returns result dict.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, LinearHead, MLPHead, MLPNoiseHead, SharedLinearNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_joint_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, fix_gas_to_chain=True) + + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_clean[pid]["chain"] + gas_usd = CHAIN_GAS_USD.get(chain, 1.0) + gas_values.append(np.log(max(gas_usd, 1e-6))) + gas_values = np.array(gas_values) + + # Build model from config + alpha_cad = config.get("alpha_cad", 0.01) + alpha_noise = config.get("alpha_noise", 0.01) + hidden = config.get("hidden", 16) + maxiter = config.get("maxiter", 500) + maxcor = config.get("maxcor", 10) + loss_type = config.get("loss_type", "l2") + cad_type = config.get("cad_type", "linear") # "linear" or "mlp" + noise_type = config.get("noise_type", "mlp") # "mlp" or "linear" + + # Cadence head + if cad_type == "mlp": + cad_head = MLPHead("cad", hidden=hidden, alpha=alpha_cad) + else: + cad_head = LinearHead("cad", alpha=alpha_cad) + + # Noise head + if noise_type == "mlp": + noise_head = MLPNoiseHead(hidden=hidden, alpha=alpha_noise) + else: + noise_head = SharedLinearNoiseHead(alpha=alpha_noise) + + model = CalibrationModel( + cadence_head=cad_head, + gas_head=FixedHead("gas", gas_values), + noise_head=noise_head, + loss_type=loss_type, + ) + + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + n_params = model.n_params(n_pools, k_attr) + + # Override maxcor in the fit method by monkey-patching options + import scipy.optimize + _orig_minimize = scipy.optimize.minimize + + def patched_minimize(fun, x0, **kwargs): + opts = kwargs.get("options", {}) + opts["maxcor"] = maxcor + opts["maxiter"] = maxiter + kwargs["options"] = opts + return _orig_minimize(fun, x0, **kwargs) + + scipy.optimize.minimize = patched_minimize + try: + t0 = time.time() + result = model.fit(jdata, maxiter=maxiter, warm_start=option_c_clean) + wall_time = time.time() - t0 + finally: + scipy.optimize.minimize = _orig_minimize + + # Compute per-pool R2 + r2s = compute_per_pool_r2(model, result, jdata, matched_clean) + + return { + "config": config, + "n_params": n_params, + "n_pools": n_pools, + "init_loss": result["init_loss"], + "final_loss": result["loss"], + "converged": result["converged"], + "wall_time_s": round(wall_time, 1), + "r2_median": round(float(np.median(r2s)), 4), + "r2_mean": round(float(np.mean(r2s)), 4), + "r2_p10": round(float(np.percentile(r2s, 10)), 4), + "r2_p25": round(float(np.percentile(r2s, 25)), 4), + "r2_p75": round(float(np.percentile(r2s, 75)), 4), + "r2_p90": round(float(np.percentile(r2s, 90)), 4), + "r2_min": round(float(np.min(r2s)), 4), + "r2_max": round(float(np.max(r2s)), 4), + "n_positive_r2": int(np.sum(r2s > 0)), + } + + +def print_result(res, idx=None): + """Print a single result row.""" + c = res["config"] + prefix = f"[{idx}] " if idx is not None else "" + label = (f"{c.get('cad_type','linear')}_cad + " + f"{c.get('noise_type','mlp')}_noise") + print(f"{prefix}{label} " + f"h={c.get('hidden',16):2d} " + f"a_c={c.get('alpha_cad',0.01):.4f} " + f"a_n={c.get('alpha_noise',0.01):.4f} " + f"maxiter={c.get('maxiter',500):5d} " + f"maxcor={c.get('maxcor',10):2d} " + f"loss={c.get('loss_type','l2'):5s} | " + f"L={res['final_loss']:7.4f} " + f"conv={str(res['converged']):5s} " + f"R2_med={res['r2_median']:+.4f} " + f"R2_mean={res['r2_mean']:+.4f} " + f"R2+={res['n_positive_r2']:2d}/{res['n_pools']} " + f"{res['wall_time_s']:5.1f}s") + + +def run_phase_1(matched_clean, option_c_clean): + """Phase 1: Sweep maxiter to diagnose convergence.""" + print("\n" + "=" * 80) + print("Phase 1: maxiter sweep (MLP noise, linear cadence)") + print("=" * 80) + + configs = [] + for maxiter in [500, 2000, 5000]: + configs.append({ + "cad_type": "linear", "noise_type": "mlp", + "maxiter": maxiter, "hidden": 16, + "alpha_cad": 0.01, "alpha_noise": 0.01, + "maxcor": 10, "loss_type": "l2", + "label": f"maxiter={maxiter}", + }) + # Also sweep maxiter for full MLP + for maxiter in [500, 2000, 5000]: + configs.append({ + "cad_type": "mlp", "noise_type": "mlp", + "maxiter": maxiter, "hidden": 16, + "alpha_cad": 0.01, "alpha_noise": 0.01, + "maxcor": 10, "loss_type": "l2", + "label": f"full_mlp_maxiter={maxiter}", + }) + + results = [] + for i, cfg in enumerate(configs): + print(f"\n Running {cfg['label']}...") + res = run_single(matched_clean, option_c_clean, cfg) + print_result(res, i) + results.append(res) + return results + + +def run_phase_2(matched_clean, option_c_clean, best_maxiter=5000): + """Phase 2: Regularization grid (alpha_noise x alpha_cad).""" + print("\n" + "=" * 80) + print("Phase 2: regularization sweep (MLP noise, linear cadence)") + print("=" * 80) + + configs = [] + for alpha_noise in [0.0001, 0.001, 0.01, 0.1]: + for alpha_cad in [0.001, 0.01, 0.1]: + configs.append({ + "cad_type": "linear", "noise_type": "mlp", + "maxiter": best_maxiter, "hidden": 16, + "alpha_cad": alpha_cad, "alpha_noise": alpha_noise, + "maxcor": 10, "loss_type": "l2", + "label": f"a_n={alpha_noise}, a_c={alpha_cad}", + }) + + results = [] + for i, cfg in enumerate(configs): + print(f"\n Running {cfg['label']}...") + res = run_single(matched_clean, option_c_clean, cfg) + print_result(res, i) + results.append(res) + return results + + +def run_phase_3(matched_clean, option_c_clean, + best_maxiter=5000, best_alpha_cad=0.01, + best_alpha_noise=0.01): + """Phase 3: Architecture sweep (hidden, maxcor, loss_type).""" + print("\n" + "=" * 80) + print("Phase 3: architecture sweep") + print("=" * 80) + + configs = [] + # Hidden size + for hidden in [8, 16, 32]: + configs.append({ + "cad_type": "linear", "noise_type": "mlp", + "maxiter": best_maxiter, "hidden": hidden, + "alpha_cad": best_alpha_cad, "alpha_noise": best_alpha_noise, + "maxcor": 10, "loss_type": "l2", + "label": f"hidden={hidden}", + }) + # maxcor + for maxcor in [10, 30, 50]: + configs.append({ + "cad_type": "linear", "noise_type": "mlp", + "maxiter": best_maxiter, "hidden": 16, + "alpha_cad": best_alpha_cad, "alpha_noise": best_alpha_noise, + "maxcor": maxcor, "loss_type": "l2", + "label": f"maxcor={maxcor}", + }) + # Loss type + for loss_type in ["l2", "huber"]: + configs.append({ + "cad_type": "linear", "noise_type": "mlp", + "maxiter": best_maxiter, "hidden": 16, + "alpha_cad": best_alpha_cad, "alpha_noise": best_alpha_noise, + "maxcor": 10, "loss_type": loss_type, + "label": f"loss={loss_type}", + }) + # Full MLP with best settings + configs.append({ + "cad_type": "mlp", "noise_type": "mlp", + "maxiter": best_maxiter, "hidden": 16, + "alpha_cad": best_alpha_cad, "alpha_noise": best_alpha_noise, + "maxcor": 10, "loss_type": "l2", + "label": "full_mlp_best", + }) + + results = [] + for i, cfg in enumerate(configs): + print(f"\n Running {cfg['label']}...") + res = run_single(matched_clean, option_c_clean, cfg) + print_result(res, i) + results.append(res) + return results + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--phase", default="all", + choices=["1", "2", "3", "all"]) + args = parser.parse_args() + + os.makedirs(OUTPUT_DIR, exist_ok=True) + matched_clean, option_c_clean = load_data() + + # Compute Option C R2 for reference + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import build_x_obs + import jax.numpy as jnp + + r2s_c = [] + for pid in sorted(matched_clean.keys()): + entry = matched_clean[pid] + rc = option_c_clean[pid] + panel = entry["panel"] + x_obs = build_x_obs(panel) + y_obs = panel["log_volume"].values.astype(float) + day_indices = entry["day_indices"] + coeffs = entry["coeffs"] + + v_arb_all = np.array(interpolate_pool_daily( + coeffs, jnp.float64(rc["log_cadence"]), + jnp.float64(np.exp(rc["log_gas"])))) + v_arb = v_arb_all[day_indices] + v_noise = np.exp(x_obs @ rc["noise_coeffs"]) + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y_obs) ** 2) + ss_tot = np.sum((y_obs - y_obs.mean()) ** 2) + r2s_c.append(1 - ss_res / max(ss_tot, 1e-10)) + + r2s_c = np.array(r2s_c) + print(f"\nOption C reference: R2 median={np.median(r2s_c):.4f}, " + f"mean={np.mean(r2s_c):.4f}, " + f"R2>0: {np.sum(r2s_c > 0)}/{len(r2s_c)}") + + all_results = {"option_c_r2_median": float(np.median(r2s_c)), + "option_c_r2_mean": float(np.mean(r2s_c)), + "phases": {}} + + if args.phase in ("1", "all"): + r1 = run_phase_1(matched_clean, option_c_clean) + all_results["phases"]["1"] = r1 + + # Pick best maxiter from phase 1 + best_maxiter = max(r1, key=lambda r: r["r2_median"])["config"]["maxiter"] + print(f"\n Best maxiter from phase 1: {best_maxiter}") + else: + best_maxiter = 5000 + + if args.phase in ("2", "all"): + r2 = run_phase_2(matched_clean, option_c_clean, best_maxiter) + all_results["phases"]["2"] = r2 + + best_r = max(r2, key=lambda r: r["r2_median"]) + best_alpha_cad = best_r["config"]["alpha_cad"] + best_alpha_noise = best_r["config"]["alpha_noise"] + print(f"\n Best from phase 2: alpha_cad={best_alpha_cad}, " + f"alpha_noise={best_alpha_noise}, R2_med={best_r['r2_median']}") + else: + best_alpha_cad = 0.01 + best_alpha_noise = 0.01 + + if args.phase in ("3", "all"): + r3 = run_phase_3(matched_clean, option_c_clean, + best_maxiter, best_alpha_cad, best_alpha_noise) + all_results["phases"]["3"] = r3 + + # Save results + out_path = os.path.join(OUTPUT_DIR, "sweep_results.json") + with open(out_path, "w") as f: + json.dump(all_results, f, indent=2, default=str) + print(f"\nResults saved to {out_path}") + + # Print summary table + print("\n" + "=" * 80) + print("SWEEP SUMMARY") + print("=" * 80) + print(f"Option C reference: R2 median={np.median(r2s_c):.4f}") + print() + for phase, results in all_results["phases"].items(): + print(f"Phase {phase}:") + for i, res in enumerate(results): + print_result(res, i) + print() + + +if __name__ == "__main__": + main() diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index 5eae07c3..11c2aa03 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -491,15 +491,17 @@ def test_init_default_gas(self): # b2 should be log(1) = 0 np.testing.assert_allclose(init[-1], 0.0) - def test_init_W2_is_zero(self): - """W2 should be zero at init so output = b2.""" + def test_init_W2_is_small(self): + """W2 should be small random at init (not zero) for L-BFGS conditioning.""" h = MLPHead("cad", hidden=4) jdata = _make_fake_jdata() init = h.init(jdata) # W2 is at [k_attr*hidden + hidden : k_attr*hidden + 2*hidden] w2_start = K_ATTR * 4 + 4 w2_end = w2_start + 4 - np.testing.assert_allclose(init[w2_start:w2_end], 0.0) + w2 = init[w2_start:w2_end] + assert np.all(np.abs(w2) < 0.1), "W2 should be small" + assert np.any(w2 != 0.0), "W2 should not be exactly zero" def test_init_warm_start(self): h = MLPHead("log_cadence", hidden=4) @@ -657,15 +659,17 @@ def test_init_default(self): assert init.shape == (n_p,) assert np.all(np.isfinite(init)) - def test_init_W2_is_zero(self): - """W2 should be zero at init so output = b2.""" + def test_init_W2_is_small(self): + """W2 should be small random at init (not zero) for L-BFGS conditioning.""" h = MLPNoiseHead(hidden=4) jdata = _make_fake_jdata() init = h.init(jdata) # W2 starts at k_attr*hidden + hidden w2_start = K_ATTR * 4 + 4 w2_end = w2_start + 4 * K_OBS - np.testing.assert_allclose(init[w2_start:w2_end], 0.0) + w2 = init[w2_start:w2_end] + assert np.all(np.abs(w2) < 0.1), "W2 should be small" + assert np.any(w2 != 0.0), "W2 should not be exactly zero" def test_init_b2_from_ols(self): """b2 should be pooled OLS noise coefficients.""" From 43024c36e236cac5315b8dea06258a418c347963 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 13:31:55 +0000 Subject: [PATCH 042/115] fix: add missing ste_temperature to test fingerprints after STE merge --- tests/pools/reCLAMM/test_reclamm_noise_volume.py | 5 +++++ tests/pools/reCLAMM/test_reclamm_reserves.py | 2 ++ 2 files changed, 7 insertions(+) diff --git a/tests/pools/reCLAMM/test_reclamm_noise_volume.py b/tests/pools/reCLAMM/test_reclamm_noise_volume.py index f413a9e3..500ae070 100644 --- a/tests/pools/reCLAMM/test_reclamm_noise_volume.py +++ b/tests/pools/reCLAMM/test_reclamm_noise_volume.py @@ -547,6 +547,7 @@ def test_tsoukalas_sqrt_from_fingerprint(self): "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "noise_model": "tsoukalas_sqrt", "reclamm_noise_params": DEFAULT_NOISE_PARAMS, + "ste_temperature": 10.0, }) # Fingerprint without noise @@ -563,6 +564,7 @@ def test_tsoukalas_sqrt_from_fingerprint(self): "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "noise_model": "arb_only", + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) @@ -618,6 +620,7 @@ def test_volatility_auto_computed_affects_fee_revenue(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, } fp_tsoukalas = Hashabledict({ @@ -946,6 +949,7 @@ def test_loglinear_from_fingerprint(self): "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "noise_model": "loglinear", "reclamm_noise_params": loglinear_params, + "ste_temperature": 10.0, }) fp_arb_only = Hashabledict({ @@ -961,6 +965,7 @@ def test_loglinear_from_fingerprint(self): "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), "noise_model": "arb_only", + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index 43263cb9..58d9daeb 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -929,6 +929,7 @@ def test_noise_trader_ratio_through_pool_class(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, } fp_no_noise = Hashabledict({**base_fp, "noise_trader_ratio": 0.0}) @@ -1291,6 +1292,7 @@ def test_lp_supply_through_pool_class(self): "tokens": ("ETH", "USDC"), "numeraire": "USDC", "all_sig_variations": tuple(map(tuple, [[1, -1], [-1, 1]])), + "ste_temperature": 10.0, }) start_index = jnp.array([0, 0]) From dbd62b2899fac71600ec7596e56c5a06f8cfd49a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 10 Mar 2026 15:50:45 +0000 Subject: [PATCH 043/115] WIP: MLP calibration with lstsq warm-start and tuned hyperparameters MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - MLPHead/MLPNoiseHead init uses lstsq warm-start for W2 instead of zeros (fixes zero-iteration L-BFGS bug) - Best hyperparameters from sweep: alpha_cad=0.001, alpha_noise=0.1, maxiter=5000 - MLP noise R²=0.575 (vs Option C 0.612) but cadence is degenerate — noise head absorbs arb volume, decomposition not identified - Add sweep script and analysis doc --- docs/joint_calibration_analysis.md | 230 +++++++++++++++++++++++++++++ quantammsim/calibration/heads.py | 23 ++- scripts/run_mlp_calibration.py | 17 ++- tests/calibration/test_heads.py | 24 +-- 4 files changed, 264 insertions(+), 30 deletions(-) create mode 100644 docs/joint_calibration_analysis.md diff --git a/docs/joint_calibration_analysis.md b/docs/joint_calibration_analysis.md new file mode 100644 index 00000000..dee27609 --- /dev/null +++ b/docs/joint_calibration_analysis.md @@ -0,0 +1,230 @@ +# Joint Calibration Analysis: MLP Capacity vs Identification + +## Training results (2026-03-10) + +### R2 progression across model architectures + +| Attempt | Architecture | Noise params | Total params | Median R2 | Joint loss | +|---|---|---|---|---|---| +| Structural MoE (numpyro) | 3 archetypes x 8 coeffs | 24 | ~40 | -0.70 | - | +| Linear joint | SharedLinearNoiseHead | 63 | 63 | -0.15 | 9.62 | +| MLP noise joint | MLPNoiseHead(hidden=16) | 255 | 255 | 0.01 | 9.39 | +| Full MLP | MLPHead(cad,16) + MLPNoiseHead(16) | 255 | 377 | -0.02 | 8.59 | +| Option C (per-pool) | PerPoolNoiseHead | 37x8=296 | 37x9=333 | 0.61 | 1.25 (median) | + +The direction is clear: more capacity on the noise side helps substantially +(-0.70 -> -0.15 -> 0.01). Adding cadence capacity (MLP noise -> full MLP) +improved joint loss (9.39 -> 8.59) but not per-pool R2 (-0.02), and didn't +converge within 500 iterations. + +### Convergence concern + +The full MLP explicitly failed to converge (scipy `success=False`). The MLP +noise model converged but only reduced loss from 12.70 to 9.39 — a 26% +reduction vs the linear baseline's 99.5% reduction (2011.77 -> 9.62). This +suggests the MLPs are undertraining. + +Current optimizer settings: +- L-BFGS-B with maxiter=500 +- ftol=1e-10, gtol=1e-8 +- maxcor=10 (L-BFGS memory, scipy default) +- alpha=0.01 for all heads (L2 regularization on weights) +- hidden=16 for all MLPs +- He init for W1, W2=0, b2=pooled OLS / mean of Option C + +### Why the MLPs may not be converging + +1. **maxiter=500 is low for 255-377 params.** L-BFGS-B typically needs + O(1000-5000) iterations for MLP-scale problems. The linear model with + 63 params converges easily in 500; the MLP with 377 params does not. + +2. **maxcor=10 may be too small.** The default L-BFGS memory of 10 past + gradients may not provide a good enough Hessian approximation for 377 + parameters. Increasing to 20-50 can help. + +3. **Regularization alpha=0.01 may be wrong.** With 37 pools and 255 noise + params, the model is overparameterized (255/37 ≈ 7 params per pool). + alpha=0.01 might be too weak (overfitting some pools, underfitting + others) or too strong (preventing the MLP from expressing the necessary + nonlinearity). This is the most important hyperparameter to sweep. + +4. **W2=0 initialization creates a flat starting surface.** Since the MLP + starts as a constant function (output = b2 everywhere), L-BFGS-B must + first learn to differentiate between pools. The initial gradients + through W1 are informative (He init + backprop through ReLU), but the + first few iterations may be slow compared to the linear model which + starts from an OLS warm-start. + +5. **Dead ReLU units.** With He init and k_attr=6 features, some hidden + units may have all-negative pre-activations across the 37 pool + attribute vectors, making them permanently dead with zero gradient. + +6. **Per-pool loss weighting.** All observations contribute equally. + USDC/WETH (1757 obs) dominates RDNT/WETH (89 obs) by 20x. The + optimizer may be fitting a few high-obs pools at the expense of many + low-obs ones. + +## Diagnosis: identification vs convergence + +Two distinct problems: + +1. **Convergence problem** (addressable via hyperparameters): + The MLP isn't reaching its minimum. Fix: more iterations, better + hyperparameters, multiple restarts. + +2. **Identification problem** (addressable via architecture): + Even at the minimum, the shared mapping can't match per-pool R2. + 37 pools is tiny for a nonlinear model. Cadence is idiosyncratic. + Fix: DeltaHead (per-pool residuals with shrinkage), better features. + +These are **independent** problems that compound. We should fix convergence +first (hyperparameter sweep) to understand the true capacity of the current +architecture before adding structural complexity. + +## Hyperparameter sweep design + +### Parameters to sweep + +| Parameter | Current | Sweep values | Rationale | +|---|---|---|---| +| maxiter | 500 | 500, 2000, 5000 | Primary convergence bottleneck | +| alpha (noise) | 0.01 | 0.0001, 0.001, 0.01, 0.1 | Controls overfitting vs underfitting | +| alpha (cadence) | 0.01 | 0.001, 0.01, 0.1 | Separate from noise reg | +| hidden | 16 | 8, 16, 32 | Capacity vs overfitting | +| maxcor | 10 | 10, 30 | L-BFGS Hessian quality | +| loss_type | l2 | l2, huber | Outlier robustness | + +### Sweep strategy + +Full grid is 3 x 4 x 3 x 3 x 2 x 2 = 432 runs. Too many. + +**Phase 1: Fix convergence (1D sweeps)** +- Sweep maxiter = [500, 2000, 5000] with defaults. Cheapest diagnostic. +- If 5000 converges, use that going forward. + +**Phase 2: Regularization (most important)** +- alpha_noise x alpha_cad grid: 4 x 3 = 12 runs at converged maxiter. +- Evaluate both joint loss AND per-pool median R2. + +**Phase 3: Architecture** +- hidden = [8, 16, 32] at best alpha settings: 3 runs. +- loss_type = [l2, huber] at best settings: 2 runs. +- maxcor = [10, 30] at best settings: 2 runs. + +Total: ~22 runs, each ~2-5 min = ~1-2 hours. + +### Metrics to track per run + +- Joint loss (final) +- Joint loss (init) — sanity check +- Converged (bool) +- Number of L-BFGS iterations used +- Per-pool median R2 +- Per-pool mean R2 +- Per-pool R2 distribution (10th, 25th, 50th, 75th, 90th percentiles) +- Wall time + +### What success looks like + +- Converged = True for the full MLP +- Joint loss < 8.0 (below current 8.59) +- Per-pool median R2 > 0.3 (closing the gap toward Option C's 0.61) +- The R2 improvement should be spread across pools, not concentrated + +## Features / data that would help + +### Missing pool attributes (from docs) + +Current features (k_attr=6 after chain dummy removal): +log_fee, mean_log_tvl, log_mcap_product, has_stable, same_asset_type, +weight_imbalance. + +These describe what the pool IS but not the market around it. Cadence is +driven by arbitrage frequency, which depends on: + +| Missing feature | Why it matters | Source | Effort | +|---|---|---|---| +| Block time | Directly limits minimum cadence. Arb=0.25s vs Main=12s | Static per chain | Trivial | +| Mean pair volatility | Pool-level (not obs-level) vol predicts arb intensity | Binance minute data (loaded) | Small | +| CEX daily volume | More CEX vol = more arb opportunities | Binance API | Medium | +| Competing DEX pools | More pools for same pair = faster arb | Balancer subgraph | Medium | +| Pool routing share | Dominant pool gets arbitraged first | DEX aggregator data | Hard | +| Mean daily swap count | Direct proxy for pool activity | Panel data | Small | + +The pair-intrinsic formula bias (1.26-2.22x) documented in +noise_calibration_review.md is the largest unexplained variance source. +It varies with pair liquidity characteristics in ways that the current +token classification doesn't capture. CEX volume/depth would help. + +### Observation-level features (x_obs, K_OBS=8) + +Current: [1, log_tvl_lag1, log_sigma, tvl*sigma, tvl*fee, sigma*fee, +dow_sin, dow_cos] + +Missing: +- Rolling CEX volume (daily) — high volume days have more noise/organic flow +- Gas price that day (mainnet) — affects whether arbs execute +- Market regime (rolling momentum) — trending vs mean-reverting +- Number of swaps that day — direct activity measure + +### Time-varying dynamics + +Panel spans 2021-2026. MEV dynamics changed dramatically: +- Flashbots launched mid-2021 +- L2s matured 2023-2024 +- EIP-4844 (March 2024) dropped L2 gas costs +The current model assumes constant cadence per pool over this period. + +## Structural improvements (post-sweep) + +### DeltaHead (per-pool residuals with shrinkage) + +Most important structural change. For cadence: +``` +log_cadence_i = f(x_attr_i) + delta_i +regularization: alpha_shared * ||W||^2 + alpha_delta * sum(delta_i^2) +``` + +At alpha_delta=0: pure per-pool (Option C) +At alpha_delta=inf: pure shared (current joint) +Cross-validate alpha_delta. + +For new pools: predict f(x_attr_new) with delta=0. + +This is essentially a mixed-effects model fitted end-to-end through the +grid interpolation loss. + +### Per-pool loss weighting + +Weight each pool's contribution by 1/sqrt(n_obs_i) to equalize pool-level +influence. Currently USDC/WETH (1757 obs) has 20x the influence of any +Sonic pool (89 obs). + +### Hybrid: per-pool cadence + shared noise + +Cadence is idiosyncratic (LOO R2 = 0.24 at best). Noise structure is +more regular (hierarchical model R2 = 0.71 on total volume). Natural split: +- Cadence: per-pool (Option C) +- Noise: shared MLP (generalizable) +- Gas: fixed to chain values + +### Sensitivity analysis (the decision point) + +Before investing more in mapping improvement: does reCLAMM optimal +concentration change materially when cadence varies +/-50%? This is +recommendation #1 in calibration_results.md, noise_calibration_review.md, +and joint_calibration_design.md. Still not done. + +If the optimum is robust, the current pipeline (Option C + Ridge LOO) is +already sufficient and further mapping improvement is nice-to-have. + +## Priority order + +1. **Hyperparameter sweep** — fix convergence before changing architecture +2. **DeltaHead** — if R2 gap persists post-sweep, this is the minimal + structural change +3. **Per-pool loss weighting** — simple fix, helps all joint models +4. **Add block_time and mean_pair_volatility** — high-signal, low-effort + features +5. **Sensitivity analysis** — the real decision point for whether any of + this matters for the downstream task diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 1191d0a0..f0e661b1 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -462,12 +462,10 @@ def init(self, jdata, warm_start=None): W1 = rng.randn(k_attr, h).astype(np.float64) * std b1 = np.zeros(h, dtype=np.float64) - # Small random W2 so L-BFGS gets a non-degenerate initial Hessian - W2 = rng.randn(h).astype(np.float64) * 0.01 b2 = np.array([self._default_bias()], dtype=np.float64) + W2 = np.zeros(h, dtype=np.float64) if warm_start is not None: - # Fit linear mapping from per-pool values, use as last-layer init vals = [] for pid in jdata.pool_ids: if pid in warm_start and self.name in warm_start[pid]: @@ -475,9 +473,15 @@ def init(self, jdata, warm_start=None): else: vals.append(self._default_bias()) y = np.array(vals) - # Use mean as b2 b2 = np.array([np.mean(y)], dtype=np.float64) + # Warm-start W2 by least-squares through hidden activations + # so the MLP init approximates the per-pool warm-start values + x_attr = np.array(jdata.x_attr) + H = np.maximum(x_attr @ W1 + b1, 0.0) # (n_pools, h) + residuals = y - float(b2) # what W2 needs to produce + W2, _, _, _ = np.linalg.lstsq(H, residuals, rcond=None) + return np.concatenate([W1.ravel(), b1, W2, b2]) def _default_bias(self): @@ -586,16 +590,21 @@ def init(self, jdata, warm_start=None): W1 = rng.randn(k_attr, h).astype(np.float64) * std b1 = np.zeros(h, dtype=np.float64) - # Small random W2 so L-BFGS gets a non-degenerate initial Hessian - W2 = rng.randn(h, K_OBS).astype(np.float64) * 0.01 + W2 = np.zeros((h, K_OBS), dtype=np.float64) if warm_start is not None: - # Use mean of per-pool noise as b2 noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) for i, pid in enumerate(jdata.pool_ids): if pid in warm_start and "noise_coeffs" in warm_start[pid]: noise_all[i] = warm_start[pid]["noise_coeffs"] b2 = np.mean(noise_all, axis=0) + + # Warm-start W2 by least-squares through hidden activations + # so the MLP init approximates the per-pool warm-start noise + x_attr = np.array(jdata.x_attr) + H = np.maximum(x_attr @ W1 + b1, 0.0) # (n_pools, h) + residuals = noise_all - b2 # (n_pools, K_OBS) + W2, _, _, _ = np.linalg.lstsq(H, residuals, rcond=None) else: # Pooled OLS noise as b2 all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) diff --git a/scripts/run_mlp_calibration.py b/scripts/run_mlp_calibration.py index 39189c49..74d455ae 100644 --- a/scripts/run_mlp_calibration.py +++ b/scripts/run_mlp_calibration.py @@ -35,9 +35,12 @@ ) OPTION_C_LOSS_CUTOFF = 5.0 OPTION_C_MAXITER = 500 -JOINT_MAXITER = 500 +JOINT_MAXITER = 5000 MLP_HIDDEN = 16 TOP_N = 50 +# Best alpha settings from sweep (phase 2) +ALPHA_CAD = 0.001 +ALPHA_NOISE = 0.1 # ---- Data loading ---- @@ -123,9 +126,9 @@ def run_linear_joint(matched_clean, option_c_clean): gas_values = _build_gas_values(jdata, matched_clean) model = CalibrationModel( - cadence_head=LinearHead("cad", alpha=0.01), + cadence_head=LinearHead("cad", alpha=ALPHA_CAD), gas_head=FixedHead("gas", gas_values), - noise_head=SharedLinearNoiseHead(alpha=0.01), + noise_head=SharedLinearNoiseHead(alpha=ALPHA_NOISE), ) n_pools = len(jdata.pool_data) @@ -150,9 +153,9 @@ def run_mlp_noise_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): gas_values = _build_gas_values(jdata, matched_clean) model = CalibrationModel( - cadence_head=LinearHead("cad", alpha=0.01), + cadence_head=LinearHead("cad", alpha=ALPHA_CAD), gas_head=FixedHead("gas", gas_values), - noise_head=MLPNoiseHead(hidden=hidden, alpha=0.01), + noise_head=MLPNoiseHead(hidden=hidden, alpha=ALPHA_NOISE), ) n_pools = len(jdata.pool_data) @@ -177,9 +180,9 @@ def run_mlp_full_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): gas_values = _build_gas_values(jdata, matched_clean) model = CalibrationModel( - cadence_head=MLPHead("cad", hidden=hidden, alpha=0.01), + cadence_head=MLPHead("cad", hidden=hidden, alpha=ALPHA_CAD), gas_head=FixedHead("gas", gas_values), - noise_head=MLPNoiseHead(hidden=hidden, alpha=0.01), + noise_head=MLPNoiseHead(hidden=hidden, alpha=ALPHA_NOISE), ) n_pools = len(jdata.pool_data) diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index 11c2aa03..b0f53000 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -491,17 +491,13 @@ def test_init_default_gas(self): # b2 should be log(1) = 0 np.testing.assert_allclose(init[-1], 0.0) - def test_init_W2_is_small(self): - """W2 should be small random at init (not zero) for L-BFGS conditioning.""" + def test_init_size(self): + """Init should return correct number of parameters.""" h = MLPHead("cad", hidden=4) jdata = _make_fake_jdata() init = h.init(jdata) - # W2 is at [k_attr*hidden + hidden : k_attr*hidden + 2*hidden] - w2_start = K_ATTR * 4 + 4 - w2_end = w2_start + 4 - w2 = init[w2_start:w2_end] - assert np.all(np.abs(w2) < 0.1), "W2 should be small" - assert np.any(w2 != 0.0), "W2 should not be exactly zero" + assert init.shape == (h.n_params(N_POOLS, K_ATTR),) + assert np.all(np.isfinite(init)) def test_init_warm_start(self): h = MLPHead("log_cadence", hidden=4) @@ -659,17 +655,13 @@ def test_init_default(self): assert init.shape == (n_p,) assert np.all(np.isfinite(init)) - def test_init_W2_is_small(self): - """W2 should be small random at init (not zero) for L-BFGS conditioning.""" + def test_init_size(self): + """Init should return correct number of parameters.""" h = MLPNoiseHead(hidden=4) jdata = _make_fake_jdata() init = h.init(jdata) - # W2 starts at k_attr*hidden + hidden - w2_start = K_ATTR * 4 + 4 - w2_end = w2_start + 4 * K_OBS - w2 = init[w2_start:w2_end] - assert np.all(np.abs(w2) < 0.1), "W2 should be small" - assert np.any(w2 != 0.0), "W2 should not be exactly zero" + assert init.shape == (h.n_params(N_POOLS, K_ATTR),) + assert np.all(np.isfinite(init)) def test_init_b2_from_ols(self): """b2 should be pooled OLS noise coefficients.""" From 49671f8d0d3a77468dda508e641d472d6a9a3208 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 13 Mar 2026 09:45:38 +0000 Subject: [PATCH 044/115] WIP: output clipping on heads, two-stage joint calibration - Add output_lo/output_hi to LinearHead and MLPHead for cadence bounds - Add run_two_stage_joint() and _extract_two_stage_per_pool() to MLP calibration script --- quantammsim/calibration/heads.py | 19 ++++- scripts/run_mlp_calibration.py | 140 ++++++++++++++++++++++++++++++- 2 files changed, 155 insertions(+), 4 deletions(-) diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index f0e661b1..6f94e604 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -186,9 +186,12 @@ class LinearHead: L2 regularization on W (not bias) with strength ``alpha``. """ - def __init__(self, name: str, alpha: float = 0.01): + def __init__(self, name: str, alpha: float = 0.01, + output_lo: float = None, output_hi: float = None): self.name = name self.alpha = alpha + self.output_lo = output_lo + self.output_hi = output_hi def n_params(self, n_pools: int, k_attr: int) -> int: return 1 + k_attr # bias + W @@ -196,7 +199,10 @@ def n_params(self, n_pools: int, k_attr: int) -> int: def predict(self, params_slice, pool_idx, x_attr_i): bias = params_slice[0] W = params_slice[1:] - return bias + jnp.dot(x_attr_i, W) + out = bias + jnp.dot(x_attr_i, W) + if self.output_lo is not None or self.output_hi is not None: + out = jnp.clip(out, self.output_lo, self.output_hi) + return out def regularization(self, params_slice): W = params_slice[1:] @@ -404,11 +410,15 @@ def __init__( hidden: int = 16, alpha: float = 0.01, seed: int = 0, + output_lo: float = None, + output_hi: float = None, ): self.name = name self.hidden = hidden self.alpha = alpha self._seed = seed + self.output_lo = output_lo + self.output_hi = output_hi def n_params(self, n_pools: int, k_attr: int) -> int: h = self.hidden @@ -431,7 +441,10 @@ def predict(self, params_slice, pool_idx, x_attr_i): k_attr = x_attr_i.shape[0] W1, b1, W2, b2 = self._unpack_weights(params_slice, k_attr) hidden = jnp.maximum(x_attr_i @ W1 + b1, 0.0) # ReLU - return hidden @ W2 + b2 + out = hidden @ W2 + b2 + if self.output_lo is not None or self.output_hi is not None: + out = jnp.clip(out, self.output_lo, self.output_hi) + return out def regularization(self, params_slice): # Regularize W1 and W2, not biases diff --git a/scripts/run_mlp_calibration.py b/scripts/run_mlp_calibration.py index 74d455ae..85db2a06 100644 --- a/scripts/run_mlp_calibration.py +++ b/scripts/run_mlp_calibration.py @@ -196,6 +196,134 @@ def run_mlp_full_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): return result, model, jdata +def run_two_stage_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): + """Two-stage joint fit to identify cadence separately from noise. + + Stage 1: LinearHead(cad) + FixedHead(gas) + PerPoolNoiseHead + Per-pool noise (8 coeffs/pool) can't fully absorb arb's daily + volatility pattern, so cadence is identified. + + Stage 2: FixedHead(cad, stage1_values) + FixedHead(gas) + MLPNoiseHead + Cadence frozen from stage 1, MLP learns shared noise mapping. + + For new-pool prediction: stage 1 linear coefficients give cadence, + stage 2 MLP gives noise. + """ + import jax.numpy as jnp + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, LinearHead, MLPNoiseHead, PerPoolNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, fix_gas_to_chain=True) + gas_values = _build_gas_values(jdata, matched_clean) + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + + # ---- Stage 1: fit cadence with per-pool noise ---- + stage1_model = CalibrationModel( + cadence_head=LinearHead("cad", alpha=ALPHA_CAD), + gas_head=FixedHead("gas", gas_values), + noise_head=PerPoolNoiseHead(), + ) + n_p1 = stage1_model.n_params(n_pools, k_attr) + print(f"\n--- Two-stage S1: LinearHead(cad) + PerPoolNoiseHead " + f"({n_pools} pools, {n_p1} params) ---") + + stage1_result = stage1_model.fit( + jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) + print(f" Loss: {stage1_result['init_loss']:.4f} -> {stage1_result['loss']:.4f}") + print(f" Converged: {stage1_result['converged']}") + + # Extract per-pool cadences from stage 1 + params1 = jnp.array(stage1_result["params_flat"]) + (cs, ce), _, _ = stage1_model._head_slices(n_pools, k_attr) + cad_slice = params1[cs:ce] + stage1_cadences = np.array([ + float(stage1_model.cadence_head.predict(cad_slice, i, jdata.x_attr[i])) + for i in range(n_pools) + ]) + print(f" Cadence range: {np.exp(stage1_cadences.min()):.1f} - " + f"{np.exp(stage1_cadences.max()):.1f} min") + + # ---- Stage 2: fit MLP noise with frozen cadence ---- + stage2_model = CalibrationModel( + cadence_head=FixedHead("cad", stage1_cadences), + gas_head=FixedHead("gas", gas_values), + noise_head=MLPNoiseHead(hidden=hidden, alpha=ALPHA_NOISE), + ) + n_p2 = stage2_model.n_params(n_pools, k_attr) + print(f"\n--- Two-stage S2: FixedHead(cad) + MLPNoiseHead(hidden={hidden}) " + f"({n_pools} pools, {n_p2} params) ---") + + # Build warm-start for stage 2 noise from stage 1 per-pool noise + (_, _), (_, _), (ns, ne) = stage1_model._head_slices(n_pools, k_attr) + noise_params1 = np.array(params1[ns:ne]) + stage2_warm = {} + for i, pid in enumerate(jdata.pool_ids): + noise_c = np.array(stage1_model.noise_head.predict( + jnp.array(noise_params1), i, jdata.x_attr[i])) + stage2_warm[pid] = {"noise_coeffs": noise_c} + + stage2_result = stage2_model.fit( + jdata, maxiter=JOINT_MAXITER, warm_start=stage2_warm) + print(f" Loss: {stage2_result['init_loss']:.4f} -> {stage2_result['loss']:.4f}") + print(f" Converged: {stage2_result['converged']}") + + # Build a composite result dict for downstream use + # Cadence comes from stage 1 linear head, noise from stage 2 MLP + result = { + "stage1_result": stage1_result, + "stage2_result": stage2_result, + "loss": stage2_result["loss"], + "init_loss": stage1_result["init_loss"], + "converged": stage1_result["converged"] and stage2_result["converged"], + "n_pools": n_pools, + "k_attr": k_attr, + "pool_ids": jdata.pool_ids, + "attr_names": jdata.attr_names, + } + + return result, stage1_model, stage2_model, jdata + + +def _extract_two_stage_per_pool(stage1_model, stage2_model, result, jdata): + """Extract per-pool params from two-stage result.""" + import jax.numpy as jnp + + stage1_result = result["stage1_result"] + stage2_result = result["stage2_result"] + n_pools = result["n_pools"] + k_attr = result["k_attr"] + + params1 = jnp.array(stage1_result["params_flat"]) + params2 = jnp.array(stage2_result["params_flat"]) + + (cs1, ce1), _, (ns1, ne1) = stage1_model._head_slices(n_pools, k_attr) + _, _, (ns2, ne2) = stage2_model._head_slices(n_pools, k_attr) + + cad_slice = params1[cs1:ce1] + noise_slice = params2[ns2:ne2] + + per_pool = [] + for i in range(n_pools): + x_attr_i = jdata.x_attr[i] + log_cad = float(stage1_model.cadence_head.predict(cad_slice, i, x_attr_i)) + log_gas = float(stage2_model.gas_head.predict( + jnp.array([]), i, x_attr_i)) # FixedHead ignores params + noise_c = np.array(stage2_model.noise_head.predict(noise_slice, i, x_attr_i)) + per_pool.append({ + "log_cadence": log_cad, + "log_gas": log_gas, + "noise_coeffs": noise_c, + "cadence_minutes": float(np.exp(log_cad)), + "gas_usd": float(np.exp(log_gas)), + }) + return per_pool + + # ---- Per-pool predictions ---- @@ -702,16 +830,23 @@ def main(): mlp_full_result, mlp_full_model, _ = run_mlp_full_joint( matched_clean, option_c_clean) + # Two-stage: cadence identified with per-pool noise, then MLP noise + two_stage_result, ts_s1_model, ts_s2_model, _ = run_two_stage_joint( + matched_clean, option_c_clean) + # Step 4: Extract per-pool params from each model linear_pp = _extract_per_pool_params(linear_model, linear_result, jdata) mlp_noise_pp = _extract_per_pool_params(mlp_noise_model, mlp_noise_result, jdata) mlp_full_pp = _extract_per_pool_params(mlp_full_model, mlp_full_result, jdata) + two_stage_pp = _extract_two_stage_per_pool( + ts_s1_model, ts_s2_model, two_stage_result, jdata) - method_labels = ["linear", "mlp_noise", "mlp_full"] + method_labels = ["linear", "mlp_noise", "mlp_full", "two_stage"] model_results_for_pred = [ ("linear", linear_pp, jdata.pool_ids), ("mlp_noise", mlp_noise_pp, jdata.pool_ids), ("mlp_full", mlp_full_pp, jdata.pool_ids), + ("two_stage", two_stage_pp, jdata.pool_ids), ] # Step 5: Per-pool predictions @@ -726,6 +861,7 @@ def main(): ("linear", linear_result), ("mlp_noise", mlp_noise_result), ("mlp_full", mlp_full_result), + ("two_stage", two_stage_result), ]) # Step 7: Plots @@ -736,6 +872,7 @@ def main(): plot_decomposition_pages(predictions, "linear", "Linear shared noise", OUTPUT_DIR) plot_decomposition_pages(predictions, "mlp_noise", "MLP noise (linear cad)", OUTPUT_DIR) plot_decomposition_pages(predictions, "mlp_full", "Full MLP (MLP cad + MLP noise)", OUTPUT_DIR) + plot_decomposition_pages(predictions, "two_stage", "Two-stage (linear cad -> MLP noise)", OUTPUT_DIR) plot_summary_distributions(predictions, method_labels, OUTPUT_DIR) plot_r2_scatter(predictions, method_labels, OUTPUT_DIR) @@ -746,6 +883,7 @@ def main(): ("linear", linear_result), ("mlp_noise", mlp_noise_result), ("mlp_full", mlp_full_result), + ("two_stage", two_stage_result), ], OUTPUT_DIR) print(f"\n{'='*70}") From 71f1a3a41468900810c02cfa6a73cb8c5b3708e7 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 01:02:31 +0000 Subject: [PATCH 045/115] feat: add reduced x_obs (k_obs=4) to calibration pipeline Remove sigma- and fee-dependent features from observation covariates so the arb channel is the only path for volatility-driven volume variation (see docs/noise_covariate_design.md). build_x_obs gains reduced=True, per_pool_fit derives k_obs from data shape, and prepare_joint_data forwards reduced_x_obs. --- quantammsim/calibration/joint_fit.py | 5 ++- quantammsim/calibration/per_pool_fit.py | 18 +++++--- quantammsim/calibration/pool_data.py | 25 ++++++++--- tests/calibration/test_joint_fit.py | 19 +++++++++ tests/calibration/test_pool_data.py | 56 +++++++++++++++++++++++++ 5 files changed, 112 insertions(+), 11 deletions(-) diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py index 6dcaf851..765312d7 100644 --- a/quantammsim/calibration/joint_fit.py +++ b/quantammsim/calibration/joint_fit.py @@ -39,6 +39,7 @@ def prepare_joint_data( matched: Dict[str, dict], drop_chain_dummies: bool = False, fix_gas_to_chain: bool = False, + reduced_x_obs: bool = False, ) -> JointData: """Build batched JAX arrays from matched pool data. @@ -46,6 +47,8 @@ def prepare_joint_data( matched: dict from match_grids_to_panel drop_chain_dummies: if True, remove chain_* columns from attributes fix_gas_to_chain: if True, store fixed_log_gas per pool from CHAIN_GAS_USD + reduced_x_obs: if True, use 4-column reduced x_obs + (removes sigma/fee terms to avoid identification problems) Returns: JointData with per-pool JAX arrays and shared attribute matrix. @@ -64,7 +67,7 @@ def prepare_joint_data( for pid in pool_ids: entry = matched[pid] panel = entry["panel"] - x_obs = build_x_obs(panel) + x_obs = build_x_obs(panel, reduced=reduced_x_obs) y_obs = panel["log_volume"].values.astype(float) d = { diff --git a/quantammsim/calibration/per_pool_fit.py b/quantammsim/calibration/per_pool_fit.py index 08140d90..be724a9b 100644 --- a/quantammsim/calibration/per_pool_fit.py +++ b/quantammsim/calibration/per_pool_fit.py @@ -31,8 +31,9 @@ def make_initial_guess(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: OLS: noise_coeffs = lstsq(x_obs, y_obs) — assumes all volume is noise. This overestimates noise but gives a reasonable starting point. """ + k_obs = x_obs.shape[1] noise_coeffs, _, _, _ = np.linalg.lstsq(x_obs, y_obs, rcond=None) - init = np.zeros(2 + K_OBS) + init = np.zeros(2 + k_obs) init[0] = np.log(12.0) # log_cadence init[1] = np.log(1.0) # log_gas (= 0.0) init[2:] = noise_coeffs @@ -41,8 +42,9 @@ def make_initial_guess(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: def make_initial_guess_fixed_gas(x_obs: np.ndarray, y_obs: np.ndarray) -> np.ndarray: """Initial params for fixed-gas mode: cadence=12min, noise_coeffs from OLS.""" + k_obs = x_obs.shape[1] noise_coeffs, _, _, _ = np.linalg.lstsq(x_obs, y_obs, rcond=None) - init = np.zeros(1 + K_OBS) + init = np.zeros(1 + k_obs) init[0] = np.log(12.0) # log_cadence init[1:] = noise_coeffs return init @@ -81,7 +83,8 @@ def fit_single_pool( if init is None: init = make_initial_guess_fixed_gas(x_obs, y_obs) - scipy_bounds = [log_cad_bounds] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + k_obs = x_obs.shape[1] + scipy_bounds = [log_cad_bounds] + [(noise_bounds[0], noise_bounds[1])] * k_obs @jax.jit def loss_and_grad(params_flat): @@ -122,10 +125,11 @@ def scipy_wrapper(params_np): if init is None: init = make_initial_guess(x_obs, y_obs) + k_obs = x_obs.shape[1] log_gas_bounds = bounds.get("log_gas", (np.log(0.001), np.log(50.0))) scipy_bounds = [ log_cad_bounds, log_gas_bounds, - ] + [(noise_bounds[0], noise_bounds[1])] * K_OBS + ] + [(noise_bounds[0], noise_bounds[1])] * k_obs @jax.jit def loss_and_grad(params_flat): @@ -165,11 +169,15 @@ def fit_all_pools( matched: Dict[str, dict], n_workers: int = 1, fix_gas_to_chain: bool = False, + reduced: bool = False, ) -> Dict[str, dict]: """Fit all matched pools. Returns prefix -> fit_result with metadata. If fix_gas_to_chain is True, gas is fixed to the known chain-level cost from CHAIN_GAS_USD, and only (log_cadence, noise_coeffs) are optimized. + + If reduced is True, uses the 4-covariate x_obs (intercept, log_tvl_lag1, + dow_sin, dow_cos) instead of the full 8-covariate set. """ results = {} @@ -178,7 +186,7 @@ def fit_all_pools( coeffs = entry["coeffs"] day_indices = entry["day_indices"] - x_obs = build_x_obs(panel) + x_obs = build_x_obs(panel, reduced=reduced) y_obs = panel["log_volume"].values.astype(float) fixed_gas = None diff --git a/quantammsim/calibration/pool_data.py b/quantammsim/calibration/pool_data.py index 61082cdc..1b8110de 100644 --- a/quantammsim/calibration/pool_data.py +++ b/quantammsim/calibration/pool_data.py @@ -18,6 +18,7 @@ ) K_OBS = 8 # observation-level covariates +K_OBS_REDUCED = 4 # [intercept, log_tvl_lag1, dow_sin, dow_cos] # Default path for cached token market caps _MCAP_PATH = os.path.join( @@ -347,11 +348,16 @@ def match_grids_to_panel( return matched -def build_x_obs(panel_rows: pd.DataFrame) -> np.ndarray: - """Build (n_obs, 8) observation covariate matrix from panel rows. +def build_x_obs(panel_rows: pd.DataFrame, reduced: bool = False) -> np.ndarray: + """Build observation covariate matrix from panel rows. - Columns: [1, log_tvl_lag1, log_sigma, tvl*sigma, tvl*fee, - sigma*fee, dow_sin, dow_cos] + Full (reduced=False): (n_obs, 8) + [1, log_tvl_lag1, log_sigma, tvl*sigma, tvl*fee, sigma*fee, dow_sin, dow_cos] + + Reduced (reduced=True): (n_obs, 4) + [1, log_tvl_lag1, dow_sin, dow_cos] + Removes sigma- and fee-dependent terms so the arb channel is the only + path for volatility-driven volume variation. Where: log_sigma = log(max(volatility, 1e-6)) @@ -361,12 +367,21 @@ def build_x_obs(panel_rows: pd.DataFrame) -> np.ndarray: weekday: Monday=0, ..., Sunday=6 """ n = len(panel_rows) + weekdays = pd.to_datetime(panel_rows["date"]).dt.weekday.values.astype(float) + + if reduced: + x = np.zeros((n, K_OBS_REDUCED)) + x[:, 0] = 1.0 + x[:, 1] = panel_rows["log_tvl_lag1"].values.astype(float) + x[:, 2] = np.sin(2 * np.pi * weekdays / 7) + x[:, 3] = np.cos(2 * np.pi * weekdays / 7) + return x + x = np.zeros((n, K_OBS)) tvl = panel_rows["log_tvl_lag1"].values.astype(float) sigma = np.log(np.maximum(panel_rows["volatility"].values.astype(float), 1e-6)) fee = panel_rows["log_fee"].values.astype(float) - weekdays = pd.to_datetime(panel_rows["date"]).dt.weekday.values.astype(float) x[:, 0] = 1.0 # intercept x[:, 1] = tvl # log_tvl_lag1 diff --git a/tests/calibration/test_joint_fit.py b/tests/calibration/test_joint_fit.py index 50d77307..edb99f4a 100644 --- a/tests/calibration/test_joint_fit.py +++ b/tests/calibration/test_joint_fit.py @@ -188,3 +188,22 @@ def test_shared_noise_predict(self, matched_data): pred = predict_new_pool_joint(result, x_attr_new) assert "noise_coeffs" in pred assert len(pred["noise_coeffs"]) == K_OBS + + +class TestPrepareJointDataReduced: + """Test prepare_joint_data with reduced_x_obs=True.""" + + def test_reduced_x_obs_shape(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata = prepare_joint_data(matched_data, reduced_x_obs=True) + for pd in jdata.pool_data: + assert pd["x_obs"].shape[1] == K_OBS_REDUCED + + def test_default_unchanged(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_joint_data + + jdata = prepare_joint_data(matched_data) + for pd in jdata.pool_data: + assert pd["x_obs"].shape[1] == K_OBS diff --git a/tests/calibration/test_pool_data.py b/tests/calibration/test_pool_data.py index b3a9cf76..55441863 100644 --- a/tests/calibration/test_pool_data.py +++ b/tests/calibration/test_pool_data.py @@ -200,6 +200,62 @@ def test_x_obs_no_nans(self, synthetic_panel): assert not np.any(np.isnan(x)) +class TestBuildXObsReduced: + """Test build_x_obs with reduced=True: 4-column pruned covariates.""" + + def test_reduced_shape(self, synthetic_panel): + from quantammsim.calibration.pool_data import K_OBS_REDUCED, build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0, reduced=True) + assert x.shape == (len(pool0), K_OBS_REDUCED) + assert K_OBS_REDUCED == 4 + + def test_reduced_columns(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x_full = build_x_obs(pool0) + x_red = build_x_obs(pool0, reduced=True) + + # col 0: intercept + np.testing.assert_array_equal(x_red[:, 0], 1.0) + # col 1: log_tvl_lag1 (same as full col 1) + np.testing.assert_allclose(x_red[:, 1], x_full[:, 1]) + # col 2: dow_sin (same as full col 6) + np.testing.assert_allclose(x_red[:, 2], x_full[:, 6]) + # col 3: dow_cos (same as full col 7) + np.testing.assert_allclose(x_red[:, 3], x_full[:, 7]) + + def test_reduced_no_sigma(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x_full = build_x_obs(pool0) + x_red = build_x_obs(pool0, reduced=True) + + # Sigma-dependent columns from full (2,3,5) should not appear + sigma_cols = x_full[:, [2, 3, 5]] + for col in range(x_red.shape[1]): + for scol in range(sigma_cols.shape[1]): + if not np.allclose(sigma_cols[:, scol], 0.0): + assert not np.allclose(x_red[:, col], sigma_cols[:, scol]) + + def test_default_unchanged(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0) + assert x.shape == (len(pool0), K_OBS) + + def test_reduced_no_nans(self, synthetic_panel): + from quantammsim.calibration.pool_data import build_x_obs + + pool0 = synthetic_panel[synthetic_panel["pool_id"] == POOL_IDS_FULL[0]] + x = build_x_obs(pool0, reduced=True) + assert not np.any(np.isnan(x)) + + class TestBuildPoolAttributes: """Test build_pool_attributes: pool-level feature matrix.""" From ba30663ce6eae64ac108d435fd863b40778ecef4 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 01:02:45 +0000 Subject: [PATCH 046/115] feat: parameterize k_obs in noise heads PerPoolNoiseHead, SharedLinearNoiseHead, and MLPNoiseHead accept k_obs=4 to match the reduced x_obs. Defaults to K_OBS=8 so existing usage is unchanged. --- quantammsim/calibration/heads.py | 80 +++++------ tests/calibration/test_calibration_model.py | 71 +++++++++ tests/calibration/test_heads.py | 150 ++++++++++++++++++++ 3 files changed, 261 insertions(+), 40 deletions(-) diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 6f94e604..6d6a3763 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -257,21 +257,22 @@ def make_bounds(self, n_pools, k_attr): class PerPoolNoiseHead: - """Per-pool noise coefficients: each pool has K_OBS free parameters. + """Per-pool noise coefficients: each pool has k_obs free parameters. Used for Option C noise or Option A with per-pool noise. """ - def __init__(self, alpha: float = 0.0): + def __init__(self, alpha: float = 0.0, k_obs: int = None): self.name = "noise" self.alpha = alpha + self.k_obs = k_obs if k_obs is not None else K_OBS def n_params(self, n_pools: int, k_attr: int) -> int: - return n_pools * K_OBS + return n_pools * self.k_obs def predict(self, params_slice, pool_idx, x_attr_i): - start = pool_idx * K_OBS - return params_slice[start:start + K_OBS] + start = pool_idx * self.k_obs + return params_slice[start:start + self.k_obs] def regularization(self, params_slice): if self.alpha == 0.0: @@ -282,13 +283,13 @@ def init(self, jdata, warm_start=None): n_pools = len(jdata.pool_data) if warm_start is not None: - noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + noise_all = np.zeros((n_pools, self.k_obs), dtype=np.float64) for i, pid in enumerate(jdata.pool_ids): if pid in warm_start and "noise_coeffs" in warm_start[pid]: noise_all[i] = warm_start[pid]["noise_coeffs"] return noise_all.ravel() - noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + noise_all = np.zeros((n_pools, self.k_obs), dtype=np.float64) for i, pd in enumerate(jdata.pool_data): x_obs_np = np.array(pd["x_obs"]) y_obs_np = np.array(pd["y_obs"]) @@ -303,11 +304,11 @@ def predict_new(self, params_slice, x_attr): def unpack_result(self, params_slice, n_pools, k_attr): return { - "noise_coeffs": np.array(params_slice).reshape(n_pools, K_OBS), + "noise_coeffs": np.array(params_slice).reshape(n_pools, self.k_obs), } def make_bounds(self, n_pools, k_attr): - return [(None, None)] * (n_pools * K_OBS) + return [(None, None)] * (n_pools * self.k_obs) # --------------------------------------------------------------------------- @@ -318,28 +319,27 @@ def make_bounds(self, n_pools, k_attr): class SharedLinearNoiseHead: """Shared linear mapping for noise: bias_noise + x_attr @ W_noise. - Output is (K_OBS,) noise coefficients, predicted from pool attributes. + Output is (k_obs,) noise coefficients, predicted from pool attributes. L2 regularization on W_noise (not bias_noise). """ - def __init__(self, alpha: float = 0.01): + def __init__(self, alpha: float = 0.01, k_obs: int = None): self.name = "noise" self.alpha = alpha + self.k_obs = k_obs if k_obs is not None else K_OBS def n_params(self, n_pools: int, k_attr: int) -> int: - return (1 + k_attr) * K_OBS + return (1 + k_attr) * self.k_obs def predict(self, params_slice, pool_idx, x_attr_i): - # params_slice is ((1+k_attr) * K_OBS,) k_attr = x_attr_i.shape[0] - W_full = params_slice.reshape(1 + k_attr, K_OBS) + W_full = params_slice.reshape(1 + k_attr, self.k_obs) bias_noise = W_full[0] W_noise = W_full[1:] return bias_noise + jnp.dot(x_attr_i, W_noise) def regularization(self, params_slice): - # Regularize W_noise only, not bias_noise - W_full = params_slice.reshape(-1, K_OBS) + W_full = params_slice.reshape(-1, self.k_obs) W_noise = W_full[1:] return self.alpha * jnp.sum(W_noise ** 2) @@ -348,8 +348,7 @@ def init(self, jdata, warm_start=None): n_pools = len(jdata.pool_data) if warm_start is not None: - # Collect per-pool noise, regress on attributes - noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + noise_all = np.zeros((n_pools, self.k_obs), dtype=np.float64) for i, pid in enumerate(jdata.pool_ids): if pid in warm_start and "noise_coeffs" in warm_start[pid]: noise_all[i] = warm_start[pid]["noise_coeffs"] @@ -357,30 +356,29 @@ def init(self, jdata, warm_start=None): params, _, _, _ = np.linalg.lstsq(X_aug, noise_all, rcond=None) return params.ravel().astype(np.float64) - # Pool OLS noise as shared bias, W_noise = 0 all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) c, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) - params = np.zeros((1 + k_attr, K_OBS), dtype=np.float64) + params = np.zeros((1 + k_attr, self.k_obs), dtype=np.float64) params[0, :] = c return params.ravel() def predict_new(self, params_slice, x_attr): k_attr = len(x_attr) - W_full = np.array(params_slice).reshape(1 + k_attr, K_OBS) + W_full = np.array(params_slice).reshape(1 + k_attr, self.k_obs) bias_noise = W_full[0] W_noise = W_full[1:] return bias_noise + x_attr @ W_noise def unpack_result(self, params_slice, n_pools, k_attr): - W_full = np.array(params_slice).reshape(1 + k_attr, K_OBS) + W_full = np.array(params_slice).reshape(1 + k_attr, self.k_obs) return { "bias_noise": W_full[0], "W_noise": W_full[1:], } def make_bounds(self, n_pools, k_attr): - return [(None, None)] * ((1 + k_attr) * K_OBS) + return [(None, None)] * ((1 + k_attr) * self.k_obs) # --------------------------------------------------------------------------- @@ -534,10 +532,10 @@ def make_bounds(self, n_pools, k_attr): class MLPNoiseHead: """Two-layer MLP mapping from pool attributes to noise coefficients. - Architecture: x_attr → Dense(hidden, ReLU) → Dense(K_OBS) + Architecture: x_attr → Dense(hidden, ReLU) → Dense(k_obs) Parameter layout (flat): - [W1(k_attr * hidden), b1(hidden), W2(hidden * K_OBS), b2(K_OBS)] + [W1(k_attr * hidden), b1(hidden), W2(hidden * k_obs), b2(k_obs)] L2 regularization on W1 and W2 (not biases). @@ -552,50 +550,53 @@ def __init__( hidden: int = 16, alpha: float = 0.01, seed: int = 0, + k_obs: int = None, ): self.name = "noise" self.hidden = hidden self.alpha = alpha self._seed = seed + self.k_obs = k_obs if k_obs is not None else K_OBS def n_params(self, n_pools: int, k_attr: int) -> int: h = self.hidden - # W1(k_attr*h) + b1(h) + W2(h*K_OBS) + b2(K_OBS) - return k_attr * h + h + h * K_OBS + K_OBS + return k_attr * h + h + h * self.k_obs + self.k_obs def _unpack_weights(self, params_slice, k_attr): """Unpack flat slice → (W1, b1, W2, b2).""" h = self.hidden + ko = self.k_obs idx = 0 W1 = params_slice[idx:idx + k_attr * h].reshape(k_attr, h) idx += k_attr * h b1 = params_slice[idx:idx + h] idx += h - W2 = params_slice[idx:idx + h * K_OBS].reshape(h, K_OBS) - idx += h * K_OBS - b2 = params_slice[idx:idx + K_OBS] + W2 = params_slice[idx:idx + h * ko].reshape(h, ko) + idx += h * ko + b2 = params_slice[idx:idx + ko] return W1, b1, W2, b2 def predict(self, params_slice, pool_idx, x_attr_i): k_attr = x_attr_i.shape[0] W1, b1, W2, b2 = self._unpack_weights(params_slice, k_attr) hidden = jnp.maximum(x_attr_i @ W1 + b1, 0.0) # ReLU - return hidden @ W2 + b2 # (K_OBS,) + return hidden @ W2 + b2 # (k_obs,) def regularization(self, params_slice): h = self.hidden + ko = self.k_obs total = params_slice.shape[0] - # Solve for k_attr: total = k*h + h + h*K_OBS + K_OBS - # k*h = total - h - h*K_OBS - K_OBS - k_attr = (total - h - h * K_OBS - K_OBS) // h + # Solve for k_attr: total = k*h + h + h*ko + ko + k_attr = (total - h - h * ko - ko) // h W1 = params_slice[:k_attr * h] - W2 = params_slice[k_attr * h + h:k_attr * h + h + h * K_OBS] + W2 = params_slice[k_attr * h + h:k_attr * h + h + h * ko] return self.alpha * (jnp.sum(W1 ** 2) + jnp.sum(W2 ** 2)) def init(self, jdata, warm_start=None): k_attr = jdata.x_attr.shape[1] n_pools = len(jdata.pool_data) h = self.hidden + ko = self.k_obs rng = np.random.RandomState(self._seed) # He initialization for W1 @@ -603,20 +604,19 @@ def init(self, jdata, warm_start=None): W1 = rng.randn(k_attr, h).astype(np.float64) * std b1 = np.zeros(h, dtype=np.float64) - W2 = np.zeros((h, K_OBS), dtype=np.float64) + W2 = np.zeros((h, ko), dtype=np.float64) if warm_start is not None: - noise_all = np.zeros((n_pools, K_OBS), dtype=np.float64) + noise_all = np.zeros((n_pools, ko), dtype=np.float64) for i, pid in enumerate(jdata.pool_ids): if pid in warm_start and "noise_coeffs" in warm_start[pid]: noise_all[i] = warm_start[pid]["noise_coeffs"] b2 = np.mean(noise_all, axis=0) # Warm-start W2 by least-squares through hidden activations - # so the MLP init approximates the per-pool warm-start noise x_attr = np.array(jdata.x_attr) H = np.maximum(x_attr @ W1 + b1, 0.0) # (n_pools, h) - residuals = noise_all - b2 # (n_pools, K_OBS) + residuals = noise_all - b2 # (n_pools, ko) W2, _, _, _ = np.linalg.lstsq(H, residuals, rcond=None) else: # Pooled OLS noise as b2 @@ -630,7 +630,7 @@ def predict_new(self, params_slice, x_attr): k_attr = len(x_attr) W1, b1, W2, b2 = self._unpack_weights(np.asarray(params_slice), k_attr) hidden = np.maximum(x_attr @ W1 + b1, 0.0) - return hidden @ W2 + b2 # (K_OBS,) + return hidden @ W2 + b2 # (k_obs,) def unpack_result(self, params_slice, n_pools, k_attr): params_np = np.array(params_slice) diff --git a/tests/calibration/test_calibration_model.py b/tests/calibration/test_calibration_model.py index 95973f47..95812f63 100644 --- a/tests/calibration/test_calibration_model.py +++ b/tests/calibration/test_calibration_model.py @@ -582,3 +582,74 @@ def test_mlp_noise_param_count(self, jdata_ppn): # MLP noise: k*h + h + h*K_OBS + K_OBS expected = (1 + k_attr) * 2 + k_attr * h + h + h * K_OBS + K_OBS assert model.n_params(n_pools, k_attr) == expected + + +# ── Reduced k_obs=4 integration tests ──────────────────────────────────── + +K_OBS_REDUCED = 4 + + +@pytest.fixture +def jdata_reduced(matched_data): + """JointData with reduced x_obs (4 columns).""" + from quantammsim.calibration.joint_fit import prepare_joint_data + return prepare_joint_data( + matched_data, drop_chain_dummies=True, + fix_gas_to_chain=True, reduced_x_obs=True, + ) + + +class TestReducedKObsIntegration: + """CalibrationModel with k_obs=4 noise heads on reduced x_obs data.""" + + def test_reduced_n_params(self, jdata_reduced): + n_pools = len(jdata_reduced.pool_data) + k_attr = jdata_reduced.x_attr.shape[1] + h = 8 + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + FixedHead("gas", np.zeros(n_pools)), + MLPNoiseHead(hidden=h, alpha=0.01, k_obs=K_OBS_REDUCED), + ) + # Linear cad: 1+k, Fixed gas: 0, + # MLP noise: k*h + h + h*4 + 4 + expected = (1 + k_attr) + 0 + k_attr * h + h + h * 4 + 4 + assert model.n_params(n_pools, k_attr) == expected + + def test_reduced_loss_runs(self, jdata_reduced): + n_pools = len(jdata_reduced.pool_data) + gas_values = np.zeros(n_pools) + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + FixedHead("gas", gas_values), + MLPNoiseHead(hidden=8, alpha=0.01, k_obs=K_OBS_REDUCED), + ) + loss_fn = model.make_joint_loss_fn(jdata_reduced) + init = jnp.array(model.pack_init(jdata_reduced)) + loss = float(loss_fn(init)) + assert np.isfinite(loss) and loss >= 0 + + def test_reduced_grad_finite(self, jdata_reduced): + n_pools = len(jdata_reduced.pool_data) + gas_values = np.zeros(n_pools) + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + FixedHead("gas", gas_values), + MLPNoiseHead(hidden=8, alpha=0.01, k_obs=K_OBS_REDUCED), + ) + loss_fn = model.make_joint_loss_fn(jdata_reduced) + init = jnp.array(model.pack_init(jdata_reduced)) + grad = jax.grad(loss_fn)(init) + assert jnp.all(jnp.isfinite(grad)) + + def test_reduced_fit_converges(self, jdata_reduced): + n_pools = len(jdata_reduced.pool_data) + gas_values = np.zeros(n_pools) + model = CalibrationModel( + LinearHead("cad", alpha=0.01), + FixedHead("gas", gas_values), + MLPNoiseHead(hidden=8, alpha=0.01, k_obs=K_OBS_REDUCED), + ) + result = model.fit(jdata_reduced, maxiter=100) + assert result["loss"] <= result["init_loss"] + assert np.isfinite(result["loss"]) diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index b0f53000..2c891e86 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -744,3 +744,153 @@ def loss(p): grad = jax.grad(loss)(params) assert grad.shape == params.shape assert jnp.all(jnp.isfinite(grad)) + + +# ── Reduced k_obs=4 tests ───────────────────────────────────────────────── + +K_OBS_REDUCED = 4 + + +def _make_fake_jdata_reduced(): + """JointData-like object with k_obs=4 x_obs for reduced noise testing.""" + from quantammsim.calibration.joint_fit import JointData + + pool_data = [] + for _ in range(N_POOLS): + n_obs = 14 + x_obs = np.random.randn(n_obs, K_OBS_REDUCED) + x_obs[:, 0] = 1.0 # intercept column + y_obs = np.random.randn(n_obs) * 0.5 + 9.0 + pool_data.append({ + "x_obs": jnp.array(x_obs), + "y_obs": jnp.array(y_obs), + "day_indices": jnp.arange(n_obs) % 10, + }) + + x_attr = jnp.array(np.random.randn(N_POOLS, K_ATTR)) + return JointData( + pool_data=pool_data, + x_attr=x_attr, + pool_ids=POOL_PREFIXES[:N_POOLS], + attr_names=[f"attr_{i}" for i in range(K_ATTR)], + ) + + +class TestPerPoolNoiseHeadReduced: + """PerPoolNoiseHead with k_obs=4.""" + + def test_n_params(self): + h = PerPoolNoiseHead(k_obs=4) + assert h.n_params(3, 5) == 3 * 4 + + def test_predict_correct_slice(self): + h = PerPoolNoiseHead(k_obs=4) + params = jnp.arange(3 * 4, dtype=float) + x_attr_i = jnp.zeros(5) + for i in range(3): + result = h.predict(params, i, x_attr_i) + expected = params[i * 4:(i + 1) * 4] + np.testing.assert_allclose(result, expected) + + def test_init_ols(self): + np.random.seed(42) + h = PerPoolNoiseHead(k_obs=4) + jdata = _make_fake_jdata_reduced() + init = h.init(jdata) + assert init.shape == (N_POOLS * 4,) + assert np.all(np.isfinite(init)) + + def test_roundtrip(self): + np.random.seed(42) + h = PerPoolNoiseHead(k_obs=4) + jdata = _make_fake_jdata_reduced() + init = h.init(jdata) + result = h.unpack_result(init, N_POOLS, K_ATTR) + assert result["noise_coeffs"].shape == (N_POOLS, 4) + + def test_default_unchanged(self): + h = PerPoolNoiseHead() + assert h.k_obs == K_OBS + assert h.n_params(3, 5) == 3 * K_OBS + + +class TestSharedLinearNoiseHeadReduced: + """SharedLinearNoiseHead with k_obs=4.""" + + def test_n_params(self): + h = SharedLinearNoiseHead(k_obs=4) + assert h.n_params(3, 5) == (1 + 5) * 4 + + def test_predict(self): + k_attr = 3 + h = SharedLinearNoiseHead(k_obs=4) + W_full = np.zeros((1 + k_attr, 4)) + W_full[0, :] = 1.0 + W_full[1, 0] = 2.0 + params = jnp.array(W_full.ravel()) + x_attr_i = jnp.array([1.0, 0.0, 0.0]) + result = h.predict(params, 0, x_attr_i) + assert result.shape == (4,) + np.testing.assert_allclose(float(result[0]), 3.0) + np.testing.assert_allclose(float(result[1]), 1.0) + + def test_init(self): + np.random.seed(42) + h = SharedLinearNoiseHead(k_obs=4) + jdata = _make_fake_jdata_reduced() + init = h.init(jdata) + assert init.shape == ((1 + K_ATTR) * 4,) + assert np.all(np.isfinite(init)) + + def test_default_unchanged(self): + h = SharedLinearNoiseHead() + assert h.k_obs == K_OBS + assert h.n_params(3, 5) == (1 + 5) * K_OBS + + +class TestMLPNoiseHeadReduced: + """MLPNoiseHead with k_obs=4.""" + + def test_n_params(self): + h = MLPNoiseHead(hidden=16, k_obs=4) + # k_attr=5: 5*16 + 16 + 16*4 + 4 = 80+16+64+4 = 164 + assert h.n_params(3, 5) == 164 + + def test_predict(self): + k_attr = 3 + h = MLPNoiseHead(hidden=4, k_obs=4) + n_p = h.n_params(1, k_attr) + params = jnp.zeros(n_p) + x_attr_i = jnp.array([1.0, 2.0, 3.0]) + result = h.predict(params, 0, x_attr_i) + assert result.shape == (4,) + + def test_init(self): + np.random.seed(42) + h = MLPNoiseHead(hidden=4, k_obs=4) + jdata = _make_fake_jdata_reduced() + init = h.init(jdata) + n_p = h.n_params(N_POOLS, K_ATTR) + assert init.shape == (n_p,) + assert np.all(np.isfinite(init)) + + def test_regularization(self): + k_attr = 2 + h = MLPNoiseHead(hidden=2, alpha=1.0, k_obs=4) + # Layout: W1(2*2=4), b1(2), W2(2*4=8), b2(4) = 18 params + n_p = h.n_params(1, k_attr) + assert n_p == 18 + params = np.zeros(n_p) + params[0] = 3.0 # W1[0,0] + params[1] = 4.0 # W1[0,1] + params[6] = 1.0 # W2[0,0] + params[7] = 2.0 # W2[0,1] + params[-1] = 999.0 # b2[-1] — not regularized + # reg = 1.0 * (9 + 16 + 1 + 4) = 30.0 + result = float(h.regularization(jnp.array(params))) + np.testing.assert_allclose(result, 30.0) + + def test_default_unchanged(self): + h = MLPNoiseHead() + assert h.k_obs == K_OBS + assert h.n_params(3, 5) == 232 From 5a2cbdf42f3ca3f45b48c0b18a7b3350fbb9e4fa Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 01:03:03 +0000 Subject: [PATCH 047/115] feat: calibrated 8-covariate noise model for reCLAMM simulator Add reclamm_calibrated_noise_volume (c_0..c_7 log-linear model with TVL, volatility, fee interactions, and DOW harmonics). Wire through all 4 reserve calculation paths with dow_sin/dow_cos scan inputs. Consolidate volatility/DOW array prep into _prepare_noise_arrays. --- quantammsim/pools/noise_trades.py | 75 ++++++++++++ quantammsim/pools/reCLAMM/reclamm.py | 107 +++++++++++------- quantammsim/pools/reCLAMM/reclamm_reserves.py | 43 +++++++ 3 files changed, 185 insertions(+), 40 deletions(-) diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index 3f45792f..b40154a0 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -272,3 +272,78 @@ def reclamm_loglinear_noise_volume( return jnp.maximum(0.0, daily_vol / 1440.0 - arb_volume_this_period) +@jit +def reclamm_calibrated_noise_volume( + effective_value_usd, + gamma, + volatility, + arb_volume_this_period, + dow_sin, + dow_cos, + noise_params=None, +): + """8-covariate calibrated noise volume from cross-pool log-linear model. + + Predicts per-minute noise volume using:: + + log(V_daily) = c_0 + c_1*log(TVL) + c_2*log(sigma) + + c_3*log(TVL)*log(sigma) + c_4*log(TVL)*fee + + c_5*log(sigma)*fee + c_6*dow_sin + c_7*dow_cos + V_noise = max(0, exp(log_daily_vol) / 1440 - arb_volume) + + where sigma is annualised daily realised volatility, fee = 1 - gamma, + and dow_sin/dow_cos encode day-of-week seasonality. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + gamma : float + Fee parameter (1 - fee_rate). + volatility : float + Annualised daily realised volatility of the price ratio. + arb_volume_this_period : float + Arb volume already accounted for this time step (USD). + dow_sin : float + sin(2*pi*weekday/7) for the current day. + dow_cos : float + cos(2*pi*weekday/7) for the current day. + noise_params : dict, optional + Calibrated coefficients: c_0 .. c_7. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + if noise_params is None: + noise_params = {} + c_0 = noise_params.get("c_0", 0.0) + c_1 = noise_params.get("c_1", 1.0) + c_2 = noise_params.get("c_2", 0.0) + c_3 = noise_params.get("c_3", 0.0) + c_4 = noise_params.get("c_4", 0.0) + c_5 = noise_params.get("c_5", 0.0) + c_6 = noise_params.get("c_6", 0.0) + c_7 = noise_params.get("c_7", 0.0) + + fee = 1.0 - gamma + log_tvl = jnp.log(jnp.maximum(effective_value_usd, 1.0)) + log_sigma = jnp.log(jnp.maximum(volatility, 1e-10)) + + log_daily_vol = ( + c_0 + + c_1 * log_tvl + + c_2 * log_sigma + + c_3 * log_tvl * log_sigma + + c_4 * log_tvl * fee + + c_5 * log_sigma * fee + + c_6 * dow_sin + + c_7 * dow_cos + ) + daily_vol = jnp.exp(log_daily_vol) + # noise_coeffs predict V_noise directly (not V_total), so no need to + # subtract arb volume — that would double-count the arb subtraction. + return jnp.maximum(0.0, daily_vol / 1440.0) + + diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 5b913ff2..f7b6d19d 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -214,6 +214,49 @@ def _resolve_fees(params, run_fingerprint): return jnp.squeeze(params["fees"]) return run_fingerprint["fees"] + def _prepare_noise_arrays(self, prices, run_fingerprint, start_index, + bout_length, arb_freq, max_len): + """Prepare volatility and dow arrays for noise models that need them. + + Returns (volatility_array, dow_sin_array, dow_cos_array) — all sliced + and decimated to match arb_prices shape. Arrays are None when the + noise model does not require them. + """ + noise_model = run_fingerprint.get("noise_model", "ratio") + needs_vol = noise_model in ( + "tsoukalas_sqrt", "tsoukalas_log", "loglinear", "calibrated", + ) + if not needs_vol: + return None, None, None + + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + vol_prepared = _prepare_dynamic_array( + volatility_array, start_index, bout_length, arb_freq, max_len, + ) + + if noise_model != "calibrated": + return vol_prepared, None, None + + # Day-of-week sin/cos arrays for the calibrated noise model. + # Compute from startDateString + minute offsets (vectorized). + import pandas as pd + start_dt = pd.Timestamp(run_fingerprint["startDateString"]) + n_minutes = prices.shape[0] + day_indices = np.arange(n_minutes) // 1440 + start_weekday = start_dt.weekday() # Monday=0 .. Sunday=6 + weekdays = ((start_weekday + day_indices) % 7).astype(np.float64) + dow_sin_full = jnp.array(np.sin(2.0 * np.pi * weekdays / 7.0)) + dow_cos_full = jnp.array(np.cos(2.0 * np.pi * weekdays / 7.0)) + dow_sin_prepared = _prepare_dynamic_array( + dow_sin_full, start_index, bout_length, arb_freq, max_len, + ) + dow_cos_prepared = _prepare_dynamic_array( + dow_cos_full, start_index, bout_length, arb_freq, max_len, + ) + return vol_prepared, dow_sin_prepared, dow_cos_prepared + @partial(jit, static_argnums=(2,)) def calculate_reserves_with_fees( self, @@ -241,16 +284,10 @@ def calculate_reserves_with_fees( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): - volatility_array = self.calculate_volatility_array( - prices, run_fingerprint, - ) - arb_vol = _prepare_dynamic_array( - volatility_array, start_index, bout_length, - arb_freq, s.arb_prices.shape[0], - ) - else: - arb_vol = None + arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, arb_freq, s.arb_prices.shape[0], + ) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_with_fees( @@ -273,6 +310,8 @@ def calculate_reserves_with_fees( noise_model=noise_model, noise_params=noise_params, volatility_array=arb_vol, + dow_sin_array=dow_sin, + dow_cos_array=dow_cos, ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -311,16 +350,10 @@ def calculate_reserves_and_fee_revenue_with_fees( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): - volatility_array = self.calculate_volatility_array( - prices, run_fingerprint, - ) - arb_vol = _prepare_dynamic_array( - volatility_array, start_index, bout_length, - arb_freq, s.arb_prices.shape[0], - ) - else: - arb_vol = None + arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, arb_freq, s.arb_prices.shape[0], + ) if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -343,6 +376,8 @@ def calculate_reserves_and_fee_revenue_with_fees( noise_model=noise_model, noise_params=noise_params, volatility_array=arb_vol, + dow_sin_array=dow_sin, + dow_cos_array=dow_cos, ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), @@ -386,16 +421,10 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): - volatility_array = self.calculate_volatility_array( - prices, run_fingerprint, - ) - arb_vol = _prepare_dynamic_array( - volatility_array, start_index, bout_length, - run_fingerprint["arb_frequency"], max_len, - ) - else: - arb_vol = None + arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, run_fingerprint["arb_frequency"], max_len, + ) return _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, @@ -418,6 +447,8 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( noise_model=noise_model, noise_params=noise_params, volatility_array=arb_vol, + dow_sin_array=dow_sin, + dow_cos_array=dow_cos, ) @partial(jit, static_argnums=(2,)) @@ -485,16 +516,10 @@ def calculate_reserves_with_dynamic_inputs( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): - volatility_array = self.calculate_volatility_array( - prices, run_fingerprint, - ) - arb_vol = _prepare_dynamic_array( - volatility_array, start_index, bout_length, - run_fingerprint["arb_frequency"], max_len, - ) - else: - arb_vol = None + arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, run_fingerprint["arb_frequency"], max_len, + ) return _jax_calc_reclamm_reserves_with_dynamic_inputs( s.initial_reserves, s.Va, s.Vb, @@ -517,6 +542,8 @@ def calculate_reserves_with_dynamic_inputs( noise_model=noise_model, noise_params=noise_params, volatility_array=arb_vol, + dow_sin_array=dow_sin, + dow_cos_array=dow_cos, ) def init_base_parameters( diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 2a46ad3d..713923c9 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -35,6 +35,7 @@ reclamm_tsoukalas_sqrt_noise_volume, reclamm_tsoukalas_log_noise_volume, reclamm_loglinear_noise_volume, + reclamm_calibrated_noise_volume, ) # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) @@ -1003,6 +1004,24 @@ def _skip_schedule_state(_): arb_volume, _np, ) + noise_fee_income = (1.0 - gamma) * noise_vol + scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) + Ra_new = Ra_new * scale + Rb_new = Rb_new * scale + elif noise_model == "calibrated": + volatility = input_list[9] + dow_sin = input_list[10] + dow_cos = input_list[11] + arb_volume = 0.5 * jnp.sum(jnp.abs(applied_trade) * prices) + real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) + effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + + _np = noise_params if noise_params is not None else {} + noise_vol = reclamm_calibrated_noise_volume( + effective_value, gamma, volatility, + arb_volume, dow_sin, dow_cos, _np, + ) + noise_fee_income = (1.0 - gamma) * noise_vol scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) Ra_new = Ra_new * scale @@ -1260,6 +1279,8 @@ def _jax_calc_reclamm_reserves_with_fees( noise_model="ratio", noise_params=None, volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, ): """Calculate reClAMM reserves over time with fees. @@ -1318,6 +1339,10 @@ def _jax_calc_reclamm_reserves_with_fees( price_ratio_updates, lp_supply_array] if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) carry_init = [ initial_reserves, @@ -1359,6 +1384,8 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( noise_model="ratio", noise_params=None, volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" if lp_supply_array is None: @@ -1426,6 +1453,10 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( price_ratio_updates, lp_supply_array] if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) carry_init = [ initial_reserves, @@ -1557,6 +1588,8 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( noise_model="ratio", noise_params=None, volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1617,6 +1650,10 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( price_ratio_updates, lp_supply_array] if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) carry_init = [ initial_reserves, @@ -1658,6 +1695,8 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( noise_model="ratio", noise_params=None, volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1731,6 +1770,10 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( price_ratio_updates, lp_supply_array] if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) carry_init = [ initial_reserves, From 246de0e7da82e736bdb6a5ce9640efa7bb3109c1 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 01:03:14 +0000 Subject: [PATCH 048/115] feat: configurable n_evaluation_points for Optuna and keep startDateString in static dict MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Read n_evaluation_points from optuna_settings instead of hardcoding 20. Keep startDateString in the static fingerprint dict — the calibrated noise model needs it to compute day-of-week arrays. --- quantammsim/runners/jax_runner_utils.py | 2 +- quantammsim/runners/jax_runners.py | 4 +++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index aa5b79d4..1c93887f 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -658,7 +658,7 @@ def __eq__(self, other): # These are excluded when creating static_dict from run_fingerprint _TRAINING_ONLY_FIELDS = frozenset({ "optimisation_settings", # Contains lr, optimizer, etc. - "startDateString", # Data loading dates + # startDateString kept in static dict — needed by calibrated noise model "endDateString", "endTestDateString", "subsidary_pools", # Handled separately diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index 178f0a04..09378845 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -1298,7 +1298,9 @@ def _extract_params_at(params_tree, j): return selected_params elif run_fingerprint["optimisation_settings"]["method"] == "optuna": - n_evaluation_points = 20 + n_evaluation_points = run_fingerprint["optimisation_settings"].get( + "optuna_settings", {} + ).get("n_evaluation_points", 20) min_spacing = data_dict["bout_length"] // 2 # E run_fingerprint["optimisation_settings"]["n_parameter_sets"] = 1 From 227fe7735c1d4caf9f098dbf76089f01f3f8d00b Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:24:08 +0000 Subject: [PATCH 049/115] feat: add token encoding for token-factored noise model Add encode_tokens() to build token index, per-pool token/chain assignments, and token covariate matrix (D_TOKEN=5) from the matched pool set. Token classification via symbol lookup for stablecoins, ETH derivatives, and L1 natives. Market cap from hardcoded values or JSON fallback. Foundation for the token-factored noise head where pool noise coefficients decompose as u[token_a] + u[token_b] + alpha[chain] + beta_fee * log(fee) + delta_i. --- quantammsim/calibration/pool_data.py | 102 +++++++++++++++++++ tests/calibration/test_pool_data.py | 145 +++++++++++++++++++++++++++ 2 files changed, 247 insertions(+) diff --git a/quantammsim/calibration/pool_data.py b/quantammsim/calibration/pool_data.py index 1b8110de..18ab8212 100644 --- a/quantammsim/calibration/pool_data.py +++ b/quantammsim/calibration/pool_data.py @@ -348,6 +348,108 @@ def match_grids_to_panel( return matched +# Token classification for token-factored model +_ETH_DERIVATIVES = { + "WETH", "ETH", "wstETH", "stETH", "rETH", "cbETH", + "waEthLidoWETH", "waEthLidowstETH", "waBasWETH", "waGnowstETH", +} +_L1_NATIVE = { + "WETH", "ETH", "WMATIC", "MATIC", "POL", "wPOL", + "WAVAX", "AVAX", "GNO", "S", "wS", "stS", +} + +D_TOKEN = 5 # [intercept, log_mcap, is_stable, is_eth_derivative, is_L1_native] + + +def _classify_token(symbol: str, mcaps: dict) -> dict: + """Classify a token into binary feature flags.""" + return { + "is_stable": 1.0 if symbol in _STABLECOINS else 0.0, + "is_eth_derivative": 1.0 if symbol in _ETH_DERIVATIVES else 0.0, + "is_L1_native": 1.0 if symbol in _L1_NATIVE else 0.0, + "log_mcap": np.log(max(mcaps.get(symbol, {}).get("mcap_usd", 1e6), 1.0)), + } + + +def encode_tokens( + matched: Dict[str, dict], + mcap_path: str = None, +) -> dict: + """Build token index, per-pool token assignments, and token covariate matrix. + + Iterates over pools in sorted key order (same ordering as build_pool_attributes). + + Returns dict with: + token_index: dict[str, int] — symbol -> integer index (sorted alphabetically) + token_a_idx: np.ndarray (n_pools,) — index of token A for each pool + token_b_idx: np.ndarray (n_pools,) — index of token B for each pool + x_token: np.ndarray (n_tokens, D_TOKEN) — token covariate matrix + chain_idx: np.ndarray (n_pools,) — chain integer index per pool + chain_index: dict[str, int] — chain name -> integer index (sorted) + log_fees: np.ndarray (n_pools,) — log(fee) per pool + n_tokens: int + n_chains: int + """ + mcaps = _load_token_mcaps(mcap_path) + pool_ids = sorted(matched.keys()) + n_pools = len(pool_ids) + + # Collect all tokens and chains + all_tokens = set() + all_chains = set() + for pid in pool_ids: + entry = matched[pid] + toks = _parse_tokens(entry["tokens"]) + all_tokens.update(toks[:2]) + all_chains.add(entry["chain"]) + + # Build sorted indices + token_list = sorted(all_tokens) + token_index = {t: i for i, t in enumerate(token_list)} + n_tokens = len(token_list) + + chain_list = sorted(all_chains) + chain_index = {c: i for i, c in enumerate(chain_list)} + n_chains = len(chain_list) + + # Build per-pool arrays + token_a_idx = np.zeros(n_pools, dtype=np.int32) + token_b_idx = np.zeros(n_pools, dtype=np.int32) + chain_idx = np.zeros(n_pools, dtype=np.int32) + log_fees = np.zeros(n_pools, dtype=np.float64) + + for i, pid in enumerate(pool_ids): + entry = matched[pid] + toks = _parse_tokens(entry["tokens"]) + token_a_idx[i] = token_index[toks[0]] + token_b_idx[i] = token_index[toks[1]] + chain_idx[i] = chain_index[entry["chain"]] + log_fees[i] = np.log(entry["fee"]) + + # Build token covariate matrix: (n_tokens, D_TOKEN) + # Columns: [intercept, log_mcap, is_stable, is_eth_derivative, is_L1_native] + x_token = np.zeros((n_tokens, D_TOKEN), dtype=np.float64) + for t, idx in token_index.items(): + cls = _classify_token(t, mcaps) + x_token[idx, 0] = 1.0 # intercept + x_token[idx, 1] = cls["log_mcap"] + x_token[idx, 2] = cls["is_stable"] + x_token[idx, 3] = cls["is_eth_derivative"] + x_token[idx, 4] = cls["is_L1_native"] + + return { + "token_index": token_index, + "token_a_idx": token_a_idx, + "token_b_idx": token_b_idx, + "x_token": x_token, + "chain_idx": chain_idx, + "chain_index": chain_index, + "log_fees": log_fees, + "n_tokens": n_tokens, + "n_chains": n_chains, + } + + def build_x_obs(panel_rows: pd.DataFrame, reduced: bool = False) -> np.ndarray: """Build observation covariate matrix from panel rows. diff --git a/tests/calibration/test_pool_data.py b/tests/calibration/test_pool_data.py index 55441863..adbf212a 100644 --- a/tests/calibration/test_pool_data.py +++ b/tests/calibration/test_pool_data.py @@ -12,6 +12,151 @@ ) +class TestEncodeTokens: + """Test encode_tokens: token index, assignments, and covariate matrix.""" + + def _get_matched(self, synthetic_daily_grid, synthetic_panel, tmp_path): + from quantammsim.calibration.pool_data import match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + return match_grids_to_panel(str(grid_dir), synthetic_panel) + + def test_returns_expected_keys( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + expected_keys = { + "token_index", "token_a_idx", "token_b_idx", + "x_token", "chain_idx", "chain_index", + "log_fees", "n_tokens", "n_chains", + } + assert expected_keys.issubset(result.keys()) + + def test_unique_tokens_discovered( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + # Synthetic panel has tokens: BTC, ETH (pool 0) and AAVE, ETH (pool 1) + # Unique tokens: AAVE, BTC, ETH (sorted) + assert result["n_tokens"] == 3 + assert set(result["token_index"].keys()) == {"AAVE", "BTC", "ETH"} + # Indices should be contiguous 0..2 + assert set(result["token_index"].values()) == {0, 1, 2} + + def test_x_token_shape_and_intercept( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + x_token = result["x_token"] + assert x_token.shape[0] == result["n_tokens"] # 3 tokens + assert x_token.shape[1] >= 4 # at least intercept + 3 binary flags + # Intercept column is all 1s + np.testing.assert_array_equal(x_token[:, 0], 1.0) + + def test_token_classifications( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + ti = result["token_index"] + x_tok = result["x_token"] + # Column layout: [intercept, log_mcap, is_stable, is_eth_deriv, is_L1_native] + # ETH: is_eth_derivative=1, is_L1_native=1 + assert x_tok[ti["ETH"], 3] == 1.0 # is_eth_derivative + assert x_tok[ti["ETH"], 4] == 1.0 # is_L1_native + # AAVE: none of the binary flags + assert x_tok[ti["AAVE"], 2] == 0.0 # not stable + assert x_tok[ti["AAVE"], 3] == 0.0 # not eth_deriv + assert x_tok[ti["AAVE"], 4] == 0.0 # not L1_native + + def test_pool_token_mapping( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + ti = result["token_index"] + + # Pool 0 (first sorted prefix): tokens = "BTC,ETH" + assert result["token_a_idx"][0] == ti["BTC"] + assert result["token_b_idx"][0] == ti["ETH"] + + # Pool 1 (second sorted prefix): tokens = "AAVE,ETH" + assert result["token_a_idx"][1] == ti["AAVE"] + assert result["token_b_idx"][1] == ti["ETH"] + + def test_chain_index( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + assert result["n_chains"] == 2 + assert set(result["chain_index"].keys()) == {"ARBITRUM", "MAINNET"} + assert set(result["chain_index"].values()) == {0, 1} + + def test_chain_idx_mapping( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + pool_ids = sorted(matched.keys()) + ci = result["chain_index"] + # Pool 0 is MAINNET, pool 1 is ARBITRUM + assert result["chain_idx"][0] == ci[matched[pool_ids[0]]["chain"]] + assert result["chain_idx"][1] == ci[matched[pool_ids[1]]["chain"]] + + def test_log_fees( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + from quantammsim.calibration.pool_data import encode_tokens + + matched = self._get_matched( + synthetic_daily_grid, synthetic_panel, tmp_path + ) + result = encode_tokens(matched) + pool_ids = sorted(matched.keys()) + for i, pid in enumerate(pool_ids): + expected_fee = matched[pid]["fee"] + np.testing.assert_allclose( + result["log_fees"][i], np.log(expected_fee), rtol=1e-6 + ) + + class TestMatchGridsToPanel: """Test match_grids_to_panel: match grid parquets to panel rows.""" From c7a0c26b22ff1d0ccc0ac2f4a2097c09a761615c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:24:18 +0000 Subject: [PATCH 050/115] feat: add TokenFactoredNoiseHead with additive token decomposition noise_coeffs_i = u[token_a] + u[token_b] + alpha[chain] + beta_fee * log(fee) + delta_i Token effects regularized toward x_token @ Gamma (population prediction from market cap and asset class). Per-pool deltas L2-regularized for partial pooling. Warm-start init decomposes Option C noise_coeffs into token/chain/fee effects via lstsq. predict_new_pool() handles seen tokens (learned u_t), unseen tokens (Gamma fallback), and unseen chains (zero alpha). Comprehensive tests: additivity, regularization, warm-start round-trip, gradient finiteness, new-pool prediction for seen/unseen tokens/chains. --- quantammsim/calibration/heads.py | 264 +++++++++++++++++++++++++++++++ tests/calibration/test_heads.py | 235 +++++++++++++++++++++++++++ 2 files changed, 499 insertions(+) diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 6d6a3763..8b40682c 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -16,6 +16,9 @@ import numpy as np from quantammsim.calibration.loss import K_OBS +from quantammsim.calibration.pool_data import ( + D_TOKEN, K_OBS_REDUCED, _classify_token, _load_token_mcaps, +) # --------------------------------------------------------------------------- @@ -524,6 +527,267 @@ def make_bounds(self, n_pools, k_attr): return [(None, None)] * self.n_params(n_pools, k_attr) +# --------------------------------------------------------------------------- +# TokenFactoredNoiseHead — additive token + chain + fee composition +# --------------------------------------------------------------------------- + + +class TokenFactoredNoiseHead: + """Noise coefficients from additive token + chain + fee composition. + + noise_coeffs_i = u[token_a_i] + u[token_b_i] + alpha[chain_i] + + beta_fee * log(fee_i) + delta_i + + Token effects u_t are regularized toward x_token_t @ Gamma (population + prediction from token covariates). Per-pool deltas are L2-regularized, + controlling the shrinkage between per-pool and population estimates. + + Parameter layout (flat): + [u (n_tokens * k_obs), + Gamma (d_token * k_obs), + alpha (n_chains * k_obs), + beta_fee (k_obs), + delta (n_pools * k_obs)] + """ + + def __init__( + self, + token_a_idx: np.ndarray, + token_b_idx: np.ndarray, + chain_idx: np.ndarray, + log_fees: np.ndarray, + x_token: np.ndarray, + n_tokens: int, + n_chains: int, + token_index: dict, + chain_index: dict, + k_obs: int = K_OBS_REDUCED, + lambda_delta: float = 1.0, + lambda_token: float = 0.1, + lambda_chain: float = 0.1, + lambda_fee: float = 0.01, + mcap_path: str = None, + ): + self.name = "noise" + self.token_a_idx = np.asarray(token_a_idx, dtype=np.int32) + self.token_b_idx = np.asarray(token_b_idx, dtype=np.int32) + self.chain_idx = np.asarray(chain_idx, dtype=np.int32) + self.log_fees = np.asarray(log_fees, dtype=np.float64) + self.x_token = np.asarray(x_token, dtype=np.float64) + self.n_tokens = n_tokens + self.n_chains = n_chains + self.d_token = x_token.shape[1] + self.k_obs = k_obs + self.token_index = dict(token_index) + self.chain_index = dict(chain_index) + self.lambda_delta = lambda_delta + self.lambda_token = lambda_token + self.lambda_chain = lambda_chain + self.lambda_fee = lambda_fee + self._mcap_path = mcap_path + # Pre-convert to JAX for predict() + self._token_a_jax = jnp.array(self.token_a_idx) + self._token_b_jax = jnp.array(self.token_b_idx) + self._chain_jax = jnp.array(self.chain_idx) + self._log_fees_jax = jnp.array(self.log_fees) + self._x_token_jax = jnp.array(self.x_token) + + def n_params(self, n_pools: int, k_attr: int) -> int: + k = self.k_obs + return (self.n_tokens * k # u + + self.d_token * k # Gamma + + self.n_chains * k # alpha + + k # beta_fee + + n_pools * k) # delta + + def _unpack(self, params_slice, n_pools): + k = self.k_obs + idx = 0 + u = params_slice[idx:idx + self.n_tokens * k].reshape(self.n_tokens, k) + idx += self.n_tokens * k + Gamma = params_slice[idx:idx + self.d_token * k].reshape(self.d_token, k) + idx += self.d_token * k + alpha = params_slice[idx:idx + self.n_chains * k].reshape(self.n_chains, k) + idx += self.n_chains * k + beta_fee = params_slice[idx:idx + k] + idx += k + delta = params_slice[idx:idx + n_pools * k].reshape(n_pools, k) + return u, Gamma, alpha, beta_fee, delta + + def _infer_n_pools(self, params_slice): + k = self.k_obs + n_shared = self.n_tokens * k + self.d_token * k + self.n_chains * k + k + return (params_slice.shape[0] - n_shared) // k + + def predict(self, params_slice, pool_idx, x_attr_i): + n_pools = self._infer_n_pools(params_slice) + u, Gamma, alpha, beta_fee, delta = self._unpack(params_slice, n_pools) + ta = self._token_a_jax[pool_idx] + tb = self._token_b_jax[pool_idx] + ch = self._chain_jax[pool_idx] + lf = self._log_fees_jax[pool_idx] + return u[ta] + u[tb] + alpha[ch] + beta_fee * lf + delta[pool_idx] + + def regularization(self, params_slice): + n_pools = self._infer_n_pools(params_slice) + u, Gamma, alpha, beta_fee, delta = self._unpack(params_slice, n_pools) + u_pred = self._x_token_jax @ Gamma + reg_token = self.lambda_token * jnp.sum((u - u_pred) ** 2) + reg_chain = self.lambda_chain * jnp.sum(alpha ** 2) + reg_fee = self.lambda_fee * jnp.sum(beta_fee ** 2) + reg_delta = self.lambda_delta * jnp.sum(delta ** 2) + return reg_token + reg_chain + reg_fee + reg_delta + + def init(self, jdata, warm_start=None): + n_pools = len(jdata.pool_data) + k = self.k_obs + + if warm_start is not None: + # Collect per-pool noise_coeffs from warm_start + noise_all = np.zeros((n_pools, k), dtype=np.float64) + for i, pid in enumerate(jdata.pool_ids): + if pid in warm_start and "noise_coeffs" in warm_start[pid]: + nc = warm_start[pid]["noise_coeffs"] + noise_all[i] = nc[:k] + + # Solve: u[ta_i] + u[tb_i] + alpha[ch_i] + beta_fee * lf_i ≈ noise_all[i] + n_cols = self.n_tokens + self.n_chains + 1 + A = np.zeros((n_pools, n_cols), dtype=np.float64) + for i in range(n_pools): + A[i, self.token_a_idx[i]] = 1.0 + A[i, self.token_b_idx[i]] += 1.0 + A[i, self.n_tokens + self.chain_idx[i]] = 1.0 + A[i, -1] = self.log_fees[i] + + lam_reg = 0.1 + AtA = A.T @ A + lam_reg * np.eye(n_cols) + u_init = np.zeros((self.n_tokens, k)) + alpha_init = np.zeros((self.n_chains, k)) + beta_fee_init = np.zeros(k) + + for j in range(k): + sol = np.linalg.solve(AtA, A.T @ noise_all[:, j]) + u_init[:, j] = sol[:self.n_tokens] + alpha_init[:, j] = sol[self.n_tokens:self.n_tokens + self.n_chains] + beta_fee_init[j] = sol[-1] + + # Delta = residuals + predicted = np.zeros_like(noise_all) + for i in range(n_pools): + predicted[i] = (u_init[self.token_a_idx[i]] + + u_init[self.token_b_idx[i]] + + alpha_init[self.chain_idx[i]] + + beta_fee_init * self.log_fees[i]) + delta_init = noise_all - predicted + + # Gamma from post-hoc regression of u on x_token + Gamma_init, _, _, _ = np.linalg.lstsq( + self.x_token, u_init, rcond=None + ) + else: + # Cold start: pooled OLS for baseline, then decompose + all_x = np.vstack([np.array(pd["x_obs"]) for pd in jdata.pool_data]) + all_y = np.concatenate([np.array(pd["y_obs"]) for pd in jdata.pool_data]) + pooled_coeffs, _, _, _ = np.linalg.lstsq(all_x, all_y, rcond=None) + pooled_coeffs = pooled_coeffs[:k] + + u_init = np.tile(pooled_coeffs / 2.0, (self.n_tokens, 1)) + Gamma_init, _, _, _ = np.linalg.lstsq( + self.x_token, u_init, rcond=None + ) + alpha_init = np.zeros((self.n_chains, k)) + beta_fee_init = np.zeros(k) + delta_init = np.zeros((n_pools, k)) + + return np.concatenate([ + u_init.ravel(), + Gamma_init.ravel(), + alpha_init.ravel(), + beta_fee_init, + delta_init.ravel(), + ]).astype(np.float64) + + def predict_new(self, params_slice, x_attr): + raise ValueError( + "TokenFactoredNoiseHead.predict_new() requires token identifiers. " + "Use predict_new_pool(params, token_a, token_b, chain, fee) instead." + ) + + def predict_new_pool( + self, params_slice, token_a, token_b, chain, fee, n_pools, + ) -> dict: + """Predict noise coefficients for a new pool from token composition. + + Seen tokens use learned u_t. Unseen tokens fall back to x_t @ Gamma. + Unseen chains use alpha = zeros. No delta for new pools. + """ + params_np = np.asarray(params_slice) + u, Gamma, alpha, beta_fee, delta = self._unpack(params_np, n_pools) + u, Gamma, alpha, beta_fee = ( + np.array(u), np.array(Gamma), np.array(alpha), np.array(beta_fee) + ) + mcaps = _load_token_mcaps(self._mcap_path) + + def _get_token_effect(token): + if token in self.token_index: + return u[self.token_index[token]] + x_t = np.zeros(self.d_token) + x_t[0] = 1.0 + cls = _classify_token(token, mcaps) + x_t[1] = cls["log_mcap"] + x_t[2] = cls["is_stable"] + x_t[3] = cls["is_eth_derivative"] + x_t[4] = cls["is_L1_native"] + return x_t @ Gamma + + u_a = _get_token_effect(token_a) + u_b = _get_token_effect(token_b) + + if chain in self.chain_index: + alpha_c = alpha[self.chain_index[chain]] + else: + alpha_c = np.zeros(self.k_obs) + + fee_effect = beta_fee * np.log(fee) + noise_coeffs = u_a + u_b + alpha_c + fee_effect + + return { + "noise_coeffs": noise_coeffs, + "components": { + "token_a": u_a, + "token_b": u_b, + "chain": alpha_c, + "fee": fee_effect, + }, + } + + def unpack_result(self, params_slice, n_pools, k_attr): + params_np = np.asarray(params_slice) + u, Gamma, alpha, beta_fee, delta = self._unpack(params_np, n_pools) + u, Gamma, alpha, beta_fee, delta = ( + np.array(u), np.array(Gamma), np.array(alpha), + np.array(beta_fee), np.array(delta), + ) + # Reconstruct per-pool noise_coeffs + noise_coeffs = np.zeros((n_pools, self.k_obs)) + for i in range(n_pools): + noise_coeffs[i] = (u[self.token_a_idx[i]] + u[self.token_b_idx[i]] + + alpha[self.chain_idx[i]] + + beta_fee * self.log_fees[i] + + delta[i]) + return { + "token_effects": u, + "Gamma": Gamma, + "chain_effects": alpha, + "beta_fee": beta_fee, + "noise_deltas": delta, + "noise_coeffs": noise_coeffs, + } + + def make_bounds(self, n_pools, k_attr): + return [(None, None)] * self.n_params(n_pools, k_attr) + + # --------------------------------------------------------------------------- # MLPNoiseHead — x_attr → Dense(hidden, relu) → Dense(K_OBS) # --------------------------------------------------------------------------- diff --git a/tests/calibration/test_heads.py b/tests/calibration/test_heads.py index 2c891e86..d5bc6449 100644 --- a/tests/calibration/test_heads.py +++ b/tests/calibration/test_heads.py @@ -16,6 +16,7 @@ PerPoolNoiseHead, SharedLinearNoiseHead, ) +from quantammsim.calibration.pool_data import K_OBS_REDUCED # ── Helpers ───────────────────────────────────────────────────────────────── @@ -894,3 +895,237 @@ def test_default_unchanged(self): h = MLPNoiseHead() assert h.k_obs == K_OBS assert h.n_params(3, 5) == 232 + + +# ── TokenFactoredNoiseHead ───────────────────────────────────────────────── + + +def _make_token_factored_head(k_obs=K_OBS_REDUCED): + """Build a TokenFactoredNoiseHead from synthetic 2-pool, 3-token data.""" + from quantammsim.calibration.heads import TokenFactoredNoiseHead + + # Pool 0: (BTC=1, ETH=2) on MAINNET=1, fee=0.003 + # Pool 1: (AAVE=0, ETH=2) on ARBITRUM=0, fee=0.01 + token_a_idx = np.array([1, 0], dtype=np.int32) # BTC, AAVE + token_b_idx = np.array([2, 2], dtype=np.int32) # ETH, ETH + chain_idx = np.array([1, 0], dtype=np.int32) # MAINNET, ARBITRUM + log_fees = np.array([np.log(0.003), np.log(0.01)]) + x_token = np.array([ + [1.0, 20.0, 0.0, 0.0, 0.0], # AAVE: volatile + [1.0, 25.0, 0.0, 0.0, 0.0], # BTC: volatile + [1.0, 26.0, 0.0, 1.0, 1.0], # ETH: eth_derivative + L1_native + ]) + token_index = {"AAVE": 0, "BTC": 1, "ETH": 2} + chain_index = {"ARBITRUM": 0, "MAINNET": 1} + + head = TokenFactoredNoiseHead( + token_a_idx=token_a_idx, + token_b_idx=token_b_idx, + chain_idx=chain_idx, + log_fees=log_fees, + x_token=x_token, + n_tokens=3, + n_chains=2, + token_index=token_index, + chain_index=chain_index, + k_obs=k_obs, + lambda_delta=1.0, + lambda_token=0.1, + lambda_chain=0.1, + lambda_fee=0.01, + ) + return head + + +class TestTokenFactoredNoiseHead: + + def test_is_head(self): + head = _make_token_factored_head() + assert isinstance(head, Head) + + def test_n_params(self): + head = _make_token_factored_head(k_obs=4) + # 3 tokens * 4 + 5 d_token * 4 + 2 chains * 4 + 4 beta_fee + 2 pools * 4 + # = 12 + 20 + 8 + 4 + 8 = 52 + assert head.n_params(2, K_ATTR) == 52 + + def test_predict_returns_k_obs_vector(self): + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + params = jnp.zeros(n_p) + x_attr_i = jnp.zeros(K_ATTR) + result = head.predict(params, 0, x_attr_i) + assert result.shape == (4,) + + def test_predict_additivity(self): + """predict(pool_0) = u[BTC] + u[ETH] + alpha[MAINNET] + + beta_fee * log(0.003) + delta[0]""" + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + + # Build params with known values + params = np.zeros(n_p) + k = 4 # k_obs + # u: (3 tokens, 4) at offset 0 + u_flat = np.array([ + 1.0, 0.0, 0.0, 0.0, # AAVE + 2.0, 0.5, 0.0, 0.0, # BTC + 3.0, 1.0, 0.0, 0.0, # ETH + ]) + params[:12] = u_flat + # Gamma: (5, 4) at offset 12 — skip (doesn't affect predict) + # alpha: (2, 4) at offset 32 + alpha_flat = np.array([ + 0.1, 0.0, 0.0, 0.0, # ARBITRUM + 0.2, 0.0, 0.0, 0.0, # MAINNET + ]) + params[32:40] = alpha_flat + # beta_fee: (4,) at offset 40 + params[40:44] = np.array([0.5, 0.0, 0.0, 0.0]) + # delta: (2, 4) at offset 44 + params[44:48] = np.array([0.05, 0.0, 0.0, 0.0]) # pool 0 delta + + result = head.predict(jnp.array(params), 0, jnp.zeros(K_ATTR)) + + # Expected for pool 0: u[BTC] + u[ETH] + alpha[MAINNET] + # + beta_fee * log(0.003) + delta[0] + expected_0 = 2.0 + 3.0 + 0.2 + 0.5 * np.log(0.003) + 0.05 + np.testing.assert_allclose(float(result[0]), expected_0, rtol=1e-5) + + def test_regularization_nonneg_and_finite(self): + head = _make_token_factored_head() + n_p = head.n_params(2, K_ATTR) + np.random.seed(42) + params = jnp.array(np.random.randn(n_p) * 0.1) + reg = float(head.regularization(params)) + assert np.isfinite(reg) + assert reg >= 0.0 + + def test_regularization_zero_when_perfect(self): + """If u = x_token @ Gamma exactly, delta=0, alpha=0, beta_fee=0, + then only the Gamma-predicted part has zero token reg.""" + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + params = np.zeros(n_p) + # Set Gamma to something, then set u = x_token @ Gamma + np.random.seed(7) + Gamma = np.random.randn(5, 4) * 0.1 + u = head.x_token @ Gamma # (3, 4) + params[:12] = u.ravel() + params[12:32] = Gamma.ravel() + # alpha, beta_fee, delta all zero + reg = float(head.regularization(jnp.array(params))) + # Only lambda_token * 0 + lambda_chain * 0 + lambda_fee * 0 + lambda_delta * 0 + np.testing.assert_allclose(reg, 0.0, atol=1e-10) + + def test_init_cold(self): + head = _make_token_factored_head(k_obs=4) + jdata = _make_fake_jdata_reduced() + init = head.init(jdata) + n_p = head.n_params(N_POOLS, K_ATTR) + assert init.shape == (n_p,) + assert np.all(np.isfinite(init)) + + def test_init_warm_start_roundtrip(self): + """init from warm_start → predict ≈ warm_start noise_coeffs.""" + head = _make_token_factored_head(k_obs=4) + jdata = _make_fake_jdata_reduced() + + # Warm start with known noise coefficients per pool + warm = { + POOL_PREFIXES[0]: {"noise_coeffs": np.array([9.0, 0.5, 0.1, -0.2])}, + POOL_PREFIXES[1]: {"noise_coeffs": np.array([8.5, 0.3, 0.2, -0.1])}, + } + init = head.init(jdata, warm_start=warm) + params = jnp.array(init) + x_attr_dummy = jnp.zeros(K_ATTR) + + for i, pid in enumerate(jdata.pool_ids): + predicted = np.array(head.predict(params, i, x_attr_dummy)) + target = warm[pid]["noise_coeffs"] + # Should approximately recover the warm-start values + # (not exact because the lstsq decomposition is underdetermined + # with 2 pools and 3 tokens) + np.testing.assert_allclose(predicted, target, atol=0.5) + + def test_gradient_finite(self): + """jax.grad of a simple loss at init produces finite gradients.""" + import jax + head = _make_token_factored_head(k_obs=4) + jdata = _make_fake_jdata_reduced() + init = jnp.array(head.init(jdata)) + x_attr_i = jnp.zeros(K_ATTR) + + def loss(p): + c = head.predict(p, 0, x_attr_i) + return jnp.sum(c ** 2) + head.regularization(p) + + grad = jax.grad(loss)(init) + assert grad.shape == init.shape + assert jnp.all(jnp.isfinite(grad)) + + def test_unpack_result_keys(self): + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + np.random.seed(42) + params = np.random.randn(n_p) + result = head.unpack_result(params, 2, K_ATTR) + for key in ["token_effects", "Gamma", "chain_effects", + "beta_fee", "noise_deltas", "noise_coeffs"]: + assert key in result, f"Missing key: {key}" + assert result["token_effects"].shape == (3, 4) + assert result["Gamma"].shape == (5, 4) + assert result["chain_effects"].shape == (2, 4) + assert result["beta_fee"].shape == (4,) + assert result["noise_deltas"].shape == (2, 4) + assert result["noise_coeffs"].shape == (2, 4) + + def test_make_bounds(self): + head = _make_token_factored_head(k_obs=4) + bounds = head.make_bounds(2, K_ATTR) + assert len(bounds) == head.n_params(2, K_ATTR) + assert all(b == (None, None) for b in bounds) + + def test_predict_new_pool_seen_tokens(self): + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + np.random.seed(42) + params = np.random.randn(n_p) * 0.1 + result = head.predict_new_pool( + params, "BTC", "AAVE", "MAINNET", 0.003, n_pools=2 + ) + assert "noise_coeffs" in result + assert "components" in result + nc = result["noise_coeffs"] + assert nc.shape == (4,) or len(nc) == 4 + # Should equal u[BTC] + u[AAVE] + alpha[MAINNET] + beta_fee*log(0.003) + # (no delta for new pool) + comps = result["components"] + reconstructed = (comps["token_a"] + comps["token_b"] + + comps["chain"] + comps["fee"]) + np.testing.assert_allclose(nc, reconstructed, rtol=1e-6) + + def test_predict_new_pool_unseen_token(self): + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + np.random.seed(42) + params = np.random.randn(n_p) * 0.1 + # "LINK" is not in token_index → should fall back to Gamma + result = head.predict_new_pool( + params, "LINK", "ETH", "MAINNET", 0.003, n_pools=2 + ) + assert "noise_coeffs" in result + assert result["noise_coeffs"].shape == (4,) or len(result["noise_coeffs"]) == 4 + + def test_predict_new_pool_unseen_chain(self): + head = _make_token_factored_head(k_obs=4) + n_p = head.n_params(2, K_ATTR) + np.random.seed(42) + params = np.random.randn(n_p) * 0.1 + # "BASE" is not in chain_index → alpha = zeros + result = head.predict_new_pool( + params, "BTC", "ETH", "BASE", 0.003, n_pools=2 + ) + np.testing.assert_allclose( + result["components"]["chain"], np.zeros(4) + ) From 7edf6ecac20ef1537a7fae0db9a69024e7bc53be Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:24:29 +0000 Subject: [PATCH 051/115] feat: integrate token-factored noise into joint calibration pipeline Add prepare_token_factored_data() combining joint data preparation with token encoding. End-to-end tests verify TokenFactoredNoiseHead fits through CalibrationModel with PerPoolHead(cadence) + FixedHead(gas), both cold-start and warm-started from Option C. --- quantammsim/calibration/joint_fit.py | 23 +++++++ tests/calibration/test_joint_fit.py | 98 ++++++++++++++++++++++++++++ 2 files changed, 121 insertions(+) diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py index 765312d7..4a237e9b 100644 --- a/quantammsim/calibration/joint_fit.py +++ b/quantammsim/calibration/joint_fit.py @@ -91,6 +91,29 @@ def prepare_joint_data( ) +def prepare_token_factored_data( + matched: Dict[str, dict], + reduced_x_obs: bool = True, + fix_gas_to_chain: bool = True, +) -> tuple: + """Prepare JointData + token encoding for TokenFactoredNoiseHead. + + Returns (jdata, token_encoding) where token_encoding is the dict from + encode_tokens() containing token/chain structure for constructing the head. + """ + from quantammsim.calibration.pool_data import encode_tokens + + jdata = prepare_joint_data( + matched, + fix_gas_to_chain=fix_gas_to_chain, + reduced_x_obs=reduced_x_obs, + ) + + token_encoding = encode_tokens(matched) + + return jdata, token_encoding + + def pack_joint_params( bias_cad: float, bias_gas: float, diff --git a/tests/calibration/test_joint_fit.py b/tests/calibration/test_joint_fit.py index edb99f4a..004ef93b 100644 --- a/tests/calibration/test_joint_fit.py +++ b/tests/calibration/test_joint_fit.py @@ -190,6 +190,104 @@ def test_shared_noise_predict(self, matched_data): assert len(pred["noise_coeffs"]) == K_OBS +class TestPrepareTokenFactoredData: + def test_returns_jdata_and_encoding(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + + jdata, token_enc = prepare_token_factored_data(matched_data) + assert hasattr(jdata, "pool_data") + assert hasattr(jdata, "x_attr") + assert "token_index" in token_enc + assert "token_a_idx" in token_enc + + def test_reduced_x_obs_by_default(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, _ = prepare_token_factored_data(matched_data) + for pd in jdata.pool_data: + assert pd["x_obs"].shape[1] == K_OBS_REDUCED + + def test_encoding_matches_pools(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + + jdata, enc = prepare_token_factored_data(matched_data) + n_pools = len(jdata.pool_data) + assert len(enc["token_a_idx"]) == n_pools + assert len(enc["token_b_idx"]) == n_pools + assert len(enc["chain_idx"]) == n_pools + assert len(enc["log_fees"]) == n_pools + + +class TestTokenFactoredEndToEnd: + """Full pipeline: TokenFactoredNoiseHead + CalibrationModel + fit.""" + + def test_model_fits_and_converges(self, matched_data): + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data(matched_data) + n_pools = len(jdata.pool_data) + + # Gas: fixed to chain defaults + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_data[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + gas_head = FixedHead("log_gas", np.array(gas_values)) + + # Cadence: per-pool + cad_head = PerPoolHead("log_cadence", default=np.log(12.0)) + + # Noise: token-factored + noise_head = TokenFactoredNoiseHead(k_obs=K_OBS_REDUCED, **enc) + + model = CalibrationModel(cad_head, gas_head, noise_head) + result = model.fit(jdata, maxiter=50) + + assert result["loss"] <= result["init_loss"] + assert "token_effects" in result + assert "noise_coeffs" in result + assert result["noise_coeffs"].shape == (n_pools, K_OBS_REDUCED) + + def test_model_with_warm_start(self, matched_data): + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.per_pool_fit import fit_all_pools + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + # Run Option C first + option_c = fit_all_pools( + matched_data, fix_gas_to_chain=True, reduced=True + ) + + jdata, enc = prepare_token_factored_data(matched_data) + n_pools = len(jdata.pool_data) + + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_data[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead(k_obs=K_OBS_REDUCED, **enc) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=100, warm_start=option_c) + assert result["loss"] <= result["init_loss"] + + class TestPrepareJointDataReduced: """Test prepare_joint_data with reduced_x_obs=True.""" From e8e9031bbfe35825a174532c083a5d2daa9b1005 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:24:38 +0000 Subject: [PATCH 052/115] feat: add reduced x_obs (k_obs=4) pipeline to calibration runner Add run_option_c_reduced() for 4-covariate per-pool fits and run_reduced_joint() for joint MLPNoiseHead with k_obs=4. Wire reduced model into the comparison pipeline with correct x_obs dispatch in compute_per_pool_predictions(). Save reduced Option C results to JSON immediately for downstream use. --- scripts/run_mlp_calibration.py | 132 +++++++++++++++++++++++++++++++-- 1 file changed, 125 insertions(+), 7 deletions(-) diff --git a/scripts/run_mlp_calibration.py b/scripts/run_mlp_calibration.py index 85db2a06..e9b377b6 100644 --- a/scripts/run_mlp_calibration.py +++ b/scripts/run_mlp_calibration.py @@ -104,6 +104,21 @@ def run_option_c(matched): return results +def run_option_c_reduced(matched): + """Per-pool fits with reduced x_obs (4 covariates) and gas fixed.""" + from quantammsim.calibration.per_pool_fit import fit_all_pools + + print(f"\n--- Option C Reduced: per-pool fits ({len(matched)} pools, " + f"4-covariate x_obs, gas fixed) ---") + results = fit_all_pools(matched, fix_gas_to_chain=True, reduced=True) + + losses = [r["loss"] for r in results.values()] + n_conv = sum(1 for r in results.values() if r["converged"]) + print(f" Converged: {n_conv}/{len(results)}") + print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") + return results + + def _build_gas_values(jdata, matched_clean): """Build fixed gas values (log-space) from chain data.""" from quantammsim.calibration.loss import CHAIN_GAS_USD @@ -289,6 +304,53 @@ def run_two_stage_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): return result, stage1_model, stage2_model, jdata +def run_reduced_joint(matched_clean, option_c_clean, hidden=MLP_HIDDEN): + """Joint fit with reduced x_obs (4 cols) to avoid noise-cadence confounding. + + Removes sigma- and fee-dependent features from the noise model's x_obs + so the arb channel (grid + cadence) is the only path for volatility-driven + volume variation. See docs/noise_covariate_design.md for theory. + + Uses LinearHead(cad) + FixedHead(gas) + MLPNoiseHead(k_obs=4). + Cadence warm-started from Option C; noise cold-started (OLS on 4-col x_obs). + """ + import jax.numpy as jnp + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import FixedHead, LinearHead, MLPNoiseHead + from quantammsim.calibration.joint_fit import prepare_joint_data + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata = prepare_joint_data( + matched_clean, drop_chain_dummies=True, + fix_gas_to_chain=True, reduced_x_obs=True) + gas_values = _build_gas_values(jdata, matched_clean) + n_pools = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + + model = CalibrationModel( + cadence_head=LinearHead("cad", alpha=ALPHA_CAD), + gas_head=FixedHead("gas", gas_values), + noise_head=MLPNoiseHead(hidden=hidden, alpha=ALPHA_NOISE, + k_obs=K_OBS_REDUCED), + ) + + n_p = model.n_params(n_pools, k_attr) + print(f"\n--- Reduced x_obs: LinearHead(cad) + MLPNoiseHead(k_obs={K_OBS_REDUCED}, " + f"hidden={hidden}) ({n_pools} pools, {n_p} params) ---") + + # Only warm-start cadence (noise dimension changed 8→4, skip noise warm-start) + warm_cad = {} + for pid in jdata.pool_ids: + if pid in option_c_clean: + warm_cad[pid] = {"cad": option_c_clean[pid]["log_cadence"]} + + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=warm_cad) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + print(f" Converged: {result['converged']}") + + return result, model, jdata + + def _extract_two_stage_per_pool(stage1_model, stage2_model, result, jdata): """Extract per-pool params from two-stage result.""" import jax.numpy as jnp @@ -357,16 +419,22 @@ def _extract_per_pool_params(model, result, jdata): def compute_per_pool_predictions(matched, option_c_results, - model_results): + model_results, reduced_models=None): """Compute V_arb, V_noise, R² per pool for Option C and each joint model. model_results: list of (label, per_pool_params, pool_ids) tuples, where per_pool_params[i] is a dict with log_cadence, log_gas, noise_coeffs. + + reduced_models: list of label strings whose noise_coeffs correspond to + reduced x_obs (4 columns). For those, build_x_obs(reduced=True) is used. """ from quantammsim.calibration.grid_interpolation import interpolate_pool_daily from quantammsim.calibration.pool_data import build_x_obs import jax.numpy as jnp + if reduced_models is None: + reduced_models = [] + pool_ids = sorted(matched.keys()) # Build lookup for each model's per-pool params @@ -388,7 +456,8 @@ def r2(v_arb, v_noise, y): coeffs = entry["coeffs"] day_indices = entry["day_indices"] - x_obs = build_x_obs(panel) + x_obs_full = build_x_obs(panel) + x_obs_red = None # lazy-build only if needed y_obs = panel["log_volume"].values.astype(float) p = { @@ -408,7 +477,7 @@ def r2(v_arb, v_noise, y): coeffs, jnp.float64(rc["log_cadence"]), jnp.float64(np.exp(rc["log_gas"])))) v_arb_c = v_arb_all[day_indices] - v_noise_c = np.exp(x_obs @ rc["noise_coeffs"]) + v_noise_c = np.exp(x_obs_full @ rc["noise_coeffs"]) p["v_arb_c"] = v_arb_c p["v_noise_c"] = v_noise_c p["r2_c"] = r2(v_arb_c, v_noise_c, y_obs) @@ -423,7 +492,15 @@ def r2(v_arb, v_noise, y): coeffs, jnp.float64(mp["log_cadence"]), jnp.float64(np.exp(mp["log_gas"])))) v_arb = v_arb_all[day_indices] - v_noise = np.exp(x_obs @ mp["noise_coeffs"]) + + # Use reduced x_obs for models that were trained with it + if label in reduced_models: + if x_obs_red is None: + x_obs_red = build_x_obs(panel, reduced=True) + v_noise = np.exp(x_obs_red @ mp["noise_coeffs"]) + else: + v_noise = np.exp(x_obs_full @ mp["noise_coeffs"]) + p[f"v_arb_{label}"] = v_arb p[f"v_noise_{label}"] = v_noise p[f"r2_{label}"] = r2(v_arb, v_noise, y_obs) @@ -807,6 +884,28 @@ def save_results_json(predictions, option_c_results, joint_results, output_dir): # ---- Main ---- +def _save_per_pool_results(results, output_path, label="option_c_reduced"): + """Save per-pool fit results to JSON immediately.""" + out = {} + for pid, r in results.items(): + out[pid] = { + "log_cadence": r["log_cadence"], + "log_gas": r["log_gas"], + "noise_coeffs": r["noise_coeffs"].tolist(), + "loss": r["loss"], + "converged": bool(r["converged"]), + "cadence_minutes": r["cadence_minutes"], + "gas_usd": r["gas_usd"], + "chain": r.get("chain", ""), + "fee": r.get("fee", 0), + "tokens": r.get("tokens", ""), + } + os.makedirs(os.path.dirname(output_path), exist_ok=True) + with open(output_path, "w") as f: + json.dump({label: out}, f, indent=2) + print(f" Saved {len(out)} pool results to {output_path}") + + def main(): os.environ.setdefault("JAX_PLATFORMS", "cpu") @@ -816,7 +915,15 @@ def main(): panel, matched = load_and_match() - # Step 1: Option C baseline + # Step 0: 4-covariate per-pool fits (fast, save immediately) + option_c_reduced = run_option_c_reduced(matched) + _save_per_pool_results( + option_c_reduced, + os.path.join(OUTPUT_DIR, "option_c_reduced.json"), + label="option_c_reduced", + ) + + # Step 1: Option C baseline (8-covariate) option_c = run_option_c(matched) # Step 2: Filter pathological pools @@ -834,25 +941,33 @@ def main(): two_stage_result, ts_s1_model, ts_s2_model, _ = run_two_stage_joint( matched_clean, option_c_clean) + # Reduced x_obs: prune sigma/fee features from noise covariates + reduced_result, reduced_model, jdata_reduced = run_reduced_joint( + matched_clean, option_c_clean) + # Step 4: Extract per-pool params from each model linear_pp = _extract_per_pool_params(linear_model, linear_result, jdata) mlp_noise_pp = _extract_per_pool_params(mlp_noise_model, mlp_noise_result, jdata) mlp_full_pp = _extract_per_pool_params(mlp_full_model, mlp_full_result, jdata) two_stage_pp = _extract_two_stage_per_pool( ts_s1_model, ts_s2_model, two_stage_result, jdata) + reduced_pp = _extract_per_pool_params( + reduced_model, reduced_result, jdata_reduced) - method_labels = ["linear", "mlp_noise", "mlp_full", "two_stage"] + method_labels = ["linear", "mlp_noise", "mlp_full", "two_stage", "reduced"] model_results_for_pred = [ ("linear", linear_pp, jdata.pool_ids), ("mlp_noise", mlp_noise_pp, jdata.pool_ids), ("mlp_full", mlp_full_pp, jdata.pool_ids), ("two_stage", two_stage_pp, jdata.pool_ids), + ("reduced", reduced_pp, jdata_reduced.pool_ids), ] # Step 5: Per-pool predictions print("\nComputing per-pool predictions...") predictions = compute_per_pool_predictions( - matched_clean, option_c_clean, model_results_for_pred) + matched_clean, option_c_clean, model_results_for_pred, + reduced_models=["reduced"]) # Step 6: Tables print_pool_table(predictions, method_labels) @@ -862,6 +977,7 @@ def main(): ("mlp_noise", mlp_noise_result), ("mlp_full", mlp_full_result), ("two_stage", two_stage_result), + ("reduced", reduced_result), ]) # Step 7: Plots @@ -873,6 +989,7 @@ def main(): plot_decomposition_pages(predictions, "mlp_noise", "MLP noise (linear cad)", OUTPUT_DIR) plot_decomposition_pages(predictions, "mlp_full", "Full MLP (MLP cad + MLP noise)", OUTPUT_DIR) plot_decomposition_pages(predictions, "two_stage", "Two-stage (linear cad -> MLP noise)", OUTPUT_DIR) + plot_decomposition_pages(predictions, "reduced", "Reduced x_obs (k_obs=4)", OUTPUT_DIR) plot_summary_distributions(predictions, method_labels, OUTPUT_DIR) plot_r2_scatter(predictions, method_labels, OUTPUT_DIR) @@ -884,6 +1001,7 @@ def main(): ("mlp_noise", mlp_noise_result), ("mlp_full", mlp_full_result), ("two_stage", two_stage_result), + ("reduced", reduced_result), ], OUTPUT_DIR) print(f"\n{'='*70}") From 229e38d83eee16790507c78d8e2e4e42e60b0e7f Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:26:39 +0000 Subject: [PATCH 053/115] feat: token-factored calibration script with Phase 0 diagnostic and LOO MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Full pipeline: Phase 0 pooled Ridge diagnostic (baseline vs pool attrs vs token dummies vs full), token-factored fit with lambda_delta sweep, token/chain/delta analysis tables, leave-one-pool-out cross-validation comparing LOO R² to Option C in-sample R², and diagnostic plots. Phase 0 results: token dummies +0.091 vs pool attrs +0.072 above baseline (R²=0.058), confirming compositional structure exists. LOO results: median R²=0.33 vs Option C 0.59, 7/36 wins — the static coefficient prediction bottleneck limits transfer to unseen pools. This motivates lagged cross-pool features. --- scripts/run_token_factored_calibration.py | 611 ++++++++++++++++++++++ 1 file changed, 611 insertions(+) create mode 100644 scripts/run_token_factored_calibration.py diff --git a/scripts/run_token_factored_calibration.py b/scripts/run_token_factored_calibration.py new file mode 100644 index 00000000..0a6a170d --- /dev/null +++ b/scripts/run_token_factored_calibration.py @@ -0,0 +1,611 @@ +"""Token-factored noise calibration: pooled diagnostic + full pipeline. + +Phase 0: Pooled Ridge diagnostic — does cross-pool signal exist? +Phase 1: Token-factored model with lambda_delta sweep +Phase 2: LOO cross-validation +Phase 3: Comparison plots and JSON export +""" + +import json +import os +import sys + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +# ---- Config ---- +PANEL_CACHE = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", +) +GRID_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "pool_grids_v2", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", +) +OPTION_C_LOSS_CUTOFF = 5.0 +JOINT_MAXITER = 5000 +LAMBDA_DELTAS = [0.01, 0.1, 0.5, 1.0, 5.0, 10.0] + + +# ---- Data loading (shared with run_mlp_calibration.py) ---- + + +def load_and_match(): + """Load panel, match to grids.""" + from quantammsim.calibration.pool_data import ( + match_grids_to_panel, + replace_panel_volatility_with_binance, + ) + + panel = pd.read_parquet(PANEL_CACHE) + + if "log_tvl_lag1" not in panel.columns: + panel = panel.sort_values(["pool_id", "date"]).reset_index(drop=True) + panel["log_tvl_lag1"] = panel.groupby("pool_id")["log_tvl"].shift(1) + panel = panel.dropna(subset=["log_tvl_lag1"]).reset_index(drop=True) + + pool_counts = panel.groupby("pool_id").size() + valid = pool_counts[pool_counts >= 10].index + panel = panel[panel["pool_id"].isin(valid)].copy() + + print("Replacing volatility with Binance minute data...") + panel = replace_panel_volatility_with_binance(panel) + + print(f"Panel: {len(panel)} obs, {panel['pool_id'].nunique()} pools, " + f"{panel['date'].min()} to {panel['date'].max()}") + + matched = match_grids_to_panel(GRID_DIR, panel) + print(f"Matched: {len(matched)} pools with grids") + return panel, matched + + +def filter_pathological(matched, option_c): + """Drop pools with high Option C loss.""" + good = {p: r for p, r in option_c.items() if r["loss"] <= OPTION_C_LOSS_CUTOFF} + dropped = set(option_c) - set(good) + matched_clean = {p: matched[p] for p in good if p in matched} + if dropped: + print(f" Dropping {len(dropped)} pools (loss > {OPTION_C_LOSS_CUTOFF}):") + for p in sorted(dropped): + print(f" {p} loss={option_c[p]['loss']:.1f}") + return matched_clean, good + + +# ---- Phase 0: Pooled Ridge Diagnostic ---- + + +def run_phase0_diagnostic(matched, option_c): + """Pooled Ridge + token-dummy Ridge — go/no-go gate.""" + from sklearn.linear_model import RidgeCV + + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + build_pool_attributes, build_x_obs, encode_tokens, _parse_tokens, + ) + import jax.numpy as jnp + + print("\n" + "=" * 70) + print("Phase 0: Pooled Ridge Diagnostic — cross-pool signal?") + print("=" * 70) + + pool_ids = sorted(matched.keys()) + X_attr, attr_names, _ = build_pool_attributes(matched) + pool_idx_map = {pid: i for i, pid in enumerate(pool_ids)} + enc = encode_tokens(matched) + + all_x, all_y, all_pool_attrs, all_token_dummies = [], [], [], [] + + for pid in pool_ids: + entry = matched[pid] + oc = option_c[pid] + coeffs = entry["coeffs"] + + v_arb_all = np.array(interpolate_pool_daily( + coeffs, jnp.float64(oc["log_cadence"]), + jnp.float64(np.exp(oc["log_gas"])))) + v_arb = v_arb_all[entry["day_indices"]] + + x_obs = build_x_obs(entry["panel"], reduced=True) + y_obs = entry["panel"]["log_volume"].values.astype(float) + y_residual = y_obs - np.log(np.maximum(v_arb, 1e-6)) + + all_x.append(x_obs) + all_y.append(y_residual) + + # Broadcast pool attrs to each obs + x_attr_row = X_attr[pool_idx_map[pid]] + all_pool_attrs.append(np.tile(x_attr_row, (len(x_obs), 1))) + + # Token dummies: one-hot for each token in the pool + n_obs = len(x_obs) + dummies = np.zeros((n_obs, enc["n_tokens"]), dtype=np.float64) + toks = _parse_tokens(entry["tokens"]) + for t in toks[:2]: + if t in enc["token_index"]: + dummies[:, enc["token_index"][t]] = 1.0 + all_token_dummies.append(dummies) + + X_obs = np.vstack(all_x) + y_combined = np.concatenate(all_y) + X_pool_attrs = np.vstack(all_pool_attrs) + X_token_dummies = np.vstack(all_token_dummies) + + # Model 1: x_obs + pool_attrs → pooled Ridge + X_combined = np.column_stack([X_obs, X_pool_attrs]) + model1 = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model1.fit(X_combined, y_combined) + r2_pooled = model1.score(X_combined, y_combined) + print(f" Pooled Ridge (x_obs + pool attrs): R² = {r2_pooled:.4f}") + + # Model 2: x_obs + token_dummies → token-dummy Ridge + X_token = np.column_stack([X_obs, X_token_dummies]) + model2 = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model2.fit(X_token, y_combined) + r2_token = model2.score(X_token, y_combined) + print(f" Token-dummy Ridge (x_obs + token dummies): R² = {r2_token:.4f}") + + # Model 3: x_obs + token_dummies + chain_dummies + log_fee + chain_dummies = np.zeros((len(y_combined), enc["n_chains"]), dtype=np.float64) + log_fees = np.zeros((len(y_combined), 1), dtype=np.float64) + offset = 0 + for pid in pool_ids: + n_obs = len(matched[pid]["day_indices"]) + ci = enc["chain_idx"][pool_idx_map[pid]] + chain_dummies[offset:offset + n_obs, ci] = 1.0 + log_fees[offset:offset + n_obs, 0] = enc["log_fees"][pool_idx_map[pid]] + offset += n_obs + + X_full = np.column_stack([X_obs, X_token_dummies, chain_dummies, log_fees]) + model3 = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model3.fit(X_full, y_combined) + r2_full = model3.score(X_full, y_combined) + print(f" Full Ridge (x_obs + tokens + chains + fee): R² = {r2_full:.4f}") + + # Baseline: x_obs only + model_base = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model_base.fit(X_obs, y_combined) + r2_base = model_base.score(X_obs, y_combined) + print(f" Baseline (x_obs only): R² = {r2_base:.4f}") + + print(f"\n Signal above baseline:") + print(f" Pool attrs: +{r2_pooled - r2_base:.4f}") + print(f" Token dummies: +{r2_token - r2_base:.4f}") + print(f" Full (tok+ch+fee): +{r2_full - r2_base:.4f}") + + if r2_full - r2_base < 0.01: + print("\n WARNING: Very weak cross-pool signal. " + "Token factoring may not improve over per-pool fits.") + + return { + "r2_baseline": r2_base, + "r2_pool_attrs": r2_pooled, + "r2_token_dummies": r2_token, + "r2_full": r2_full, + "n_obs_total": len(y_combined), + "n_pools": len(pool_ids), + "n_tokens": enc["n_tokens"], + "n_chains": enc["n_chains"], + } + + +# ---- Phase 1: Token-Factored Model ---- + + +def _build_gas_values(jdata, matched_clean): + """Build fixed gas values (log-space) from chain data.""" + from quantammsim.calibration.loss import CHAIN_GAS_USD + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_clean[pid]["chain"] + gas_usd = CHAIN_GAS_USD.get(chain, 1.0) + gas_values.append(np.log(max(gas_usd, 1e-6))) + return np.array(gas_values) + + +def run_token_factored(matched_clean, option_c_clean, lambda_delta=1.0): + """Fit TokenFactoredNoiseHead with PerPoolHead(cadence) + FixedHead(gas).""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data(matched_clean) + n_pools = len(jdata.pool_data) + + gas_values = _build_gas_values(jdata, matched_clean) + gas_head = FixedHead("log_gas", gas_values) + cad_head = PerPoolHead("log_cadence", default=np.log(12.0)) + noise_head = TokenFactoredNoiseHead( + k_obs=K_OBS_REDUCED, + lambda_delta=lambda_delta, + **enc, + ) + + model = CalibrationModel(cad_head, gas_head, noise_head) + n_p = model.n_params(n_pools, jdata.x_attr.shape[1]) + print(f"\n--- Token-factored (lambda_delta={lambda_delta}) ---") + print(f" {n_pools} pools, {enc['n_tokens']} tokens, " + f"{enc['n_chains']} chains, {n_p} params") + + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + print(f" Converged: {result['converged']}") + + return result, model, jdata, enc + + +# ---- Phase 2: Analysis & Visualization ---- + + +def print_token_effects(result, enc): + """Print token effect table.""" + u = result["token_effects"] + Gamma = result["Gamma"] + x_token = enc["x_token"] + token_index = enc["token_index"] + inv_index = {v: k for k, v in token_index.items()} + + u_pred = x_token @ Gamma # population prediction + + print(f"\n{'='*70}") + print("Token effects (u_t) vs population prediction (x_t @ Gamma)") + print(f"{'='*70}") + print(f"{'Token':<12} {'u[0]':>8} {'pred[0]':>8} {'delta[0]':>8} " + f"{'u[1]':>8} {'pred[1]':>8}") + print("-" * 60) + for idx in range(len(inv_index)): + name = inv_index[idx] + print(f"{name:<12} {u[idx,0]:>8.3f} {u_pred[idx,0]:>8.3f} " + f"{u[idx,0]-u_pred[idx,0]:>8.3f} " + f"{u[idx,1]:>8.3f} {u_pred[idx,1]:>8.3f}") + + +def print_chain_effects(result, enc): + """Print chain effect table.""" + alpha = result["chain_effects"] + chain_index = enc["chain_index"] + inv_index = {v: k for k, v in chain_index.items()} + + print(f"\n{'='*70}") + print("Chain effects (alpha)") + print(f"{'='*70}") + k = alpha.shape[1] + header = f"{'Chain':<12}" + "".join(f" {'a['+str(j)+']':>8}" for j in range(k)) + print(header) + print("-" * (12 + 9 * k)) + for idx in range(len(inv_index)): + name = inv_index[idx] + vals = " ".join(f"{alpha[idx, j]:>8.3f}" for j in range(k)) + print(f"{name:<12} {vals}") + + +def print_delta_analysis(result, enc, jdata, matched_clean): + """Print per-pool delta analysis.""" + delta = result["noise_deltas"] + pool_ids = jdata.pool_ids + token_index = enc["token_index"] + inv_token = {v: k for k, v in token_index.items()} + + print(f"\n{'='*70}") + print("Per-pool deltas (unexplained residual)") + print(f"{'='*70}") + print(f"{'Pool':<24} {'Tokens':<16} {'Chain':<10} " + f"{'|delta|':>8} {'delta[0]':>8}") + print("-" * 70) + + delta_norms = np.linalg.norm(delta, axis=1) + order = np.argsort(-delta_norms) + for i in order: + pid = pool_ids[i] + entry = matched_clean[pid] + print(f"{pid[:24]:<24} {entry['tokens']:<16} {entry['chain']:<10} " + f"{delta_norms[i]:>8.3f} {delta[i, 0]:>8.3f}") + + +def run_lambda_sweep(matched_clean, option_c_clean): + """Sweep lambda_delta values and report loss + delta shrinkage.""" + print(f"\n{'='*70}") + print("Lambda_delta sweep") + print(f"{'='*70}") + print(f"{'lambda':>10} {'loss':>10} {'delta_norm':>12} {'mean_|d|':>10}") + print("-" * 45) + + results = [] + for lam in LAMBDA_DELTAS: + result, model, jdata, enc = run_token_factored( + matched_clean, option_c_clean, lambda_delta=lam) + delta = result["noise_deltas"] + delta_norm = float(np.linalg.norm(delta)) + mean_abs_d = float(np.mean(np.abs(delta))) + print(f"{lam:>10.2f} {result['loss']:>10.4f} " + f"{delta_norm:>12.4f} {mean_abs_d:>10.4f}") + results.append({ + "lambda_delta": lam, + "loss": result["loss"], + "delta_norm": delta_norm, + "mean_abs_delta": mean_abs_d, + "converged": result["converged"], + }) + + return results + + +# ---- Phase 3: LOO Cross-Validation ---- + + +def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): + """Leave-one-pool-out cross-validation via predict_new_pool.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.pool_data import K_OBS_REDUCED, build_x_obs, _parse_tokens + import jax.numpy as jnp + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + print(f"\n{'='*70}") + print(f"LOO Cross-Validation (lambda_delta={lambda_delta})") + print(f"{'='*70}") + + loo_results = [] + for hold_out_pid in pool_ids: + # Build training set without hold-out pool + train_matched = {p: matched_clean[p] for p in pool_ids if p != hold_out_pid} + train_oc = {p: option_c_clean[p] for p in pool_ids if p != hold_out_pid} + + if len(train_matched) < 3: + continue + + # Fit on training set + jdata, enc = prepare_token_factored_data(train_matched) + gas_values = _build_gas_values(jdata, train_matched) + + noise_head = TokenFactoredNoiseHead( + k_obs=K_OBS_REDUCED, + lambda_delta=lambda_delta, + **enc, + ) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", gas_values), + noise_head, + ) + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=train_oc) + + # Extract noise params and predict for hold-out pool + n_train = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + (_, _), (_, _), (ns, ne) = model._head_slices(n_train, k_attr) + noise_params = result["params_flat"][ns:ne] + + ho_entry = matched_clean[hold_out_pid] + toks = _parse_tokens(ho_entry["tokens"]) + ho_pred = noise_head.predict_new_pool( + noise_params, toks[0], toks[1], + ho_entry["chain"], ho_entry["fee"], + n_pools=n_train, + ) + + # Evaluate hold-out R² + ho_panel = ho_entry["panel"] + x_obs_ho = build_x_obs(ho_panel, reduced=True) + y_obs_ho = ho_panel["log_volume"].values.astype(float) + + # Use Option C cadence for the hold-out pool (not predicting cadence) + oc_ho = option_c_clean[hold_out_pid] + v_arb_all = np.array(interpolate_pool_daily( + ho_entry["coeffs"], + jnp.float64(oc_ho["log_cadence"]), + jnp.float64(np.exp(oc_ho["log_gas"])), + )) + v_arb = v_arb_all[ho_entry["day_indices"]] + v_noise = np.exp(x_obs_ho @ ho_pred["noise_coeffs"]) + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y_obs_ho) ** 2) + ss_tot = np.sum((y_obs_ho - y_obs_ho.mean()) ** 2) + r2_loo = 1 - ss_res / max(ss_tot, 1e-10) + + # Compare with Option C in-sample R² + v_noise_c = np.exp(build_x_obs(ho_panel, reduced=True) @ oc_ho["noise_coeffs"][:K_OBS_REDUCED]) + log_pred_c = np.log(np.maximum(v_arb + v_noise_c, 1e-6)) + ss_res_c = np.sum((log_pred_c - y_obs_ho) ** 2) + r2_c = 1 - ss_res_c / max(ss_tot, 1e-10) + + loo_results.append({ + "pool_id": hold_out_pid, + "r2_loo": r2_loo, + "r2_option_c": r2_c, + "tokens": ho_entry["tokens"], + "chain": ho_entry["chain"], + }) + + print(f" {hold_out_pid[:16]} ({ho_entry['tokens']:<14}) " + f"R²_LOO={r2_loo:.3f} R²_C={r2_c:.3f} " + f"{'BETTER' if r2_loo > r2_c else 'worse'}") + + if loo_results: + r2s_loo = [r["r2_loo"] for r in loo_results] + r2s_c = [r["r2_option_c"] for r in loo_results] + n_better = sum(1 for r in loo_results if r["r2_loo"] > r["r2_option_c"]) + print(f"\n LOO median R²: {np.median(r2s_loo):.4f} " + f"(Option C: {np.median(r2s_c):.4f})") + print(f" LOO wins: {n_better}/{len(loo_results)}") + + return loo_results + + +# ---- Plots ---- + + +def plot_lambda_sweep(sweep_results, output_dir): + """Plot loss and delta norm vs lambda_delta.""" + lambdas = [r["lambda_delta"] for r in sweep_results] + losses = [r["loss"] for r in sweep_results] + delta_norms = [r["delta_norm"] for r in sweep_results] + + fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5)) + + ax1.semilogx(lambdas, losses, "o-", color="steelblue") + ax1.set_xlabel("lambda_delta") + ax1.set_ylabel("Loss") + ax1.set_title("Loss vs lambda_delta") + + ax2.semilogx(lambdas, delta_norms, "o-", color="orangered") + ax2.set_xlabel("lambda_delta") + ax2.set_ylabel("||delta||") + ax2.set_title("Delta norm vs lambda_delta") + + fig.tight_layout() + out = os.path.join(output_dir, "lambda_sweep.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_token_effects(result, enc, output_dir): + """Bar chart of token effects (intercept coefficient).""" + u = result["token_effects"] + token_index = enc["token_index"] + inv_index = {v: k for k, v in token_index.items()} + names = [inv_index[i] for i in range(len(inv_index))] + + fig, ax = plt.subplots(figsize=(max(8, len(names) * 0.5), 5)) + x = np.arange(len(names)) + ax.bar(x, u[:, 0], color="steelblue", alpha=0.8) + ax.set_xticks(x) + ax.set_xticklabels(names, rotation=45, ha="right", fontsize=8) + ax.set_ylabel("u_t[0] (intercept effect)") + ax.set_title("Token effects on noise intercept") + ax.axhline(0, color="black", linewidth=0.5, linestyle="--") + fig.tight_layout() + out = os.path.join(output_dir, "token_effects.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_loo_scatter(loo_results, output_dir): + """Scatter: Option C R² vs LOO R².""" + if not loo_results: + return + + r2_c = [r["r2_option_c"] for r in loo_results] + r2_loo = [r["r2_loo"] for r in loo_results] + + fig, ax = plt.subplots(figsize=(7, 6)) + ax.scatter(r2_c, r2_loo, alpha=0.7, s=40, edgecolors="k", linewidth=0.5) + lo = min(min(r2_c), min(r2_loo)) + hi = max(max(r2_c), max(r2_loo)) + margin = (hi - lo) * 0.05 + 0.01 + ax.plot([lo - margin, hi + margin], [lo - margin, hi + margin], + "k--", alpha=0.3, linewidth=1) + ax.set_xlabel("Option C R² (in-sample)") + ax.set_ylabel("Token-factored R² (LOO)") + ax.set_title("LOO Cross-Validation: Token-Factored vs Option C") + + n_better = sum(1 for c, l in zip(r2_c, r2_loo) if l > c) + ax.text(0.05, 0.95, f"LOO wins: {n_better}/{len(r2_c)}", + transform=ax.transAxes, fontsize=10, va="top") + + fig.tight_layout() + out = os.path.join(output_dir, "loo_scatter.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +# ---- Main ---- + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Token-Factored Noise Calibration") + print("=" * 70) + + panel, matched = load_and_match() + + # Step 1: Option C baseline (reduced x_obs) + from quantammsim.calibration.per_pool_fit import fit_all_pools + print(f"\n--- Option C Reduced: per-pool fits ({len(matched)} pools) ---") + option_c = fit_all_pools(matched, fix_gas_to_chain=True, reduced=True) + losses = [r["loss"] for r in option_c.values()] + print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") + + # Step 2: Filter pathological pools + matched_clean, option_c_clean = filter_pathological(matched, option_c) + + # Phase 0: Diagnostic + diag = run_phase0_diagnostic(matched_clean, option_c_clean) + + # Phase 1: Token-factored model (default lambda) + result, model, jdata, enc = run_token_factored( + matched_clean, option_c_clean, lambda_delta=1.0) + + # Analysis + print_token_effects(result, enc) + print_chain_effects(result, enc) + print_delta_analysis(result, enc, jdata, matched_clean) + + # Lambda sweep + sweep_results = run_lambda_sweep(matched_clean, option_c_clean) + + # Phase 2: LOO cross-validation + loo_results = run_loo_validation( + matched_clean, option_c_clean, lambda_delta=1.0) + + # Phase 3: Plots & export + print("\nGenerating plots...") + os.makedirs(OUTPUT_DIR, exist_ok=True) + + plot_lambda_sweep(sweep_results, OUTPUT_DIR) + plot_token_effects(result, enc, OUTPUT_DIR) + plot_loo_scatter(loo_results, OUTPUT_DIR) + + # JSON export + export = { + "phase0_diagnostic": diag, + "token_factored": { + "loss": result["loss"], + "init_loss": result["init_loss"], + "converged": result["converged"], + "n_pools": result["n_pools"], + "n_tokens": enc["n_tokens"], + "n_chains": enc["n_chains"], + "token_index": enc["token_index"], + "chain_index": enc["chain_index"], + "token_effects": result["token_effects"].tolist(), + "Gamma": result["Gamma"].tolist(), + "chain_effects": result["chain_effects"].tolist(), + "beta_fee": result["beta_fee"].tolist(), + "noise_deltas": result["noise_deltas"].tolist(), + "noise_coeffs": result["noise_coeffs"].tolist(), + }, + "lambda_sweep": sweep_results, + "loo_results": loo_results, + } + json_path = os.path.join(OUTPUT_DIR, "token_factored_results.json") + with open(json_path, "w") as f: + json.dump(export, f, indent=2, default=str) + print(f" Saved: {json_path}") + + print(f"\n{'='*70}") + print(f"Done. Output in: {OUTPUT_DIR}") + + +if __name__ == "__main__": + main() From 5db391905d309385fd6d8352a3c1f5507a113d4a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 11:28:31 +0000 Subject: [PATCH 054/115] feat: support calibrated noise model in Optuna parameter tuning Load per-pool noise_coeffs from calibration JSON, derive arb_frequency from calibrated log_cadence, and pick up token pair, fee, and gas from pool metadata. Supports both 4-covariate (reduced) and 8-covariate (full) noise coefficient formats. Add --n-eval-points flag for evaluation sub-window control. --- experiments/tune_reclamm_params.py | 76 +++++++++++++++++++++++++++++- 1 file changed, 74 insertions(+), 2 deletions(-) diff --git a/experiments/tune_reclamm_params.py b/experiments/tune_reclamm_params.py index 49699230..e2448bfe 100644 --- a/experiments/tune_reclamm_params.py +++ b/experiments/tune_reclamm_params.py @@ -13,9 +13,16 @@ # More trials, custom fees python experiments/tune_reclamm_params.py --n-trials 200 --fees 0.005 + + # With calibrated 8-covariate noise model and arb frequency from calibration + python experiments/tune_reclamm_params.py --noise-model calibrated \ + --noise-params-json results/mlp_calibration/option_c_reduced.json \ + --noise-pool-id 0x9d1fcf346ea1b0 """ import argparse +import json +import math from quantammsim.runners.jax_runners import train_on_historic_data PARAMETER_CONFIG = { @@ -39,6 +46,15 @@ def main(): choices=["geometric", "constant_arc_length"]) parser.add_argument("--centeredness-scaling", action="store_true") parser.add_argument("--noise-trader-ratio", type=float, default=0.0) + parser.add_argument("--noise-model", default=None, + choices=["ratio", "loglinear", "calibrated", "arb_only"], + help="Noise volume model (default: ratio via noise-trader-ratio)") + parser.add_argument("--noise-params-json", default=None, + help="JSON file with per-pool calibration results") + parser.add_argument("--noise-pool-id", default=None, + help="Pool ID to load noise params for (from --noise-params-json)") + parser.add_argument("--arb-frequency", type=int, default=None, + help="Arb frequency in minutes (default: from calibrated cadence or 1)") parser.add_argument("--start-date", default="2024-06-01 00:00:00") parser.add_argument("--end-date", default="2025-01-01 00:00:00", help="End of training / start of test") @@ -49,6 +65,8 @@ def main(): help="Validation holdout fraction (default: 0.2, use 0 to disable)") parser.add_argument("--overfitting-penalty", type=float, default=None, help="Overfitting penalty weight (default: 0.2)") + parser.add_argument("--n-eval-points", type=int, default=None, + help="Number of evaluation sub-windows (default: 20, use 1 for full-window)") args = parser.parse_args() learn_speed = args.interpolation == "constant_arc_length" @@ -56,19 +74,72 @@ def main(): if learn_speed: param_config.update(ARC_LENGTH_SPEED_CONFIG) + # --- Noise model setup --- + pool_tokens = ["AAVE", "ETH"] # default + noise_fp = {"noise_trader_ratio": args.noise_trader_ratio} + if args.noise_model: + noise_fp["noise_model"] = args.noise_model + if args.noise_params_json and args.noise_pool_id: + with open(args.noise_params_json) as f: + all_results = json.load(f) + # Support both {"option_c_reduced": {pid: ...}} and {pid: ...} formats + pool_results = all_results + for key in all_results: + if isinstance(all_results[key], dict) and args.noise_pool_id in all_results[key]: + pool_results = all_results[key] + break + pool_data = pool_results[args.noise_pool_id] + coeffs = pool_data["noise_coeffs"] + if len(coeffs) == 8: + # Full 8-covariate model: [intercept, log_tvl, log_sigma, + # tvl*sigma, tvl*fee, sigma*fee, dow_sin, dow_cos] + noise_fp["reclamm_noise_params"] = { + f"c_{i}": c for i, c in enumerate(coeffs) + } + elif len(coeffs) == 4: + # Reduced 4-covariate model: [intercept, log_tvl, dow_sin, dow_cos] + # Map to c_0, c_1, c_6, c_7 (sigma/fee terms stay at 0) + noise_fp["reclamm_noise_params"] = { + "c_0": coeffs[0], "c_1": coeffs[1], + "c_6": coeffs[2], "c_7": coeffs[3], + } + else: + raise ValueError(f"Expected 4 or 8 noise_coeffs, got {len(coeffs)}") + # Derive arb_frequency from calibrated cadence if not explicitly set + if args.arb_frequency is None: + log_cad = pool_data["log_cadence"] + args.arb_frequency = max(1, round(math.exp(log_cad))) + print(f" arb_frequency={args.arb_frequency} " + f"(from log_cadence={log_cad:.2f}, " + f"cadence={math.exp(log_cad):.1f} min)") + # Use pool's fee and gas from calibration as defaults + if "fee" in pool_data: + args.fees = pool_data["fee"] + if "gas_usd" in pool_data: + args.gas_cost = pool_data["gas_usd"] + # Pick up token pair from calibration + # Map on-chain names (WETH, WBTC) to data-file names (ETH, BTC) + _TOKEN_MAP = {"WETH": "ETH", "WBTC": "BTC"} + if "tokens" in pool_data: + pool_tokens = [ + _TOKEN_MAP.get(t, t) for t in pool_data["tokens"].split(",") + ] + print(f" tokens={pool_tokens}, fee={args.fees}, gas={args.gas_cost}") + fp = { "rule": "reclamm", - "tokens": ["AAVE", "ETH"], + "tokens": pool_tokens, "startDateString": args.start_date, "endDateString": args.end_date, "endTestDateString": args.end_test_date, "initial_pool_value": 1_000_000.0, "do_arb": True, + **({"arb_frequency": args.arb_frequency} if args.arb_frequency is not None else {}), "fees": args.fees, "gas_cost": args.gas_cost, "arb_fees": 0.0, "protocol_fee_split": 0.5, - "noise_trader_ratio": args.noise_trader_ratio, + **noise_fp, "return_val": args.objective, "reclamm_interpolation_method": args.interpolation, "reclamm_centeredness_scaling": args.centeredness_scaling, @@ -86,6 +157,7 @@ def main(): "multi_objective": False, "parameter_config": param_config, **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), + **({"n_evaluation_points": args.n_eval_points} if args.n_eval_points is not None else {}), }, }, } From 5398c3b5483d7600c66909767600467f039b0344 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 12:20:34 +0000 Subject: [PATCH 055/115] feat: add token canonicalization and cross-pool lagged volume features MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Token canonicalization (_CANON_MAP) maps wrapped/derivative tokens to their base symbols (WETH→ETH, waBasWETH→ETH, WBTC→BTC, etc.), reducing the token graph from ~32 to ~22 unique tokens and thickening peer groups for cross-pool information sharing. Cross-pool lag features (build_cross_pool_x_obs, K_OBS_CROSS=7) enrich observation-level covariates with lagged peer volume averages for token A, token B, and chain — so daily noise predictions can adapt to market conditions without autoregressive cold-start issues. --- quantammsim/calibration/heads.py | 8 +- quantammsim/calibration/pool_data.py | 176 +++++++++++++++- tests/calibration/test_pool_data.py | 303 +++++++++++++++++++++++++++ 3 files changed, 481 insertions(+), 6 deletions(-) diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index 8b40682c..ff86dfc9 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -17,7 +17,8 @@ from quantammsim.calibration.loss import K_OBS from quantammsim.calibration.pool_data import ( - D_TOKEN, K_OBS_REDUCED, _classify_token, _load_token_mcaps, + D_TOKEN, K_OBS_REDUCED, _canonicalize_token, _classify_token, + _load_token_mcaps, ) @@ -720,6 +721,7 @@ def predict_new_pool( Seen tokens use learned u_t. Unseen tokens fall back to x_t @ Gamma. Unseen chains use alpha = zeros. No delta for new pools. + Input token names are canonicalized before lookup. """ params_np = np.asarray(params_slice) u, Gamma, alpha, beta_fee, delta = self._unpack(params_np, n_pools) @@ -728,6 +730,10 @@ def predict_new_pool( ) mcaps = _load_token_mcaps(self._mcap_path) + # Canonicalize input tokens + token_a = _canonicalize_token(token_a) + token_b = _canonicalize_token(token_b) + def _get_token_effect(token): if token in self.token_index: return u[self.token_index[token]] diff --git a/quantammsim/calibration/pool_data.py b/quantammsim/calibration/pool_data.py index 18ab8212..3b05bbb7 100644 --- a/quantammsim/calibration/pool_data.py +++ b/quantammsim/calibration/pool_data.py @@ -19,6 +19,9 @@ K_OBS = 8 # observation-level covariates K_OBS_REDUCED = 4 # [intercept, log_tvl_lag1, dow_sin, dow_cos] +K_OBS_CROSS = 7 # [intercept, log_tvl_lag1, dow_sin, dow_cos, + # cross_vol_token_a_{t-1}, cross_vol_token_b_{t-1}, + # cross_vol_chain_{t-1}] # Default path for cached token market caps _MCAP_PATH = os.path.join( @@ -348,6 +351,24 @@ def match_grids_to_panel( return matched +# Token canonicalization — map wrapped/LST variants to base tokens +_CANON_MAP = { + "WETH": "ETH", "waBasWETH": "ETH", "waEthLidoWETH": "ETH", + "waEthLidowstETH": "wstETH", "waGnowstETH": "wstETH", + "waBasUSDC": "USDC", "scUSD": "USDC", "USDC.e": "USDC", + "USDbC": "USDC", "waEthUSDC": "USDC", + "sDAI": "DAI", "WXDAI": "DAI", + "WBTC": "BTC", "cbBTC": "BTC", + "stS": "S", "wS": "S", + "waGnoGNO": "GNO", "osGNO": "GNO", +} + + +def _canonicalize_token(symbol: str) -> str: + """Map wrapped/derivative token to its canonical base symbol.""" + return _CANON_MAP.get(symbol, symbol) + + # Token classification for token-factored model _ETH_DERIVATIVES = { "WETH", "ETH", "wstETH", "stETH", "rETH", "cbETH", @@ -374,11 +395,16 @@ def _classify_token(symbol: str, mcaps: dict) -> dict: def encode_tokens( matched: Dict[str, dict], mcap_path: str = None, + canonicalize: bool = True, ) -> dict: """Build token index, per-pool token assignments, and token covariate matrix. Iterates over pools in sorted key order (same ordering as build_pool_attributes). + When canonicalize=True (default), wrappd/derivative tokens are mapped to + their canonical base symbol via _CANON_MAP before building the index. + Raw symbols are still used for market cap lookup. + Returns dict with: token_index: dict[str, int] — symbol -> integer index (sorted alphabetically) token_a_idx: np.ndarray (n_pools,) — index of token A for each pool @@ -394,14 +420,19 @@ def encode_tokens( pool_ids = sorted(matched.keys()) n_pools = len(pool_ids) - # Collect all tokens and chains + # Collect all tokens and chains; store per-pool canonical pairs all_tokens = set() all_chains = set() + pool_canon_toks = [] # (canon_a, canon_b) per pool in sorted order for pid in pool_ids: entry = matched[pid] toks = _parse_tokens(entry["tokens"]) - all_tokens.update(toks[:2]) + raw_a, raw_b = toks[0], toks[1] + canon_a = _canonicalize_token(raw_a) if canonicalize else raw_a + canon_b = _canonicalize_token(raw_b) if canonicalize else raw_b + all_tokens.update([canon_a, canon_b]) all_chains.add(entry["chain"]) + pool_canon_toks.append((canon_a, canon_b)) # Build sorted indices token_list = sorted(all_tokens) @@ -420,9 +451,9 @@ def encode_tokens( for i, pid in enumerate(pool_ids): entry = matched[pid] - toks = _parse_tokens(entry["tokens"]) - token_a_idx[i] = token_index[toks[0]] - token_b_idx[i] = token_index[toks[1]] + canon_a, canon_b = pool_canon_toks[i] + token_a_idx[i] = token_index[canon_a] + token_b_idx[i] = token_index[canon_b] chain_idx[i] = chain_index[entry["chain"]] log_fees[i] = np.log(entry["fee"]) @@ -497,6 +528,141 @@ def build_x_obs(panel_rows: pd.DataFrame, reduced: bool = False) -> np.ndarray: return x +def build_cross_pool_x_obs( + panel_rows: pd.DataFrame, + matched: Dict[str, dict], + pool_id: str, + exclude_pool: Optional[str] = None, + canonicalize: bool = True, +) -> np.ndarray: + """Build x_obs with cross-pool lagged volume features. + + Columns 0-3: same as build_x_obs(reduced=True) + Column 4: mean log_volume at t-1 across pools sharing token A (excl self) + Column 5: mean log_volume at t-1 across pools sharing token B (excl self) + Column 6: mean log_volume at t-1 across pools on same chain (excl self) + + The first observation (day 0) is dropped because there is no lag available. + + Args: + panel_rows: DataFrame for this pool + matched: full matched dict (all pools) + pool_id: this pool's key in matched (prefix) + exclude_pool: optional pool to exclude from peer averages (for LOO) + canonicalize: if True, canonicalize tokens before peer matching + + Returns: + (n_obs - 1, K_OBS_CROSS) array + """ + # Get this pool's tokens and chain + entry = matched[pool_id] + toks = _parse_tokens(entry["tokens"]) + tok_a_raw, tok_b_raw = toks[0], toks[1] + tok_a = _canonicalize_token(tok_a_raw) if canonicalize else tok_a_raw + tok_b = _canonicalize_token(tok_b_raw) if canonicalize else tok_b_raw + this_chain = entry["chain"] + + # Build peer sets: token→set of pool_ids, chain→set of pool_ids + token_peers = {} # canonical_token → set of (prefix, panel_df) + chain_peers = {} # chain → set of (prefix, panel_df) + all_pool_ids = sorted(matched.keys()) + + for pid in all_pool_ids: + if pid == pool_id: + continue # always exclude self + if pid == exclude_pool: + continue + peer_entry = matched[pid] + peer_toks = _parse_tokens(peer_entry["tokens"]) + peer_canonical = set() + for t in peer_toks[:2]: + ct = _canonicalize_token(t) if canonicalize else t + peer_canonical.add(ct) + + for ct in peer_canonical: + if ct not in token_peers: + token_peers[ct] = [] + token_peers[ct].append(pid) + + peer_chain = peer_entry["chain"] + if peer_chain not in chain_peers: + chain_peers[peer_chain] = [] + chain_peers[peer_chain].append(pid) + + # Build (pool_id, date_ordinal) → log_volume lookup from all pools + vol_lookup = {} # (pid, date_ordinal) → log_volume + for pid in all_pool_ids: + if pid == pool_id or pid == exclude_pool: + continue + peer_panel = matched[pid]["panel"] + peer_dates = pd.to_datetime(peer_panel["date"]) + peer_ords = np.array([d.toordinal() for d in peer_dates]) + peer_vols = peer_panel["log_volume"].values.astype(float) + for ord_val, vol_val in zip(peer_ords, peer_vols): + vol_lookup[(pid, int(ord_val))] = vol_val + + # Compute global lagged mean for fallback + all_vols = list(vol_lookup.values()) + global_mean_vol = float(np.mean(all_vols)) if all_vols else 0.0 + + # Get this pool's dates + dates = pd.to_datetime(panel_rows["date"]) + date_ords = np.array([d.toordinal() for d in dates]) + n_obs = len(panel_rows) + + def _peer_mean_at_lag(peer_pids, date_ord_prev): + """Mean log_volume of peer pools at date_ord_prev.""" + vals = [] + for pid in peer_pids: + key = (pid, date_ord_prev) + if key in vol_lookup: + vals.append(vol_lookup[key]) + if vals: + return float(np.mean(vals)) + return np.nan + + # Build cross-pool features for each obs (starting from day 1) + cross_vol_a = np.full(n_obs, np.nan) + cross_vol_b = np.full(n_obs, np.nan) + cross_vol_chain = np.full(n_obs, np.nan) + + tok_a_peers = token_peers.get(tok_a, []) + tok_b_peers = token_peers.get(tok_b, []) + chain_peer_list = chain_peers.get(this_chain, []) + + for i in range(1, n_obs): + prev_ord = int(date_ords[i - 1]) + + if tok_a_peers: + cross_vol_a[i] = _peer_mean_at_lag(tok_a_peers, prev_ord) + if tok_b_peers: + cross_vol_b[i] = _peer_mean_at_lag(tok_b_peers, prev_ord) + if chain_peer_list: + cross_vol_chain[i] = _peer_mean_at_lag(chain_peer_list, prev_ord) + + # Drop first day, fill NaN with global mean + cross_vol_a = cross_vol_a[1:] + cross_vol_b = cross_vol_b[1:] + cross_vol_chain = cross_vol_chain[1:] + + cross_vol_a = np.where(np.isnan(cross_vol_a), global_mean_vol, cross_vol_a) + cross_vol_b = np.where(np.isnan(cross_vol_b), global_mean_vol, cross_vol_b) + cross_vol_chain = np.where(np.isnan(cross_vol_chain), global_mean_vol, cross_vol_chain) + + # Build base x_obs (reduced) and drop first row + x_base = build_x_obs(panel_rows, reduced=True) + x_base = x_base[1:] # drop first day + + # Assemble + x = np.zeros((n_obs - 1, K_OBS_CROSS)) + x[:, :4] = x_base + x[:, 4] = cross_vol_a + x[:, 5] = cross_vol_b + x[:, 6] = cross_vol_chain + + return x + + def build_pool_attributes( matched: Dict[str, dict], mcap_path: str = None, diff --git a/tests/calibration/test_pool_data.py b/tests/calibration/test_pool_data.py index adbf212a..4864c19f 100644 --- a/tests/calibration/test_pool_data.py +++ b/tests/calibration/test_pool_data.py @@ -502,3 +502,306 @@ def test_attributes_returns_pool_order( assert isinstance(pool_ids, list) assert len(pool_ids) == len(matched) assert set(pool_ids) == set(matched.keys()) + + +class TestTokenCanonicalization: + """Test _CANON_MAP and canonicalization in encode_tokens.""" + + def test_canon_map_exists(self): + from quantammsim.calibration.pool_data import _CANON_MAP + assert isinstance(_CANON_MAP, dict) + + def test_canon_map_expected_mappings(self): + from quantammsim.calibration.pool_data import _CANON_MAP + assert _CANON_MAP["WETH"] == "ETH" + assert _CANON_MAP["waBasWETH"] == "ETH" + assert _CANON_MAP["waEthLidoWETH"] == "ETH" + assert _CANON_MAP["waEthLidowstETH"] == "wstETH" + assert _CANON_MAP["waGnowstETH"] == "wstETH" + assert _CANON_MAP["waBasUSDC"] == "USDC" + assert _CANON_MAP["scUSD"] == "USDC" + assert _CANON_MAP["sDAI"] == "DAI" + assert _CANON_MAP["WBTC"] == "BTC" + assert _CANON_MAP["waGnoGNO"] == "GNO" + assert _CANON_MAP["stS"] == "S" + + def test_canonicalize_passthrough(self): + from quantammsim.calibration.pool_data import _CANON_MAP + for tok in ["AAVE", "BTC", "ETH", "LINK", "ARB"]: + assert tok not in _CANON_MAP + + def test_canonicalize_function(self): + from quantammsim.calibration.pool_data import _canonicalize_token + assert _canonicalize_token("WETH") == "ETH" + assert _canonicalize_token("WBTC") == "BTC" + assert _canonicalize_token("ETH") == "ETH" + assert _canonicalize_token("AAVE") == "AAVE" + + def test_encode_tokens_canonicalize_false( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """With canonicalize=False, same result as v1 (no merging).""" + from quantammsim.calibration.pool_data import encode_tokens, match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + result = encode_tokens(matched, canonicalize=False) + # Synthetic uses BTC, ETH, AAVE — none in canon map, same either way + assert result["n_tokens"] == 3 + assert set(result["token_index"].keys()) == {"AAVE", "BTC", "ETH"} + + def test_encode_tokens_canonicalize_default( + self, synthetic_daily_grid, synthetic_panel, tmp_path + ): + """Default canonicalize=True. Synthetic data unaffected (BTC, ETH, AAVE not in map).""" + from quantammsim.calibration.pool_data import encode_tokens, match_grids_to_panel + + grid_dir = tmp_path / "grids" + grid_dir.mkdir() + for prefix in POOL_PREFIXES: + synthetic_daily_grid.to_parquet( + grid_dir / f"{prefix}_daily.parquet", index=False + ) + matched = match_grids_to_panel(str(grid_dir), synthetic_panel) + result = encode_tokens(matched) # canonicalize=True by default + assert result["n_tokens"] == 3 + assert set(result["token_index"].keys()) == {"AAVE", "BTC", "ETH"} + + def test_encode_tokens_merges_wrapped_tokens(self): + """Synthetic matched dict with WETH+USDC and waBasWETH+WBTC → ETH, BTC, USDC.""" + from quantammsim.calibration.pool_data import encode_tokens + + # Minimal matched dict — encode_tokens only uses 'tokens' and 'fee' keys + matched = { + "pool_a": { + "tokens": "WETH,USDC", + "fee": 0.003, + "chain": "MAINNET", + }, + "pool_b": { + "tokens": "waBasWETH,WBTC", + "fee": 0.003, + "chain": "BASE", + }, + } + result = encode_tokens(matched, canonicalize=True) + # WETH→ETH, waBasWETH→ETH, WBTC→BTC → unique: {BTC, ETH, USDC} + assert result["n_tokens"] == 3 + assert set(result["token_index"].keys()) == {"BTC", "ETH", "USDC"} + + # Both pools should have ETH as token A (canonicalized) + ti = result["token_index"] + assert result["token_a_idx"][0] == ti["ETH"] # pool_a: WETH→ETH + assert result["token_a_idx"][1] == ti["ETH"] # pool_b: waBasWETH→ETH + + def test_encode_tokens_canon_false_keeps_wrapped(self): + """canonicalize=False keeps WETH and waBasWETH as separate tokens.""" + from quantammsim.calibration.pool_data import encode_tokens + + matched = { + "pool_a": { + "tokens": "WETH,USDC", + "fee": 0.003, + "chain": "MAINNET", + }, + "pool_b": { + "tokens": "waBasWETH,WBTC", + "fee": 0.003, + "chain": "BASE", + }, + } + result = encode_tokens(matched, canonicalize=False) + # No merging: WETH, USDC, waBasWETH, WBTC → 4 unique tokens + assert result["n_tokens"] == 4 + assert "WETH" in result["token_index"] + assert "waBasWETH" in result["token_index"] + + +class TestCrossPoolFeatures: + """Test build_cross_pool_x_obs: cross-pool lagged volume features.""" + + @pytest.fixture + def three_pool_panel(self): + """3 pools sharing tokens, 10 days each.""" + np.random.seed(42) + dates = pd.date_range("2025-12-01", periods=10, freq="D") + rows = [] + pool_configs = [ + ("0xpool_a_full_id_padding_to_66_chars_aaaaaaaaaaaaaaaaaaaaaaaaa", + "MAINNET", "ETH,USDC", 0.003), + ("0xpool_b_full_id_padding_to_66_chars_bbbbbbbbbbbbbbbbbbbbbbbbb", + "MAINNET", "ETH,AAVE", 0.003), + ("0xpool_c_full_id_padding_to_66_chars_ccccccccccccccccccccccccc", + "ARBITRUM", "AAVE,USDC", 0.01), + ] + for full_id, chain, tokens, fee in pool_configs: + for di, date in enumerate(dates): + tvl = 12.0 + 0.05 * np.sin(2 * np.pi * di / 7) + vol = 9.0 + 0.3 * np.random.randn() + rows.append({ + "pool_id": full_id, + "chain": chain, + "date": date, + "log_volume": vol, + "log_tvl": tvl, + "log_tvl_lag1": tvl - 0.01, + "volatility": 0.4, + "log_fee": np.log(fee), + "swap_fee": fee, + "tokens": tokens, + }) + return pd.DataFrame(rows) + + @pytest.fixture + def three_pool_matched(self, three_pool_panel): + """Minimal matched dict for 3 pools (no grid needed for x_obs tests).""" + matched = {} + for full_id in three_pool_panel["pool_id"].unique(): + prefix = full_id[:16] + rows = three_pool_panel[three_pool_panel["pool_id"] == full_id].copy() + rows = rows.reset_index(drop=True) + matched[prefix] = { + "panel": rows, + "pool_id": full_id, + "chain": rows.iloc[0]["chain"], + "fee": float(np.exp(rows.iloc[0]["log_fee"])), + "tokens": rows.iloc[0]["tokens"], + "weights": [0.5, 0.5], + } + return matched + + def test_build_cross_pool_x_obs_shape(self, three_pool_matched): + from quantammsim.calibration.pool_data import ( + K_OBS_CROSS, build_cross_pool_x_obs, + ) + + pid = sorted(three_pool_matched.keys())[0] + entry = three_pool_matched[pid] + x = build_cross_pool_x_obs(entry["panel"], three_pool_matched, pid) + # Drops first day → n_obs - 1 rows, K_OBS_CROSS=7 columns + assert x.shape[1] == K_OBS_CROSS + assert x.shape[0] == len(entry["panel"]) - 1 + + def test_first_four_cols_match_reduced(self, three_pool_matched): + from quantammsim.calibration.pool_data import ( + build_cross_pool_x_obs, build_x_obs, + ) + + pid = sorted(three_pool_matched.keys())[0] + entry = three_pool_matched[pid] + x_cross = build_cross_pool_x_obs(entry["panel"], three_pool_matched, pid) + x_reduced = build_x_obs(entry["panel"], reduced=True) + # First 4 columns should match (after dropping first row) + np.testing.assert_allclose(x_cross[:, :4], x_reduced[1:, :4]) + + def test_cross_vol_token_a_excludes_self(self, three_pool_matched): + """Peer average for token A excludes pool i itself.""" + from quantammsim.calibration.pool_data import build_cross_pool_x_obs + + pool_ids = sorted(three_pool_matched.keys()) + pid_a = pool_ids[0] # ETH,USDC + pid_b = pool_ids[1] # ETH,AAVE — shares ETH with pool_a + + x_a = build_cross_pool_x_obs( + three_pool_matched[pid_a]["panel"], + three_pool_matched, pid_a, + ) + # Column 4 = cross_vol_token_a (ETH peers excl self) + # Pool b also has ETH, so pool_a's cross_vol_token_a should use pool_b's volume + panel_b = three_pool_matched[pid_b]["panel"] + log_vol_b_lagged = panel_b["log_volume"].values[:-1] # lag by 1 + np.testing.assert_allclose(x_a[:, 4], log_vol_b_lagged, rtol=1e-6) + + def test_cross_vol_is_lagged(self, three_pool_matched): + """Features at day t use log_volume at day t-1.""" + from quantammsim.calibration.pool_data import build_cross_pool_x_obs + + pool_ids = sorted(three_pool_matched.keys()) + pid = pool_ids[0] + x = build_cross_pool_x_obs( + three_pool_matched[pid]["panel"], + three_pool_matched, pid, + ) + # x has n_obs - 1 rows (first day dropped) + # Row 0 of x corresponds to day 1 and should use day 0 volume + assert x.shape[0] > 0 + + def test_cross_vol_nan_free_after_first_day(self, three_pool_matched): + from quantammsim.calibration.pool_data import build_cross_pool_x_obs + + for pid in three_pool_matched: + x = build_cross_pool_x_obs( + three_pool_matched[pid]["panel"], + three_pool_matched, pid, + ) + assert not np.any(np.isnan(x)), f"NaNs in cross-pool x_obs for {pid}" + + def test_exclude_pool_changes_features(self, three_pool_matched): + """exclude_pool removes that pool from peer averages.""" + from quantammsim.calibration.pool_data import build_cross_pool_x_obs + + pool_ids = sorted(three_pool_matched.keys()) + pid = pool_ids[0] # ETH,USDC + + x_normal = build_cross_pool_x_obs( + three_pool_matched[pid]["panel"], + three_pool_matched, pid, + ) + x_excluded = build_cross_pool_x_obs( + three_pool_matched[pid]["panel"], + three_pool_matched, pid, + exclude_pool=pool_ids[1], # exclude the ETH peer + ) + # Chain feature (col 6) may change too; token A col (4) definitely changes + # since pool_b is the only ETH peer + # With only peer excluded, cross_vol_token_a should be NaN→fallback + assert not np.allclose(x_normal[:, 4], x_excluded[:, 4]) + + def test_single_token_pool_fallback(self): + """When a token appears in only one pool, its cross_vol uses global mean.""" + from quantammsim.calibration.pool_data import build_cross_pool_x_obs + + np.random.seed(42) + dates = pd.date_range("2025-12-01", periods=5, freq="D") + rows = [] + # Pool A: LINK,USDC — LINK is unique + for di, date in enumerate(dates): + rows.append({ + "pool_id": "0xsolo_link_pool_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "chain": "MAINNET", "date": date, + "log_volume": 9.0 + 0.1 * di, + "log_tvl_lag1": 12.0, "volatility": 0.4, + "log_fee": np.log(0.003), "tokens": "LINK,USDC", + }) + # Pool B: ETH,USDC + for di, date in enumerate(dates): + rows.append({ + "pool_id": "0xpeer_eth_pool__bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "chain": "MAINNET", "date": date, + "log_volume": 10.0 + 0.1 * di, + "log_tvl_lag1": 13.0, "volatility": 0.4, + "log_fee": np.log(0.003), "tokens": "ETH,USDC", + }) + panel = pd.DataFrame(rows) + + matched = {} + for full_id in panel["pool_id"].unique(): + prefix = full_id[:16] + sub = panel[panel["pool_id"] == full_id].reset_index(drop=True) + matched[prefix] = { + "panel": sub, "pool_id": full_id, + "chain": sub.iloc[0]["chain"], + "fee": float(np.exp(sub.iloc[0]["log_fee"])), + "tokens": sub.iloc[0]["tokens"], "weights": [0.5, 0.5], + } + + pid_link = [p for p in matched if matched[p]["tokens"] == "LINK,USDC"][0] + x = build_cross_pool_x_obs(panel[panel["pool_id"] == matched[pid_link]["pool_id"]].reset_index(drop=True), + matched, pid_link) + # LINK has no peers → col 4 should be a fallback (global mean), not NaN + assert not np.any(np.isnan(x[:, 4])) From e99d3e25d1b86884563016c7425a88d6da046e43 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 12:20:48 +0000 Subject: [PATCH 056/115] feat: report data_loss and reg_loss separately in CalibrationModel.fit Re-evaluates pool loss functions at the optimum to decompose total loss into data_loss (mean per-pool MSE) and reg_loss (head regularization). Enables tracking whether lambda annealing is reducing data fit or just shrinking regularization. --- quantammsim/calibration/calibration_model.py | 10 +++ tests/calibration/test_joint_fit.py | 87 ++++++++++++++++++++ 2 files changed, 97 insertions(+) diff --git a/quantammsim/calibration/calibration_model.py b/quantammsim/calibration/calibration_model.py index 88e940cb..86471ec1 100644 --- a/quantammsim/calibration/calibration_model.py +++ b/quantammsim/calibration/calibration_model.py @@ -179,6 +179,7 @@ def loss_fn(params_flat): return data_loss + reg # Attach for the scipy wrapper + loss_fn._pool_loss_fns = pool_loss_fns loss_fn._pool_val_and_grad_fns = pool_val_and_grad_fns loss_fn._n_pools = n_pools loss_fn._head_slices = (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) @@ -282,9 +283,18 @@ def scipy_wrapper(params_np): (cad_s, cad_e), (gas_s, gas_e), (noise_s, noise_e) = \ self._head_slices(n_pools, k_attr) + # Compute data_loss and reg_loss at optimum + fitted_j = jnp.array(result.x) + data_loss_val = sum( + float(fn(fitted_j)) for fn in loss_fn._pool_loss_fns + ) / n_pools + reg_loss_val = float(result.fun) - data_loss_val + out = { "init_loss": init_loss, "loss": float(result.fun), + "data_loss": data_loss_val, + "reg_loss": reg_loss_val, "converged": result.success, "params_flat": np.array(result.x), } diff --git a/tests/calibration/test_joint_fit.py b/tests/calibration/test_joint_fit.py index 004ef93b..3a3971d9 100644 --- a/tests/calibration/test_joint_fit.py +++ b/tests/calibration/test_joint_fit.py @@ -288,6 +288,93 @@ def test_model_with_warm_start(self, matched_data): assert result["loss"] <= result["init_loss"] +class TestDataRegLossSeparation: + """Test that fit() reports data_loss and reg_loss separately.""" + + def test_fit_reports_data_and_reg_loss(self, matched_data): + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data(matched_data) + n_pools = len(jdata.pool_data) + + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_data[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead(k_obs=K_OBS_REDUCED, **enc) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=50) + + assert "data_loss" in result + assert "reg_loss" in result + + def test_data_plus_reg_equals_total(self, matched_data): + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data(matched_data) + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_data[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead(k_obs=K_OBS_REDUCED, **enc) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=50) + + np.testing.assert_allclose( + result["data_loss"] + result["reg_loss"], + result["loss"], + rtol=1e-6, + ) + + def test_data_loss_leq_total(self, matched_data): + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data(matched_data) + gas_values = [] + for pid in jdata.pool_ids: + chain = matched_data[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead(k_obs=K_OBS_REDUCED, **enc) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=50) + + assert result["data_loss"] <= result["loss"] + 1e-10 + assert result["reg_loss"] >= -1e-10 + + class TestPrepareJointDataReduced: """Test prepare_joint_data with reduced_x_obs=True.""" From 34751981dc460c64c070264ba37b288c0fe12e0c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 12:24:22 +0000 Subject: [PATCH 057/115] feat: v2 runner with lambda annealing, cross-pool ablation, and LOO MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Lambda sweep now runs descending (high→low regularization) with each fit warm-starting from the previous result. Runner runs two ablations side-by-side: baseline (K_OBS_REDUCED=4) vs cross-pool (K_OBS_CROSS=7), reporting separated data/reg loss and LOO R² for each configuration. prepare_token_factored_data() gains cross_pool parameter to swap in cross-pool lag features automatically. --- quantammsim/calibration/joint_fit.py | 35 ++- scripts/run_token_factored_calibration.py | 292 ++++++++++++++++------ 2 files changed, 246 insertions(+), 81 deletions(-) diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py index 4a237e9b..d71be73c 100644 --- a/quantammsim/calibration/joint_fit.py +++ b/quantammsim/calibration/joint_fit.py @@ -95,21 +95,54 @@ def prepare_token_factored_data( matched: Dict[str, dict], reduced_x_obs: bool = True, fix_gas_to_chain: bool = True, + canonicalize: bool = True, + cross_pool: bool = False, ) -> tuple: """Prepare JointData + token encoding for TokenFactoredNoiseHead. + Args: + matched: dict from match_grids_to_panel + reduced_x_obs: if True, use 4-column reduced x_obs + fix_gas_to_chain: if True, fix gas to chain-level costs + canonicalize: if True, canonicalize token names before building index + cross_pool: if True, use cross-pool lag features (K_OBS_CROSS=7) + Returns (jdata, token_encoding) where token_encoding is the dict from encode_tokens() containing token/chain structure for constructing the head. """ from quantammsim.calibration.pool_data import encode_tokens + if cross_pool: + from quantammsim.calibration.pool_data import ( + K_OBS_CROSS, build_cross_pool_x_obs, + ) + jdata = prepare_joint_data( matched, fix_gas_to_chain=fix_gas_to_chain, reduced_x_obs=reduced_x_obs, ) - token_encoding = encode_tokens(matched) + if cross_pool: + # Replace x_obs with cross-pool version for each pool + pool_ids = jdata.pool_ids + new_pool_data = [] + for i, pid in enumerate(pool_ids): + entry = matched[pid] + x_obs_cross = build_cross_pool_x_obs( + entry["panel"], matched, pid, canonicalize=canonicalize, + ) + d = dict(jdata.pool_data[i]) + d["x_obs"] = jnp.array(x_obs_cross) + new_pool_data.append(d) + jdata = JointData( + pool_data=new_pool_data, + x_attr=jdata.x_attr, + pool_ids=jdata.pool_ids, + attr_names=jdata.attr_names, + ) + + token_encoding = encode_tokens(matched, canonicalize=canonicalize) return jdata, token_encoding diff --git a/scripts/run_token_factored_calibration.py b/scripts/run_token_factored_calibration.py index 0a6a170d..1ccf15e3 100644 --- a/scripts/run_token_factored_calibration.py +++ b/scripts/run_token_factored_calibration.py @@ -1,14 +1,13 @@ -"""Token-factored noise calibration: pooled diagnostic + full pipeline. +"""Token-factored noise calibration v2: canonicalization + cross-pool lag features. Phase 0: Pooled Ridge diagnostic — does cross-pool signal exist? -Phase 1: Token-factored model with lambda_delta sweep -Phase 2: LOO cross-validation +Phase 1: Token-factored model with lambda_delta annealing sweep +Phase 2: LOO cross-validation (baseline vs cross-pool ablation) Phase 3: Comparison plots and JSON export """ import json import os -import sys import matplotlib matplotlib.use("Agg") @@ -31,7 +30,8 @@ ) OPTION_C_LOSS_CUTOFF = 5.0 JOINT_MAXITER = 5000 -LAMBDA_DELTAS = [0.01, 0.1, 0.5, 1.0, 5.0, 10.0] +# Sorted descending for warm-start annealing (highest regularization first) +LAMBDA_DELTAS = [10.0, 5.0, 1.0, 0.5, 0.1, 0.01] # ---- Data loading (shared with run_mlp_calibration.py) ---- @@ -96,7 +96,7 @@ def run_phase0_diagnostic(matched, option_c): print("=" * 70) pool_ids = sorted(matched.keys()) - X_attr, attr_names, _ = build_pool_attributes(matched) + X_attr, _, _ = build_pool_attributes(matched) pool_idx_map = {pid: i for i, pid in enumerate(pool_ids)} enc = encode_tokens(matched) @@ -137,14 +137,14 @@ def run_phase0_diagnostic(matched, option_c): X_pool_attrs = np.vstack(all_pool_attrs) X_token_dummies = np.vstack(all_token_dummies) - # Model 1: x_obs + pool_attrs → pooled Ridge + # Model 1: x_obs + pool_attrs X_combined = np.column_stack([X_obs, X_pool_attrs]) model1 = RidgeCV(alphas=np.logspace(-2, 4, 50)) model1.fit(X_combined, y_combined) r2_pooled = model1.score(X_combined, y_combined) print(f" Pooled Ridge (x_obs + pool attrs): R² = {r2_pooled:.4f}") - # Model 2: x_obs + token_dummies → token-dummy Ridge + # Model 2: x_obs + token_dummies X_token = np.column_stack([X_obs, X_token_dummies]) model2 = RidgeCV(alphas=np.logspace(-2, 4, 50)) model2.fit(X_token, y_combined) @@ -209,35 +209,65 @@ def _build_gas_values(jdata, matched_clean): return np.array(gas_values) -def run_token_factored(matched_clean, option_c_clean, lambda_delta=1.0): +def _result_to_warm_start(result): + """Extract per-pool warm_start dict from a CalibrationModel fit result. + + Returns dict: pool_id -> {log_cadence, noise_coeffs} suitable for + passing as warm_start to CalibrationModel.fit(). + """ + pool_ids = result["pool_ids"] + warm = {} + for i, pid in enumerate(pool_ids): + entry = {} + # Cadence: from PerPoolHead + if "log_cadence_per_pool" in result: + entry["log_cadence"] = float(result["log_cadence_per_pool"][i]) + # Noise: per-pool coefficients + if "noise_coeffs" in result: + entry["noise_coeffs"] = result["noise_coeffs"][i] + warm[pid] = entry + return warm + + +def run_token_factored( + matched_clean, option_c_clean, lambda_delta=1.0, + cross_pool=False, warm_start=None, +): """Fit TokenFactoredNoiseHead with PerPoolHead(cadence) + FixedHead(gas).""" from quantammsim.calibration.calibration_model import CalibrationModel from quantammsim.calibration.heads import ( FixedHead, PerPoolHead, TokenFactoredNoiseHead, ) from quantammsim.calibration.joint_fit import prepare_token_factored_data - from quantammsim.calibration.pool_data import K_OBS_REDUCED + from quantammsim.calibration.pool_data import K_OBS_CROSS, K_OBS_REDUCED - jdata, enc = prepare_token_factored_data(matched_clean) + k_obs = K_OBS_CROSS if cross_pool else K_OBS_REDUCED + + jdata, enc = prepare_token_factored_data( + matched_clean, cross_pool=cross_pool, + ) n_pools = len(jdata.pool_data) gas_values = _build_gas_values(jdata, matched_clean) gas_head = FixedHead("log_gas", gas_values) cad_head = PerPoolHead("log_cadence", default=np.log(12.0)) noise_head = TokenFactoredNoiseHead( - k_obs=K_OBS_REDUCED, + k_obs=k_obs, lambda_delta=lambda_delta, **enc, ) model = CalibrationModel(cad_head, gas_head, noise_head) n_p = model.n_params(n_pools, jdata.x_attr.shape[1]) - print(f"\n--- Token-factored (lambda_delta={lambda_delta}) ---") + cp_tag = " [cross-pool]" if cross_pool else "" + print(f"\n--- Token-factored (lambda_delta={lambda_delta}){cp_tag} ---") print(f" {n_pools} pools, {enc['n_tokens']} tokens, " - f"{enc['n_chains']} chains, {n_p} params") + f"{enc['n_chains']} chains, {n_p} params, k_obs={k_obs}") - result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=option_c_clean) - print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}") + ws = warm_start if warm_start is not None else option_c_clean + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=ws) + print(f" Loss: {result['init_loss']:.4f} -> {result['loss']:.4f}" + f" (data={result['data_loss']:.4f}, reg={result['reg_loss']:.4f})") print(f" Converged: {result['converged']}") return result, model, jdata, enc @@ -292,8 +322,6 @@ def print_delta_analysis(result, enc, jdata, matched_clean): """Print per-pool delta analysis.""" delta = result["noise_deltas"] pool_ids = jdata.pool_ids - token_index = enc["token_index"] - inv_token = {v: k for k, v in token_index.items()} print(f"\n{'='*70}") print("Per-pool deltas (unexplained residual)") @@ -311,30 +339,44 @@ def print_delta_analysis(result, enc, jdata, matched_clean): f"{delta_norms[i]:>8.3f} {delta[i, 0]:>8.3f}") -def run_lambda_sweep(matched_clean, option_c_clean): - """Sweep lambda_delta values and report loss + delta shrinkage.""" +def run_lambda_sweep(matched_clean, option_c_clean, cross_pool=False): + """Sweep lambda_delta with warm-start annealing (descending lambda). + + Each fit warm-starts from the previous result, so the sweep is + effectively a continuation path from high to low regularization. + """ print(f"\n{'='*70}") - print("Lambda_delta sweep") + cp_tag = " [cross-pool]" if cross_pool else "" + print(f"Lambda_delta sweep{cp_tag}") print(f"{'='*70}") - print(f"{'lambda':>10} {'loss':>10} {'delta_norm':>12} {'mean_|d|':>10}") - print("-" * 45) + print(f"{'lambda':>10} {'loss':>10} {'data_loss':>10} {'reg_loss':>10} " + f"{'delta_norm':>12} {'mean_|d|':>10}") + print("-" * 65) results = [] + warm_start = option_c_clean for lam in LAMBDA_DELTAS: result, model, jdata, enc = run_token_factored( - matched_clean, option_c_clean, lambda_delta=lam) + matched_clean, option_c_clean, lambda_delta=lam, + cross_pool=cross_pool, warm_start=warm_start, + ) delta = result["noise_deltas"] delta_norm = float(np.linalg.norm(delta)) mean_abs_d = float(np.mean(np.abs(delta))) print(f"{lam:>10.2f} {result['loss']:>10.4f} " + f"{result['data_loss']:>10.4f} {result['reg_loss']:>10.4f} " f"{delta_norm:>12.4f} {mean_abs_d:>10.4f}") results.append({ "lambda_delta": lam, "loss": result["loss"], + "data_loss": result["data_loss"], + "reg_loss": result["reg_loss"], "delta_norm": delta_norm, "mean_abs_delta": mean_abs_d, "converged": result["converged"], }) + # Warm-start next iteration from this result + warm_start = _result_to_warm_start(result) return results @@ -342,7 +384,9 @@ def run_lambda_sweep(matched_clean, option_c_clean): # ---- Phase 3: LOO Cross-Validation ---- -def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): +def run_loo_validation( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False, +): """Leave-one-pool-out cross-validation via predict_new_pool.""" from quantammsim.calibration.calibration_model import CalibrationModel from quantammsim.calibration.heads import ( @@ -350,14 +394,18 @@ def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): ) from quantammsim.calibration.grid_interpolation import interpolate_pool_daily from quantammsim.calibration.joint_fit import prepare_token_factored_data - from quantammsim.calibration.pool_data import K_OBS_REDUCED, build_x_obs, _parse_tokens + from quantammsim.calibration.pool_data import ( + K_OBS_CROSS, K_OBS_REDUCED, build_cross_pool_x_obs, + build_x_obs, _parse_tokens, + ) import jax.numpy as jnp + k_obs = K_OBS_CROSS if cross_pool else K_OBS_REDUCED pool_ids = sorted(matched_clean.keys()) - n_pools = len(pool_ids) + cp_tag = " [cross-pool]" if cross_pool else "" print(f"\n{'='*70}") - print(f"LOO Cross-Validation (lambda_delta={lambda_delta})") + print(f"LOO Cross-Validation (lambda_delta={lambda_delta}){cp_tag}") print(f"{'='*70}") loo_results = [] @@ -370,11 +418,13 @@ def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): continue # Fit on training set - jdata, enc = prepare_token_factored_data(train_matched) + jdata, enc = prepare_token_factored_data( + train_matched, cross_pool=cross_pool, + ) gas_values = _build_gas_values(jdata, train_matched) noise_head = TokenFactoredNoiseHead( - k_obs=K_OBS_REDUCED, + k_obs=k_obs, lambda_delta=lambda_delta, **enc, ) @@ -401,9 +451,22 @@ def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): # Evaluate hold-out R² ho_panel = ho_entry["panel"] - x_obs_ho = build_x_obs(ho_panel, reduced=True) y_obs_ho = ho_panel["log_volume"].values.astype(float) + if cross_pool: + # Build cross-pool x_obs for held-out pool. + # Use matched_clean so pool's own entry is accessible; + # build_cross_pool_x_obs auto-excludes pool_id from its own peers. + x_obs_ho = build_cross_pool_x_obs( + ho_panel, matched_clean, hold_out_pid, + ) + # Trim y_obs to match (first day dropped) + y_obs_ho = y_obs_ho[1:] + day_indices_ho = ho_entry["day_indices"][1:] + else: + x_obs_ho = build_x_obs(ho_panel, reduced=True) + day_indices_ho = ho_entry["day_indices"] + # Use Option C cadence for the hold-out pool (not predicting cadence) oc_ho = option_c_clean[hold_out_pid] v_arb_all = np.array(interpolate_pool_daily( @@ -411,18 +474,25 @@ def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): jnp.float64(oc_ho["log_cadence"]), jnp.float64(np.exp(oc_ho["log_gas"])), )) - v_arb = v_arb_all[ho_entry["day_indices"]] - v_noise = np.exp(x_obs_ho @ ho_pred["noise_coeffs"]) + v_arb = v_arb_all[day_indices_ho] + + # Noise coefficients are k_obs-dimensional; x_obs_ho has k_obs columns + noise_coeffs = ho_pred["noise_coeffs"][:k_obs] + v_noise = np.exp(x_obs_ho @ noise_coeffs) log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) ss_res = np.sum((log_pred - y_obs_ho) ** 2) ss_tot = np.sum((y_obs_ho - y_obs_ho.mean()) ** 2) r2_loo = 1 - ss_res / max(ss_tot, 1e-10) # Compare with Option C in-sample R² - v_noise_c = np.exp(build_x_obs(ho_panel, reduced=True) @ oc_ho["noise_coeffs"][:K_OBS_REDUCED]) - log_pred_c = np.log(np.maximum(v_arb + v_noise_c, 1e-6)) - ss_res_c = np.sum((log_pred_c - y_obs_ho) ** 2) - r2_c = 1 - ss_res_c / max(ss_tot, 1e-10) + x_obs_c = build_x_obs(ho_panel, reduced=True) + v_noise_c = np.exp(x_obs_c @ oc_ho["noise_coeffs"][:K_OBS_REDUCED]) + v_arb_c = v_arb_all[ho_entry["day_indices"]] + log_pred_c = np.log(np.maximum(v_arb_c + v_noise_c, 1e-6)) + y_obs_full = ho_panel["log_volume"].values.astype(float) + ss_res_c = np.sum((log_pred_c - y_obs_full) ** 2) + ss_tot_c = np.sum((y_obs_full - y_obs_full.mean()) ** 2) + r2_c = 1 - ss_res_c / max(ss_tot_c, 1e-10) loo_results.append({ "pool_id": hold_out_pid, @@ -450,18 +520,21 @@ def run_loo_validation(matched_clean, option_c_clean, lambda_delta=1.0): # ---- Plots ---- -def plot_lambda_sweep(sweep_results, output_dir): - """Plot loss and delta norm vs lambda_delta.""" +def plot_lambda_sweep(sweep_results, output_dir, suffix=""): + """Plot loss (data/reg separated) and delta norm vs lambda_delta.""" lambdas = [r["lambda_delta"] for r in sweep_results] - losses = [r["loss"] for r in sweep_results] + data_losses = [r["data_loss"] for r in sweep_results] + reg_losses = [r["reg_loss"] for r in sweep_results] delta_norms = [r["delta_norm"] for r in sweep_results] fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 5)) - ax1.semilogx(lambdas, losses, "o-", color="steelblue") + ax1.semilogx(lambdas, data_losses, "o-", color="steelblue", label="data_loss") + ax1.semilogx(lambdas, reg_losses, "s--", color="orangered", label="reg_loss") ax1.set_xlabel("lambda_delta") ax1.set_ylabel("Loss") - ax1.set_title("Loss vs lambda_delta") + ax1.set_title("Data + Reg Loss vs lambda_delta") + ax1.legend() ax2.semilogx(lambdas, delta_norms, "o-", color="orangered") ax2.set_xlabel("lambda_delta") @@ -469,7 +542,7 @@ def plot_lambda_sweep(sweep_results, output_dir): ax2.set_title("Delta norm vs lambda_delta") fig.tight_layout() - out = os.path.join(output_dir, "lambda_sweep.png") + out = os.path.join(output_dir, f"lambda_sweep{suffix}.png") fig.savefig(out, dpi=150, bbox_inches="tight") plt.close(fig) print(f" Saved: {out}") @@ -497,7 +570,7 @@ def plot_token_effects(result, enc, output_dir): print(f" Saved: {out}") -def plot_loo_scatter(loo_results, output_dir): +def plot_loo_scatter(loo_results, output_dir, suffix=""): """Scatter: Option C R² vs LOO R².""" if not loo_results: return @@ -514,14 +587,14 @@ def plot_loo_scatter(loo_results, output_dir): "k--", alpha=0.3, linewidth=1) ax.set_xlabel("Option C R² (in-sample)") ax.set_ylabel("Token-factored R² (LOO)") - ax.set_title("LOO Cross-Validation: Token-Factored vs Option C") + ax.set_title(f"LOO: Token-Factored vs Option C{suffix}") n_better = sum(1 for c, l in zip(r2_c, r2_loo) if l > c) ax.text(0.05, 0.95, f"LOO wins: {n_better}/{len(r2_c)}", transform=ax.transAxes, fontsize=10, va="top") fig.tight_layout() - out = os.path.join(output_dir, "loo_scatter.png") + out = os.path.join(output_dir, f"loo_scatter{suffix}.png") fig.savefig(out, dpi=150, bbox_inches="tight") plt.close(fig) print(f" Saved: {out}") @@ -534,7 +607,8 @@ def main(): os.environ.setdefault("JAX_PLATFORMS", "cpu") print("=" * 70) - print("Token-Factored Noise Calibration") + print("Token-Factored Noise Calibration v2") + print(" Canonicalization + Cross-Pool Lag Features") print("=" * 70) panel, matched = load_and_match() @@ -552,53 +626,111 @@ def main(): # Phase 0: Diagnostic diag = run_phase0_diagnostic(matched_clean, option_c_clean) - # Phase 1: Token-factored model (default lambda) - result, model, jdata, enc = run_token_factored( - matched_clean, option_c_clean, lambda_delta=1.0) + # ---- Ablation: Baseline (K_OBS_REDUCED=4, no cross-pool) ---- + print("\n" + "=" * 70) + print("ABLATION 1: Baseline (K_OBS_REDUCED=4, no cross-pool features)") + print("=" * 70) + + result_base, model_base, jdata_base, enc_base = run_token_factored( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) # Analysis - print_token_effects(result, enc) - print_chain_effects(result, enc) - print_delta_analysis(result, enc, jdata, matched_clean) + print_token_effects(result_base, enc_base) + print_chain_effects(result_base, enc_base) + print_delta_analysis(result_base, enc_base, jdata_base, matched_clean) - # Lambda sweep - sweep_results = run_lambda_sweep(matched_clean, option_c_clean) + # Lambda sweep with annealing + sweep_baseline = run_lambda_sweep(matched_clean, option_c_clean, cross_pool=False) - # Phase 2: LOO cross-validation - loo_results = run_loo_validation( - matched_clean, option_c_clean, lambda_delta=1.0) + # LOO + loo_baseline = run_loo_validation( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) - # Phase 3: Plots & export + # ---- Ablation: Cross-pool (K_OBS_CROSS=7) ---- + print("\n" + "=" * 70) + print("ABLATION 2: Cross-pool lag features (K_OBS_CROSS=7)") + print("=" * 70) + + result_cross, model_cross, jdata_cross, enc_cross = run_token_factored( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=True) + + print_token_effects(result_cross, enc_cross) + print_chain_effects(result_cross, enc_cross) + print_delta_analysis(result_cross, enc_cross, jdata_cross, matched_clean) + + sweep_cross = run_lambda_sweep(matched_clean, option_c_clean, cross_pool=True) + + loo_cross = run_loo_validation( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=True) + + # ---- Ablation summary ---- + print("\n" + "=" * 70) + print("ABLATION COMPARISON") + print("=" * 70) + for label, loo in [("Baseline (k=4)", loo_baseline), ("Cross-pool (k=7)", loo_cross)]: + if loo: + r2s = [r["r2_loo"] for r in loo] + r2s_c = [r["r2_option_c"] for r in loo] + wins = sum(1 for r in loo if r["r2_loo"] > r["r2_option_c"]) + print(f" {label}: median R²_LOO={np.median(r2s):.4f}, " + f"median R²_C={np.median(r2s_c):.4f}, " + f"wins={wins}/{len(loo)}") + + # Plots print("\nGenerating plots...") os.makedirs(OUTPUT_DIR, exist_ok=True) - plot_lambda_sweep(sweep_results, OUTPUT_DIR) - plot_token_effects(result, enc, OUTPUT_DIR) - plot_loo_scatter(loo_results, OUTPUT_DIR) + plot_lambda_sweep(sweep_baseline, OUTPUT_DIR, suffix="_baseline") + plot_lambda_sweep(sweep_cross, OUTPUT_DIR, suffix="_crosspool") + plot_token_effects(result_base, enc_base, OUTPUT_DIR) + plot_loo_scatter(loo_baseline, OUTPUT_DIR, suffix="_baseline") + plot_loo_scatter(loo_cross, OUTPUT_DIR, suffix="_crosspool") # JSON export export = { "phase0_diagnostic": diag, - "token_factored": { - "loss": result["loss"], - "init_loss": result["init_loss"], - "converged": result["converged"], - "n_pools": result["n_pools"], - "n_tokens": enc["n_tokens"], - "n_chains": enc["n_chains"], - "token_index": enc["token_index"], - "chain_index": enc["chain_index"], - "token_effects": result["token_effects"].tolist(), - "Gamma": result["Gamma"].tolist(), - "chain_effects": result["chain_effects"].tolist(), - "beta_fee": result["beta_fee"].tolist(), - "noise_deltas": result["noise_deltas"].tolist(), - "noise_coeffs": result["noise_coeffs"].tolist(), + "baseline": { + "loss": result_base["loss"], + "data_loss": result_base["data_loss"], + "reg_loss": result_base["reg_loss"], + "init_loss": result_base["init_loss"], + "converged": result_base["converged"], + "n_pools": result_base["n_pools"], + "n_tokens": enc_base["n_tokens"], + "n_chains": enc_base["n_chains"], + "token_index": enc_base["token_index"], + "chain_index": enc_base["chain_index"], + "token_effects": result_base["token_effects"].tolist(), + "Gamma": result_base["Gamma"].tolist(), + "chain_effects": result_base["chain_effects"].tolist(), + "beta_fee": result_base["beta_fee"].tolist(), + "noise_deltas": result_base["noise_deltas"].tolist(), + "noise_coeffs": result_base["noise_coeffs"].tolist(), + }, + "cross_pool": { + "loss": result_cross["loss"], + "data_loss": result_cross["data_loss"], + "reg_loss": result_cross["reg_loss"], + "init_loss": result_cross["init_loss"], + "converged": result_cross["converged"], + "n_pools": result_cross["n_pools"], + "n_tokens": enc_cross["n_tokens"], + "n_chains": enc_cross["n_chains"], + "token_index": enc_cross["token_index"], + "chain_index": enc_cross["chain_index"], + "token_effects": result_cross["token_effects"].tolist(), + "Gamma": result_cross["Gamma"].tolist(), + "chain_effects": result_cross["chain_effects"].tolist(), + "beta_fee": result_cross["beta_fee"].tolist(), + "noise_deltas": result_cross["noise_deltas"].tolist(), + "noise_coeffs": result_cross["noise_coeffs"].tolist(), }, - "lambda_sweep": sweep_results, - "loo_results": loo_results, + "lambda_sweep_baseline": sweep_baseline, + "lambda_sweep_crosspool": sweep_cross, + "loo_baseline": loo_baseline, + "loo_crosspool": loo_cross, } - json_path = os.path.join(OUTPUT_DIR, "token_factored_results.json") + json_path = os.path.join(OUTPUT_DIR, "token_factored_v2_results.json") with open(json_path, "w") as f: json.dump(export, f, indent=2, default=str) print(f" Saved: {json_path}") From 31f882918ee61a1a65054be1225d5d232440ce09 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 16 Mar 2026 22:18:53 +0000 Subject: [PATCH 058/115] fix: cross-pool x_obs shape mismatch and warm-start k_obs padding TokenFactoredNoiseHead.init() now zero-pads when warm_start noise_coeffs are shorter than k_obs (e.g. warm-starting k_obs=7 from k_obs=4 Option C results). prepare_token_factored_data(cross_pool=True) now trims y_obs and day_indices to match the first-day-dropped x_obs from build_cross_pool_x_obs. Runner gains --cross-pool-only flag with pickle caching of stage 1 (Option C + filtering) and baseline results so ablation 2 can run independently. Baseline-missing paths handled gracefully. Adds TestPrepareTokenFactoredCrossPool with shape consistency tests that would have caught the broadcast error. --- quantammsim/calibration/heads.py | 5 +- quantammsim/calibration/joint_fit.py | 4 + scripts/run_token_factored_calibration.py | 230 +++++++++++++++------- tests/calibration/test_joint_fit.py | 28 +++ 4 files changed, 189 insertions(+), 78 deletions(-) diff --git a/quantammsim/calibration/heads.py b/quantammsim/calibration/heads.py index ff86dfc9..b0513152 100644 --- a/quantammsim/calibration/heads.py +++ b/quantammsim/calibration/heads.py @@ -648,8 +648,9 @@ def init(self, jdata, warm_start=None): noise_all = np.zeros((n_pools, k), dtype=np.float64) for i, pid in enumerate(jdata.pool_ids): if pid in warm_start and "noise_coeffs" in warm_start[pid]: - nc = warm_start[pid]["noise_coeffs"] - noise_all[i] = nc[:k] + nc = np.asarray(warm_start[pid]["noise_coeffs"]) + n_copy = min(len(nc), k) + noise_all[i, :n_copy] = nc[:n_copy] # Solve: u[ta_i] + u[tb_i] + alpha[ch_i] + beta_fee * lf_i ≈ noise_all[i] n_cols = self.n_tokens + self.n_chains + 1 diff --git a/quantammsim/calibration/joint_fit.py b/quantammsim/calibration/joint_fit.py index d71be73c..8eaa7b86 100644 --- a/quantammsim/calibration/joint_fit.py +++ b/quantammsim/calibration/joint_fit.py @@ -132,8 +132,12 @@ def prepare_token_factored_data( x_obs_cross = build_cross_pool_x_obs( entry["panel"], matched, pid, canonicalize=canonicalize, ) + # x_obs_cross has n_obs-1 rows (first day dropped); + # trim y_obs and day_indices to match d = dict(jdata.pool_data[i]) d["x_obs"] = jnp.array(x_obs_cross) + d["y_obs"] = d["y_obs"][1:] + d["day_indices"] = d["day_indices"][1:] new_pool_data.append(d) jdata = JointData( pool_data=new_pool_data, diff --git a/scripts/run_token_factored_calibration.py b/scripts/run_token_factored_calibration.py index 1ccf15e3..1f1cbf58 100644 --- a/scripts/run_token_factored_calibration.py +++ b/scripts/run_token_factored_calibration.py @@ -6,8 +6,10 @@ Phase 3: Comparison plots and JSON export """ +import argparse import json import os +import pickle import matplotlib matplotlib.use("Agg") @@ -600,10 +602,97 @@ def plot_loo_scatter(loo_results, output_dir, suffix=""): print(f" Saved: {out}") +# ---- Intermediate state caching ---- + +_CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def _save_stage1(matched_clean, option_c_clean, diag): + """Cache Option C fits, filtering, and Phase 0 diagnostic.""" + os.makedirs(_CACHE_DIR, exist_ok=True) + path = os.path.join(_CACHE_DIR, "stage1.pkl") + with open(path, "wb") as f: + pickle.dump({ + "matched_clean": matched_clean, + "option_c_clean": option_c_clean, + "diag": diag, + }, f) + print(f" Cached stage 1 to {path}") + + +def _load_stage1(): + """Load cached stage 1 results. Returns None if missing.""" + path = os.path.join(_CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + return None + with open(path, "rb") as f: + data = pickle.load(f) + print(f" Loaded stage 1 cache from {path}") + return data + + +def _save_baseline(result_base, enc_base, sweep_baseline, loo_baseline): + """Cache ablation 1 results.""" + os.makedirs(_CACHE_DIR, exist_ok=True) + path = os.path.join(_CACHE_DIR, "baseline.pkl") + with open(path, "wb") as f: + pickle.dump({ + "result_base": result_base, + "enc_base": enc_base, + "sweep_baseline": sweep_baseline, + "loo_baseline": loo_baseline, + }, f) + print(f" Cached baseline results to {path}") + + +def _load_baseline(): + """Load cached baseline results. Returns None if missing.""" + path = os.path.join(_CACHE_DIR, "baseline.pkl") + if not os.path.exists(path): + return None + with open(path, "rb") as f: + data = pickle.load(f) + print(f" Loaded baseline cache from {path}") + return data + + +def _export_ablation_result(result, enc): + """Build JSON-serializable dict from ablation result + encoding.""" + return { + "loss": result["loss"], + "data_loss": result["data_loss"], + "reg_loss": result["reg_loss"], + "init_loss": result["init_loss"], + "converged": result["converged"], + "n_pools": result["n_pools"], + "n_tokens": enc["n_tokens"], + "n_chains": enc["n_chains"], + "token_index": enc["token_index"], + "chain_index": enc["chain_index"], + "token_effects": result["token_effects"].tolist(), + "Gamma": result["Gamma"].tolist(), + "chain_effects": result["chain_effects"].tolist(), + "beta_fee": result["beta_fee"].tolist(), + "noise_deltas": result["noise_deltas"].tolist(), + "noise_coeffs": result["noise_coeffs"].tolist(), + } + + # ---- Main ---- def main(): + parser = argparse.ArgumentParser( + description="Token-factored noise calibration v2") + parser.add_argument( + "--cross-pool-only", action="store_true", + help="Skip baseline ablation, load from cache, run only cross-pool", + ) + args = parser.parse_args() + os.environ.setdefault("JAX_PLATFORMS", "cpu") print("=" * 70) @@ -611,55 +700,71 @@ def main(): print(" Canonicalization + Cross-Pool Lag Features") print("=" * 70) - panel, matched = load_and_match() - - # Step 1: Option C baseline (reduced x_obs) - from quantammsim.calibration.per_pool_fit import fit_all_pools - print(f"\n--- Option C Reduced: per-pool fits ({len(matched)} pools) ---") - option_c = fit_all_pools(matched, fix_gas_to_chain=True, reduced=True) - losses = [r["loss"] for r in option_c.values()] - print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") - - # Step 2: Filter pathological pools - matched_clean, option_c_clean = filter_pathological(matched, option_c) - - # Phase 0: Diagnostic - diag = run_phase0_diagnostic(matched_clean, option_c_clean) - - # ---- Ablation: Baseline (K_OBS_REDUCED=4, no cross-pool) ---- - print("\n" + "=" * 70) - print("ABLATION 1: Baseline (K_OBS_REDUCED=4, no cross-pool features)") - print("=" * 70) + # ---- Stage 1: Option C + filtering + Phase 0 ---- + cached_s1 = _load_stage1() if args.cross_pool_only else None + + if cached_s1 is not None: + matched_clean = cached_s1["matched_clean"] + option_c_clean = cached_s1["option_c_clean"] + diag = cached_s1["diag"] + print(f" Using cached stage 1: {len(matched_clean)} pools") + else: + panel, matched = load_and_match() + + from quantammsim.calibration.per_pool_fit import fit_all_pools + print(f"\n--- Option C Reduced: per-pool fits ({len(matched)} pools) ---") + option_c = fit_all_pools(matched, fix_gas_to_chain=True, reduced=True) + losses = [r["loss"] for r in option_c.values()] + print(f" Loss: median={np.median(losses):.4f}, mean={np.mean(losses):.4f}") + + matched_clean, option_c_clean = filter_pathological(matched, option_c) + diag = run_phase0_diagnostic(matched_clean, option_c_clean) + _save_stage1(matched_clean, option_c_clean, diag) + + # ---- Ablation 1: Baseline ---- + if args.cross_pool_only: + cached_bl = _load_baseline() + if cached_bl is not None: + result_base = cached_bl["result_base"] + enc_base = cached_bl["enc_base"] + sweep_baseline = cached_bl["sweep_baseline"] + loo_baseline = cached_bl["loo_baseline"] + else: + print(" No baseline cache found — skipping baseline ablation.") + result_base = enc_base = sweep_baseline = loo_baseline = None + else: + print("\n" + "=" * 70) + print("ABLATION 1: Baseline (K_OBS_REDUCED=4, no cross-pool features)") + print("=" * 70) - result_base, model_base, jdata_base, enc_base = run_token_factored( - matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) + result_base, _, jdata_base, enc_base = run_token_factored( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) - # Analysis - print_token_effects(result_base, enc_base) - print_chain_effects(result_base, enc_base) - print_delta_analysis(result_base, enc_base, jdata_base, matched_clean) + print_token_effects(result_base, enc_base) + print_chain_effects(result_base, enc_base) + print_delta_analysis(result_base, enc_base, jdata_base, matched_clean) - # Lambda sweep with annealing - sweep_baseline = run_lambda_sweep(matched_clean, option_c_clean, cross_pool=False) + sweep_baseline = run_lambda_sweep( + matched_clean, option_c_clean, cross_pool=False) + loo_baseline = run_loo_validation( + matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) - # LOO - loo_baseline = run_loo_validation( - matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=False) + _save_baseline(result_base, enc_base, sweep_baseline, loo_baseline) - # ---- Ablation: Cross-pool (K_OBS_CROSS=7) ---- + # ---- Ablation 2: Cross-pool ---- print("\n" + "=" * 70) print("ABLATION 2: Cross-pool lag features (K_OBS_CROSS=7)") print("=" * 70) - result_cross, model_cross, jdata_cross, enc_cross = run_token_factored( + result_cross, _, jdata_cross, enc_cross = run_token_factored( matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=True) print_token_effects(result_cross, enc_cross) print_chain_effects(result_cross, enc_cross) print_delta_analysis(result_cross, enc_cross, jdata_cross, matched_clean) - sweep_cross = run_lambda_sweep(matched_clean, option_c_clean, cross_pool=True) - + sweep_cross = run_lambda_sweep( + matched_clean, option_c_clean, cross_pool=True) loo_cross = run_loo_validation( matched_clean, option_c_clean, lambda_delta=1.0, cross_pool=True) @@ -667,7 +772,10 @@ def main(): print("\n" + "=" * 70) print("ABLATION COMPARISON") print("=" * 70) - for label, loo in [("Baseline (k=4)", loo_baseline), ("Cross-pool (k=7)", loo_cross)]: + ablations = [("Cross-pool (k=7)", loo_cross)] + if loo_baseline is not None: + ablations.insert(0, ("Baseline (k=4)", loo_baseline)) + for label, loo in ablations: if loo: r2s = [r["r2_loo"] for r in loo] r2s_c = [r["r2_option_c"] for r in loo] @@ -680,56 +788,26 @@ def main(): print("\nGenerating plots...") os.makedirs(OUTPUT_DIR, exist_ok=True) - plot_lambda_sweep(sweep_baseline, OUTPUT_DIR, suffix="_baseline") + if sweep_baseline is not None: + plot_lambda_sweep(sweep_baseline, OUTPUT_DIR, suffix="_baseline") plot_lambda_sweep(sweep_cross, OUTPUT_DIR, suffix="_crosspool") - plot_token_effects(result_base, enc_base, OUTPUT_DIR) - plot_loo_scatter(loo_baseline, OUTPUT_DIR, suffix="_baseline") + if result_base is not None: + plot_token_effects(result_base, enc_base, OUTPUT_DIR) + if loo_baseline is not None: + plot_loo_scatter(loo_baseline, OUTPUT_DIR, suffix="_baseline") plot_loo_scatter(loo_cross, OUTPUT_DIR, suffix="_crosspool") # JSON export export = { "phase0_diagnostic": diag, - "baseline": { - "loss": result_base["loss"], - "data_loss": result_base["data_loss"], - "reg_loss": result_base["reg_loss"], - "init_loss": result_base["init_loss"], - "converged": result_base["converged"], - "n_pools": result_base["n_pools"], - "n_tokens": enc_base["n_tokens"], - "n_chains": enc_base["n_chains"], - "token_index": enc_base["token_index"], - "chain_index": enc_base["chain_index"], - "token_effects": result_base["token_effects"].tolist(), - "Gamma": result_base["Gamma"].tolist(), - "chain_effects": result_base["chain_effects"].tolist(), - "beta_fee": result_base["beta_fee"].tolist(), - "noise_deltas": result_base["noise_deltas"].tolist(), - "noise_coeffs": result_base["noise_coeffs"].tolist(), - }, - "cross_pool": { - "loss": result_cross["loss"], - "data_loss": result_cross["data_loss"], - "reg_loss": result_cross["reg_loss"], - "init_loss": result_cross["init_loss"], - "converged": result_cross["converged"], - "n_pools": result_cross["n_pools"], - "n_tokens": enc_cross["n_tokens"], - "n_chains": enc_cross["n_chains"], - "token_index": enc_cross["token_index"], - "chain_index": enc_cross["chain_index"], - "token_effects": result_cross["token_effects"].tolist(), - "Gamma": result_cross["Gamma"].tolist(), - "chain_effects": result_cross["chain_effects"].tolist(), - "beta_fee": result_cross["beta_fee"].tolist(), - "noise_deltas": result_cross["noise_deltas"].tolist(), - "noise_coeffs": result_cross["noise_coeffs"].tolist(), - }, - "lambda_sweep_baseline": sweep_baseline, + "cross_pool": _export_ablation_result(result_cross, enc_cross), "lambda_sweep_crosspool": sweep_cross, - "loo_baseline": loo_baseline, "loo_crosspool": loo_cross, } + if result_base is not None: + export["baseline"] = _export_ablation_result(result_base, enc_base) + export["lambda_sweep_baseline"] = sweep_baseline + export["loo_baseline"] = loo_baseline json_path = os.path.join(OUTPUT_DIR, "token_factored_v2_results.json") with open(json_path, "w") as f: json.dump(export, f, indent=2, default=str) diff --git a/tests/calibration/test_joint_fit.py b/tests/calibration/test_joint_fit.py index 3a3971d9..1fe02f27 100644 --- a/tests/calibration/test_joint_fit.py +++ b/tests/calibration/test_joint_fit.py @@ -375,6 +375,34 @@ def test_data_loss_leq_total(self, matched_data): assert result["reg_loss"] >= -1e-10 +class TestPrepareTokenFactoredCrossPool: + """Test prepare_token_factored_data with cross_pool=True.""" + + def test_cross_pool_jdata_shapes_consistent(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + + jdata, enc = prepare_token_factored_data(matched_data, cross_pool=True) + for pd in jdata.pool_data: + assert pd["x_obs"].shape[0] == pd["y_obs"].shape[0] + assert pd["x_obs"].shape[0] == pd["day_indices"].shape[0] + + def test_cross_pool_x_obs_has_7_cols(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.pool_data import K_OBS_CROSS + + jdata, _ = prepare_token_factored_data(matched_data, cross_pool=True) + for pd in jdata.pool_data: + assert pd["x_obs"].shape[1] == K_OBS_CROSS + + def test_cross_pool_drops_first_obs(self, matched_data): + from quantammsim.calibration.joint_fit import prepare_token_factored_data + + jdata_base, _ = prepare_token_factored_data(matched_data, cross_pool=False) + jdata_cross, _ = prepare_token_factored_data(matched_data, cross_pool=True) + for pd_base, pd_cross in zip(jdata_base.pool_data, jdata_cross.pool_data): + assert pd_cross["y_obs"].shape[0] == pd_base["y_obs"].shape[0] - 1 + + class TestPrepareJointDataReduced: """Test prepare_joint_data with reduced_x_obs=True.""" From 21fb161832df852ef827c03533e82cb12f058598 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 17 Mar 2026 13:14:06 +0000 Subject: [PATCH 059/115] feat: cross-pool volume prediction experiments MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Diagnostic experiments establishing the cross-pool prediction landscape: - run_cross_pool_diagnostics: lambda_token sweep, leave-one-in, AR1 baseline, pool connectivity analysis - run_cross_pool_linear: ridge regression (peers only, peers+own lag, LOO with overlap transfer, 30d burn-in, peer mean) - run_cross_pool_noise_linear: same battery on noise residuals (log_vol - log_V_arb) - run_residual_comparison: apples-to-apples R² on noise residual target across all methods including Option C - run_deepsets_volume: DeepSets on total volume (v1, raw) - run_deepsets_noise: DeepSets with V_arb decomposition and Optuna - run_deepsets_v2: full feature menu with Optuna feature selection, trains on total volume, evaluates on noise residual Key findings: ridge in-sample peers+own = 0.599 (matching Option C), but cross-pool signal is almost entirely shared arb response — noise residual ridge ceiling is 0.098. Option C noise residual R² = 0.060. --- experiments/run_cross_pool_diagnostics.py | 423 +++++++++++++ experiments/run_cross_pool_linear.py | 457 ++++++++++++++ experiments/run_cross_pool_noise_linear.py | 392 ++++++++++++ experiments/run_deepsets_noise.py | 595 ++++++++++++++++++ experiments/run_deepsets_v2.py | 697 +++++++++++++++++++++ experiments/run_deepsets_volume.py | 467 ++++++++++++++ experiments/run_residual_comparison.py | 235 +++++++ 7 files changed, 3266 insertions(+) create mode 100644 experiments/run_cross_pool_diagnostics.py create mode 100644 experiments/run_cross_pool_linear.py create mode 100644 experiments/run_cross_pool_noise_linear.py create mode 100644 experiments/run_deepsets_noise.py create mode 100644 experiments/run_deepsets_v2.py create mode 100644 experiments/run_deepsets_volume.py create mode 100644 experiments/run_residual_comparison.py diff --git a/experiments/run_cross_pool_diagnostics.py b/experiments/run_cross_pool_diagnostics.py new file mode 100644 index 00000000..ae9e2ed7 --- /dev/null +++ b/experiments/run_cross_pool_diagnostics.py @@ -0,0 +1,423 @@ +"""Diagnostic experiments for cross-pool noise calibration. + +Runs cheap experiments to bound the value of learned cross-pool aggregation: +1. Lambda_token sweep — is the LOO failure due to overfitting token effects? +2. Leave-one-in — how much pool-specific data closes the gap? +3. Naive AR baseline — is the model barely beating lag-1? +4. Pool connectivity — which pools are predictable at all? + +Uses cached stage1 data from run_token_factored_calibration.py. +""" + +import os +import pickle +import sys + +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +JOINT_MAXITER = 3000 # reduced for sweep speed + + +def load_stage1(): + """Load cached stage 1 (matched_clean + option_c_clean).""" + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache. Run run_token_factored_calibration.py first.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + print(f"Loaded {len(data['matched_clean'])} pools from cache") + return data["matched_clean"], data["option_c_clean"] + + +# ---- Diagnostic 1: Lambda_token sweep ---- + + +def run_lambda_token_sweep(matched_clean, option_c_clean): + """LOO with varying lambda_token to test whether overfitting is the problem.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED, build_x_obs, _parse_tokens + import jax.numpy as jnp + + pool_ids = sorted(matched_clean.keys()) + lambda_tokens = [0.1, 0.5, 1.0, 5.0, 10.0] + + print("\n" + "=" * 70) + print("Diagnostic 1: Lambda_token sweep (LOO)") + print("=" * 70) + print(f" lambda_delta=1.0 fixed, sweeping lambda_token") + print(f" maxiter={JOINT_MAXITER}") + + all_results = {} + + for lt in lambda_tokens: + print(f"\n--- lambda_token={lt} ---") + loo_r2s = [] + + for hold_out_pid in pool_ids: + train_matched = {p: matched_clean[p] for p in pool_ids if p != hold_out_pid} + train_oc = {p: option_c_clean[p] for p in pool_ids if p != hold_out_pid} + + if len(train_matched) < 3: + continue + + jdata, enc = prepare_token_factored_data(train_matched) + + gas_values = [] + for pid in jdata.pool_ids: + chain = train_matched[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead( + k_obs=K_OBS_REDUCED, + lambda_delta=1.0, + lambda_token=lt, + **enc, + ) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=train_oc) + + # Predict for held-out pool + n_train = len(jdata.pool_data) + k_attr = jdata.x_attr.shape[1] + (_, _), (_, _), (ns, ne) = model._head_slices(n_train, k_attr) + noise_params = result["params_flat"][ns:ne] + + ho_entry = matched_clean[hold_out_pid] + toks = _parse_tokens(ho_entry["tokens"]) + ho_pred = noise_head.predict_new_pool( + noise_params, toks[0], toks[1], + ho_entry["chain"], ho_entry["fee"], + n_pools=n_train, + ) + + # Evaluate + ho_panel = ho_entry["panel"] + x_obs_ho = build_x_obs(ho_panel, reduced=True) + y_obs_ho = ho_panel["log_volume"].values.astype(float) + + oc_ho = option_c_clean[hold_out_pid] + v_arb_all = np.array(interpolate_pool_daily( + ho_entry["coeffs"], + jnp.float64(oc_ho["log_cadence"]), + jnp.float64(np.exp(oc_ho["log_gas"])), + )) + v_arb = v_arb_all[ho_entry["day_indices"]] + v_noise = np.exp(x_obs_ho @ ho_pred["noise_coeffs"][:K_OBS_REDUCED]) + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y_obs_ho) ** 2) + ss_tot = np.sum((y_obs_ho - y_obs_ho.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + loo_r2s.append(r2) + + tag = "OK" if r2 > 0 else "NEG" + print(f" {hold_out_pid[:16]} R²={r2:.3f} [{tag}]") + + median_r2 = np.median(loo_r2s) + wins = sum(1 for r2, pid in zip(loo_r2s, pool_ids) + if r2 > option_c_clean[pid].get("r2", 0)) + all_results[lt] = { + "median_r2": median_r2, + "mean_r2": np.mean(loo_r2s), + "r2s": loo_r2s, + "n_negative": sum(1 for r in loo_r2s if r < 0), + } + print(f" lambda_token={lt}: median R²={median_r2:.4f}, " + f"mean={np.mean(loo_r2s):.4f}, " + f"n_negative={sum(1 for r in loo_r2s if r < 0)}") + + # Summary table + print(f"\n{'='*60}") + print(f"{'lambda_token':>12} {'median_R²':>10} {'mean_R²':>10} {'n_neg':>6}") + print("-" * 42) + for lt in lambda_tokens: + r = all_results[lt] + print(f"{lt:>12.1f} {r['median_r2']:>10.4f} {r['mean_r2']:>10.4f} " + f"{r['n_negative']:>6}") + + return all_results + + +# ---- Diagnostic 2: Leave-one-in ---- + + +def run_leave_one_in(matched_clean, option_c_clean, n_days_in=30): + """LOO but give held-out pool n_days_in days of data for adaptation.""" + from quantammsim.calibration.calibration_model import CalibrationModel + from quantammsim.calibration.heads import ( + FixedHead, PerPoolHead, TokenFactoredNoiseHead, + ) + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.joint_fit import prepare_token_factored_data + from quantammsim.calibration.loss import CHAIN_GAS_USD + from quantammsim.calibration.pool_data import K_OBS_REDUCED, build_x_obs, _parse_tokens + import jax.numpy as jnp + + pool_ids = sorted(matched_clean.keys()) + + print("\n" + "=" * 70) + print(f"Diagnostic 2: Leave-one-in ({n_days_in} days of held-out data)") + print("=" * 70) + + results = [] + + for hold_out_pid in pool_ids: + ho_entry = matched_clean[hold_out_pid] + ho_panel = ho_entry["panel"] + n_obs = len(ho_panel) + + if n_obs <= n_days_in + 10: + print(f" {hold_out_pid[:16]} — too few obs ({n_obs}), skipping") + continue + + # Split: first n_days_in for training, rest for evaluation + train_panel = ho_panel.iloc[:n_days_in].copy() + eval_panel = ho_panel.iloc[n_days_in:].copy() + train_day_indices = ho_entry["day_indices"][:n_days_in] + eval_day_indices = ho_entry["day_indices"][n_days_in:] + + # Build training matched: all other pools + truncated held-out pool + train_matched = {} + for p in pool_ids: + if p != hold_out_pid: + train_matched[p] = matched_clean[p] + + # Add truncated held-out pool + ho_train_entry = dict(ho_entry) + ho_train_entry["panel"] = train_panel.reset_index(drop=True) + ho_train_entry["day_indices"] = train_day_indices + train_matched[hold_out_pid] = ho_train_entry + + train_oc = dict(option_c_clean) # all pools including held-out + + # Fit with held-out pool included (gets its own delta from 30 days) + jdata, enc = prepare_token_factored_data(train_matched) + + gas_values = [] + for pid in jdata.pool_ids: + chain = train_matched[pid]["chain"] + gas_values.append(np.log(max(CHAIN_GAS_USD.get(chain, 1.0), 1e-6))) + + noise_head = TokenFactoredNoiseHead( + k_obs=K_OBS_REDUCED, + lambda_delta=1.0, + lambda_token=0.1, + **enc, + ) + model = CalibrationModel( + PerPoolHead("log_cadence", default=np.log(12.0)), + FixedHead("log_gas", np.array(gas_values)), + noise_head, + ) + result = model.fit(jdata, maxiter=JOINT_MAXITER, warm_start=train_oc) + + # Find held-out pool's index in training set and extract noise_coeffs + ho_idx = jdata.pool_ids.index(hold_out_pid) + noise_coeffs = result["noise_coeffs"][ho_idx] + + # Evaluate on held-out days + x_obs_eval = build_x_obs(eval_panel, reduced=True) + y_obs_eval = eval_panel["log_volume"].values.astype(float) + + oc_ho = option_c_clean[hold_out_pid] + v_arb_all = np.array(interpolate_pool_daily( + ho_entry["coeffs"], + jnp.float64(oc_ho["log_cadence"]), + jnp.float64(np.exp(oc_ho["log_gas"])), + )) + v_arb = v_arb_all[eval_day_indices] + v_noise = np.exp(x_obs_eval @ noise_coeffs[:K_OBS_REDUCED]) + log_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + ss_res = np.sum((log_pred - y_obs_eval) ** 2) + ss_tot = np.sum((y_obs_eval - y_obs_eval.mean()) ** 2) + r2_in = 1 - ss_res / max(ss_tot, 1e-10) + + # Also compute Option C R² on eval days for comparison + v_noise_c = np.exp(x_obs_eval @ oc_ho["noise_coeffs"][:K_OBS_REDUCED]) + log_pred_c = np.log(np.maximum(v_arb + v_noise_c, 1e-6)) + ss_res_c = np.sum((log_pred_c - y_obs_eval) ** 2) + r2_c_eval = 1 - ss_res_c / max(ss_tot, 1e-10) + + results.append({ + "pool_id": hold_out_pid, + "r2_leave_one_in": r2_in, + "r2_option_c_eval": r2_c_eval, + "n_train_days": n_days_in, + "n_eval_days": len(eval_panel), + "tokens": ho_entry["tokens"], + }) + + print(f" {hold_out_pid[:16]} ({ho_entry['tokens']:<14}) " + f"R²_in={r2_in:.3f} R²_C_eval={r2_c_eval:.3f} " + f"n_eval={len(eval_panel)}") + + if results: + r2s_in = [r["r2_leave_one_in"] for r in results] + r2s_c = [r["r2_option_c_eval"] for r in results] + print(f"\n Leave-one-in ({n_days_in}d): median R²={np.median(r2s_in):.4f}") + print(f" Option C (eval days): median R²={np.median(r2s_c):.4f}") + print(f" Recall: zero-shot LOO: median R²=0.362") + + return results + + +# ---- Diagnostic 3: Naive AR baseline ---- + + +def run_naive_ar_baseline(matched_clean): + """Compute R² of vol_tomorrow = vol_today (no model, no cross-pool).""" + print("\n" + "=" * 70) + print("Diagnostic 3: Naive autoregressive baseline (lag-1 copy)") + print("=" * 70) + + pool_r2s = [] + for pid in sorted(matched_clean.keys()): + panel = matched_clean[pid]["panel"] + y = panel["log_volume"].values.astype(float) + + if len(y) < 3: + continue + + # Predict day t from day t-1 + y_true = y[1:] + y_pred = y[:-1] + + ss_res = np.sum((y_pred - y_true) ** 2) + ss_tot = np.sum((y_true - y_true.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²_AR1={r2:.3f} n_obs={len(y)}") + + print(f"\n Naive AR1: median R²={np.median(pool_r2s):.4f}, " + f"mean={np.mean(pool_r2s):.4f}") + print(f" Recall: zero-shot LOO = 0.362, Option C in-sample = 0.589") + + return pool_r2s + + +# ---- Diagnostic 4: Pool connectivity analysis ---- + + +def run_connectivity_analysis(matched_clean, option_c_clean): + """Analyze token overlap and partition LOO R² by connectivity.""" + from quantammsim.calibration.pool_data import _parse_tokens, _canonicalize_token + + print("\n" + "=" * 70) + print("Diagnostic 4: Pool connectivity analysis") + print("=" * 70) + + pool_ids = sorted(matched_clean.keys()) + + # Build canonical token sets per pool + pool_tokens = {} + for pid in pool_ids: + toks = _parse_tokens(matched_clean[pid]["tokens"]) + canon = {_canonicalize_token(t) for t in toks[:2]} + pool_tokens[pid] = canon + + # Count: for each pool, how many other pools share at least 1 token? + # And how many share both tokens? + print(f"\n{'Pool':<18} {'Tokens':<16} {'1+ shared':>10} {'2 shared':>10} " + f"{'R²_C':>8}") + print("-" * 66) + + connectivity = [] + for pid in pool_ids: + my_toks = pool_tokens[pid] + n_one_shared = 0 + n_both_shared = 0 + for other in pool_ids: + if other == pid: + continue + overlap = len(my_toks & pool_tokens[other]) + if overlap >= 1: + n_one_shared += 1 + if overlap >= 2: + n_both_shared += 1 + + oc = option_c_clean[pid] + r2_c = 1 - oc["loss"] / max( + np.var(matched_clean[pid]["panel"]["log_volume"].values) * + len(matched_clean[pid]["panel"]) / + max(len(matched_clean[pid]["panel"]) - 1, 1), + 1e-10, + ) + + connectivity.append({ + "pool_id": pid, + "tokens": matched_clean[pid]["tokens"], + "n_one_shared": n_one_shared, + "n_both_shared": n_both_shared, + }) + + print(f" {pid[:16]} {matched_clean[pid]['tokens']:<16} " + f"{n_one_shared:>10} {n_both_shared:>10}") + + # Partition: well-connected (1+ shared ≥ 3) vs isolated + well_connected = [c for c in connectivity if c["n_one_shared"] >= 3] + isolated = [c for c in connectivity if c["n_one_shared"] < 3] + + print(f"\n Well-connected (≥3 pools share a token): {len(well_connected)}") + print(f" Isolated (<3 pools share a token): {len(isolated)}") + + if isolated: + print(f"\n Isolated pools:") + for c in isolated: + print(f" {c['pool_id'][:16]} {c['tokens']}") + + return connectivity + + +# ---- Main ---- + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Cross-Pool Calibration Diagnostics") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + # Run all diagnostics + ar_results = run_naive_ar_baseline(matched_clean) + connectivity = run_connectivity_analysis(matched_clean, option_c_clean) + leave_one_in = run_leave_one_in(matched_clean, option_c_clean, n_days_in=30) + lambda_sweep = run_lambda_token_sweep(matched_clean, option_c_clean) + + # Final summary + print("\n" + "=" * 70) + print("SUMMARY") + print("=" * 70) + print(f" Naive AR1 baseline: median R² = {np.median(ar_results):.4f}") + if leave_one_in: + r2s_in = [r["r2_leave_one_in"] for r in leave_one_in] + print(f" Leave-one-in (30 days): median R² = {np.median(r2s_in):.4f}") + print(f" Zero-shot LOO (current): median R² = 0.362") + print(f" Option C in-sample: median R² = 0.589") + print(f"\n Lambda_token sweep:") + for lt, r in sorted(lambda_sweep.items()): + print(f" lambda_token={lt:>5.1f}: median R² = {r['median_r2']:.4f} " + f"(n_neg={r['n_negative']})") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_cross_pool_linear.py b/experiments/run_cross_pool_linear.py new file mode 100644 index 00000000..acd5c48f --- /dev/null +++ b/experiments/run_cross_pool_linear.py @@ -0,0 +1,457 @@ +"""Cross-pool linear volume prediction baselines. + +1. Ridge cross-pool regression: log_vol_i_t = W_i @ log_vol_{-i, t-1} + - In-sample: fit full 36x36 W, evaluate on training data + - LOO: hold out pool i, fit W on 35 pools, predict pool i using + token-overlap-weighted average of learned rows (transfer via similarity) + - LOO with burn-in: use 30 days of pool i to learn its row of W directly + +2. Zero-parameter peer-mean: predicted_vol_i_t = mean(log_vol_{j,t-1}) + for peers sharing a canonical token with pool i +""" + +import os +import pickle +import sys + +import numpy as np +from sklearn.linear_model import RidgeCV + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache. Run run_token_factored_calibration.py first.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + print(f"Loaded {len(data['matched_clean'])} pools from cache") + return data["matched_clean"], data["option_c_clean"] + + +def build_volume_matrix(matched_clean): + """Build (n_dates, n_pools) aligned volume matrix. + + Returns vol_matrix, date_list, pool_ids. + Dates are the intersection of all pools' date ranges. + Missing values filled with NaN. + """ + pool_ids = sorted(matched_clean.keys()) + + # Collect all (pool, date) -> log_volume + pool_date_vol = {} + all_dates = set() + for pid in pool_ids: + panel = matched_clean[pid]["panel"] + dates = panel["date"].values + vols = panel["log_volume"].values.astype(float) + pool_date_vol[pid] = dict(zip(dates, vols)) + all_dates.update(dates) + + date_list = sorted(all_dates) + n_dates = len(date_list) + n_pools = len(pool_ids) + + vol_matrix = np.full((n_dates, n_pools), np.nan) + for j, pid in enumerate(pool_ids): + dv = pool_date_vol[pid] + for t, date in enumerate(date_list): + if date in dv: + vol_matrix[t, j] = dv[date] + + return vol_matrix, date_list, pool_ids + + +def build_token_overlap(matched_clean, pool_ids): + """Build (n_pools, n_pools) token overlap matrix (0, 1, or 2).""" + from quantammsim.calibration.pool_data import _parse_tokens, _canonicalize_token + + n = len(pool_ids) + overlap = np.zeros((n, n), dtype=np.int32) + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + for i in range(n): + for j in range(n): + overlap[i, j] = len(pool_tokens[i] & pool_tokens[j]) + + return overlap + + +def r2_score(y_true, y_pred): + ss_res = np.sum((y_true - y_pred) ** 2) + ss_tot = np.sum((y_true - y_true.mean()) ** 2) + return 1 - ss_res / max(ss_tot, 1e-10) + + +# ---- 1. Ridge cross-pool regression ---- + + +def run_ridge_cross_pool(matched_clean): + """Full cross-pool ridge: predict each pool from all others' lag-1.""" + vol_matrix, date_list, pool_ids = build_volume_matrix(matched_clean) + n_dates, n_pools = vol_matrix.shape + + # Check if fully-observed rows exist + X_lag = vol_matrix[:-1, :] + Y_cur = vol_matrix[1:, :] + valid = ~np.any(np.isnan(X_lag), axis=1) & ~np.any(np.isnan(Y_cur), axis=1) + n_valid = int(valid.sum()) + + # Peers only + print("\n" + "=" * 70) + print("1a. Ridge cross-pool regression (in-sample, peers only)") + print("=" * 70) + print(f" {n_pools} pools, {n_valid} fully-observed day pairs " + f"(of {n_dates-1} total)") + print(" Using per-pool valid rows with NaN imputation.") + r2_peers, _, _, _ = _run_ridge_per_pool_valid( + vol_matrix, pool_ids, matched_clean, include_own_lag=False) + + # Peers + own lag + print("\n" + "=" * 70) + print("1a+. Ridge cross-pool regression (in-sample, peers + own lag)") + print("=" * 70) + r2_both, _, _, _ = _run_ridge_per_pool_valid( + vol_matrix, pool_ids, matched_clean, include_own_lag=True) + + return r2_peers, r2_both, vol_matrix, date_list, pool_ids + + +def _run_ridge_per_pool_valid(vol_matrix, pool_ids, matched_clean, + include_own_lag=False): + """Fallback: per-pool ridge using only rows where pool i AND predictors have data.""" + n_dates, n_pools = vol_matrix.shape + tag = " + own_lag" if include_own_lag else "" + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + X_lag = vol_matrix[:-1, :] + y_cur = vol_matrix[1:, i] + own_lag = X_lag[:, i] # pool i's own lag + + # Valid: pool i has data today AND own lag exists (if used) + valid_y = ~np.isnan(y_cur) + if include_own_lag: + valid_y = valid_y & ~np.isnan(own_lag) + + X_others = np.delete(X_lag, i, axis=1) + + # For each predictor, fill NaN with that predictor's mean (simple imputation) + X_filled = X_others.copy() + for j in range(X_filled.shape[1]): + col = X_filled[:, j] + col_mean = np.nanmean(col) + col[np.isnan(col)] = col_mean + X_filled[:, j] = col + + if include_own_lag: + X_full = np.column_stack([X_filled, own_lag[:, None]]) + else: + X_full = X_filled + + X_i = X_full[valid_y] + y_i = y_cur[valid_y] + + if len(y_i) < 10: + pool_r2s.append(np.nan) + continue + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_i, y_i) + y_pred = model.predict(X_i) + r2 = r2_score(y_i, y_pred) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n_obs={len(y_i)} alpha={model.alpha_:.1f}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n In-sample ridge{tag}: median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + + return pool_r2s, vol_matrix, None, pool_ids + + +def run_ridge_loo(matched_clean): + """LOO cross-pool ridge with token-overlap transfer.""" + print("\n" + "=" * 70) + print("1b. Ridge cross-pool LOO (transfer via token overlap)") + print("=" * 70) + + vol_matrix, date_list, pool_ids = build_volume_matrix(matched_clean) + n_dates, n_pools = vol_matrix.shape + overlap = build_token_overlap(matched_clean, pool_ids) + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + # Training pools: all except i + train_idx = [j for j in range(n_pools) if j != i] + n_train = len(train_idx) + + # Build training data: for each training pool k, predict from others' lag + # Use per-pool valid rows with NaN imputation + X_lag_all = vol_matrix[:-1, :] + Y_cur_all = vol_matrix[1:, :] + + # Fit a ridge model for each training pool + train_models = {} + train_weights = {} # weight vectors (excluding self) + for k_pos, k in enumerate(train_idx): + # Predictors: all pools except k (including pool i's historical data!) + pred_idx = [j for j in range(n_pools) if j != k] + X_k = X_lag_all[:, pred_idx].copy() + y_k = Y_cur_all[:, k] + + valid = ~np.isnan(y_k) + for c in range(X_k.shape[1]): + col = X_k[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_k[:, c] = col + + X_k = X_k[valid] + y_k = y_k[valid] + + if len(y_k) < 10: + continue + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_k, y_k) + train_models[k] = model + + # Store full weight vector (n_pools-1,) with mapping to pool indices + w = np.zeros(n_pools) + for widx, pidx in enumerate(pred_idx): + w[pidx] = model.coef_[widx] + w_intercept = model.intercept_ + train_weights[k] = (w, w_intercept) + + if not train_weights: + pool_r2s.append(np.nan) + continue + + # Transfer to held-out pool i: weighted average of training pools' weight vectors + # Weight by token overlap with pool i + w_transfer = np.zeros(n_pools) + intercept_transfer = 0.0 + total_sim = 0.0 + for k in train_weights: + sim = overlap[i, k] + if sim == 0: + sim = 0.1 # small weight for unrelated pools + w_k, b_k = train_weights[k] + w_transfer += sim * w_k + intercept_transfer += sim * b_k + total_sim += sim + + w_transfer /= total_sim + intercept_transfer /= total_sim + + # Zero out pool i's own weight (shouldn't predict from self) + w_transfer[i] = 0.0 + + # Predict pool i + X_lag_i = vol_matrix[:-1, :].copy() + y_true_i = vol_matrix[1:, i] + valid = ~np.isnan(y_true_i) + + # Impute NaN predictors + for c in range(X_lag_i.shape[1]): + col = X_lag_i[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_lag_i[:, c] = col + + y_pred_i = X_lag_i[valid] @ w_transfer + intercept_transfer + y_true_i = y_true_i[valid] + + r2 = r2_score(y_true_i, y_pred_i) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n_eval={len(y_true_i)}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n LOO ridge (overlap transfer): median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + print(f" Recall: AR1={0.397:.3f}, zero-shot token-factored={0.362:.3f}") + + return pool_r2s + + +def run_ridge_loo_burnin(matched_clean, n_burnin=30): + """LOO with burn-in: learn pool i's weight row from n_burnin days.""" + print("\n" + "=" * 70) + print(f"1c. Ridge cross-pool LOO with {n_burnin}-day burn-in") + print("=" * 70) + + vol_matrix, date_list, pool_ids = build_volume_matrix(matched_clean) + n_dates, n_pools = vol_matrix.shape + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + # Pool i's data + y_all = vol_matrix[:, i] + valid_days = ~np.isnan(y_all) + valid_indices = np.where(valid_days)[0] + + if len(valid_indices) < n_burnin + 10: + print(f" {pid[:16]} — too few obs, skipping") + pool_r2s.append(np.nan) + continue + + # Split: first n_burnin valid days for training, rest for eval + burn_indices = valid_indices[:n_burnin] + eval_indices = valid_indices[n_burnin:] + + # Training: predict pool i from all others' lag using burn-in days + # Need (day, day-1) pairs where day is in burn_indices and day >= 1 + burn_pairs = burn_indices[burn_indices >= 1] + + pred_idx = [j for j in range(n_pools) if j != i] + X_burn = vol_matrix[burn_pairs - 1][:, pred_idx].copy() + y_burn = vol_matrix[burn_pairs, i] + + # Impute NaN + for c in range(X_burn.shape[1]): + col = X_burn[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_burn[:, c] = col + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_burn, y_burn) + + # Evaluate on remaining days + eval_pairs = eval_indices[eval_indices >= 1] + X_eval = vol_matrix[eval_pairs - 1][:, pred_idx].copy() + y_eval = vol_matrix[eval_pairs, i] + + for c in range(X_eval.shape[1]): + col = X_eval[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_eval[:, c] = col + + y_pred = model.predict(X_eval) + r2 = r2_score(y_eval, y_pred) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n_burn={len(burn_pairs)} n_eval={len(eval_pairs)} " + f"alpha={model.alpha_:.1f}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n LOO ridge ({n_burnin}d burn-in): " + f"median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + + return pool_r2s + + +# ---- 2. Zero-parameter peer-mean ---- + + +def run_peer_mean_baseline(matched_clean): + """Predict pool i's volume as mean of token-peer lagged volumes.""" + from quantammsim.calibration.pool_data import _parse_tokens, _canonicalize_token + + print("\n" + "=" * 70) + print("2. Zero-parameter peer-mean baseline") + print("=" * 70) + + vol_matrix, date_list, pool_ids = build_volume_matrix(matched_clean) + n_dates, n_pools = vol_matrix.shape + overlap = build_token_overlap(matched_clean, pool_ids) + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + # Peers: pools sharing at least 1 token + peers = [j for j in range(n_pools) if j != i and overlap[i, j] >= 1] + + if not peers: + pool_r2s.append(np.nan) + continue + + y_true = vol_matrix[1:, i] + valid = ~np.isnan(y_true) + + # Peer mean at t-1 + peer_lag = vol_matrix[:-1, :][:, peers] + peer_mean = np.nanmean(peer_lag, axis=1) + + y_pred = peer_mean[valid] + y_true = y_true[valid] + + # Remove any remaining NaN + both_valid = ~np.isnan(y_pred) + y_pred = y_pred[both_valid] + y_true = y_true[both_valid] + + r2 = r2_score(y_true, y_pred) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n_peers={len(peers)} n_obs={len(y_true)}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n Peer-mean baseline: median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + print(f" Recall: AR1={0.397:.3f}, zero-shot token-factored={0.362:.3f}") + + return pool_r2s + + +# ---- Main ---- + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Cross-Pool Linear Volume Prediction Baselines") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + # 1a. In-sample ridge (peers only + peers + own lag) + r2_peers, r2_both, vol_matrix, date_list, pool_ids = run_ridge_cross_pool(matched_clean) + + # 1b. LOO ridge with token-overlap transfer + loo_r2s = run_ridge_loo(matched_clean) + + # 1c. LOO ridge with 30-day burn-in + burnin_r2s = run_ridge_loo_burnin(matched_clean, n_burnin=30) + + # 2. Zero-parameter peer mean + peer_r2s = run_peer_mean_baseline(matched_clean) + + # Summary + def safe_median(xs): + v = [x for x in xs if not np.isnan(x)] + return np.median(v) if v else float("nan") + + print("\n" + "=" * 70) + print("SUMMARY") + print("=" * 70) + print(f" Ridge in-sample (peers): median R² = {safe_median(r2_peers):.4f}") + print(f" Ridge in-sample (+own): median R² = {safe_median(r2_both):.4f}") + print(f" Ridge LOO (overlap xfer): median R² = {safe_median(loo_r2s):.4f}") + print(f" Ridge LOO (30d burn-in): median R² = {safe_median(burnin_r2s):.4f}") + print(f" Peer-mean (0 params): median R² = {safe_median(peer_r2s):.4f}") + print(f" ---") + print(f" Naive AR1: median R² = 0.397") + print(f" Token-factored LOO: median R² = 0.362") + print(f" Option C in-sample: median R² = 0.589") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_cross_pool_noise_linear.py b/experiments/run_cross_pool_noise_linear.py new file mode 100644 index 00000000..08ead10a --- /dev/null +++ b/experiments/run_cross_pool_noise_linear.py @@ -0,0 +1,392 @@ +"""Cross-pool linear prediction of NOISE residuals. + +Decomposes total volume into V_arb (from grid + Option C cadence/gas) +and noise residual, then tests whether peer pools' lagged noise +residuals predict this pool's noise residual. + +1. Ridge in-sample: noise_resid_i_t = W_i @ noise_resid_{-i, t-1} [+ own_lag] +2. Ridge LOO with overlap transfer +3. Ridge LOO with 30-day burn-in +""" + +import os +import pickle +import sys + +import numpy as np +from sklearn.linear_model import RidgeCV + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + print(f"Loaded {len(data['matched_clean'])} pools from cache") + return data["matched_clean"], data["option_c_clean"] + + +def build_noise_residual_matrix(matched_clean, option_c_clean): + """Build (n_dates, n_pools) noise residual matrix. + + noise_resid_i_t = log_volume_i_t - log(V_arb_i_t) + + V_arb computed from grid interpolation at Option C cadence/gas. + Returns residual matrix (NaN where missing), date list, pool ids. + """ + import jax.numpy as jnp + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # Collect all dates + all_dates = set() + for pid in pool_ids: + panel = matched_clean[pid]["panel"] + all_dates.update(panel["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + # Build matrices + vol_matrix = np.full((n_dates, n_pools), np.nan) + resid_matrix = np.full((n_dates, n_pools), np.nan) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + panel = entry["panel"] + coeffs = entry["coeffs"] + day_indices = entry["day_indices"] + + # Compute V_arb from grid + v_arb_all = np.array(interpolate_pool_daily( + coeffs, + jnp.float64(oc["log_cadence"]), + jnp.float64(np.exp(oc["log_gas"])), + )) + v_arb = v_arb_all[day_indices] + log_v_arb = np.log(np.maximum(v_arb, 1e-6)) + + # Fill matrices + dates = panel["date"].values + log_vols = panel["log_volume"].values.astype(float) + + for k, date in enumerate(dates): + t = date_to_idx[date] + vol_matrix[t, j] = log_vols[k] + resid_matrix[t, j] = log_vols[k] - log_v_arb[k] + + print(f" Built noise residual matrix: {n_dates} dates x {n_pools} pools") + print(f" Residual stats: mean={np.nanmean(resid_matrix):.3f}, " + f"std={np.nanstd(resid_matrix):.3f}") + + return vol_matrix, resid_matrix, date_list, pool_ids + + +def build_token_overlap(matched_clean, pool_ids): + from quantammsim.calibration.pool_data import _parse_tokens, _canonicalize_token + n = len(pool_ids) + overlap = np.zeros((n, n), dtype=np.int32) + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + for i in range(n): + for j in range(n): + overlap[i, j] = len(pool_tokens[i] & pool_tokens[j]) + return overlap + + +def r2_score(y_true, y_pred): + ss_res = np.sum((y_true - y_pred) ** 2) + ss_tot = np.sum((y_true - y_true.mean()) ** 2) + return 1 - ss_res / max(ss_tot, 1e-10) + + +# ---- In-sample ridge on noise residuals ---- + + +def run_ridge_insample(resid_matrix, pool_ids, matched_clean): + """Per-pool ridge: predict noise_resid from peers' lagged residuals.""" + n_dates, n_pools = resid_matrix.shape + + for include_own in [False, True]: + tag = "peers + own_lag" if include_own else "peers only" + print(f"\n{'='*70}") + print(f"Ridge in-sample on noise residuals ({tag})") + print(f"{'='*70}") + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + X_lag = resid_matrix[:-1, :] + y_cur = resid_matrix[1:, i] + own_lag = X_lag[:, i] + + valid = ~np.isnan(y_cur) + if include_own: + valid = valid & ~np.isnan(own_lag) + + X_others = np.delete(X_lag, i, axis=1) + X_filled = X_others.copy() + for c in range(X_filled.shape[1]): + col = X_filled[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_filled[:, c] = col + + if include_own: + X_full = np.column_stack([X_filled, own_lag[:, None]]) + else: + X_full = X_filled + + X_i = X_full[valid] + y_i = y_cur[valid] + + if len(y_i) < 10: + pool_r2s.append(np.nan) + continue + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_i, y_i) + y_pred = model.predict(X_i) + r2 = r2_score(y_i, y_pred) + pool_r2s.append(r2) + + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n={len(y_i)} alpha={model.alpha_:.1f}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n In-sample ridge ({tag}): median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + + return pool_r2s + + +def run_ar1_noise_baseline(resid_matrix, pool_ids, matched_clean): + """Naive AR1 on noise residuals: resid_tomorrow = resid_today.""" + print(f"\n{'='*70}") + print("AR1 baseline on noise residuals") + print(f"{'='*70}") + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + y = resid_matrix[:, i] + valid = ~np.isnan(y[:-1]) & ~np.isnan(y[1:]) + y_true = y[1:][valid] + y_pred = y[:-1][valid] + + if len(y_true) < 3: + pool_r2s.append(np.nan) + continue + + r2 = r2_score(y_true, y_pred) + pool_r2s.append(r2) + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n={len(y_true)}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n AR1 noise residual: median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + return pool_r2s + + +# ---- LOO with overlap transfer ---- + + +def run_ridge_loo(resid_matrix, pool_ids, matched_clean): + """LOO on noise residuals with token-overlap weight transfer.""" + print(f"\n{'='*70}") + print("Ridge LOO on noise residuals (overlap transfer, peers + own_lag)") + print(f"{'='*70}") + + n_dates, n_pools = resid_matrix.shape + overlap = build_token_overlap(matched_clean, pool_ids) + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + train_idx = [j for j in range(n_pools) if j != i] + + # Fit ridge for each training pool + train_weights = {} + for k in train_idx: + pred_idx = [j for j in range(n_pools) if j != k] + X_lag = resid_matrix[:-1, pred_idx].copy() + own_lag_k = resid_matrix[:-1, k] + y_k = resid_matrix[1:, k] + + valid = ~np.isnan(y_k) & ~np.isnan(own_lag_k) + for c in range(X_lag.shape[1]): + col = X_lag[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_lag[:, c] = col + + X_k = np.column_stack([X_lag[valid], own_lag_k[valid, None]]) + y_k = y_k[valid] + + if len(y_k) < 10: + continue + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_k, y_k) + + # Store weights mapped to pool indices + w = np.zeros(n_pools + 1) # +1 for own_lag + for widx, pidx in enumerate(pred_idx): + w[pidx] = model.coef_[widx] + w[-1] = model.coef_[-1] # own_lag weight + train_weights[k] = (w, model.intercept_) + + if not train_weights: + pool_r2s.append(np.nan) + continue + + # Transfer: overlap-weighted average of training pool weight vectors + w_transfer = np.zeros(n_pools + 1) + b_transfer = 0.0 + total_sim = 0.0 + for k in train_weights: + sim = max(overlap[i, k], 0.1) + w_k, b_k = train_weights[k] + w_transfer += sim * w_k + b_transfer += sim * b_k + total_sim += sim + w_transfer /= total_sim + b_transfer /= total_sim + w_transfer[i] = 0.0 # no self-prediction from peers + + # Predict held-out pool + X_lag_all = resid_matrix[:-1, :].copy() + own_lag_i = resid_matrix[:-1, i] + y_true = resid_matrix[1:, i] + valid = ~np.isnan(y_true) & ~np.isnan(own_lag_i) + + for c in range(X_lag_all.shape[1]): + col = X_lag_all[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_lag_all[:, c] = col + + X_i = np.column_stack([X_lag_all[valid], own_lag_i[valid, None]]) + y_pred = X_i @ w_transfer + b_transfer + y_true = y_true[valid] + + r2 = r2_score(y_true, y_pred) + pool_r2s.append(r2) + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n={len(y_true)}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n LOO ridge (overlap transfer): median R²={np.median(valid_r2s):.4f}, " + f"mean={np.mean(valid_r2s):.4f}") + return pool_r2s + + +# ---- LOO with burn-in ---- + + +def run_ridge_burnin(resid_matrix, pool_ids, matched_clean, n_burnin=30): + """LOO with burn-in: learn pool i's weights from first n_burnin days.""" + print(f"\n{'='*70}") + print(f"Ridge LOO on noise residuals ({n_burnin}d burn-in, peers + own_lag)") + print(f"{'='*70}") + + n_dates, n_pools = resid_matrix.shape + + pool_r2s = [] + for i, pid in enumerate(pool_ids): + y_all = resid_matrix[:, i] + own_lag_all = np.full(n_dates, np.nan) + own_lag_all[1:] = y_all[:-1] + + valid_days = ~np.isnan(y_all) & ~np.isnan(own_lag_all) + valid_indices = np.where(valid_days)[0] + + if len(valid_indices) < n_burnin + 10: + pool_r2s.append(np.nan) + continue + + burn_idx = valid_indices[:n_burnin] + eval_idx = valid_indices[n_burnin:] + + pred_idx = [j for j in range(n_pools) if j != i] + + def build_X(indices): + X_peers = resid_matrix[indices - 1][:, pred_idx].copy() + for c in range(X_peers.shape[1]): + col = X_peers[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_peers[:, c] = col + own = y_all[indices - 1] + return np.column_stack([X_peers, own[:, None]]) + + X_burn = build_X(burn_idx) + y_burn = y_all[burn_idx] + + model = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model.fit(X_burn, y_burn) + + X_eval = build_X(eval_idx) + y_eval = y_all[eval_idx] + y_pred = model.predict(X_eval) + + r2 = r2_score(y_eval, y_pred) + pool_r2s.append(r2) + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"R²={r2:.3f} n_eval={len(eval_idx)} alpha={model.alpha_:.1f}") + + valid_r2s = [r for r in pool_r2s if not np.isnan(r)] + print(f"\n LOO ridge ({n_burnin}d burn-in): " + f"median R²={np.median(valid_r2s):.4f}, mean={np.mean(valid_r2s):.4f}") + return pool_r2s + + +# ---- Main ---- + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Cross-Pool Linear Prediction of Noise Residuals") + print(" (total volume - grid arb volume)") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + vol_matrix, resid_matrix, date_list, pool_ids = build_noise_residual_matrix( + matched_clean, option_c_clean) + + ar1_r2s = run_ar1_noise_baseline(resid_matrix, pool_ids, matched_clean) + insample_r2s = run_ridge_insample(resid_matrix, pool_ids, matched_clean) + loo_r2s = run_ridge_loo(resid_matrix, pool_ids, matched_clean) + burnin_r2s = run_ridge_burnin(resid_matrix, pool_ids, matched_clean, n_burnin=30) + + def safe_median(xs): + v = [x for x in xs if x is not None and not np.isnan(x)] + return np.median(v) if v else float("nan") + + print("\n" + "=" * 70) + print("SUMMARY (noise residuals)") + print("=" * 70) + print(f" AR1 noise residual: median R² = {safe_median(ar1_r2s):.4f}") + print(f" Ridge in-sample (+own): median R² = {safe_median(insample_r2s):.4f}") + print(f" Ridge LOO (overlap xfer): median R² = {safe_median(loo_r2s):.4f}") + print(f" Ridge LOO (30d burn-in): median R² = {safe_median(burnin_r2s):.4f}") + print(f" ---") + print(f" (total vol) AR1: median R² = 0.397") + print(f" (total vol) Ridge +own: median R² = 0.599") + print(f" Option C in-sample: median R² = 0.589") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_deepsets_noise.py b/experiments/run_deepsets_noise.py new file mode 100644 index 00000000..a6392563 --- /dev/null +++ b/experiments/run_deepsets_noise.py @@ -0,0 +1,595 @@ +"""DeepSets noise volume prediction with V_arb decomposition. + +Predicts V_noise via a shared encoder-decoder over peer pools. +V_arb is precomputed from grids at Option C cadence/gas. +Loss: mean((log(V_arb + V_noise_predicted) - log_volume)^2) + +Usage: + python experiments/run_deepsets_noise.py # default hparams + python experiments/run_deepsets_noise.py --tune 50 # Optuna, 50 trials +""" + +import argparse +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + +# Default hyperparameters +DEFAULTS = dict( + hidden=16, + d_embed=8, + lr=3e-4, + l2_alpha=1e-3, + n_epochs=1000, + include_own_lag=True, +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +# ---- Data construction ---- + + +def build_data(matched_clean, option_c_clean, exclude_pool_idx=None): + """Build training arrays with V_arb decomposition. + + For each (pool i, day t) sample: + - peer_vols: other pools' log_volume at t-1 + - v_arb: precomputed arb volume for pool i at day t + - local_features: [log_tvl_lag1, dow_sin, dow_cos] + - own_lag: pool i's log_volume at t-1 + - y: log_volume at day t + """ + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + build_pool_attributes, _parse_tokens, _canonicalize_token, + ) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # ---- Collect all dates, build volume + V_arb matrices ---- + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + vol_matrix = np.full((n_dates, n_pools), np.nan) + v_arb_matrix = np.full((n_dates, n_pools), np.nan) + tvl_matrix = np.full((n_dates, n_pools), np.nan) + weekday_matrix = np.full(n_dates, np.nan) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + panel = entry["panel"] + + # V_arb from grid + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], + jnp.float64(oc["log_cadence"]), + jnp.float64(np.exp(oc["log_gas"])), + )) + v_arb_day = v_arb_all[entry["day_indices"]] + + dates = panel["date"].values + log_vols = panel["log_volume"].values.astype(float) + tvl_vals = panel["log_tvl_lag1"].values.astype(float) + + for k, date in enumerate(dates): + t = date_to_idx[date] + vol_matrix[t, j] = log_vols[k] + v_arb_matrix[t, j] = v_arb_day[k] + tvl_matrix[t, j] = tvl_vals[k] + + # Weekdays + for t, date in enumerate(date_list): + dt = pd.Timestamp(date) + weekday_matrix[t] = dt.weekday() + + # ---- Pool attributes ---- + X_attr, attr_names, _ = build_pool_attributes(matched_clean) + attr_mean = np.mean(X_attr, axis=0) + attr_std = np.std(X_attr, axis=0) + attr_std[attr_std < 1e-6] = 1.0 + X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) + k_attr = X_attr_norm.shape[1] + + # ---- Token overlap ---- + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + n_peers = n_pools - 1 + peer_attrs = np.zeros((n_pools, n_peers, k_attr), dtype=np.float32) + peer_overlap = np.zeros((n_pools, n_peers), dtype=np.float32) + peer_col_idx = np.zeros((n_pools, n_peers), dtype=np.int32) + + for i in range(n_pools): + peers = [j for j in range(n_pools) if j != i] + for p, j in enumerate(peers): + peer_attrs[i, p] = X_attr_norm[j] + peer_overlap[i, p] = len(pool_tokens[i] & pool_tokens[j]) + peer_col_idx[i, p] = j + + target_attrs = X_attr_norm + + # ---- Standardize volumes for encoder input ---- + vol_mean = float(np.nanmean(vol_matrix)) + vol_std = float(np.nanstd(vol_matrix)) + + # ---- Build samples ---- + sample_pools, sample_days = [], [] + for i in range(n_pools): + if i == exclude_pool_idx: + continue + for t in range(1, n_dates): + if (np.isnan(vol_matrix[t, i]) or np.isnan(vol_matrix[t - 1, i]) + or np.isnan(v_arb_matrix[t, i]) or np.isnan(tvl_matrix[t, i])): + continue + sample_pools.append(i) + sample_days.append(t) + + sample_pools = np.array(sample_pools, dtype=np.int32) + sample_days = np.array(sample_days, dtype=np.int32) + n_samples = len(sample_pools) + + peer_vols_arr = np.zeros((n_samples, n_peers), dtype=np.float32) + peer_mask_arr = np.zeros((n_samples, n_peers), dtype=np.float32) + own_lag_arr = np.zeros(n_samples, dtype=np.float32) + v_arb_arr = np.zeros(n_samples, dtype=np.float32) + local_arr = np.zeros((n_samples, 3), dtype=np.float32) # tvl, dow_sin, dow_cos + y_arr = np.zeros(n_samples, dtype=np.float32) + + for s in range(n_samples): + i = sample_pools[s] + t = sample_days[s] + cols = peer_col_idx[i] + + pvols_raw = vol_matrix[t - 1, cols] + valid = ~np.isnan(pvols_raw) + pvols_norm = (pvols_raw - vol_mean) / vol_std + peer_vols_arr[s] = np.where(valid, pvols_norm, 0.0) + peer_mask_arr[s] = valid.astype(np.float32) + + own_lag_arr[s] = (vol_matrix[t - 1, i] - vol_mean) / vol_std + v_arb_arr[s] = v_arb_matrix[t, i] + y_arr[s] = vol_matrix[t, i] # raw log_volume (not standardized) + + wd = weekday_matrix[t] + local_arr[s, 0] = tvl_matrix[t, i] + local_arr[s, 1] = np.sin(2 * np.pi * wd / 7) + local_arr[s, 2] = np.cos(2 * np.pi * wd / 7) + + # Standardize local features + local_mean = np.mean(local_arr, axis=0) + local_std = np.std(local_arr, axis=0) + local_std[local_std < 1e-6] = 1.0 + local_arr = ((local_arr - local_mean) / local_std).astype(np.float32) + + return { + "peer_attrs": jnp.array(peer_attrs), + "target_attrs": jnp.array(target_attrs), + "peer_overlap": jnp.array(peer_overlap), + "peer_vols": jnp.array(peer_vols_arr), + "peer_mask": jnp.array(peer_mask_arr), + "own_lag": jnp.array(own_lag_arr), + "v_arb": jnp.array(v_arb_arr), + "local": jnp.array(local_arr), + "y": jnp.array(y_arr), + "pool_idx": jnp.array(sample_pools), + "day_idx": sample_days, + "n_pools": n_pools, + "n_peers": n_peers, + "k_attr": k_attr, + "k_local": 3, + "pool_ids": pool_ids, + } + + +# ---- Model ---- + + +def init_params(key, k_attr, k_local, hidden, d_embed, include_own_lag): + k1, k2, k3, k4 = jax.random.split(key, 4) + enc_in = 2 * k_attr + 2 # peer_attr + target_attr + peer_vol + overlap + dec_in = d_embed + k_attr + k_local + (1 if include_own_lag else 0) + + return { + "enc_W1": jax.random.normal(k1, (enc_in, hidden)) * np.sqrt(2.0 / enc_in), + "enc_b1": jnp.zeros(hidden), + "enc_W2": jax.random.normal(k2, (hidden, d_embed)) * np.sqrt(2.0 / hidden), + "enc_b2": jnp.zeros(d_embed), + "dec_W1": jax.random.normal(k3, (dec_in, hidden)) * np.sqrt(2.0 / dec_in), + "dec_b1": jnp.zeros(hidden), + "dec_W2": jax.random.normal(k4, (hidden, 1)) * 0.01, + "dec_b2": jnp.zeros(1), + } + + +def forward(params, peer_attrs_all, target_attrs_all, peer_overlap_all, + peer_vols, peer_mask, own_lag, local_feat, pool_idx, + include_own_lag=True): + """Returns log_v_noise per sample.""" + batch = peer_vols.shape[0] + + pa = peer_attrs_all[pool_idx] + ta = target_attrs_all[pool_idx] + ov = peer_overlap_all[pool_idx] + ta_broad = jnp.broadcast_to(ta[:, None, :], pa.shape) + + enc_in = jnp.concatenate([ + pa, ta_broad, + peer_vols[:, :, None], + ov[:, :, None], + ], axis=-1) + + flat = enc_in.reshape(-1, enc_in.shape[-1]) + h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) + h = h @ params["enc_W2"] + params["enc_b2"] + h = h.reshape(batch, peer_vols.shape[1], -1) + + h_masked = h * peer_mask[:, :, None] + n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) + summary = jnp.sum(h_masked, axis=1) / n_valid + + dec_parts = [summary, ta, local_feat] + if include_own_lag: + dec_parts.append(own_lag[:, None]) + dec_in = jnp.concatenate(dec_parts, axis=-1) + + h_dec = jnp.maximum(dec_in @ params["dec_W1"] + params["dec_b1"], 0.0) + log_v_noise = (h_dec @ params["dec_W2"] + params["dec_b2"])[:, 0] + return log_v_noise + + +def loss_fn(params, static, peer_vols, peer_mask, own_lag, local_feat, + pool_idx, v_arb, y, l2_alpha, include_own_lag): + """Log-space V_arb + V_noise loss matching the calibration pipeline.""" + log_v_noise = forward( + params, static["peer_attrs"], static["target_attrs"], + static["peer_overlap"], peer_vols, peer_mask, own_lag, local_feat, + pool_idx, include_own_lag, + ) + v_noise = jnp.exp(log_v_noise) + log_v_pred = jnp.log(jnp.maximum(v_arb + v_noise, 1e-6)) + mse = jnp.mean((log_v_pred - y) ** 2) + reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) + return mse + alpha * reg if (alpha := l2_alpha) else mse + l2_alpha * reg + + +@jax.jit +def _loss_and_grad(params, static, peer_vols, peer_mask, own_lag, local_feat, + pool_idx, v_arb, y, l2_alpha, include_own_lag): + return jax.value_and_grad(loss_fn)( + params, static, peer_vols, peer_mask, own_lag, local_feat, + pool_idx, v_arb, y, l2_alpha, include_own_lag, + ) + + +# ---- Training ---- + + +def train(params, data, hparams, verbose=True): + """Full-batch Adam.""" + static = {k: data[k] for k in ["peer_attrs", "target_attrs", "peer_overlap"]} + include_own_lag = hparams["include_own_lag"] + lr = hparams["lr"] + l2_alpha = hparams["l2_alpha"] + n_epochs = hparams["n_epochs"] + + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + + for epoch in range(n_epochs): + loss_val, grads = _loss_and_grad( + params, static, data["peer_vols"], data["peer_mask"], + data["own_lag"], data["local"], data["pool_idx"], + data["v_arb"], data["y"], l2_alpha, include_own_lag, + ) + + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): + print(f" epoch {epoch:4d} loss={float(loss_val):.6f}") + + return params, float(loss_val) + + +# ---- Evaluation ---- + + +def per_pool_r2(params, data, hparams): + """Per-pool R² on the log(V_arb + V_noise) prediction.""" + static = {k: data[k] for k in ["peer_attrs", "target_attrs", "peer_overlap"]} + log_v_noise = np.array(forward( + params, static["peer_attrs"], static["target_attrs"], + static["peer_overlap"], data["peer_vols"], data["peer_mask"], + data["own_lag"], data["local"], data["pool_idx"], + hparams["include_own_lag"], + )) + v_noise = np.exp(log_v_noise) + v_arb = np.array(data["v_arb"]) + log_v_pred = np.log(np.maximum(v_arb + v_noise, 1e-6)) + y = np.array(data["y"]) + pool_idx = np.array(data["pool_idx"]) + + r2s = {} + for i in range(data["n_pools"]): + mask = pool_idx == i + if mask.sum() < 2: + continue + yi = y[mask] + pi = log_v_pred[mask] + ss_res = np.sum((yi - pi) ** 2) + ss_tot = np.sum((yi - yi.mean()) ** 2) + r2s[i] = 1 - ss_res / max(ss_tot, 1e-10) + return r2s + + +def subset_data(data, mask): + """Subset data arrays by boolean mask.""" + jmask = jnp.array(mask) + static_keys = ["peer_attrs", "target_attrs", "peer_overlap", + "n_pools", "n_peers", "k_attr", "k_local", "pool_ids"] + out = {k: data[k] for k in static_keys} + for k in ["peer_vols", "peer_mask", "own_lag", "v_arb", "local", "y", "pool_idx"]: + out[k] = data[k][jmask] + out["day_idx"] = np.array(data["day_idx"])[mask] + return out + + +# ---- Experiments ---- + + +def run_single(matched_clean, option_c_clean, hparams, split_frac=0.7): + """Train with temporal split, report in-sample and eval R².""" + data = build_data(matched_clean, option_c_clean) + n_params = sum(v.size for v in init_params( + jax.random.PRNGKey(0), data["k_attr"], data["k_local"], + hparams["hidden"], hparams["d_embed"], hparams["include_own_lag"], + ).values()) + + print(f" {data['peer_vols'].shape[0]} samples, {data['n_pools']} pools, " + f"{n_params} params") + + day_idx = np.array(data["day_idx"]) + split_day = int(day_idx.max() * split_frac) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = subset_data(data, train_mask) + eval_data = subset_data(data, eval_mask) + + print(f" Train: {int(train_mask.sum())} samples, " + f"Eval: {int(eval_mask.sum())} samples") + + params = init_params( + jax.random.PRNGKey(42), data["k_attr"], data["k_local"], + hparams["hidden"], hparams["d_embed"], hparams["include_own_lag"], + ) + t0 = time.time() + params, final_loss = train(params, train_data, hparams) + print(f" Training: {time.time() - t0:.1f}s, final loss={final_loss:.6f}") + + r2_train = per_pool_r2(params, train_data, hparams) + r2_eval = per_pool_r2(params, eval_data, hparams) + + pool_ids = data["pool_ids"] + for i, pid in enumerate(pool_ids): + r_tr = r2_train.get(i, float("nan")) + r_ev = r2_eval.get(i, float("nan")) + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"train={r_tr:.3f} eval={r_ev:.3f}") + + vals_train = [v for v in r2_train.values() if np.isfinite(v)] + vals_eval = [v for v in r2_eval.values() if np.isfinite(v)] + med_train = np.median(vals_train) if vals_train else float("nan") + med_eval = np.median(vals_eval) if vals_eval else float("nan") + + print(f"\n Train: median R²={med_train:.4f}") + print(f" Eval: median R²={med_eval:.4f}") + print(f" (Option C in-sample: 0.589)") + + return med_eval, params, data + + +def run_loo(matched_clean, option_c_clean, hparams): + """Full LOO.""" + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + loo_r2s = [] + + print(f"\n{'='*70}") + print("LOO DeepSets Noise") + print(f"{'='*70}") + + # Use fewer epochs for LOO + loo_hparams = dict(hparams, n_epochs=min(hparams["n_epochs"], 500)) + + for hold_out_idx in range(n_pools): + hold_out_pid = pool_ids[hold_out_idx] + train_data = build_data(matched_clean, option_c_clean, + exclude_pool_idx=hold_out_idx) + + params = init_params( + jax.random.PRNGKey(42), train_data["k_attr"], train_data["k_local"], + loo_hparams["hidden"], loo_hparams["d_embed"], + loo_hparams["include_own_lag"], + ) + params, _ = train(params, train_data, loo_hparams, verbose=False) + + # Eval on held-out pool + full_data = build_data(matched_clean, option_c_clean) + ho_mask = np.array(full_data["pool_idx"]) == hold_out_idx + if ho_mask.sum() < 2: + loo_r2s.append(float("nan")) + continue + + eval_data = subset_data(full_data, ho_mask) + r2s = per_pool_r2(params, eval_data, loo_hparams) + r2 = r2s.get(hold_out_idx, float("nan")) + loo_r2s.append(r2) + + tag = "OK" if r2 > 0 else "NEG" + print(f" {hold_out_pid[:16]} ({matched_clean[hold_out_pid]['tokens']:<14}) " + f"R²={r2:.3f} [{tag}]") + + valid = [r for r in loo_r2s if np.isfinite(r)] + med = np.median(valid) if valid else float("nan") + print(f"\n LOO: median R²={med:.4f}, " + f"mean={np.mean(valid):.4f}, " + f"n_neg={sum(1 for r in valid if r < 0)}") + return loo_r2s + + +# ---- Optuna ---- + + +def run_optuna(matched_clean, option_c_clean, n_trials): + """Hyperparameter optimization with Optuna.""" + import optuna + + # Precompute data once (shared across trials) + data = build_data(matched_clean, option_c_clean) + day_idx = np.array(data["day_idx"]) + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = subset_data(data, train_mask) + eval_data = subset_data(data, eval_mask) + + def objective(trial): + hp = { + "hidden": trial.suggest_categorical("hidden", [8, 16, 32]), + "d_embed": trial.suggest_categorical("d_embed", [4, 8, 16]), + "lr": trial.suggest_float("lr", 1e-4, 1e-2, log=True), + "l2_alpha": trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True), + "n_epochs": trial.suggest_categorical("n_epochs", [500, 1000, 2000]), + "include_own_lag": trial.suggest_categorical("include_own_lag", [True, False]), + } + + params = init_params( + jax.random.PRNGKey(42), data["k_attr"], data["k_local"], + hp["hidden"], hp["d_embed"], hp["include_own_lag"], + ) + params, final_loss = train(params, train_data, hp, verbose=False) + + r2s = per_pool_r2(params, eval_data, hp) + vals = [v for v in r2s.values() if np.isfinite(v)] + med_r2 = float(np.median(vals)) if vals else -10.0 + + # Report train R² too for diagnostics + r2s_tr = per_pool_r2(params, train_data, hp) + vals_tr = [v for v in r2s_tr.values() if np.isfinite(v)] + med_tr = float(np.median(vals_tr)) if vals_tr else -10.0 + + trial.set_user_attr("train_median_r2", med_tr) + trial.set_user_attr("final_loss", final_loss) + + print(f" Trial {trial.number}: eval={med_r2:.4f} train={med_tr:.4f} " + f"h={hp['hidden']} d={hp['d_embed']} lr={hp['lr']:.1e} " + f"alpha={hp['l2_alpha']:.1e} epochs={hp['n_epochs']} " + f"own_lag={hp['include_own_lag']}") + + return med_r2 + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print("Optuna Results") + print(f"{'='*70}") + print(f" Best trial: {study.best_trial.number}") + print(f" Best eval median R²: {study.best_value:.4f}") + print(f" Best params: {study.best_params}") + print(f" Train median R²: {study.best_trial.user_attrs['train_median_r2']:.4f}") + + # Show top 5 + print(f"\n Top 5 trials:") + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + for t in trials[:5]: + if t.value is not None: + print(f" #{t.number}: eval={t.value:.4f} " + f"train={t.user_attrs.get('train_median_r2', '?'):.4f} " + f"{t.params}") + + return study + + +# ---- Main ---- + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--tune", type=int, default=0, + help="Run Optuna with N trials") + parser.add_argument("--loo", action="store_true", + help="Run LOO evaluation") + parser.add_argument("--hidden", type=int, default=DEFAULTS["hidden"]) + parser.add_argument("--d-embed", type=int, default=DEFAULTS["d_embed"]) + parser.add_argument("--lr", type=float, default=DEFAULTS["lr"]) + parser.add_argument("--l2-alpha", type=float, default=DEFAULTS["l2_alpha"]) + parser.add_argument("--epochs", type=int, default=DEFAULTS["n_epochs"]) + parser.add_argument("--no-own-lag", action="store_true") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + hparams = { + "hidden": args.hidden, + "d_embed": args.d_embed, + "lr": args.lr, + "l2_alpha": args.l2_alpha, + "n_epochs": args.epochs, + "include_own_lag": not args.no_own_lag, + } + + print("=" * 70) + print("DeepSets Noise Volume Prediction (V_arb decomposition)") + print(f" {hparams}") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + if args.tune > 0: + run_optuna(matched_clean, option_c_clean, args.tune) + else: + print(f"\n{'='*70}") + print("Temporal split (70/30)") + print(f"{'='*70}") + med_eval, params, data = run_single(matched_clean, option_c_clean, hparams) + + if args.loo: + run_loo(matched_clean, option_c_clean, hparams) + + +if __name__ == "__main__": + main() diff --git a/experiments/run_deepsets_v2.py b/experiments/run_deepsets_v2.py new file mode 100644 index 00000000..1ba3485f --- /dev/null +++ b/experiments/run_deepsets_v2.py @@ -0,0 +1,697 @@ +"""DeepSets v2: full feature menu with Optuna feature selection. + +Trains on total log_volume, evaluates on both total volume and noise +residual (log_vol - log_V_arb). V_arb precomputed from Option C fits. + +Feature menu: + Peer (encoder) — always: peer_attr, target_attr, vol_lag1, overlap + optional: vol_lag2, vol_change, tvl, volatility + Local (decoder) — always: target_attr, own_vol_lag1, dow_sin, dow_cos + optional: own_vol_lag2, own_vol_change, own_tvl, own_volatility + +Usage: + python experiments/run_deepsets_v2.py # defaults + python experiments/run_deepsets_v2.py --tune 50 # Optuna + python experiments/run_deepsets_v2.py --loo # LOO eval +""" + +import argparse +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +# ---- Data construction ---- + + +def build_all_features(matched_clean, option_c_clean): + """Build all possible feature matrices. Called once.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + build_pool_attributes, _parse_tokens, _canonicalize_token, + ) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # Collect dates + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + # Daily matrices: (n_dates, n_pools) + vol_matrix = np.full((n_dates, n_pools), np.nan) + tvl_matrix = np.full((n_dates, n_pools), np.nan) + volatility_matrix = np.full((n_dates, n_pools), np.nan) + v_arb_matrix = np.full((n_dates, n_pools), np.nan) + weekday_arr = np.zeros(n_dates) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + panel = entry["panel"] + + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], + jnp.float64(oc["log_cadence"]), + jnp.float64(np.exp(oc["log_gas"])), + )) + v_arb = v_arb_all[entry["day_indices"]] + + dates = panel["date"].values + for k, date in enumerate(dates): + t = date_to_idx[date] + vol_matrix[t, j] = panel["log_volume"].values[k] + tvl_matrix[t, j] = panel["log_tvl_lag1"].values[k] + volatility_matrix[t, j] = panel["volatility"].values[k] + v_arb_matrix[t, j] = v_arb[k] + + for t, date in enumerate(date_list): + weekday_arr[t] = pd.Timestamp(date).weekday() + + # Pool attributes (static) + X_attr, attr_names, _ = build_pool_attributes(matched_clean) + attr_mean = np.mean(X_attr, axis=0) + attr_std = np.std(X_attr, axis=0) + attr_std[attr_std < 1e-6] = 1.0 + X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) + k_attr = X_attr_norm.shape[1] + + # Token overlap + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + n_peers = n_pools - 1 + peer_attrs = np.zeros((n_pools, n_peers, k_attr), dtype=np.float32) + peer_overlap = np.zeros((n_pools, n_peers), dtype=np.float32) + peer_col_idx = np.zeros((n_pools, n_peers), dtype=np.int32) + + for i in range(n_pools): + peers = [j for j in range(n_pools) if j != i] + for p, j in enumerate(peers): + peer_attrs[i, p] = X_attr_norm[j] + peer_overlap[i, p] = len(pool_tokens[i] & pool_tokens[j]) + peer_col_idx[i, p] = j + + # Standardization stats for volumes + vol_mean = float(np.nanmean(vol_matrix)) + vol_std = float(np.nanstd(vol_matrix)) + tvl_mean = float(np.nanmean(tvl_matrix)) + tvl_std = float(np.nanstd(tvl_matrix)) + vola_mean = float(np.nanmean(volatility_matrix)) + vola_std = float(np.nanstd(volatility_matrix)) + + # Build samples: require t >= 2 (for lag-2), valid vol at t, t-1, t-2 + sample_pools, sample_days = [], [] + for i in range(n_pools): + for t in range(2, n_dates): + if (np.isnan(vol_matrix[t, i]) or np.isnan(vol_matrix[t - 1, i]) + or np.isnan(vol_matrix[t - 2, i])): + continue + sample_pools.append(i) + sample_days.append(t) + + sample_pools = np.array(sample_pools, dtype=np.int32) + sample_days = np.array(sample_days, dtype=np.int32) + n_samples = len(sample_pools) + + def _norm_vol(x): + return (x - vol_mean) / vol_std + + def _norm_tvl(x): + return (x - tvl_mean) / tvl_std + + def _norm_vola(x): + return (x - vola_mean) / vola_std + + # Per-sample arrays + # Peer features (per peer) + pf_vol_lag1 = np.zeros((n_samples, n_peers), dtype=np.float32) + pf_vol_lag2 = np.zeros((n_samples, n_peers), dtype=np.float32) + pf_vol_change = np.zeros((n_samples, n_peers), dtype=np.float32) + pf_tvl = np.zeros((n_samples, n_peers), dtype=np.float32) + pf_volatility = np.zeros((n_samples, n_peers), dtype=np.float32) + peer_mask = np.zeros((n_samples, n_peers), dtype=np.float32) + + # Local features + lf_own_vol_lag1 = np.zeros(n_samples, dtype=np.float32) + lf_own_vol_lag2 = np.zeros(n_samples, dtype=np.float32) + lf_own_vol_change = np.zeros(n_samples, dtype=np.float32) + lf_own_tvl = np.zeros(n_samples, dtype=np.float32) + lf_own_volatility = np.zeros(n_samples, dtype=np.float32) + lf_dow_sin = np.zeros(n_samples, dtype=np.float32) + lf_dow_cos = np.zeros(n_samples, dtype=np.float32) + + # Targets + y_total = np.zeros(n_samples, dtype=np.float32) + v_arb_samples = np.zeros(n_samples, dtype=np.float32) + + for s in range(n_samples): + i = sample_pools[s] + t = sample_days[s] + cols = peer_col_idx[i] + + # Peer features at t-1 + pvols1 = vol_matrix[t - 1, cols] + pvols2 = vol_matrix[t - 2, cols] + valid = ~np.isnan(pvols1) + peer_mask[s] = valid.astype(np.float32) + + pf_vol_lag1[s] = np.where(valid, _norm_vol(pvols1), 0.0) + pf_vol_lag2[s] = np.where(valid & ~np.isnan(pvols2), _norm_vol(pvols2), 0.0) + pf_vol_change[s] = np.where( + valid & ~np.isnan(pvols2), + _norm_vol(pvols1) - _norm_vol(pvols2), 0.0) + + ptvl = tvl_matrix[t - 1, cols] + pf_tvl[s] = np.where(valid & ~np.isnan(ptvl), _norm_tvl(ptvl), 0.0) + + pvola = volatility_matrix[t - 1, cols] + pf_volatility[s] = np.where(valid & ~np.isnan(pvola), _norm_vola(pvola), 0.0) + + # Local features + lf_own_vol_lag1[s] = _norm_vol(vol_matrix[t - 1, i]) + lf_own_vol_lag2[s] = _norm_vol(vol_matrix[t - 2, i]) + lf_own_vol_change[s] = lf_own_vol_lag1[s] - lf_own_vol_lag2[s] + + tvl_val = tvl_matrix[t, i] + lf_own_tvl[s] = _norm_tvl(tvl_val) if np.isfinite(tvl_val) else 0.0 + + vola_val = volatility_matrix[t, i] + lf_own_volatility[s] = _norm_vola(vola_val) if np.isfinite(vola_val) else 0.0 + + wd = weekday_arr[t] + lf_dow_sin[s] = np.sin(2 * np.pi * wd / 7) + lf_dow_cos[s] = np.cos(2 * np.pi * wd / 7) + + y_total[s] = vol_matrix[t, i] + v_arb_val = v_arb_matrix[t, i] + v_arb_samples[s] = v_arb_val if np.isfinite(v_arb_val) else 1e-6 + + return { + # Static per-pool + "peer_attrs": peer_attrs, # (n_pools, n_peers, k_attr) + "target_attrs": X_attr_norm, # (n_pools, k_attr) + "peer_overlap": peer_overlap, # (n_pools, n_peers) + # Per-sample peer features + "pf_vol_lag1": pf_vol_lag1, + "pf_vol_lag2": pf_vol_lag2, + "pf_vol_change": pf_vol_change, + "pf_tvl": pf_tvl, + "pf_volatility": pf_volatility, + "peer_mask": peer_mask, + # Per-sample local features + "lf_own_vol_lag1": lf_own_vol_lag1, + "lf_own_vol_lag2": lf_own_vol_lag2, + "lf_own_vol_change": lf_own_vol_change, + "lf_own_tvl": lf_own_tvl, + "lf_own_volatility": lf_own_volatility, + "lf_dow_sin": lf_dow_sin, + "lf_dow_cos": lf_dow_cos, + # Targets + "y_total": y_total, + "v_arb": v_arb_samples, + # Indices + "pool_idx": sample_pools, + "day_idx": sample_days, + # Meta + "n_pools": n_pools, + "n_peers": n_peers, + "k_attr": k_attr, + "pool_ids": pool_ids, + "vol_mean": vol_mean, + "vol_std": vol_std, + } + + +def assemble_inputs(data, feat_cfg): + """Assemble encoder/decoder inputs based on feature config. + + Returns (peer_input, local_input, peer_mask, pool_idx, y, v_arb) + all as JAX arrays ready for training. + """ + n_samples = len(data["pool_idx"]) + n_peers = data["n_peers"] + + # ---- Peer encoder input: (n_samples, n_peers, n_feat) ---- + # Always: peer_attr, target_attr, vol_lag1, overlap + pool_idx = data["pool_idx"] + pa = data["peer_attrs"][pool_idx] # (n_samples, n_peers, k_attr) + ta = data["target_attrs"][pool_idx] # (n_samples, k_attr) + ta_broad = np.broadcast_to(ta[:, None, :], pa.shape) + + peer_parts = [ + pa, ta_broad, + data["pf_vol_lag1"][:, :, None], + data["peer_overlap"][pool_idx][:, :, None], + ] + + if feat_cfg.get("peer_vol_lag2"): + peer_parts.append(data["pf_vol_lag2"][:, :, None]) + if feat_cfg.get("peer_vol_change"): + peer_parts.append(data["pf_vol_change"][:, :, None]) + if feat_cfg.get("peer_tvl"): + peer_parts.append(data["pf_tvl"][:, :, None]) + if feat_cfg.get("peer_volatility"): + peer_parts.append(data["pf_volatility"][:, :, None]) + + peer_input = np.concatenate(peer_parts, axis=-1).astype(np.float32) + + # ---- Local decoder input: (n_samples, n_feat) ---- + # Always: target_attr, own_vol_lag1, dow_sin, dow_cos + local_parts = [ + ta, + data["lf_own_vol_lag1"][:, None], + data["lf_dow_sin"][:, None], + data["lf_dow_cos"][:, None], + ] + + if feat_cfg.get("own_vol_lag2"): + local_parts.append(data["lf_own_vol_lag2"][:, None]) + if feat_cfg.get("own_vol_change"): + local_parts.append(data["lf_own_vol_change"][:, None]) + if feat_cfg.get("own_tvl"): + local_parts.append(data["lf_own_tvl"][:, None]) + if feat_cfg.get("own_volatility"): + local_parts.append(data["lf_own_volatility"][:, None]) + + local_input = np.concatenate(local_parts, axis=-1).astype(np.float32) + + return { + "peer_input": jnp.array(peer_input), + "local_input": jnp.array(local_input), + "peer_mask": jnp.array(data["peer_mask"]), + "y": jnp.array(data["y_total"]), + "v_arb": jnp.array(data["v_arb"]), + "pool_idx": jnp.array(pool_idx), + "n_peer_feat": peer_input.shape[-1], + "n_local_feat": local_input.shape[-1], + } + + +# ---- Model ---- + + +def init_params(key, n_peer_feat, n_local_feat, hidden, d_embed): + k1, k2, k3, k4 = jax.random.split(key, 4) + dec_in = d_embed + n_local_feat + return { + "enc_W1": jax.random.normal(k1, (n_peer_feat, hidden)) * np.sqrt(2.0 / n_peer_feat), + "enc_b1": jnp.zeros(hidden), + "enc_W2": jax.random.normal(k2, (hidden, d_embed)) * np.sqrt(2.0 / hidden), + "enc_b2": jnp.zeros(d_embed), + "dec_W1": jax.random.normal(k3, (dec_in, hidden)) * np.sqrt(2.0 / dec_in), + "dec_b1": jnp.zeros(hidden), + "dec_W2": jax.random.normal(k4, (hidden, 1)) * 0.01, + "dec_b2": jnp.zeros(1), + } + + +def forward(params, peer_input, peer_mask, local_input): + """Returns predicted log_volume (total) per sample.""" + batch, n_peers, _ = peer_input.shape + + flat = peer_input.reshape(-1, peer_input.shape[-1]) + h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) + h = h @ params["enc_W2"] + params["enc_b2"] + h = h.reshape(batch, n_peers, -1) + + h_masked = h * peer_mask[:, :, None] + n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) + summary = jnp.sum(h_masked, axis=1) / n_valid + + dec_in = jnp.concatenate([summary, local_input], axis=-1) + h_dec = jnp.maximum(dec_in @ params["dec_W1"] + params["dec_b1"], 0.0) + return (h_dec @ params["dec_W2"] + params["dec_b2"])[:, 0] + + +def loss_fn(params, peer_input, peer_mask, local_input, y, l2_alpha): + """MSE on total log_volume + L2 reg.""" + pred = forward(params, peer_input, peer_mask, local_input) + mse = jnp.mean((pred - y) ** 2) + reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) + return mse + l2_alpha * reg + + +_grad_fn = jax.jit(jax.value_and_grad(loss_fn)) + + +# ---- Training ---- + + +def train(params, inputs, n_epochs, lr, l2_alpha, verbose=True): + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + final_loss = float("inf") + + for epoch in range(n_epochs): + loss_val, grads = _grad_fn( + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], inputs["y"], l2_alpha, + ) + final_loss = float(loss_val) + + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): + print(f" epoch {epoch:4d} loss={final_loss:.6f}") + + return params, final_loss + + +# ---- Evaluation ---- + + +def evaluate(params, inputs, data, label=""): + """Per-pool R² on total volume and noise residual.""" + pred_total = np.array(forward( + params, inputs["peer_input"], inputs["peer_mask"], inputs["local_input"], + )) + y_total = np.array(inputs["y"]) + v_arb = np.array(inputs["v_arb"]) + pool_idx = np.array(data["pool_idx"]) if "pool_idx" in data else np.array(inputs["pool_idx"]) + + # Noise residual: compare (pred - log(v_arb)) vs (y - log(v_arb)) + log_v_arb = np.log(np.maximum(v_arb, 1e-6)) + resid_true = y_total - log_v_arb + resid_pred = pred_total - log_v_arb + + r2_total = {} + r2_resid = {} + pool_ids = data.get("pool_ids", []) + + for i in range(data["n_pools"]): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res_t = np.sum((yt - pt) ** 2) + ss_tot_t = np.sum((yt - yt.mean()) ** 2) + r2_total[i] = 1 - ss_res_t / max(ss_tot_t, 1e-10) + + rt = resid_true[mask] + rp = resid_pred[mask] + ss_res_r = np.sum((rt - rp) ** 2) + ss_tot_r = np.sum((rt - rt.mean()) ** 2) + r2_resid[i] = 1 - ss_res_r / max(ss_tot_r, 1e-10) + + def _med(d): + v = [x for x in d.values() if np.isfinite(x)] + return np.median(v) if v else float("nan") + + if label: + print(f"\n {label}:") + for i in range(data["n_pools"]): + if i in r2_total and i < len(pool_ids): + pid = pool_ids[i] + print(f" {pid[:16]} total={r2_total[i]:.3f} resid={r2_resid[i]:.3f}") + + med_total = _med(r2_total) + med_resid = _med(r2_resid) + print(f" Median R² total={med_total:.4f} resid={med_resid:.4f}") + return med_total, med_resid, r2_total, r2_resid + + +# ---- Temporal split ---- + + +def run_temporal(data, feat_cfg, hparams, split_frac=0.7): + """Train on first split_frac of days, eval on rest.""" + day_idx = data["day_idx"] + split_day = int(day_idx.max() * split_frac) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + def _subset(d, mask): + out = {} + for k, v in d.items(): + if isinstance(v, np.ndarray) and len(v) == len(mask): + out[k] = v[mask] + else: + out[k] = v + return out + + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + train_inputs = assemble_inputs(train_data, feat_cfg) + eval_inputs = assemble_inputs(eval_data, feat_cfg) + + n_pf = train_inputs["n_peer_feat"] + n_lf = train_inputs["n_local_feat"] + n_params = sum(v.size for v in init_params( + jax.random.PRNGKey(0), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], + ).values()) + + print(f" Train: {int(train_mask.sum())}, Eval: {int(eval_mask.sum())}, " + f"peer_feat={n_pf}, local_feat={n_lf}, params={n_params}") + + params = init_params( + jax.random.PRNGKey(42), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], + ) + t0 = time.time() + params, final_loss = train( + params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], + ) + print(f" Training: {time.time() - t0:.1f}s") + + print("\n --- Train ---") + evaluate(params, train_inputs, data) + print("\n --- Eval ---") + _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data) + + return med_resid_eval + + +# ---- Optuna ---- + + +def run_optuna(data, n_trials): + import optuna + + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + def _subset(d, mask): + out = {} + for k, v in d.items(): + if isinstance(v, np.ndarray) and len(v) == len(mask): + out[k] = v[mask] + else: + out[k] = v + return out + + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + def objective(trial): + feat_cfg = { + "peer_vol_lag2": trial.suggest_categorical("peer_vol_lag2", [True, False]), + "peer_vol_change": trial.suggest_categorical("peer_vol_change", [True, False]), + "peer_tvl": trial.suggest_categorical("peer_tvl", [True, False]), + "peer_volatility": trial.suggest_categorical("peer_volatility", [True, False]), + "own_vol_lag2": trial.suggest_categorical("own_vol_lag2", [True, False]), + "own_vol_change": trial.suggest_categorical("own_vol_change", [True, False]), + "own_tvl": trial.suggest_categorical("own_tvl", [True, False]), + "own_volatility": trial.suggest_categorical("own_volatility", [True, False]), + } + hparams = { + "hidden": trial.suggest_categorical("hidden", [8, 16, 32]), + "d_embed": trial.suggest_categorical("d_embed", [4, 8, 16]), + "lr": trial.suggest_float("lr", 1e-4, 1e-2, log=True), + "l2_alpha": trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True), + "n_epochs": trial.suggest_categorical("n_epochs", [500, 1000, 2000]), + } + + train_inputs = assemble_inputs(train_data, feat_cfg) + eval_inputs = assemble_inputs(eval_data, feat_cfg) + + params = init_params( + jax.random.PRNGKey(42), + train_inputs["n_peer_feat"], train_inputs["n_local_feat"], + hparams["hidden"], hparams["d_embed"], + ) + params, _ = train(params, train_inputs, hparams["n_epochs"], + hparams["lr"], hparams["l2_alpha"], verbose=False) + + # Eval R² on noise residual + pred = np.array(forward( + params, eval_inputs["peer_input"], eval_inputs["peer_mask"], + eval_inputs["local_input"], + )) + y = np.array(eval_inputs["y"]) + v_arb = np.array(eval_inputs["v_arb"]) + log_v_arb = np.log(np.maximum(v_arb, 1e-6)) + pool_idx = np.array(eval_data["pool_idx"]) + + r2_resids = [] + r2_totals = [] + for i in range(data["n_pools"]): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y[mask] + pt = pred[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2_totals.append(1 - ss_res / max(ss_tot, 1e-10)) + + rt = yt - log_v_arb[mask] + rp = pt - log_v_arb[mask] + ss_res_r = np.sum((rt - rp) ** 2) + ss_tot_r = np.sum((rt - rt.mean()) ** 2) + r2_resids.append(1 - ss_res_r / max(ss_tot_r, 1e-10)) + + med_resid = float(np.median(r2_resids)) if r2_resids else -10.0 + med_total = float(np.median(r2_totals)) if r2_totals else -10.0 + + trial.set_user_attr("med_total_r2", med_total) + n_feat = sum(feat_cfg.values()) + print(f" Trial {trial.number}: resid={med_resid:.4f} total={med_total:.4f} " + f"h={hparams['hidden']} d={hparams['d_embed']} " + f"lr={hparams['lr']:.1e} a={hparams['l2_alpha']:.1e} " + f"ep={hparams['n_epochs']} feat={n_feat}/8") + + return med_resid + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print("Optuna Results") + print(f"{'='*70}") + print(f" Best eval noise resid R²: {study.best_value:.4f}") + print(f" Best total R²: {study.best_trial.user_attrs['med_total_r2']:.4f}") + print(f" Best params:") + for k, v in sorted(study.best_params.items()): + print(f" {k}: {v}") + + print(f"\n Top 10:") + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + for t in trials[:10]: + if t.value is not None: + feats = sum(1 for k in ["peer_vol_lag2", "peer_vol_change", + "peer_tvl", "peer_volatility", + "own_vol_lag2", "own_vol_change", + "own_tvl", "own_volatility"] + if t.params.get(k)) + print(f" #{t.number}: resid={t.value:.4f} " + f"total={t.user_attrs.get('med_total_r2', '?'):.4f} " + f"h={t.params['hidden']} d={t.params['d_embed']} " + f"feat={feats}/8") + + return study + + +# ---- Main ---- + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--tune", type=int, default=0) + parser.add_argument("--loo", action="store_true") + # Architecture + parser.add_argument("--hidden", type=int, default=16) + parser.add_argument("--d-embed", type=int, default=8) + parser.add_argument("--lr", type=float, default=3e-4) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--epochs", type=int, default=1000) + # Feature flags + parser.add_argument("--peer-vol-lag2", action="store_true") + parser.add_argument("--peer-vol-change", action="store_true") + parser.add_argument("--peer-tvl", action="store_true") + parser.add_argument("--peer-volatility", action="store_true") + parser.add_argument("--own-vol-lag2", action="store_true") + parser.add_argument("--own-vol-change", action="store_true") + parser.add_argument("--own-tvl", action="store_true") + parser.add_argument("--own-volatility", action="store_true") + parser.add_argument("--all-features", action="store_true", + help="Enable all optional features") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + feat_cfg = { + "peer_vol_lag2": args.peer_vol_lag2 or args.all_features, + "peer_vol_change": args.peer_vol_change or args.all_features, + "peer_tvl": args.peer_tvl or args.all_features, + "peer_volatility": args.peer_volatility or args.all_features, + "own_vol_lag2": args.own_vol_lag2 or args.all_features, + "own_vol_change": args.own_vol_change or args.all_features, + "own_tvl": args.own_tvl or args.all_features, + "own_volatility": args.own_volatility or args.all_features, + } + hparams = { + "hidden": args.hidden, + "d_embed": args.d_embed, + "lr": args.lr, + "l2_alpha": args.l2_alpha, + "n_epochs": args.epochs, + } + + print("=" * 70) + print("DeepSets v2: Total Volume Target + Noise Residual Eval") + feat_on = [k for k, v in feat_cfg.items() if v] + print(f" Optional features: {feat_on or 'none'}") + print(f" Architecture: {hparams}") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding features...") + t0 = time.time() + data = build_all_features(matched_clean, option_c_clean) + print(f" {len(data['pool_idx'])} samples, {data['n_pools']} pools, " + f"{time.time() - t0:.1f}s") + + if args.tune > 0: + run_optuna(data, args.tune) + else: + print(f"\n{'='*70}") + print("Temporal split (70/30)") + print(f"{'='*70}") + run_temporal(data, feat_cfg, hparams) + + print(f"\n Baselines for comparison:") + print(f" Option C on residual: median R² = 0.060") + print(f" Ridge+own on residual: median R² = 0.098") + print(f" Constant zero: median R² = -0.083") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_deepsets_volume.py b/experiments/run_deepsets_volume.py new file mode 100644 index 00000000..dc174d22 --- /dev/null +++ b/experiments/run_deepsets_volume.py @@ -0,0 +1,467 @@ +"""DeepSets cross-pool volume prediction. + +Architecture: + For pool i at day t: + For each peer j != i with valid data at t-1: + h_j = Encoder(attr_j, attr_i, vol_j_{t-1}, overlap_ij) + peer_summary = masked_mean(h_j) + pred_i_t = Decoder(peer_summary, attr_i, own_vol_{t-1}) + +Evaluation: + 1. In-sample R² (all data) + 2. Temporal split (70/30) + 3. LOO (hold out one pool, retrain, evaluate) +""" + +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +HIDDEN = 8 +D_EMBED = 4 +LR = 1e-3 +N_EPOCHS = 500 +N_EPOCHS_LOO = 200 +L2_ALPHA = 0.001 + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +def build_volume_matrix(matched_clean): + """Build (n_dates, n_pools) volume matrix. NaN where missing.""" + pool_ids = sorted(matched_clean.keys()) + pool_date_vol = {} + all_dates = set() + for pid in pool_ids: + panel = matched_clean[pid]["panel"] + dates = panel["date"].values + vols = panel["log_volume"].values.astype(float) + pool_date_vol[pid] = dict(zip(dates, vols)) + all_dates.update(dates) + + date_list = sorted(all_dates) + n_dates = len(date_list) + n_pools = len(pool_ids) + vol_matrix = np.full((n_dates, n_pools), np.nan) + for j, pid in enumerate(pool_ids): + dv = pool_date_vol[pid] + for t, date in enumerate(date_list): + if date in dv: + vol_matrix[t, j] = dv[date] + return vol_matrix, date_list, pool_ids + + +def build_data(matched_clean, exclude_pool_idx=None): + """Build all arrays for DeepSets training. + + If exclude_pool_idx is set, that pool is excluded from training + samples but kept as a peer (its volume data is still available). + """ + from quantammsim.calibration.pool_data import ( + build_pool_attributes, _parse_tokens, _canonicalize_token, + ) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + vol_matrix, date_list, _ = build_volume_matrix(matched_clean) + X_attr, attr_names, _ = build_pool_attributes(matched_clean) + + # Standardize attributes + attr_mean = np.mean(X_attr, axis=0) + attr_std = np.std(X_attr, axis=0) + attr_std[attr_std < 1e-6] = 1.0 + X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) + + # Standardize volumes + vol_mean = float(np.nanmean(vol_matrix)) + vol_std = float(np.nanstd(vol_matrix)) + vol_norm = ((vol_matrix - vol_mean) / vol_std).astype(np.float32) + + # Token overlap + k_attr = X_attr_norm.shape[1] + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + # Per-pool peer structures + n_peers = n_pools - 1 + peer_attrs = np.zeros((n_pools, n_peers, k_attr), dtype=np.float32) + peer_overlap = np.zeros((n_pools, n_peers), dtype=np.float32) + peer_col_idx = np.zeros((n_pools, n_peers), dtype=np.int32) + + for i in range(n_pools): + peers = [j for j in range(n_pools) if j != i] + for p, j in enumerate(peers): + peer_attrs[i, p] = X_attr_norm[j] + peer_overlap[i, p] = len(pool_tokens[i] & pool_tokens[j]) + peer_col_idx[i, p] = j + + target_attrs = X_attr_norm + + # Build samples + sample_pools, sample_days = [], [] + for i in range(n_pools): + if i == exclude_pool_idx: + continue + for t in range(1, len(date_list)): + if np.isnan(vol_matrix[t, i]) or np.isnan(vol_matrix[t - 1, i]): + continue + sample_pools.append(i) + sample_days.append(t) + + sample_pools = np.array(sample_pools, dtype=np.int32) + sample_days = np.array(sample_days, dtype=np.int32) + n_samples = len(sample_pools) + + # Vectorized: gather peer volumes and masks + peer_vols = np.zeros((n_samples, n_peers), dtype=np.float32) + peer_mask = np.zeros((n_samples, n_peers), dtype=np.float32) + own_lag = np.zeros(n_samples, dtype=np.float32) + y = np.zeros(n_samples, dtype=np.float32) + + for s in range(n_samples): + i = sample_pools[s] + t = sample_days[s] + cols = peer_col_idx[i] + pvols = vol_norm[t - 1, cols] + valid = ~np.isnan(pvols) + peer_vols[s] = np.where(valid, pvols, 0.0) + peer_mask[s] = valid.astype(np.float32) + own_lag[s] = vol_norm[t - 1, i] + y[s] = vol_norm[t, i] + + return { + "peer_attrs": jnp.array(peer_attrs), + "target_attrs": jnp.array(target_attrs), + "peer_overlap": jnp.array(peer_overlap), + "peer_vols": jnp.array(peer_vols), + "peer_mask": jnp.array(peer_mask), + "own_lag": jnp.array(own_lag), + "y": jnp.array(y), + "pool_idx": jnp.array(sample_pools), + "day_idx": sample_days, + "n_pools": n_pools, + "n_peers": n_peers, + "k_attr": k_attr, + "pool_ids": pool_ids, + "vol_mean": vol_mean, + "vol_std": vol_std, + } + + +# ---- Model ---- + + +def init_params(key, k_attr, hidden=HIDDEN, d=D_EMBED): + k1, k2, k3, k4 = jax.random.split(key, 4) + enc_in = 2 * k_attr + 2 # peer_attr + target_attr + peer_vol + overlap + dec_in = d + k_attr + 1 # summary + target_attr + own_lag + return { + "enc_W1": jax.random.normal(k1, (enc_in, hidden)) * np.sqrt(2.0 / enc_in), + "enc_b1": jnp.zeros(hidden), + "enc_W2": jax.random.normal(k2, (hidden, d)) * np.sqrt(2.0 / hidden), + "enc_b2": jnp.zeros(d), + "dec_W1": jax.random.normal(k3, (dec_in, hidden)) * np.sqrt(2.0 / dec_in), + "dec_b1": jnp.zeros(hidden), + "dec_W2": jax.random.normal(k4, (hidden, 1)) * 0.01, + "dec_b2": jnp.zeros(1), + } + + +def forward(params, peer_attrs_all, target_attrs_all, peer_overlap_all, + peer_vols, peer_mask, own_lag, pool_idx): + """Batched DeepSets forward pass.""" + batch = peer_vols.shape[0] + n_peers = peer_vols.shape[1] + + pa = peer_attrs_all[pool_idx] # (batch, n_peers, k_attr) + ta = target_attrs_all[pool_idx] # (batch, k_attr) + ov = peer_overlap_all[pool_idx] # (batch, n_peers) + + ta_broad = jnp.broadcast_to(ta[:, None, :], pa.shape) + + enc_in = jnp.concatenate([ + pa, ta_broad, + peer_vols[:, :, None], + ov[:, :, None], + ], axis=-1) + + # Encoder MLP + flat = enc_in.reshape(-1, enc_in.shape[-1]) + h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) + h = h @ params["enc_W2"] + params["enc_b2"] + h = h.reshape(batch, n_peers, -1) + + # Masked mean + h_masked = h * peer_mask[:, :, None] + n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) + summary = jnp.sum(h_masked, axis=1) / n_valid + + # Decoder MLP + dec_in = jnp.concatenate([summary, ta, own_lag[:, None]], axis=-1) + h_dec = jnp.maximum(dec_in @ params["dec_W1"] + params["dec_b1"], 0.0) + return (h_dec @ params["dec_W2"] + params["dec_b2"])[:, 0] + + +def loss_fn(params, static, peer_vols, peer_mask, own_lag, pool_idx, y, alpha): + pred = forward(params, static["peer_attrs"], static["target_attrs"], + static["peer_overlap"], peer_vols, peer_mask, own_lag, pool_idx) + mse = jnp.mean((pred - y) ** 2) + reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) + return mse + alpha * reg + + +grad_fn = jax.jit(jax.value_and_grad(loss_fn)) + + +# ---- Training ---- + + +def train(params, data, n_epochs=N_EPOCHS, lr=LR, alpha=L2_ALPHA, verbose=True): + """Full-batch Adam training.""" + static = { + "peer_attrs": data["peer_attrs"], + "target_attrs": data["target_attrs"], + "peer_overlap": data["peer_overlap"], + } + + # Adam state + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + + for epoch in range(n_epochs): + loss_val, grads = grad_fn( + params, static, data["peer_vols"], data["peer_mask"], + data["own_lag"], data["pool_idx"], data["y"], alpha, + ) + + # Adam update + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 100 == 0 or epoch == n_epochs - 1): + print(f" epoch {epoch:4d} loss={float(loss_val):.6f}") + + return params + + +# ---- Evaluation ---- + + +def per_pool_r2(params, data): + """Compute per-pool R² from trained model.""" + static = { + "peer_attrs": data["peer_attrs"], + "target_attrs": data["target_attrs"], + "peer_overlap": data["peer_overlap"], + } + pred = np.array(forward( + params, static["peer_attrs"], static["target_attrs"], + static["peer_overlap"], data["peer_vols"], data["peer_mask"], + data["own_lag"], data["pool_idx"], + )) + y = np.array(data["y"]) + pool_idx = np.array(data["pool_idx"]) + + r2s = {} + for i in range(data["n_pools"]): + mask = pool_idx == i + if mask.sum() < 2: + continue + yi = y[mask] + pi = pred[mask] + ss_res = np.sum((yi - pi) ** 2) + ss_tot = np.sum((yi - yi.mean()) ** 2) + r2s[i] = 1 - ss_res / max(ss_tot, 1e-10) + return r2s + + +# ---- Main experiments ---- + + +def run_insample(matched_clean): + print("\n" + "=" * 70) + print("1. In-sample DeepSets") + print("=" * 70) + + data = build_data(matched_clean) + n_params = sum(v.size for v in init_params(jax.random.PRNGKey(0), data["k_attr"]).values()) + print(f" {data['peer_vols'].shape[0]} samples, {data['n_pools']} pools, " + f"{data['k_attr']} attrs, {n_params} params") + + params = init_params(jax.random.PRNGKey(42), data["k_attr"]) + t0 = time.time() + params = train(params, data) + print(f" Training: {time.time() - t0:.1f}s") + + r2s = per_pool_r2(params, data) + pool_ids = data["pool_ids"] + for i, pid in enumerate(pool_ids): + if i in r2s: + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) R²={r2s[i]:.3f}") + + vals = list(r2s.values()) + print(f"\n In-sample: median R²={np.median(vals):.4f}, mean={np.mean(vals):.4f}") + return params, data, r2s + + +def run_temporal_split(matched_clean, split_frac=0.7): + print("\n" + "=" * 70) + print(f"2. Temporal split ({int(split_frac*100)}/{int((1-split_frac)*100)})") + print("=" * 70) + + data_all = build_data(matched_clean) + day_idx = np.array(data_all["day_idx"]) + max_day = day_idx.max() + split_day = int(max_day * split_frac) + + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + def subset(data, mask): + jmask = jnp.array(mask) + return { + **{k: data[k] for k in ["peer_attrs", "target_attrs", "peer_overlap", + "n_pools", "n_peers", "k_attr", "pool_ids", + "vol_mean", "vol_std"]}, + "peer_vols": data["peer_vols"][jmask], + "peer_mask": data["peer_mask"][jmask], + "own_lag": data["own_lag"][jmask], + "y": data["y"][jmask], + "pool_idx": data["pool_idx"][jmask], + "day_idx": data_all["day_idx"][mask], + } + + train_data = subset(data_all, train_mask) + eval_data = subset(data_all, eval_mask) + + print(f" Train: {int(train_mask.sum())} samples, Eval: {int(eval_mask.sum())} samples") + + params = init_params(jax.random.PRNGKey(42), data_all["k_attr"]) + params = train(params, train_data) + + r2s_train = per_pool_r2(params, train_data) + r2s_eval = per_pool_r2(params, eval_data) + + pool_ids = data_all["pool_ids"] + for i, pid in enumerate(pool_ids): + r_tr = r2s_train.get(i, float("nan")) + r_ev = r2s_eval.get(i, float("nan")) + print(f" {pid[:16]} ({matched_clean[pid]['tokens']:<14}) " + f"train={r_tr:.3f} eval={r_ev:.3f}") + + vals_eval = [v for v in r2s_eval.values() if np.isfinite(v)] + print(f"\n Temporal eval: median R²={np.median(vals_eval):.4f}, " + f"mean={np.mean(vals_eval):.4f}") + return r2s_eval + + +def run_loo(matched_clean): + print("\n" + "=" * 70) + print("3. LOO DeepSets") + print("=" * 70) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + loo_r2s = [] + + for hold_out_idx in range(n_pools): + hold_out_pid = pool_ids[hold_out_idx] + + # Build training data excluding held-out pool's samples + # (but keeping its volume data for peers) + train_data = build_data(matched_clean, exclude_pool_idx=hold_out_idx) + + params = init_params(jax.random.PRNGKey(42), train_data["k_attr"]) + params = train(params, train_data, n_epochs=N_EPOCHS_LOO, verbose=False) + + # Build eval data: only held-out pool's samples + eval_data = build_data(matched_clean) + ho_mask = np.array(eval_data["pool_idx"]) == hold_out_idx + if ho_mask.sum() < 2: + loo_r2s.append(float("nan")) + continue + + jmask = jnp.array(ho_mask) + eval_sub = { + **{k: eval_data[k] for k in ["peer_attrs", "target_attrs", "peer_overlap", + "n_pools", "n_peers", "k_attr", "pool_ids", + "vol_mean", "vol_std"]}, + "peer_vols": eval_data["peer_vols"][jmask], + "peer_mask": eval_data["peer_mask"][jmask], + "own_lag": eval_data["own_lag"][jmask], + "y": eval_data["y"][jmask], + "pool_idx": eval_data["pool_idx"][jmask], + "day_idx": np.array(eval_data["day_idx"])[ho_mask], + } + + r2s = per_pool_r2(params, eval_sub) + r2 = r2s.get(hold_out_idx, float("nan")) + loo_r2s.append(r2) + + tag = "OK" if r2 > 0 else "NEG" + print(f" {hold_out_pid[:16]} ({matched_clean[hold_out_pid]['tokens']:<14}) " + f"R²={r2:.3f} [{tag}]") + + valid = [r for r in loo_r2s if np.isfinite(r)] + print(f"\n LOO DeepSets: median R²={np.median(valid):.4f}, " + f"mean={np.mean(valid):.4f}, " + f"n_neg={sum(1 for r in valid if r < 0)}") + return loo_r2s + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("DeepSets Cross-Pool Volume Prediction") + print(f" hidden={HIDDEN}, d={D_EMBED}, lr={LR}, " + f"alpha={L2_ALPHA}, epochs={N_EPOCHS}") + print("=" * 70) + + matched_clean, _ = load_stage1() + + params, data, r2_insample = run_insample(matched_clean) + r2_temporal = run_temporal_split(matched_clean) + r2_loo = run_loo(matched_clean) + + print("\n" + "=" * 70) + print("SUMMARY") + print("=" * 70) + vals_in = list(r2_insample.values()) + vals_temp = [v for v in r2_temporal.values() if np.isfinite(v)] + vals_loo = [r for r in r2_loo if np.isfinite(r)] + print(f" DeepSets in-sample: median R² = {np.median(vals_in):.4f}") + print(f" DeepSets temporal (30%): median R² = {np.median(vals_temp):.4f}") + print(f" DeepSets LOO: median R² = {np.median(vals_loo):.4f}") + print(f" ---") + print(f" Ridge in-sample: median R² = 0.441") + print(f" Naive AR1: median R² = 0.397") + print(f" Token-factored LOO: median R² = 0.362") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_residual_comparison.py b/experiments/run_residual_comparison.py new file mode 100644 index 00000000..201c4039 --- /dev/null +++ b/experiments/run_residual_comparison.py @@ -0,0 +1,235 @@ +"""Apples-to-apples R² comparison on noise residuals. + +Target for all methods: r_it = log(V_total_it) - log(V_arb_it) + +Methods: + 1. Option C: log(1 + exp(x_obs @ noise_coeffs) / V_arb) + 2. AR1 on residuals: r_{i, t-1} + 3. Ridge on residuals (peers only, in-sample) + 4. Ridge on residuals (peers + own lag, in-sample) + 5. Constant zero (predict r=0, i.e. V_total = V_arb) +""" + +import os +import pickle +import sys + +import numpy as np +from sklearn.linear_model import RidgeCV + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +def r2_score(y_true, y_pred): + ss_res = np.sum((y_true - y_pred) ** 2) + ss_tot = np.sum((y_true - y_true.mean()) ** 2) + return 1 - ss_res / max(ss_tot, 1e-10) + + +def main(): + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + import jax.numpy as jnp + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + K_OBS_REDUCED, build_x_obs, _parse_tokens, _canonicalize_token, + ) + + print("=" * 70) + print("Apples-to-Apples: All methods on noise residual target") + print(" target = log(V_total) - log(V_arb)") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # ---- Build aligned data per pool ---- + # For each pool: residual, Option C prediction of residual, dates + pool_data = {} + all_dates = set() + + for pid in pool_ids: + entry = matched_clean[pid] + oc = option_c_clean[pid] + panel = entry["panel"] + + # V_arb + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], + jnp.float64(oc["log_cadence"]), + jnp.float64(np.exp(oc["log_gas"])), + )) + v_arb = v_arb_all[entry["day_indices"]] + log_v_arb = np.log(np.maximum(v_arb, 1e-6)) + + # Observed + log_vol = panel["log_volume"].values.astype(float) + dates = panel["date"].values + + # Noise residual target + resid = log_vol - log_v_arb + + # Option C noise prediction (in residual space) + x_obs = build_x_obs(panel, reduced=True) + noise_coeffs = oc["noise_coeffs"][:K_OBS_REDUCED] + v_noise_oc = np.exp(x_obs @ noise_coeffs) + resid_pred_oc = np.log(np.maximum(1.0 + v_noise_oc / np.maximum(v_arb, 1e-6), 1e-10)) + + pool_data[pid] = { + "dates": dates, + "resid": resid, + "resid_pred_oc": resid_pred_oc, + "log_vol": log_vol, + "v_arb": v_arb, + } + all_dates.update(dates) + + # ---- Build residual matrix for cross-pool methods ---- + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + resid_matrix = np.full((n_dates, n_pools), np.nan) + for j, pid in enumerate(pool_ids): + pd = pool_data[pid] + for k, date in enumerate(pd["dates"]): + resid_matrix[date_to_idx[date], j] = pd["resid"][k] + + # ---- Token overlap for peer identification ---- + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + # ---- Compute R² for each method, per pool ---- + results = {m: [] for m in [ + "option_c", "ar1", "ridge_peers", "ridge_peers_own", + "constant_zero", "peer_mean", + ]} + + print(f"\n{'Pool':<18} {'Tokens':<14} {'OptC':>7} {'AR1':>7} " + f"{'R_peer':>7} {'R_p+own':>7} {'zero':>7} {'pmean':>7} {'n':>5}") + print("-" * 90) + + for i, pid in enumerate(pool_ids): + pd = pool_data[pid] + resid = pd["resid"] + n_obs = len(resid) + + # --- Option C --- + r2_oc = r2_score(resid, pd["resid_pred_oc"]) + + # --- Constant zero (V_total = V_arb) --- + r2_zero = r2_score(resid, np.zeros_like(resid)) + + # --- AR1 on residuals --- + if n_obs >= 3: + r2_ar1 = r2_score(resid[1:], resid[:-1]) + else: + r2_ar1 = np.nan + + # --- Ridge peers only (in-sample) --- + X_lag = resid_matrix[:-1, :] + y_cur = resid_matrix[1:, i] + own_lag = X_lag[:, i] + valid = ~np.isnan(y_cur) + + X_others = np.delete(X_lag, i, axis=1) + X_filled = X_others.copy() + for c in range(X_filled.shape[1]): + col = X_filled[:, c] + m = np.nanmean(col) + col[np.isnan(col)] = m if np.isfinite(m) else 0.0 + X_filled[:, c] = col + + X_peers = X_filled[valid] + y_i = y_cur[valid] + + if len(y_i) >= 10: + model_p = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model_p.fit(X_peers, y_i) + r2_rp = r2_score(y_i, model_p.predict(X_peers)) + else: + r2_rp = np.nan + + # --- Ridge peers + own lag (in-sample) --- + valid_own = valid & ~np.isnan(own_lag) + X_both = np.column_stack([X_filled, own_lag[:, None]]) + X_both_v = X_both[valid_own] + y_both = y_cur[valid_own] + + if len(y_both) >= 10: + model_po = RidgeCV(alphas=np.logspace(-2, 4, 50)) + model_po.fit(X_both_v, y_both) + r2_rpo = r2_score(y_both, model_po.predict(X_both_v)) + else: + r2_rpo = np.nan + + # --- Peer mean (zero parameter) --- + peers = [j for j in range(n_pools) if j != i + and len(pool_tokens[i] & pool_tokens[j]) >= 1] + if peers: + peer_lag = resid_matrix[:-1, :][:, peers] + peer_mean = np.nanmean(peer_lag, axis=1) + y_pm = y_cur[valid] + pm_pred = peer_mean[valid] + pm_valid = ~np.isnan(pm_pred) + if pm_valid.sum() >= 3: + r2_pm = r2_score(y_pm[pm_valid], pm_pred[pm_valid]) + else: + r2_pm = np.nan + else: + r2_pm = np.nan + + results["option_c"].append(r2_oc) + results["ar1"].append(r2_ar1) + results["ridge_peers"].append(r2_rp) + results["ridge_peers_own"].append(r2_rpo) + results["constant_zero"].append(r2_zero) + results["peer_mean"].append(r2_pm) + + tokens = matched_clean[pid]["tokens"] + print(f" {pid[:16]} {tokens:<14} {r2_oc:>7.3f} {r2_ar1:>7.3f} " + f"{r2_rp:>7.3f} {r2_rpo:>7.3f} {r2_zero:>7.3f} " + f"{r2_pm:>7.3f} {n_obs:>5}") + + # ---- Summary ---- + def safe_median(xs): + v = [x for x in xs if np.isfinite(x)] + return np.median(v) if v else float("nan") + + print(f"\n{'='*70}") + print("SUMMARY — all on noise residual target") + print(f"{'='*70}") + for name, label in [ + ("option_c", "Option C (per-pool fitted)"), + ("ar1", "AR1 on residuals"), + ("ridge_peers", "Ridge peers only (in-sample)"), + ("ridge_peers_own", "Ridge peers + own lag (in-sample)"), + ("peer_mean", "Peer mean (0 params)"), + ("constant_zero", "Constant zero (V_total=V_arb)"), + ]: + vals = results[name] + med = safe_median(vals) + mean = np.nanmean([x for x in vals if np.isfinite(x)]) + n_neg = sum(1 for x in vals if np.isfinite(x) and x < 0) + print(f" {label:<35} median R² = {med:>7.4f} " + f"mean = {mean:>7.4f} n_neg = {n_neg}") + + +if __name__ == "__main__": + main() From 1056ee03f28eb77e994a039e278c49b1cce9b7df Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 17 Mar 2026 13:27:25 +0000 Subject: [PATCH 060/115] wip on deepsets --- experiments/run_deepsets_v2.py | 212 ++++++++++++++++++++++++++------- 1 file changed, 167 insertions(+), 45 deletions(-) diff --git a/experiments/run_deepsets_v2.py b/experiments/run_deepsets_v2.py index 1ba3485f..7e9a41cd 100644 --- a/experiments/run_deepsets_v2.py +++ b/experiments/run_deepsets_v2.py @@ -6,12 +6,21 @@ Feature menu: Peer (encoder) — always: peer_attr, target_attr, vol_lag1, overlap optional: vol_lag2, vol_change, tvl, volatility + relational: same_chain, log_tvl_ratio, log_fee_ratio Local (decoder) — always: target_attr, own_vol_lag1, dow_sin, dow_cos optional: own_vol_lag2, own_vol_change, own_tvl, own_volatility +Model variants: + encoder_type: "mlp" (2-layer ReLU) or "linear" (single affine) + no_peers: decoder-only ablation (zero peer summary) + huber_delta: Huber loss transition point (default 1.0) + Per-pool loss weighting (equal weight per pool regardless of sample count) + Usage: python experiments/run_deepsets_v2.py # defaults python experiments/run_deepsets_v2.py --tune 50 # Optuna + python experiments/run_deepsets_v2.py --no-peers # decoder-only + python experiments/run_deepsets_v2.py --encoder-type linear python experiments/run_deepsets_v2.py --loo # LOO eval """ @@ -101,6 +110,13 @@ def build_all_features(matched_clean, option_c_clean): X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) k_attr = X_attr_norm.shape[1] + # Raw per-pool values for relational features + fee_idx = attr_names.index("log_fee") + tvl_idx = attr_names.index("mean_log_tvl") + raw_log_fee = X_attr[:, fee_idx] + raw_mean_log_tvl = X_attr[:, tvl_idx] + pool_chains = [matched_clean[pid]["chain"] for pid in pool_ids] + # Token overlap pool_tokens = {} for i, pid in enumerate(pool_ids): @@ -111,6 +127,9 @@ def build_all_features(matched_clean, option_c_clean): peer_attrs = np.zeros((n_pools, n_peers, k_attr), dtype=np.float32) peer_overlap = np.zeros((n_pools, n_peers), dtype=np.float32) peer_col_idx = np.zeros((n_pools, n_peers), dtype=np.int32) + rel_same_chain = np.zeros((n_pools, n_peers), dtype=np.float32) + rel_log_tvl_ratio = np.zeros((n_pools, n_peers), dtype=np.float32) + rel_log_fee_ratio = np.zeros((n_pools, n_peers), dtype=np.float32) for i in range(n_pools): peers = [j for j in range(n_pools) if j != i] @@ -118,6 +137,15 @@ def build_all_features(matched_clean, option_c_clean): peer_attrs[i, p] = X_attr_norm[j] peer_overlap[i, p] = len(pool_tokens[i] & pool_tokens[j]) peer_col_idx[i, p] = j + rel_same_chain[i, p] = float(pool_chains[i] == pool_chains[j]) + rel_log_tvl_ratio[i, p] = abs(raw_mean_log_tvl[i] - raw_mean_log_tvl[j]) + rel_log_fee_ratio[i, p] = abs(raw_log_fee[i] - raw_log_fee[j]) + + # Standardize ratio features + for arr in [rel_log_tvl_ratio, rel_log_fee_ratio]: + mu = np.mean(arr) + sigma = max(np.std(arr), 1e-6) + arr[:] = ((arr - mu) / sigma).astype(np.float32) # Standardization stats for volumes vol_mean = float(np.nanmean(vol_matrix)) @@ -219,6 +247,9 @@ def _norm_vola(x): "peer_attrs": peer_attrs, # (n_pools, n_peers, k_attr) "target_attrs": X_attr_norm, # (n_pools, k_attr) "peer_overlap": peer_overlap, # (n_pools, n_peers) + "rel_same_chain": rel_same_chain, # (n_pools, n_peers) + "rel_log_tvl_ratio": rel_log_tvl_ratio, # (n_pools, n_peers) + "rel_log_fee_ratio": rel_log_fee_ratio, # (n_pools, n_peers) # Per-sample peer features "pf_vol_lag1": pf_vol_lag1, "pf_vol_lag2": pf_vol_lag2, @@ -253,8 +284,7 @@ def _norm_vola(x): def assemble_inputs(data, feat_cfg): """Assemble encoder/decoder inputs based on feature config. - Returns (peer_input, local_input, peer_mask, pool_idx, y, v_arb) - all as JAX arrays ready for training. + Returns dict with JAX arrays ready for training. """ n_samples = len(data["pool_idx"]) n_peers = data["n_peers"] @@ -281,6 +311,14 @@ def assemble_inputs(data, feat_cfg): if feat_cfg.get("peer_volatility"): peer_parts.append(data["pf_volatility"][:, :, None]) + # Relational features (optional via feat_cfg) + if feat_cfg.get("rel_same_chain", True): + peer_parts.append(data["rel_same_chain"][pool_idx][:, :, None]) + if feat_cfg.get("rel_tvl_ratio", True): + peer_parts.append(data["rel_log_tvl_ratio"][pool_idx][:, :, None]) + if feat_cfg.get("rel_fee_ratio", True): + peer_parts.append(data["rel_log_fee_ratio"][pool_idx][:, :, None]) + peer_input = np.concatenate(peer_parts, axis=-1).astype(np.float32) # ---- Local decoder input: (n_samples, n_feat) ---- @@ -310,6 +348,7 @@ def assemble_inputs(data, feat_cfg): "y": jnp.array(data["y_total"]), "v_arb": jnp.array(data["v_arb"]), "pool_idx": jnp.array(pool_idx), + "n_pools": data["n_pools"], "n_peer_feat": peer_input.shape[-1], "n_local_feat": local_input.shape[-1], } @@ -318,62 +357,102 @@ def assemble_inputs(data, feat_cfg): # ---- Model ---- -def init_params(key, n_peer_feat, n_local_feat, hidden, d_embed): +def init_params(key, n_peer_feat, n_local_feat, hidden, d_embed, + encoder_type="mlp"): + """Initialize model parameters. + + encoder_type: "mlp" (2-layer ReLU) or "linear" (single affine). + Presence of "enc_W2" in params dict distinguishes the two at forward time. + """ k1, k2, k3, k4 = jax.random.split(key, 4) dec_in = d_embed + n_local_feat - return { - "enc_W1": jax.random.normal(k1, (n_peer_feat, hidden)) * np.sqrt(2.0 / n_peer_feat), - "enc_b1": jnp.zeros(hidden), - "enc_W2": jax.random.normal(k2, (hidden, d_embed)) * np.sqrt(2.0 / hidden), - "enc_b2": jnp.zeros(d_embed), - "dec_W1": jax.random.normal(k3, (dec_in, hidden)) * np.sqrt(2.0 / dec_in), - "dec_b1": jnp.zeros(hidden), - "dec_W2": jax.random.normal(k4, (hidden, 1)) * 0.01, - "dec_b2": jnp.zeros(1), - } + params = {} + + if encoder_type == "mlp": + params["enc_W1"] = jax.random.normal(k1, (n_peer_feat, hidden)) * np.sqrt(2.0 / n_peer_feat) + params["enc_b1"] = jnp.zeros(hidden) + params["enc_W2"] = jax.random.normal(k2, (hidden, d_embed)) * np.sqrt(2.0 / hidden) + params["enc_b2"] = jnp.zeros(d_embed) + else: # linear + params["enc_W1"] = jax.random.normal(k1, (n_peer_feat, d_embed)) * np.sqrt(2.0 / n_peer_feat) + params["enc_b1"] = jnp.zeros(d_embed) + + params["dec_W1"] = jax.random.normal(k3, (dec_in, hidden)) * np.sqrt(2.0 / dec_in) + params["dec_b1"] = jnp.zeros(hidden) + params["dec_W2"] = jax.random.normal(k4, (hidden, 1)) * 0.01 + params["dec_b2"] = jnp.zeros(1) + return params -def forward(params, peer_input, peer_mask, local_input): +def forward(params, peer_input, peer_mask, local_input, no_peers=False): """Returns predicted log_volume (total) per sample.""" batch, n_peers, _ = peer_input.shape - flat = peer_input.reshape(-1, peer_input.shape[-1]) - h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) - h = h @ params["enc_W2"] + params["enc_b2"] - h = h.reshape(batch, n_peers, -1) - - h_masked = h * peer_mask[:, :, None] - n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) - summary = jnp.sum(h_masked, axis=1) / n_valid + if no_peers: + d_embed = params["dec_W1"].shape[0] - local_input.shape[-1] + summary = jnp.zeros((batch, d_embed)) + else: + flat = peer_input.reshape(-1, peer_input.shape[-1]) + if "enc_W2" in params: + # MLP encoder: 2-layer with ReLU + h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) + h = h @ params["enc_W2"] + params["enc_b2"] + else: + # Linear encoder: single affine + h = flat @ params["enc_W1"] + params["enc_b1"] + h = h.reshape(batch, n_peers, -1) + + h_masked = h * peer_mask[:, :, None] + n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) + summary = jnp.sum(h_masked, axis=1) / n_valid dec_in = jnp.concatenate([summary, local_input], axis=-1) h_dec = jnp.maximum(dec_in @ params["dec_W1"] + params["dec_b1"], 0.0) return (h_dec @ params["dec_W2"] + params["dec_b2"])[:, 0] -def loss_fn(params, peer_input, peer_mask, local_input, y, l2_alpha): - """MSE on total log_volume + L2 reg.""" - pred = forward(params, peer_input, peer_mask, local_input) - mse = jnp.mean((pred - y) ** 2) +def loss_fn(params, peer_input, peer_mask, local_input, y, l2_alpha, + pool_idx, n_pools, huber_delta, no_peers): + """Huber loss with per-pool weighting + L2 reg.""" + pred = forward(params, peer_input, peer_mask, local_input, no_peers) + residuals = pred - y + abs_r = jnp.abs(residuals) + huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + # Per-pool mean loss, then average across pools + total = 0.0 + for i in range(n_pools): + mask_i = (pool_idx == i).astype(jnp.float32) + n_i = jnp.maximum(jnp.sum(mask_i), 1.0) + total = total + jnp.sum(huber_vals * mask_i) / n_i + data_loss = total / n_pools + reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) - return mse + l2_alpha * reg + return data_loss + l2_alpha * reg -_grad_fn = jax.jit(jax.value_and_grad(loss_fn)) +# n_pools (arg 7) and no_peers (arg 9) must be static for Python control flow +_grad_fn = jax.jit(jax.value_and_grad(loss_fn), static_argnums=(7, 9)) # ---- Training ---- -def train(params, inputs, n_epochs, lr, l2_alpha, verbose=True): +def train(params, inputs, n_epochs, lr, l2_alpha, huber_delta=1.0, + no_peers=False, verbose=True): m = {k: jnp.zeros_like(v) for k, v in params.items()} v = {k: jnp.zeros_like(v) for k, v in params.items()} final_loss = float("inf") + n_pools = int(inputs["n_pools"]) + pool_idx = inputs["pool_idx"] + for epoch in range(n_epochs): loss_val, grads = _grad_fn( params, inputs["peer_input"], inputs["peer_mask"], inputs["local_input"], inputs["y"], l2_alpha, + pool_idx, n_pools, huber_delta, no_peers, ) final_loss = float(loss_val) @@ -393,10 +472,11 @@ def train(params, inputs, n_epochs, lr, l2_alpha, verbose=True): # ---- Evaluation ---- -def evaluate(params, inputs, data, label=""): +def evaluate(params, inputs, data, label="", no_peers=False): """Per-pool R² on total volume and noise residual.""" pred_total = np.array(forward( - params, inputs["peer_input"], inputs["peer_mask"], inputs["local_input"], + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], no_peers=no_peers, )) y_total = np.array(inputs["y"]) v_arb = np.array(inputs["v_arb"]) @@ -469,28 +549,41 @@ def _subset(d, mask): train_inputs = assemble_inputs(train_data, feat_cfg) eval_inputs = assemble_inputs(eval_data, feat_cfg) + encoder_type = hparams.get("encoder_type", "mlp") + no_peers = hparams.get("no_peers", False) + huber_delta = hparams.get("huber_delta", 1.0) + n_pf = train_inputs["n_peer_feat"] n_lf = train_inputs["n_local_feat"] n_params = sum(v.size for v in init_params( jax.random.PRNGKey(0), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], + encoder_type=encoder_type, ).values()) print(f" Train: {int(train_mask.sum())}, Eval: {int(eval_mask.sum())}, " f"peer_feat={n_pf}, local_feat={n_lf}, params={n_params}") + if encoder_type != "mlp": + print(f" encoder_type={encoder_type}") + if no_peers: + print(f" no_peers=True (decoder-only ablation)") + if huber_delta != 1.0: + print(f" huber_delta={huber_delta}") params = init_params( jax.random.PRNGKey(42), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], + encoder_type=encoder_type, ) t0 = time.time() params, final_loss = train( params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], + huber_delta=huber_delta, no_peers=no_peers, ) print(f" Training: {time.time() - t0:.1f}s") print("\n --- Train ---") - evaluate(params, train_inputs, data) + evaluate(params, train_inputs, data, no_peers=no_peers) print("\n --- Eval ---") - _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data) + _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data, no_peers=no_peers) return med_resid_eval @@ -498,6 +591,13 @@ def _subset(d, mask): # ---- Optuna ---- +_FEAT_KEYS = [ + "peer_vol_lag2", "peer_vol_change", "peer_tvl", "peer_volatility", + "own_vol_lag2", "own_vol_change", "own_tvl", "own_volatility", + "rel_same_chain", "rel_tvl_ratio", "rel_fee_ratio", +] + + def run_optuna(data, n_trials): import optuna @@ -528,6 +628,9 @@ def objective(trial): "own_vol_change": trial.suggest_categorical("own_vol_change", [True, False]), "own_tvl": trial.suggest_categorical("own_tvl", [True, False]), "own_volatility": trial.suggest_categorical("own_volatility", [True, False]), + "rel_same_chain": trial.suggest_categorical("rel_same_chain", [True, False]), + "rel_tvl_ratio": trial.suggest_categorical("rel_tvl_ratio", [True, False]), + "rel_fee_ratio": trial.suggest_categorical("rel_fee_ratio", [True, False]), } hparams = { "hidden": trial.suggest_categorical("hidden", [8, 16, 32]), @@ -535,6 +638,9 @@ def objective(trial): "lr": trial.suggest_float("lr", 1e-4, 1e-2, log=True), "l2_alpha": trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True), "n_epochs": trial.suggest_categorical("n_epochs", [500, 1000, 2000]), + "encoder_type": trial.suggest_categorical("encoder_type", ["mlp", "linear"]), + "huber_delta": trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5, 2.0]), + "no_peers": trial.suggest_categorical("no_peers", [True, False]), } train_inputs = assemble_inputs(train_data, feat_cfg) @@ -544,14 +650,20 @@ def objective(trial): jax.random.PRNGKey(42), train_inputs["n_peer_feat"], train_inputs["n_local_feat"], hparams["hidden"], hparams["d_embed"], + encoder_type=hparams["encoder_type"], + ) + params, _ = train( + params, train_inputs, hparams["n_epochs"], + hparams["lr"], hparams["l2_alpha"], + huber_delta=hparams["huber_delta"], + no_peers=hparams["no_peers"], + verbose=False, ) - params, _ = train(params, train_inputs, hparams["n_epochs"], - hparams["lr"], hparams["l2_alpha"], verbose=False) # Eval R² on noise residual pred = np.array(forward( params, eval_inputs["peer_input"], eval_inputs["peer_mask"], - eval_inputs["local_input"], + eval_inputs["local_input"], no_peers=hparams["no_peers"], )) y = np.array(eval_inputs["y"]) v_arb = np.array(eval_inputs["v_arb"]) @@ -580,11 +692,13 @@ def objective(trial): med_total = float(np.median(r2_totals)) if r2_totals else -10.0 trial.set_user_attr("med_total_r2", med_total) - n_feat = sum(feat_cfg.values()) + n_feat = sum(1 for k in _FEAT_KEYS if feat_cfg.get(k)) print(f" Trial {trial.number}: resid={med_resid:.4f} total={med_total:.4f} " - f"h={hparams['hidden']} d={hparams['d_embed']} " + f"enc={hparams['encoder_type']} h={hparams['hidden']} d={hparams['d_embed']} " + f"hub={hparams['huber_delta']} " + f"{'no_peers ' if hparams['no_peers'] else ''}" f"lr={hparams['lr']:.1e} a={hparams['l2_alpha']:.1e} " - f"ep={hparams['n_epochs']} feat={n_feat}/8") + f"ep={hparams['n_epochs']} feat={n_feat}/11") return med_resid @@ -605,15 +719,12 @@ def objective(trial): reverse=True) for t in trials[:10]: if t.value is not None: - feats = sum(1 for k in ["peer_vol_lag2", "peer_vol_change", - "peer_tvl", "peer_volatility", - "own_vol_lag2", "own_vol_change", - "own_tvl", "own_volatility"] - if t.params.get(k)) + feats = sum(1 for k in _FEAT_KEYS if t.params.get(k)) print(f" #{t.number}: resid={t.value:.4f} " f"total={t.user_attrs.get('med_total_r2', '?'):.4f} " + f"enc={t.params['encoder_type']} " f"h={t.params['hidden']} d={t.params['d_embed']} " - f"feat={feats}/8") + f"feat={feats}/11") return study @@ -631,6 +742,10 @@ def main(): parser.add_argument("--lr", type=float, default=3e-4) parser.add_argument("--l2-alpha", type=float, default=1e-3) parser.add_argument("--epochs", type=int, default=1000) + parser.add_argument("--encoder-type", choices=["mlp", "linear"], default="mlp") + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--no-peers", action="store_true", + help="Decoder-only ablation (zero peer summary)") # Feature flags parser.add_argument("--peer-vol-lag2", action="store_true") parser.add_argument("--peer-vol-change", action="store_true") @@ -655,6 +770,10 @@ def main(): "own_vol_change": args.own_vol_change or args.all_features, "own_tvl": args.own_tvl or args.all_features, "own_volatility": args.own_volatility or args.all_features, + # Relational features: always on for CLI, searchable in Optuna + "rel_same_chain": True, + "rel_tvl_ratio": True, + "rel_fee_ratio": True, } hparams = { "hidden": args.hidden, @@ -662,6 +781,9 @@ def main(): "lr": args.lr, "l2_alpha": args.l2_alpha, "n_epochs": args.epochs, + "encoder_type": args.encoder_type, + "huber_delta": args.huber_delta, + "no_peers": args.no_peers, } print("=" * 70) From 7ab989ab519bfbbb4e2a6d67dd62cc02b99982fb Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 17 Mar 2026 13:32:09 +0000 Subject: [PATCH 061/115] =?UTF-8?q?feat:=20deepsets=20v2=20improvements=20?= =?UTF-8?q?=E2=80=94=20relational=20features,=20Huber=20loss,=20encoder=20?= =?UTF-8?q?variants?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add six improvements to the DeepSets cross-pool volume prediction model: - Explicit relational features in encoder: same_chain, log_tvl_ratio, log_fee_ratio precomputed per (target, peer) pair - Huber loss with configurable delta replaces MSE for outlier robustness - Per-pool loss weighting: equal weight per pool regardless of sample count - Linear encoder variant (single affine) as alternative to 2-layer MLP - Decoder-only ablation (--no-peers) to isolate cross-pool signal value - Relational features searchable in Optuna (11 feature booleans total) --- experiments/run_deepsets_v2.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/experiments/run_deepsets_v2.py b/experiments/run_deepsets_v2.py index 7e9a41cd..3d899b87 100644 --- a/experiments/run_deepsets_v2.py +++ b/experiments/run_deepsets_v2.py @@ -286,9 +286,6 @@ def assemble_inputs(data, feat_cfg): Returns dict with JAX arrays ready for training. """ - n_samples = len(data["pool_idx"]) - n_peers = data["n_peers"] - # ---- Peer encoder input: (n_samples, n_peers, n_feat) ---- # Always: peer_attr, target_attr, vol_lag1, overlap pool_idx = data["pool_idx"] @@ -574,7 +571,7 @@ def _subset(d, mask): encoder_type=encoder_type, ) t0 = time.time() - params, final_loss = train( + params, _ = train( params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], huber_delta=huber_delta, no_peers=no_peers, ) From d6d1e7b7a59f8b9416853956dc9cce31a253f09f Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 17 Mar 2026 18:07:29 +0000 Subject: [PATCH 062/115] feat: LOO evaluation, warm-start decoder, residual target, minimal encoder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - LOO cross-validation: train on N-1 pools, evaluate held-out pool (tests cross-pool generalization, the actual use case) - Decoder warm-start via OLS through hidden activations — predictions start in the right volume range instead of near-zero - --target-residual: train on noise residual (log_vol - log_V_arb) instead of total log_volume - --minimal-encoder: 7-feature encoder (fee, tvl, overlap, same_chain) to prevent pool identification through full attribute vectors - Loss function handles variable active pool counts (for LOO gaps) - Scatter-add loss replaces Python loop over pools - Module-level _subset with explicit _SAMPLE_KEYS - Fix in-place mutation of ratio feature arrays LOO results (default hparams): median total R²=0.39, resid R²=-0.22 Temporal split with --target-residual: train resid R²~0.06 Optuna sweep in progress, best eval resid R²=-0.108 (LR 20x too low at default — Optuna finds ~6e-3 optimal) --- experiments/run_deepsets_v2.py | 318 ++++++++++++++++++++++++++------- 1 file changed, 256 insertions(+), 62 deletions(-) diff --git a/experiments/run_deepsets_v2.py b/experiments/run_deepsets_v2.py index 3d899b87..4e021416 100644 --- a/experiments/run_deepsets_v2.py +++ b/experiments/run_deepsets_v2.py @@ -141,11 +141,14 @@ def build_all_features(matched_clean, option_c_clean): rel_log_tvl_ratio[i, p] = abs(raw_mean_log_tvl[i] - raw_mean_log_tvl[j]) rel_log_fee_ratio[i, p] = abs(raw_log_fee[i] - raw_log_fee[j]) - # Standardize ratio features - for arr in [rel_log_tvl_ratio, rel_log_fee_ratio]: + # Standardize ratio features (new arrays, no in-place mutation) + def _standardize(arr): mu = np.mean(arr) sigma = max(np.std(arr), 1e-6) - arr[:] = ((arr - mu) / sigma).astype(np.float32) + return ((arr - mu) / sigma).astype(np.float32) + + rel_log_tvl_ratio = _standardize(rel_log_tvl_ratio) + rel_log_fee_ratio = _standardize(rel_log_fee_ratio) # Standardization stats for volumes vol_mean = float(np.nanmean(vol_matrix)) @@ -267,6 +270,7 @@ def _norm_vola(x): "lf_dow_cos": lf_dow_cos, # Targets "y_total": y_total, + "y_residual": (y_total - np.log(np.maximum(v_arb_samples, 1e-6))).astype(np.float32), "v_arb": v_arb_samples, # Indices "pool_idx": sample_pools, @@ -278,6 +282,8 @@ def _norm_vola(x): "pool_ids": pool_ids, "vol_mean": vol_mean, "vol_std": vol_std, + "fee_attr_idx": fee_idx, + "tvl_attr_idx": tvl_idx, } @@ -287,18 +293,37 @@ def assemble_inputs(data, feat_cfg): Returns dict with JAX arrays ready for training. """ # ---- Peer encoder input: (n_samples, n_peers, n_feat) ---- - # Always: peer_attr, target_attr, vol_lag1, overlap pool_idx = data["pool_idx"] pa = data["peer_attrs"][pool_idx] # (n_samples, n_peers, k_attr) ta = data["target_attrs"][pool_idx] # (n_samples, k_attr) - ta_broad = np.broadcast_to(ta[:, None, :], pa.shape) - - peer_parts = [ - pa, ta_broad, - data["pf_vol_lag1"][:, :, None], - data["peer_overlap"][pool_idx][:, :, None], - ] + if feat_cfg.get("minimal_encoder"): + # 7-feature encoder: peer_fee, peer_tvl, target_fee, target_tvl, + # vol_lag1, overlap, same_chain — prevents pool identification + fi = data["fee_attr_idx"] + ti = data["tvl_attr_idx"] + pa_min = np.stack([pa[:, :, fi], pa[:, :, ti]], axis=-1) + ta_min = np.stack([ta[:, fi], ta[:, ti]], axis=-1) + ta_min_broad = np.broadcast_to( + ta_min[:, None, :], (pa_min.shape[0], pa_min.shape[1], 2)) + peer_parts = [ + pa_min, ta_min_broad, + data["pf_vol_lag1"][:, :, None], + data["peer_overlap"][pool_idx][:, :, None], + data["rel_same_chain"][pool_idx][:, :, None], + ] + else: + ta_broad = np.broadcast_to(ta[:, None, :], pa.shape) + peer_parts = [ + pa, ta_broad, + data["pf_vol_lag1"][:, :, None], + data["peer_overlap"][pool_idx][:, :, None], + ] + # Relational features (optional via feat_cfg, default on) + if feat_cfg.get("rel_same_chain", True): + peer_parts.append(data["rel_same_chain"][pool_idx][:, :, None]) + + # Optional temporal peer features (both modes) if feat_cfg.get("peer_vol_lag2"): peer_parts.append(data["pf_vol_lag2"][:, :, None]) if feat_cfg.get("peer_vol_change"): @@ -308,9 +333,7 @@ def assemble_inputs(data, feat_cfg): if feat_cfg.get("peer_volatility"): peer_parts.append(data["pf_volatility"][:, :, None]) - # Relational features (optional via feat_cfg) - if feat_cfg.get("rel_same_chain", True): - peer_parts.append(data["rel_same_chain"][pool_idx][:, :, None]) + # Relational ratio features (both modes) if feat_cfg.get("rel_tvl_ratio", True): peer_parts.append(data["rel_log_tvl_ratio"][pool_idx][:, :, None]) if feat_cfg.get("rel_fee_ratio", True): @@ -342,7 +365,7 @@ def assemble_inputs(data, feat_cfg): "peer_input": jnp.array(peer_input), "local_input": jnp.array(local_input), "peer_mask": jnp.array(data["peer_mask"]), - "y": jnp.array(data["y_total"]), + "y": jnp.array(data["y_residual"] if feat_cfg.get("target_residual") else data["y_total"]), "v_arb": jnp.array(data["v_arb"]), "pool_idx": jnp.array(pool_idx), "n_pools": data["n_pools"], @@ -381,6 +404,33 @@ def init_params(key, n_peer_feat, n_local_feat, hidden, d_embed, return params +def warm_start_decoder(params, inputs, d_embed): + """Set decoder output layer via OLS through hidden activations. + + Fits y ~ h(local_input) with zero peer summary, so the decoder + starts predicting in the right volume range (~10-17 log scale). + """ + local = np.array(inputs["local_input"]) + y = np.array(inputs["y"]) + n = local.shape[0] + + # Simulate decoder input with zero peer summary + dec_in = np.concatenate( + [np.zeros((n, d_embed), dtype=np.float32), local], axis=1) + + # Forward through first decoder layer with current (random) weights + h = np.maximum( + dec_in @ np.array(params["dec_W1"]) + np.array(params["dec_b1"]), 0.0) + + # OLS: y ≈ h @ W2 + b2 + h_bias = np.concatenate([h, np.ones((n, 1), dtype=np.float32)], axis=1) + sol, _, _, _ = np.linalg.lstsq(h_bias, y[:, None], rcond=None) + + params["dec_W2"] = jnp.array(sol[:-1].astype(np.float32)) + params["dec_b2"] = jnp.array(sol[-1:].astype(np.float32)) + return params + + def forward(params, peer_input, peer_mask, local_input, no_peers=False): """Returns predicted log_volume (total) per sample.""" batch, n_peers, _ = peer_input.shape @@ -417,19 +467,19 @@ def loss_fn(params, peer_input, peer_mask, local_input, y, l2_alpha, huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, huber_delta * (abs_r - 0.5 * huber_delta)) - # Per-pool mean loss, then average across pools - total = 0.0 - for i in range(n_pools): - mask_i = (pool_idx == i).astype(jnp.float32) - n_i = jnp.maximum(jnp.sum(mask_i), 1.0) - total = total + jnp.sum(huber_vals * mask_i) / n_i - data_loss = total / n_pools + # Per-pool mean loss, then average across active pools (handles LOO gaps) + pool_counts = jnp.zeros(n_pools).at[pool_idx].add(jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) + data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) return data_loss + l2_alpha * reg -# n_pools (arg 7) and no_peers (arg 9) must be static for Python control flow +# n_pools (arg 7): static for jnp.zeros shape; no_peers (arg 9): static for if/else in forward _grad_fn = jax.jit(jax.value_and_grad(loss_fn), static_argnums=(7, 9)) @@ -469,20 +519,31 @@ def train(params, inputs, n_epochs, lr, l2_alpha, huber_delta=1.0, # ---- Evaluation ---- -def evaluate(params, inputs, data, label="", no_peers=False): +def evaluate(params, inputs, data, label="", no_peers=False, + target_residual=False): """Per-pool R² on total volume and noise residual.""" - pred_total = np.array(forward( + pred = np.array(forward( params, inputs["peer_input"], inputs["peer_mask"], inputs["local_input"], no_peers=no_peers, )) - y_total = np.array(inputs["y"]) + y = np.array(inputs["y"]) v_arb = np.array(inputs["v_arb"]) - pool_idx = np.array(data["pool_idx"]) if "pool_idx" in data else np.array(inputs["pool_idx"]) + pool_idx = np.array(inputs["pool_idx"]) - # Noise residual: compare (pred - log(v_arb)) vs (y - log(v_arb)) log_v_arb = np.log(np.maximum(v_arb, 1e-6)) - resid_true = y_total - log_v_arb - resid_pred = pred_total - log_v_arb + + if target_residual: + # Model predicts noise residual directly + resid_true = y + resid_pred = pred + y_total = y + log_v_arb # reconstruct total for total R² + pred_total = pred + log_v_arb + else: + # Model predicts total log_volume + y_total = y + pred_total = pred + resid_true = y - log_v_arb + resid_pred = pred - log_v_arb r2_total = {} r2_resid = {} @@ -521,6 +582,26 @@ def _med(d): return med_total, med_resid, r2_total, r2_resid +# Keys indexed by sample (shape[0] == n_samples) +_SAMPLE_KEYS = { + "pf_vol_lag1", "pf_vol_lag2", "pf_vol_change", "pf_tvl", "pf_volatility", + "peer_mask", "lf_own_vol_lag1", "lf_own_vol_lag2", "lf_own_vol_change", + "lf_own_tvl", "lf_own_volatility", "lf_dow_sin", "lf_dow_cos", + "y_total", "y_residual", "v_arb", "pool_idx", "day_idx", +} + + +def _subset(d, mask): + """Subset sample-indexed arrays by boolean mask.""" + out = {} + for k, v in d.items(): + if k in _SAMPLE_KEYS and isinstance(v, np.ndarray): + out[k] = v[mask] + else: + out[k] = v + return out + + # ---- Temporal split ---- @@ -531,15 +612,6 @@ def run_temporal(data, feat_cfg, hparams, split_frac=0.7): train_mask = day_idx <= split_day eval_mask = day_idx > split_day - def _subset(d, mask): - out = {} - for k, v in d.items(): - if isinstance(v, np.ndarray) and len(v) == len(mask): - out[k] = v[mask] - else: - out[k] = v - return out - train_data = _subset(data, train_mask) eval_data = _subset(data, eval_mask) @@ -549,6 +621,7 @@ def _subset(d, mask): encoder_type = hparams.get("encoder_type", "mlp") no_peers = hparams.get("no_peers", False) huber_delta = hparams.get("huber_delta", 1.0) + target_residual = feat_cfg.get("target_residual", False) n_pf = train_inputs["n_peer_feat"] n_lf = train_inputs["n_local_feat"] @@ -559,6 +632,8 @@ def _subset(d, mask): print(f" Train: {int(train_mask.sum())}, Eval: {int(eval_mask.sum())}, " f"peer_feat={n_pf}, local_feat={n_lf}, params={n_params}") + if target_residual: + print(f" target=residual (log_vol - log_V_arb)") if encoder_type != "mlp": print(f" encoder_type={encoder_type}") if no_peers: @@ -570,6 +645,7 @@ def _subset(d, mask): jax.random.PRNGKey(42), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], encoder_type=encoder_type, ) + params = warm_start_decoder(params, train_inputs, hparams["d_embed"]) t0 = time.time() params, _ = train( params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], @@ -577,25 +653,127 @@ def _subset(d, mask): ) print(f" Training: {time.time() - t0:.1f}s") + eval_kw = dict(no_peers=no_peers, target_residual=target_residual) print("\n --- Train ---") - evaluate(params, train_inputs, data, no_peers=no_peers) + evaluate(params, train_inputs, data, **eval_kw) print("\n --- Eval ---") - _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data, no_peers=no_peers) + _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data, **eval_kw) return med_resid_eval +# ---- LOO cross-validation ---- + + +def run_loo(data, feat_cfg, hparams): + """Leave-one-pool-out: train on N-1 pools, evaluate on held-out pool. + + Tests cross-pool generalization — can the shared encoder+decoder predict + volume for a pool it has never optimized on? The held-out pool's volume + is still observable as peer features for training pools. + + Note: normalization stats are computed on the full dataset. With N=36 + the leakage from including one held-out pool is ~3% on mean/std. + """ + n_pools = data["n_pools"] + pool_idx = data["pool_idx"] + pool_ids = data.get("pool_ids", []) + + encoder_type = hparams.get("encoder_type", "mlp") + no_peers = hparams.get("no_peers", False) + huber_delta = hparams.get("huber_delta", 1.0) + target_residual = feat_cfg.get("target_residual", False) + d_embed = hparams["d_embed"] + + r2_total_all = {} + r2_resid_all = {} + + for held_out in range(n_pools): + pid = pool_ids[held_out] if held_out < len(pool_ids) else f"pool_{held_out}" + + train_mask = pool_idx != held_out + eval_mask = pool_idx == held_out + n_eval = int(eval_mask.sum()) + + if n_eval < 2: + print(f" [{held_out:2d}] {pid[:16]}: skipped ({n_eval} samples)") + continue + + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + train_inputs = assemble_inputs(train_data, feat_cfg) + eval_inputs = assemble_inputs(eval_data, feat_cfg) + + params = init_params( + jax.random.PRNGKey(42), + train_inputs["n_peer_feat"], train_inputs["n_local_feat"], + hparams["hidden"], d_embed, + encoder_type=encoder_type, + ) + params = warm_start_decoder(params, train_inputs, d_embed) + + params, _ = train( + params, train_inputs, hparams["n_epochs"], + hparams["lr"], hparams["l2_alpha"], + huber_delta=huber_delta, no_peers=no_peers, + verbose=False, + ) + + # Evaluate on held-out pool + pred = np.array(forward( + params, eval_inputs["peer_input"], eval_inputs["peer_mask"], + eval_inputs["local_input"], no_peers=no_peers, + )) + y = np.array(eval_inputs["y"]) + v_arb = np.array(eval_inputs["v_arb"]) + log_v_arb = np.log(np.maximum(v_arb, 1e-6)) + + if target_residual: + resid_true, resid_pred = y, pred + y_total = y + log_v_arb + pred_total = pred + log_v_arb + else: + y_total, pred_total = y, pred + resid_true = y - log_v_arb + resid_pred = pred - log_v_arb + + ss_res = np.sum((y_total - pred_total) ** 2) + ss_tot = np.sum((y_total - y_total.mean()) ** 2) + r2_t = 1 - ss_res / max(ss_tot, 1e-10) + + ss_res_r = np.sum((resid_true - resid_pred) ** 2) + ss_tot_r = np.sum((resid_true - resid_true.mean()) ** 2) + r2_r = 1 - ss_res_r / max(ss_tot_r, 1e-10) + + r2_total_all[held_out] = r2_t + r2_resid_all[held_out] = r2_r + + print(f" [{held_out:2d}] {pid[:16]}: total={r2_t:.3f} resid={r2_r:.3f} (n={n_eval})") + + def _med(d): + v = [x for x in d.values() if np.isfinite(x)] + return np.median(v) if v else float("nan") + + med_total = _med(r2_total_all) + med_resid = _med(r2_resid_all) + print(f"\n LOO Median R² total={med_total:.4f} resid={med_resid:.4f}") + print(f" ({len(r2_total_all)} pools evaluated)") + + return med_total, med_resid + + # ---- Optuna ---- _FEAT_KEYS = [ "peer_vol_lag2", "peer_vol_change", "peer_tvl", "peer_volatility", "own_vol_lag2", "own_vol_change", "own_tvl", "own_volatility", - "rel_same_chain", "rel_tvl_ratio", "rel_fee_ratio", + "rel_same_chain", "rel_tvl_ratio", "rel_fee_ratio", "minimal_encoder", ] -def run_optuna(data, n_trials): +def run_optuna(data, n_trials, target_residual=False): import optuna day_idx = data["day_idx"] @@ -603,15 +781,6 @@ def run_optuna(data, n_trials): train_mask = day_idx <= split_day eval_mask = day_idx > split_day - def _subset(d, mask): - out = {} - for k, v in d.items(): - if isinstance(v, np.ndarray) and len(v) == len(mask): - out[k] = v[mask] - else: - out[k] = v - return out - train_data = _subset(data, train_mask) eval_data = _subset(data, eval_mask) @@ -628,6 +797,8 @@ def objective(trial): "rel_same_chain": trial.suggest_categorical("rel_same_chain", [True, False]), "rel_tvl_ratio": trial.suggest_categorical("rel_tvl_ratio", [True, False]), "rel_fee_ratio": trial.suggest_categorical("rel_fee_ratio", [True, False]), + "minimal_encoder": trial.suggest_categorical("minimal_encoder", [True, False]), + "target_residual": target_residual, } hparams = { "hidden": trial.suggest_categorical("hidden", [8, 16, 32]), @@ -649,6 +820,7 @@ def objective(trial): hparams["hidden"], hparams["d_embed"], encoder_type=hparams["encoder_type"], ) + params = warm_start_decoder(params, train_inputs, hparams["d_embed"]) params, _ = train( params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], @@ -658,6 +830,7 @@ def objective(trial): ) # Eval R² on noise residual + _tgt_resid = feat_cfg.get("target_residual", False) pred = np.array(forward( params, eval_inputs["peer_input"], eval_inputs["peer_mask"], eval_inputs["local_input"], no_peers=hparams["no_peers"], @@ -673,16 +846,23 @@ def objective(trial): mask = pool_idx == i if mask.sum() < 2: continue - yt = y[mask] - pt = pred[mask] + yi = y[mask] + pi = pred[mask] + lva = log_v_arb[mask] + + if _tgt_resid: + resid_true, resid_pred = yi, pi + yt, pt = yi + lva, pi + lva + else: + yt, pt = yi, pi + resid_true, resid_pred = yi - lva, pi - lva + ss_res = np.sum((yt - pt) ** 2) ss_tot = np.sum((yt - yt.mean()) ** 2) r2_totals.append(1 - ss_res / max(ss_tot, 1e-10)) - rt = yt - log_v_arb[mask] - rp = pt - log_v_arb[mask] - ss_res_r = np.sum((rt - rp) ** 2) - ss_tot_r = np.sum((rt - rt.mean()) ** 2) + ss_res_r = np.sum((resid_true - resid_pred) ** 2) + ss_tot_r = np.sum((resid_true - resid_true.mean()) ** 2) r2_resids.append(1 - ss_res_r / max(ss_tot_r, 1e-10)) med_resid = float(np.median(r2_resids)) if r2_resids else -10.0 @@ -695,7 +875,8 @@ def objective(trial): f"hub={hparams['huber_delta']} " f"{'no_peers ' if hparams['no_peers'] else ''}" f"lr={hparams['lr']:.1e} a={hparams['l2_alpha']:.1e} " - f"ep={hparams['n_epochs']} feat={n_feat}/11") + f"ep={hparams['n_epochs']} feat={n_feat}/12" + f"{' minimal' if feat_cfg.get('minimal_encoder') else ''}") return med_resid @@ -721,7 +902,7 @@ def objective(trial): f"total={t.user_attrs.get('med_total_r2', '?'):.4f} " f"enc={t.params['encoder_type']} " f"h={t.params['hidden']} d={t.params['d_embed']} " - f"feat={feats}/11") + f"feat={feats}/12") return study @@ -743,6 +924,12 @@ def main(): parser.add_argument("--huber-delta", type=float, default=1.0) parser.add_argument("--no-peers", action="store_true", help="Decoder-only ablation (zero peer summary)") + parser.add_argument("--minimal-encoder", action="store_true", + help="7-feature encoder (fee, tvl, overlap, same_chain) " + "instead of full attributes") + parser.add_argument("--target-residual", action="store_true", + help="Train on noise residual (log_vol - log_V_arb) " + "instead of total log_volume") # Feature flags parser.add_argument("--peer-vol-lag2", action="store_true") parser.add_argument("--peer-vol-change", action="store_true") @@ -771,6 +958,8 @@ def main(): "rel_same_chain": True, "rel_tvl_ratio": True, "rel_fee_ratio": True, + "minimal_encoder": args.minimal_encoder, + "target_residual": args.target_residual, } hparams = { "hidden": args.hidden, @@ -798,8 +987,13 @@ def main(): print(f" {len(data['pool_idx'])} samples, {data['n_pools']} pools, " f"{time.time() - t0:.1f}s") - if args.tune > 0: - run_optuna(data, args.tune) + if args.loo: + print(f"\n{'='*70}") + print("Leave-One-Pool-Out Cross-Validation") + print(f"{'='*70}") + run_loo(data, feat_cfg, hparams) + elif args.tune > 0: + run_optuna(data, args.tune, target_residual=args.target_residual) else: print(f"\n{'='*70}") print("Temporal split (70/30)") From 5c04a47fb19312f935ec7117d07228e74db24098 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 19 Mar 2026 13:52:55 +0000 Subject: [PATCH 063/115] feat: learnable cadence via PCHIP, linear market noise model, hybrid encoder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - DeepSets v2: add --learn-cadence flag that jointly optimizes per-pool arb cadence with the noise model via Adam through differentiable PCHIP. Per-pool PCHIP coefficients closed over in JIT'd loss function. Add --x-obs flag (none/reduced/cross) for Option C covariates. Full decomposition diagnostics (arb%, noise%, cadence bounds, train/eval gap). Optuna cadence sweep (run_optuna_cadence). Best eval total R²=0.50. - New: market_features.py — derives daily features from Binance minute data. BTC regime (log_price, return, realized vol, trends), token-level (return, realized vol, trends), pair realized volatility (A/B ratio). Token map for wrapped/derivative tokens. - New: run_linear_market_noise.py — linear noise model with market features and learnable cadence. V_noise = exp(x @ shared_coeffs). Optional per-pool intercepts. Eval R²=0.39 with 59 params. - New: run_hybrid_noise.py — DeepSets encoder produces scalar peer_effect, fed as covariate (+ tvl interaction) into linear noise model. Encoder learns peer aggregation, linear model keeps TVL coefficient interpretable for counterfactual analysis. Eval R²=0.40 with 478 params. --- experiments/run_deepsets_v2.py | 698 ++++++++++++++++++++- experiments/run_hybrid_noise.py | 584 +++++++++++++++++ experiments/run_linear_market_noise.py | 463 ++++++++++++++ quantammsim/calibration/market_features.py | 297 +++++++++ 4 files changed, 2016 insertions(+), 26 deletions(-) create mode 100644 experiments/run_hybrid_noise.py create mode 100644 experiments/run_linear_market_noise.py create mode 100644 quantammsim/calibration/market_features.py diff --git a/experiments/run_deepsets_v2.py b/experiments/run_deepsets_v2.py index 4e021416..bc3661b7 100644 --- a/experiments/run_deepsets_v2.py +++ b/experiments/run_deepsets_v2.py @@ -59,6 +59,7 @@ def build_all_features(matched_clean, option_c_clean): from quantammsim.calibration.grid_interpolation import interpolate_pool_daily from quantammsim.calibration.pool_data import ( build_pool_attributes, _parse_tokens, _canonicalize_token, + build_x_obs, build_cross_pool_x_obs, ) pool_ids = sorted(matched_clean.keys()) @@ -99,6 +100,22 @@ def build_all_features(matched_clean, option_c_clean): volatility_matrix[t, j] = panel["volatility"].values[k] v_arb_matrix[t, j] = v_arb[k] + # Per-pool coeffs, gas, and day mapping for learnable cadence + pool_coeffs = [] + pool_gas = [] + init_log_cadences = np.zeros(n_pools, dtype=np.float32) + common_to_grid = np.full((n_pools, n_dates), 0, dtype=np.int32) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + pool_coeffs.append(entry["coeffs"]) + pool_gas.append(jnp.float64(np.exp(oc["log_gas"]))) + init_log_cadences[j] = oc["log_cadence"] + dates_j = entry["panel"]["date"].values + for k, date in enumerate(dates_j): + common_to_grid[j, date_to_idx[date]] = entry["day_indices"][k] + for t, date in enumerate(date_list): weekday_arr[t] = pd.Timestamp(date).weekday() @@ -150,6 +167,31 @@ def _standardize(arr): rel_log_tvl_ratio = _standardize(rel_log_tvl_ratio) rel_log_fee_ratio = _standardize(rel_log_fee_ratio) + # Cross-pool peer maps: which pools share tokens / chain + from collections import defaultdict + pool_tokens_ordered = {} + token_to_pools = defaultdict(set) + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + ordered = [_canonicalize_token(t) for t in toks[:2]] + pool_tokens_ordered[i] = ordered + for tok in ordered: + token_to_pools[tok].add(i) + + token_a_peers = {} + token_b_peers = {} + chain_peer_map = {} + for i in range(n_pools): + toks = pool_tokens_ordered[i] + token_a_peers[i] = sorted(token_to_pools[toks[0]] - {i}) + token_b_peers[i] = sorted(token_to_pools[toks[1]] - {i}) if len(toks) > 1 else [] + chain_peer_map[i] = [j for j in range(n_pools) if j != i and pool_chains[j] == pool_chains[i]] + + # Per-pool log_fee for interaction features + pool_log_fee = raw_log_fee.copy() + fee_mean = float(np.mean(pool_log_fee)) + fee_std = max(float(np.std(pool_log_fee)), 1e-6) + # Standardization stats for volumes vol_mean = float(np.nanmean(vol_matrix)) vol_std = float(np.nanstd(vol_matrix)) @@ -158,6 +200,27 @@ def _standardize(arr): vola_mean = float(np.nanmean(volatility_matrix)) vola_std = float(np.nanstd(volatility_matrix)) + # Build x_obs per pool, mapped to common date grid + # x_obs_reduced: (n_dates, n_pools, 4), x_obs_cross: (n_dates, n_pools, 7) + from quantammsim.calibration.pool_data import K_OBS_REDUCED, K_OBS_CROSS + x_obs_reduced_grid = np.full((n_dates, n_pools, K_OBS_REDUCED), np.nan) + x_obs_cross_grid = np.full((n_dates, n_pools, K_OBS_CROSS), np.nan) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + panel = entry["panel"] + dates_j = panel["date"].values + + # Reduced x_obs (4 features) + xr = build_x_obs(panel, reduced=True) # (n_obs, 4) + for k, date in enumerate(dates_j): + x_obs_reduced_grid[date_to_idx[date], j] = xr[k] + + # Cross-pool x_obs (7 features) — drops first day + xc = build_cross_pool_x_obs(panel, matched_clean, pid) # (n_obs-1, 7) + for k, date in enumerate(dates_j[1:]): + x_obs_cross_grid[date_to_idx[date], j] = xc[k] + # Build samples: require t >= 2 (for lag-2), valid vol at t, t-1, t-2 sample_pools, sample_days = [], [] for i in range(n_pools): @@ -198,6 +261,19 @@ def _norm_vola(x): lf_own_volatility = np.zeros(n_samples, dtype=np.float32) lf_dow_sin = np.zeros(n_samples, dtype=np.float32) lf_dow_cos = np.zeros(n_samples, dtype=np.float32) + # Interaction features (from calibration pipeline's x_obs) + lf_tvl_x_vola = np.zeros(n_samples, dtype=np.float32) + lf_tvl_x_fee = np.zeros(n_samples, dtype=np.float32) + lf_vola_x_fee = np.zeros(n_samples, dtype=np.float32) + # Cross-pool volume aggregates + lf_cross_vol_tok_a = np.zeros(n_samples, dtype=np.float32) + lf_cross_vol_tok_b = np.zeros(n_samples, dtype=np.float32) + lf_cross_vol_chain = np.zeros(n_samples, dtype=np.float32) + lf_market_vol = np.zeros(n_samples, dtype=np.float32) + # Cross-pool momentum (peer volume changes) + lf_cross_mom_tok_a = np.zeros(n_samples, dtype=np.float32) + lf_cross_mom_tok_b = np.zeros(n_samples, dtype=np.float32) + lf_cross_mom_chain = np.zeros(n_samples, dtype=np.float32) # Targets y_total = np.zeros(n_samples, dtype=np.float32) @@ -241,10 +317,64 @@ def _norm_vola(x): lf_dow_sin[s] = np.sin(2 * np.pi * wd / 7) lf_dow_cos[s] = np.cos(2 * np.pi * wd / 7) + # Interaction features (raw products, standardized after loop) + norm_fee_i = (raw_log_fee[i] - fee_mean) / fee_std + lf_tvl_x_vola[s] = lf_own_tvl[s] * lf_own_volatility[s] + lf_tvl_x_fee[s] = lf_own_tvl[s] * norm_fee_i + lf_vola_x_fee[s] = lf_own_volatility[s] * norm_fee_i + + # Cross-pool volume aggregates at t-1 + def _peer_vol_mean(peer_list, t_lag): + if not peer_list: + return vol_mean # global fallback + vals = vol_matrix[t_lag, peer_list] + valid = vals[~np.isnan(vals)] + return float(np.mean(valid)) if len(valid) > 0 else vol_mean + + def _peer_vol_change_mean(peer_list, t_lag): + if not peer_list: + return 0.0 + v1 = vol_matrix[t_lag, peer_list] + v2 = vol_matrix[t_lag - 1, peer_list] + valid = ~np.isnan(v1) & ~np.isnan(v2) + if valid.sum() == 0: + return 0.0 + return float(np.mean(v1[valid] - v2[valid])) + + lf_cross_vol_tok_a[s] = _norm_vol(_peer_vol_mean(token_a_peers[i], t - 1)) + lf_cross_vol_tok_b[s] = _norm_vol(_peer_vol_mean(token_b_peers[i], t - 1)) + lf_cross_vol_chain[s] = _norm_vol(_peer_vol_mean(chain_peer_map[i], t - 1)) + lf_market_vol[s] = _norm_vol(float(np.nanmean(vol_matrix[t - 1, :]))) + + # Cross-pool momentum: mean volume change of peers (t-1 vs t-2) + lf_cross_mom_tok_a[s] = _peer_vol_change_mean(token_a_peers[i], t - 1) + lf_cross_mom_tok_b[s] = _peer_vol_change_mean(token_b_peers[i], t - 1) + lf_cross_mom_chain[s] = _peer_vol_change_mean(chain_peer_map[i], t - 1) + y_total[s] = vol_matrix[t, i] v_arb_val = v_arb_matrix[t, i] v_arb_samples[s] = v_arb_val if np.isfinite(v_arb_val) else 1e-6 + # Per-sample grid day indices for learnable cadence + sample_grid_days = common_to_grid[sample_pools, sample_days] + + # Per-sample x_obs arrays + x_obs_reduced = np.zeros((n_samples, K_OBS_REDUCED), dtype=np.float32) + x_obs_cross = np.zeros((n_samples, K_OBS_CROSS), dtype=np.float32) + for s in range(n_samples): + xr = x_obs_reduced_grid[sample_days[s], sample_pools[s]] + if np.all(np.isfinite(xr)): + x_obs_reduced[s] = xr + xc = x_obs_cross_grid[sample_days[s], sample_pools[s]] + if np.all(np.isfinite(xc)): + x_obs_cross[s] = xc + + # Standardize momentum features (raw volume differences) + for arr in [lf_cross_mom_tok_a, lf_cross_mom_tok_b, lf_cross_mom_chain]: + mu = np.mean(arr) + sigma = max(np.std(arr), 1e-6) + arr[:] = ((arr - mu) / sigma).astype(np.float32) + return { # Static per-pool "peer_attrs": peer_attrs, # (n_pools, n_peers, k_attr) @@ -268,10 +398,30 @@ def _norm_vola(x): "lf_own_volatility": lf_own_volatility, "lf_dow_sin": lf_dow_sin, "lf_dow_cos": lf_dow_cos, + # Interaction features + "lf_tvl_x_vola": lf_tvl_x_vola, + "lf_tvl_x_fee": lf_tvl_x_fee, + "lf_vola_x_fee": lf_vola_x_fee, + # Cross-pool volume aggregates + "lf_cross_vol_tok_a": lf_cross_vol_tok_a, + "lf_cross_vol_tok_b": lf_cross_vol_tok_b, + "lf_cross_vol_chain": lf_cross_vol_chain, + "lf_market_vol": lf_market_vol, + # Cross-pool momentum + "lf_cross_mom_tok_a": lf_cross_mom_tok_a, + "lf_cross_mom_tok_b": lf_cross_mom_tok_b, + "lf_cross_mom_chain": lf_cross_mom_chain, # Targets "y_total": y_total, "y_residual": (y_total - np.log(np.maximum(v_arb_samples, 1e-6))).astype(np.float32), "v_arb": v_arb_samples, + # Cadence learning (per-pool, not subject to _subset) + "pool_coeffs": pool_coeffs, # list of PoolCoeffsDaily + "pool_gas": pool_gas, # list of jnp scalars + "init_log_cadences": init_log_cadences, # (n_pools,) + "sample_grid_days": sample_grid_days, # (n_samples,) + "x_obs_reduced": x_obs_reduced, # (n_samples, 4) + "x_obs_cross": x_obs_cross, # (n_samples, 7) # Indices "pool_idx": sample_pools, "day_idx": sample_days, @@ -359,19 +509,50 @@ def assemble_inputs(data, feat_cfg): if feat_cfg.get("own_volatility"): local_parts.append(data["lf_own_volatility"][:, None]) + # Interaction features (tvl×vola, tvl×fee, vola×fee) + if feat_cfg.get("interactions"): + local_parts.append(data["lf_tvl_x_vola"][:, None]) + local_parts.append(data["lf_tvl_x_fee"][:, None]) + local_parts.append(data["lf_vola_x_fee"][:, None]) + + # Cross-pool volume aggregates (token-peer, chain-peer, market) + if feat_cfg.get("cross_pool_vol"): + local_parts.append(data["lf_cross_vol_tok_a"][:, None]) + local_parts.append(data["lf_cross_vol_tok_b"][:, None]) + local_parts.append(data["lf_cross_vol_chain"][:, None]) + local_parts.append(data["lf_market_vol"][:, None]) + + # Cross-pool momentum (peer volume changes) + if feat_cfg.get("cross_pool_momentum"): + local_parts.append(data["lf_cross_mom_tok_a"][:, None]) + local_parts.append(data["lf_cross_mom_tok_b"][:, None]) + local_parts.append(data["lf_cross_mom_chain"][:, None]) + + # Option C x_obs covariates (none / reduced=4 / cross=7) + x_obs_mode = feat_cfg.get("x_obs_mode", "none") + if x_obs_mode == "reduced" and "x_obs_reduced" in data: + local_parts.append(data["x_obs_reduced"]) + elif x_obs_mode == "cross" and "x_obs_cross" in data: + local_parts.append(data["x_obs_cross"]) + local_input = np.concatenate(local_parts, axis=-1).astype(np.float32) - return { + result = { "peer_input": jnp.array(peer_input), "local_input": jnp.array(local_input), "peer_mask": jnp.array(data["peer_mask"]), "y": jnp.array(data["y_residual"] if feat_cfg.get("target_residual") else data["y_total"]), + "y_total": jnp.array(data["y_total"]), "v_arb": jnp.array(data["v_arb"]), "pool_idx": jnp.array(pool_idx), "n_pools": data["n_pools"], "n_peer_feat": peer_input.shape[-1], "n_local_feat": local_input.shape[-1], } + # Cadence learning arrays + if "sample_grid_days" in data: + result["sample_grid_days"] = jnp.array(data["sample_grid_days"]) + return result # ---- Model ---- @@ -483,24 +664,89 @@ def loss_fn(params, peer_input, peer_mask, local_input, y, l2_alpha, _grad_fn = jax.jit(jax.value_and_grad(loss_fn), static_argnums=(7, 9)) +# ---- Learnable cadence ---- + + +def make_cadence_loss_fn(pool_coeffs, pool_gas, n_pools, no_peers): + """Build a loss function with per-pool PCHIP coefficients closed over. + + The returned function is JIT-compiled. The Python loop over pools is + unrolled at trace time, so each pool's coefficients are constants. + + The neural net predicts log(V_noise). V_arb comes from PCHIP at the + current learnable log_cadence. Loss is Huber on log(V_arb + V_noise) + vs log(V_obs). + """ + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + def loss_fn_cadence(params, peer_input, peer_mask, local_input, y_total, + sample_grid_days, pool_idx, l2_alpha, huber_delta): + # Neural net predicts log(V_noise) + log_v_noise = forward(params, peer_input, peer_mask, local_input, no_peers) + + # Compute V_arb per sample via PCHIP (loop unrolled at trace time) + log_cadence = params["log_cadence"] + n_samples = y_total.shape[0] + v_arb = jnp.zeros(n_samples) + + for i in range(n_pools): + v_arb_all = interpolate_pool_daily( + pool_coeffs[i], log_cadence[i], pool_gas[i]) + # Index into this pool's daily V_arb; clip for safety on other pools' samples + safe_days = jnp.clip(sample_grid_days, 0, v_arb_all.shape[0] - 1) + v_arb = jnp.where(pool_idx == i, v_arb_all[safe_days], v_arb) + + # Combine: log(V_arb + V_noise) via numerically stable logaddexp + log_v_arb = jnp.log(jnp.maximum(v_arb, 1e-10)) + log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) + + # Huber loss with per-pool weighting + residuals = log_v_total - y_total + abs_r = jnp.abs(residuals) + huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + pool_counts = jnp.zeros(n_pools).at[pool_idx].add( + jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) + data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active + + reg = sum(jnp.sum(v ** 2) for k, v in params.items() if "W" in k) + return data_loss + l2_alpha * reg + + grad_fn = jax.jit(jax.value_and_grad(loss_fn_cadence)) + return grad_fn + + # ---- Training ---- def train(params, inputs, n_epochs, lr, l2_alpha, huber_delta=1.0, - no_peers=False, verbose=True): + no_peers=False, verbose=True, grad_fn_override=None): m = {k: jnp.zeros_like(v) for k, v in params.items()} v = {k: jnp.zeros_like(v) for k, v in params.items()} final_loss = float("inf") n_pools = int(inputs["n_pools"]) pool_idx = inputs["pool_idx"] + use_cadence = grad_fn_override is not None for epoch in range(n_epochs): - loss_val, grads = _grad_fn( - params, inputs["peer_input"], inputs["peer_mask"], - inputs["local_input"], inputs["y"], l2_alpha, - pool_idx, n_pools, huber_delta, no_peers, - ) + if use_cadence: + loss_val, grads = grad_fn_override( + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], inputs["y_total"], + inputs["sample_grid_days"], pool_idx, l2_alpha, huber_delta, + ) + else: + loss_val, grads = _grad_fn( + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], inputs["y"], l2_alpha, + pool_idx, n_pools, huber_delta, no_peers, + ) final_loss = float(loss_val) for k in params: @@ -511,7 +757,24 @@ def train(params, inputs, n_epochs, lr, l2_alpha, huber_delta=1.0, params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): - print(f" epoch {epoch:4d} loss={final_loss:.6f}") + if use_cadence: + cads = np.exp(np.array(params["log_cadence"])) + # Quick decomposition check: forward pass + V_arb at current cadences + _lvn = np.array(forward( + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], no_peers)) + _vn = np.exp(_lvn) + _vo = np.exp(np.array(inputs["y_total"])) + # Approximate arb fraction (use V_obs - V_noise as proxy to avoid PCHIP call) + _arb_proxy = np.clip(1.0 - _vn / _vo, 0, None) + _n_pathological = np.sum(_arb_proxy < -0.5) # noise > 1.5x observed + _n_bound = np.sum((cads <= 1.01) | (cads >= 59.9)) + print(f" epoch {epoch:4d} loss={final_loss:.6f}" + f" cad=[{cads.min():.1f}-{np.median(cads):.1f}-{cads.max():.1f}]" + f" |logVn|={np.mean(np.abs(_lvn)):.1f}" + f" bound={_n_bound}") + else: + print(f" epoch {epoch:4d} loss={final_loss:.6f}") return params, final_loss @@ -582,12 +845,173 @@ def _med(d): return med_total, med_resid, r2_total, r2_resid +def _compute_cadence_decomposition(params, inputs, data, no_peers=False): + """Compute V_arb, V_noise, and predictions for cadence mode. Returns numpy arrays.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + log_v_noise = np.array(forward( + params, inputs["peer_input"], inputs["peer_mask"], + inputs["local_input"], no_peers=no_peers, + )) + y_total = np.array(inputs["y_total"]) + pool_idx = np.array(inputs["pool_idx"]) + sample_grid_days = np.array(inputs["sample_grid_days"]) + + pool_coeffs = data["pool_coeffs"] + pool_gas = data["pool_gas"] + log_cadence = np.array(params["log_cadence"]) + n_pools = data["n_pools"] + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + pool_coeffs[i], jnp.float64(log_cadence[i]), pool_gas[i])) + v_arb[mask] = v_arb_all[sample_grid_days[mask]] + + v_obs = np.exp(y_total) + v_noise = np.exp(log_v_noise) + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + return { + "v_arb": v_arb, "v_noise": v_noise, "v_obs": v_obs, + "log_v_noise": log_v_noise, "log_v_arb": log_v_arb, + "pred_total": pred_total, "y_total": y_total, + "pool_idx": pool_idx, "log_cadence": log_cadence, + } + + +def evaluate_cadence(params, inputs, data, label="", no_peers=False): + """Evaluate with learned cadence: per-pool R², decomposition diagnostics.""" + dec = _compute_cadence_decomposition(params, inputs, data, no_peers) + pool_ids = data.get("pool_ids", []) + init_cads = data["init_log_cadences"] + n_pools = data["n_pools"] + + if label: + print(f"\n {label}:") + print(f" {'Pool'[:16]:16s} {'R²':>6s} {'Cad init':>8s} {'→learn':>7s}" + f" {'Arb%':>6s} {'Noise%':>7s} {'logVn μ':>7s} {'logVn σ':>7s} {'Flag':>5s}") + print(f" {'-'*80}") + + r2_total = {} + pool_diag = [] + for i in range(n_pools): + mask = dec["pool_idx"] == i + if mask.sum() < 2: + continue + yt = dec["y_total"][mask] + pt = dec["pred_total"][mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2_total[i] = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] if i < len(pool_ids) else f"pool_{i}" + cad_init = np.exp(init_cads[i]) + cad_learned = np.exp(dec["log_cadence"][i]) + + va = dec["v_arb"][mask] + vo = dec["v_obs"][mask] + vn = dec["v_noise"][mask] + lvn = dec["log_v_noise"][mask] + + arb_pct = np.median(va / vo) * 100 + noise_pct = np.median(vn / vo) * 100 + lvn_mu = np.mean(lvn) + lvn_std = np.std(lvn) + + # Flags + flags = [] + if arb_pct > 150: + flags.append("A") # arb dominates + if cad_learned <= 1.01 or cad_learned >= 59.9: + flags.append("B") # cadence at bound + if r2_total[i] < 0: + flags.append("X") # negative R² + + flag_str = "".join(flags) if flags else "" + pool_diag.append({ + "idx": i, "pid": pid, "r2": r2_total[i], + "cad_init": cad_init, "cad_learned": cad_learned, + "arb_pct": arb_pct, "noise_pct": noise_pct, + "lvn_mu": lvn_mu, "lvn_std": lvn_std, "flags": flag_str, + }) + + print(f" {pid[:16]:16s} {r2_total[i]:6.3f} {cad_init:7.1f}m {cad_learned:6.1f}m" + f" {arb_pct:6.0f}% {noise_pct:6.0f}% {lvn_mu:7.1f} {lvn_std:7.2f}" + f" {flag_str:>5s}") + + # ── Summary statistics ── + vals = [x for x in r2_total.values() if np.isfinite(x)] + med_r2 = np.median(vals) if vals else float("nan") + cads = np.exp(dec["log_cadence"]) + + n_pathological = sum(1 for d in pool_diag if d["arb_pct"] > 150) + n_at_bound = sum(1 for d in pool_diag + if d["cad_learned"] <= 1.01 or d["cad_learned"] >= 59.9) + n_negative_r2 = sum(1 for d in pool_diag if d["r2"] < 0) + healthy = [d for d in pool_diag if d["arb_pct"] <= 150 and d["r2"] > 0] + med_r2_healthy = (np.median([d["r2"] for d in healthy]) + if healthy else float("nan")) + + print(f"\n ── Summary ──") + print(f" Median R² total: {med_r2:.4f} (healthy only: {med_r2_healthy:.4f})") + print(f" Cadence range: {cads.min():.1f} - {np.median(cads):.1f}" + f" - {cads.max():.1f} min") + print(f" Decomposition: {len(pool_diag) - n_pathological}/{len(pool_diag)}" + f" healthy (arb≤150%), {n_pathological} pathological") + print(f" Cadence at bounds: {n_at_bound}/{len(pool_diag)}" + f" (≤1min or ≥60min)") + print(f" Negative R²: {n_negative_r2}/{len(pool_diag)}") + print(f" Flags: A=arb>150%, B=cadence at bound, X=negative R²") + + return med_r2, r2_total, pool_diag + + +def print_cadence_comparison(train_diag, eval_diag): + """Print train vs eval diagnostic comparison.""" + train_map = {d["pid"]: d for d in train_diag} + eval_map = {d["pid"]: d for d in eval_diag} + all_pids = sorted(set(train_map) | set(eval_map)) + + print(f"\n ── Train vs Eval Gap ──") + print(f" {'Pool'[:16]:16s} {'R² trn':>7s} {'R² eval':>7s} {'Gap':>6s}" + f" {'ArbTrn%':>7s} {'ArbEval%':>8s}") + print(f" {'-'*55}") + + gaps = [] + for pid in all_pids: + td = train_map.get(pid) + ed = eval_map.get(pid) + if td is None or ed is None: + continue + gap = td["r2"] - ed["r2"] + gaps.append(gap) + flag = " ***" if abs(gap) > 0.5 else "" + print(f" {pid[:16]:16s} {td['r2']:7.3f} {ed['r2']:7.3f} {gap:+6.3f}" + f" {td['arb_pct']:6.0f}% {ed['arb_pct']:7.0f}%{flag}") + + if gaps: + print(f" Median gap: {np.median(gaps):+.3f} " + f"Mean gap: {np.mean(gaps):+.3f} " + f"Max gap: {max(gaps):+.3f}") + + # Keys indexed by sample (shape[0] == n_samples) _SAMPLE_KEYS = { "pf_vol_lag1", "pf_vol_lag2", "pf_vol_change", "pf_tvl", "pf_volatility", "peer_mask", "lf_own_vol_lag1", "lf_own_vol_lag2", "lf_own_vol_change", "lf_own_tvl", "lf_own_volatility", "lf_dow_sin", "lf_dow_cos", - "y_total", "y_residual", "v_arb", "pool_idx", "day_idx", + "lf_tvl_x_vola", "lf_tvl_x_fee", "lf_vola_x_fee", + "lf_cross_vol_tok_a", "lf_cross_vol_tok_b", "lf_cross_vol_chain", + "lf_market_vol", "lf_cross_mom_tok_a", "lf_cross_mom_tok_b", + "lf_cross_mom_chain", + "y_total", "y_residual", "v_arb", "sample_grid_days", + "x_obs_reduced", "x_obs_cross", + "pool_idx", "day_idx", } @@ -622,6 +1046,7 @@ def run_temporal(data, feat_cfg, hparams, split_frac=0.7): no_peers = hparams.get("no_peers", False) huber_delta = hparams.get("huber_delta", 1.0) target_residual = feat_cfg.get("target_residual", False) + learn_cadence = hparams.get("learn_cadence", False) n_pf = train_inputs["n_peer_feat"] n_lf = train_inputs["n_local_feat"] @@ -632,6 +1057,8 @@ def run_temporal(data, feat_cfg, hparams, split_frac=0.7): print(f" Train: {int(train_mask.sum())}, Eval: {int(eval_mask.sum())}, " f"peer_feat={n_pf}, local_feat={n_lf}, params={n_params}") + if learn_cadence: + print(f" learn_cadence=True (joint cadence+noise optimization)") if target_residual: print(f" target=residual (log_vol - log_V_arb)") if encoder_type != "mlp": @@ -645,21 +1072,57 @@ def run_temporal(data, feat_cfg, hparams, split_frac=0.7): jax.random.PRNGKey(42), n_pf, n_lf, hparams["hidden"], hparams["d_embed"], encoder_type=encoder_type, ) - params = warm_start_decoder(params, train_inputs, hparams["d_embed"]) - t0 = time.time() - params, _ = train( - params, train_inputs, hparams["n_epochs"], hparams["lr"], hparams["l2_alpha"], - huber_delta=huber_delta, no_peers=no_peers, - ) - print(f" Training: {time.time() - t0:.1f}s") - eval_kw = dict(no_peers=no_peers, target_residual=target_residual) - print("\n --- Train ---") - evaluate(params, train_inputs, data, **eval_kw) - print("\n --- Eval ---") - _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data, **eval_kw) + if learn_cadence: + # Add learnable cadence, initialized from Option C + params["log_cadence"] = jnp.array(data["init_log_cadences"]) + init_cads = np.exp(data["init_log_cadences"]) + print(f" Init cadence: {init_cads.min():.1f}-{np.median(init_cads):.1f}" + f"-{init_cads.max():.1f} min") + + # Warm-start decoder to predict noise residual (log_vol - log_V_arb) + # using the Option C V_arb as the initial target + ws_inputs = dict(train_inputs) + ws_inputs["y"] = train_inputs["y_total"] - jnp.log( + jnp.maximum(train_inputs["v_arb"], 1e-6)) + params = warm_start_decoder(params, ws_inputs, hparams["d_embed"]) + + grad_fn = make_cadence_loss_fn( + data["pool_coeffs"], data["pool_gas"], + data["n_pools"], no_peers) + + print(" Compiling cadence loss (may take a moment)...") + t0 = time.time() + params, _ = train( + params, train_inputs, hparams["n_epochs"], hparams["lr"], + hparams["l2_alpha"], huber_delta=huber_delta, no_peers=no_peers, + grad_fn_override=grad_fn, + ) + print(f" Training: {time.time() - t0:.1f}s") + + print("\n --- Train ---") + _, _, train_diag = evaluate_cadence( + params, train_inputs, data, no_peers=no_peers) + print("\n --- Eval ---") + _, _, eval_diag = evaluate_cadence( + params, eval_inputs, data, no_peers=no_peers) + print_cadence_comparison(train_diag, eval_diag) + else: + params = warm_start_decoder(params, train_inputs, hparams["d_embed"]) + t0 = time.time() + params, _ = train( + params, train_inputs, hparams["n_epochs"], hparams["lr"], + hparams["l2_alpha"], huber_delta=huber_delta, no_peers=no_peers, + ) + print(f" Training: {time.time() - t0:.1f}s") + + eval_kw = dict(no_peers=no_peers, target_residual=target_residual) + print("\n --- Train ---") + evaluate(params, train_inputs, data, **eval_kw) + print("\n --- Eval ---") + _, med_resid_eval, _, _ = evaluate(params, eval_inputs, data, **eval_kw) - return med_resid_eval + return params # ---- LOO cross-validation ---- @@ -770,6 +1233,7 @@ def _med(d): "peer_vol_lag2", "peer_vol_change", "peer_tvl", "peer_volatility", "own_vol_lag2", "own_vol_change", "own_tvl", "own_volatility", "rel_same_chain", "rel_tvl_ratio", "rel_fee_ratio", "minimal_encoder", + "interactions", "cross_pool_vol", "cross_pool_momentum", ] @@ -798,11 +1262,14 @@ def objective(trial): "rel_tvl_ratio": trial.suggest_categorical("rel_tvl_ratio", [True, False]), "rel_fee_ratio": trial.suggest_categorical("rel_fee_ratio", [True, False]), "minimal_encoder": trial.suggest_categorical("minimal_encoder", [True, False]), + "interactions": trial.suggest_categorical("interactions", [True, False]), + "cross_pool_vol": trial.suggest_categorical("cross_pool_vol", [True, False]), + "cross_pool_momentum": trial.suggest_categorical("cross_pool_momentum", [True, False]), "target_residual": target_residual, } hparams = { - "hidden": trial.suggest_categorical("hidden", [8, 16, 32]), - "d_embed": trial.suggest_categorical("d_embed", [4, 8, 16]), + "hidden": trial.suggest_categorical("hidden", [16, 32, 64, 128]), + "d_embed": trial.suggest_categorical("d_embed", [4, 8, 16, 32]), "lr": trial.suggest_float("lr", 1e-4, 1e-2, log=True), "l2_alpha": trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True), "n_epochs": trial.suggest_categorical("n_epochs", [500, 1000, 2000]), @@ -875,7 +1342,7 @@ def objective(trial): f"hub={hparams['huber_delta']} " f"{'no_peers ' if hparams['no_peers'] else ''}" f"lr={hparams['lr']:.1e} a={hparams['l2_alpha']:.1e} " - f"ep={hparams['n_epochs']} feat={n_feat}/12" + f"ep={hparams['n_epochs']} feat={n_feat}/15" f"{' minimal' if feat_cfg.get('minimal_encoder') else ''}") return med_resid @@ -902,7 +1369,166 @@ def objective(trial): f"total={t.user_attrs.get('med_total_r2', '?'):.4f} " f"enc={t.params['encoder_type']} " f"h={t.params['hidden']} d={t.params['d_embed']} " - f"feat={feats}/12") + f"feat={feats}/15") + + return study + + +def run_optuna_cadence(data, n_trials): + """Optuna sweep with learnable cadence. Optimizes median eval total R².""" + import optuna + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + pool_coeffs = data["pool_coeffs"] + pool_gas = data["pool_gas"] + n_pools = data["n_pools"] + + # Pre-build grad_fn closures for (no_peers=True, no_peers=False) + # to avoid recompiling on every trial with the same no_peers setting + _grad_fn_cache = {} + + def _get_grad_fn(no_peers): + if no_peers not in _grad_fn_cache: + _grad_fn_cache[no_peers] = make_cadence_loss_fn( + pool_coeffs, pool_gas, n_pools, no_peers) + return _grad_fn_cache[no_peers] + + def objective(trial): + feat_cfg = { + "peer_vol_lag2": trial.suggest_categorical("peer_vol_lag2", [True, False]), + "peer_vol_change": trial.suggest_categorical("peer_vol_change", [True, False]), + "peer_tvl": trial.suggest_categorical("peer_tvl", [True, False]), + "peer_volatility": trial.suggest_categorical("peer_volatility", [True, False]), + "own_vol_lag2": trial.suggest_categorical("own_vol_lag2", [True, False]), + "own_vol_change": trial.suggest_categorical("own_vol_change", [True, False]), + "own_tvl": trial.suggest_categorical("own_tvl", [True, False]), + "own_volatility": trial.suggest_categorical("own_volatility", [True, False]), + "rel_same_chain": trial.suggest_categorical("rel_same_chain", [True, False]), + "rel_tvl_ratio": trial.suggest_categorical("rel_tvl_ratio", [True, False]), + "rel_fee_ratio": trial.suggest_categorical("rel_fee_ratio", [True, False]), + "minimal_encoder": trial.suggest_categorical("minimal_encoder", [True, False]), + "interactions": trial.suggest_categorical("interactions", [True, False]), + "cross_pool_vol": trial.suggest_categorical("cross_pool_vol", [True, False]), + "cross_pool_momentum": trial.suggest_categorical("cross_pool_momentum", [True, False]), + "x_obs_mode": trial.suggest_categorical("x_obs_mode", ["none", "reduced", "cross"]), + } + hparams = { + "hidden": trial.suggest_categorical("hidden", [16, 32, 64]), + "d_embed": trial.suggest_categorical("d_embed", [4, 8, 16]), + "lr": trial.suggest_float("lr", 3e-4, 3e-3, log=True), + "l2_alpha": trial.suggest_float("l2_alpha", 1e-5, 1e-2, log=True), + "n_epochs": trial.suggest_categorical("n_epochs", [500, 1000, 2000]), + "encoder_type": trial.suggest_categorical("encoder_type", ["mlp", "linear"]), + "huber_delta": trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5]), + "no_peers": trial.suggest_categorical("no_peers", [True, False]), + } + + no_peers = hparams["no_peers"] + train_inputs = assemble_inputs(train_data, feat_cfg) + eval_inputs = assemble_inputs(eval_data, feat_cfg) + + params = init_params( + jax.random.PRNGKey(42), + train_inputs["n_peer_feat"], train_inputs["n_local_feat"], + hparams["hidden"], hparams["d_embed"], + encoder_type=hparams["encoder_type"], + ) + # Learnable cadence from Option C init + params["log_cadence"] = jnp.array(data["init_log_cadences"]) + + # Warm-start decoder on noise residual + ws_inputs = dict(train_inputs) + ws_inputs["y"] = train_inputs["y_total"] - jnp.log( + jnp.maximum(train_inputs["v_arb"], 1e-6)) + params = warm_start_decoder(params, ws_inputs, hparams["d_embed"]) + + grad_fn = _get_grad_fn(no_peers) + params, _ = train( + params, train_inputs, hparams["n_epochs"], + hparams["lr"], hparams["l2_alpha"], + huber_delta=hparams["huber_delta"], no_peers=no_peers, + verbose=False, grad_fn_override=grad_fn, + ) + + # Eval: compute V_arb at learned cadences, combine with net + log_v_noise = np.array(forward( + params, eval_inputs["peer_input"], eval_inputs["peer_mask"], + eval_inputs["local_input"], no_peers=no_peers, + )) + y_total = np.array(eval_inputs["y_total"]) + pool_idx = np.array(eval_data["pool_idx"]) + sample_grid_days = np.array(eval_inputs["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + pool_coeffs[i], jnp.float64(log_cadence[i]), pool_gas[i])) + v_arb[mask] = v_arb_all[sample_grid_days[mask]] + + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + r2_totals = [] + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2_totals.append(1 - ss_res / max(ss_tot, 1e-10)) + + med_total = float(np.median(r2_totals)) if r2_totals else -10.0 + + cads = np.exp(log_cadence) + trial.set_user_attr("med_total_r2", med_total) + trial.set_user_attr("cad_median", float(np.median(cads))) + n_feat = sum(1 for k in _FEAT_KEYS if feat_cfg.get(k)) + print(f" Trial {trial.number}: total={med_total:.4f} " + f"enc={hparams['encoder_type']} h={hparams['hidden']} d={hparams['d_embed']} " + f"hub={hparams['huber_delta']} " + f"{'no_peers ' if no_peers else ''}" + f"lr={hparams['lr']:.1e} a={hparams['l2_alpha']:.1e} " + f"ep={hparams['n_epochs']} feat={n_feat}/15" + f" cad=[{cads.min():.0f}-{np.median(cads):.0f}-{cads.max():.0f}]") + + return med_total + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print("Optuna Results (learn_cadence)") + print(f"{'='*70}") + print(f" Best eval total R²: {study.best_value:.4f}") + print(f" Best params:") + for k, v in sorted(study.best_params.items()): + print(f" {k}: {v}") + + print(f"\n Top 10:") + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + for t in trials[:10]: + if t.value is not None: + feats = sum(1 for k in _FEAT_KEYS if t.params.get(k)) + cad_med = t.user_attrs.get("cad_median", "?") + print(f" #{t.number}: total={t.value:.4f} " + f"enc={t.params['encoder_type']} " + f"h={t.params['hidden']} d={t.params['d_embed']} " + f"feat={feats}/15 cad_med={cad_med:.0f}") return study @@ -924,6 +1550,13 @@ def main(): parser.add_argument("--huber-delta", type=float, default=1.0) parser.add_argument("--no-peers", action="store_true", help="Decoder-only ablation (zero peer summary)") + parser.add_argument("--learn-cadence", action="store_true", + help="Jointly optimize per-pool arb cadence via PCHIP") + parser.add_argument("--x-obs", choices=["none", "reduced", "cross"], + default="none", + help="Append Option C x_obs covariates to decoder: " + "none, reduced (4: intercept,tvl,dow), " + "cross (7: +peer volumes)") parser.add_argument("--minimal-encoder", action="store_true", help="7-feature encoder (fee, tvl, overlap, same_chain) " "instead of full attributes") @@ -939,6 +1572,12 @@ def main(): parser.add_argument("--own-vol-change", action="store_true") parser.add_argument("--own-tvl", action="store_true") parser.add_argument("--own-volatility", action="store_true") + parser.add_argument("--interactions", action="store_true", + help="tvl×vola, tvl×fee, vola×fee interaction terms") + parser.add_argument("--cross-pool-vol", action="store_true", + help="Token-peer, chain-peer, market volume aggregates") + parser.add_argument("--cross-pool-momentum", action="store_true", + help="Peer volume change momentum features") parser.add_argument("--all-features", action="store_true", help="Enable all optional features") args = parser.parse_args() @@ -958,8 +1597,12 @@ def main(): "rel_same_chain": True, "rel_tvl_ratio": True, "rel_fee_ratio": True, + "interactions": args.interactions or args.all_features, + "cross_pool_vol": args.cross_pool_vol or args.all_features, + "cross_pool_momentum": args.cross_pool_momentum or args.all_features, "minimal_encoder": args.minimal_encoder, "target_residual": args.target_residual, + "x_obs_mode": args.x_obs, } hparams = { "hidden": args.hidden, @@ -970,6 +1613,7 @@ def main(): "encoder_type": args.encoder_type, "huber_delta": args.huber_delta, "no_peers": args.no_peers, + "learn_cadence": args.learn_cadence, } print("=" * 70) @@ -992,6 +1636,8 @@ def main(): print("Leave-One-Pool-Out Cross-Validation") print(f"{'='*70}") run_loo(data, feat_cfg, hparams) + elif args.tune > 0 and args.learn_cadence: + run_optuna_cadence(data, args.tune) elif args.tune > 0: run_optuna(data, args.tune, target_residual=args.target_residual) else: diff --git a/experiments/run_hybrid_noise.py b/experiments/run_hybrid_noise.py new file mode 100644 index 00000000..be26b138 --- /dev/null +++ b/experiments/run_hybrid_noise.py @@ -0,0 +1,584 @@ +"""Hybrid noise model: DeepSets peer encoder + linear noise model. + +Architecture: + peer_effect = DeepSets_encoder(peer_data, current_pool_attrs) → scalar + log(V_noise) = [x_obs, market, peer_effect, peer_effect×tvl, ...] @ coeffs + V_total = V_arb(cadence) + exp(log_v_noise) + +The encoder learns how to aggregate peer information. The linear model +learns how that aggregate (plus market/pool features) drives noise volume. +Cadence is learnable per-pool via PCHIP. + +Usage: + python experiments/run_hybrid_noise.py + python experiments/run_hybrid_noise.py --encoder-hidden 16 --epochs 2000 + python experiments/run_hybrid_noise.py --n-peer-outputs 3 # multi-dim peer effect +""" + +import argparse +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30)): + """Build all features: x_obs, market, peer encoder inputs.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + build_x_obs, build_cross_pool_x_obs, build_pool_attributes, + _parse_tokens, _canonicalize_token, K_OBS_CROSS, + ) + from quantammsim.calibration.market_features import ( + build_pool_market_features, pool_market_features_to_matrix, + ) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # Common date grid + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + # Volume matrix and per-pool metadata + vol_matrix = np.full((n_dates, n_pools), np.nan) + pool_coeffs = [] + pool_gas = [] + init_log_cadences = np.zeros(n_pools, dtype=np.float32) + common_to_grid = np.full((n_pools, n_dates), 0, dtype=np.int32) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + pool_coeffs.append(entry["coeffs"]) + pool_gas.append(jnp.float64(np.exp(oc["log_gas"]))) + init_log_cadences[j] = oc["log_cadence"] + dates = entry["panel"]["date"].values + log_vols = entry["panel"]["log_volume"].values.astype(float) + for k, date in enumerate(dates): + t = date_to_idx[date] + vol_matrix[t, j] = log_vols[k] + common_to_grid[j, t] = entry["day_indices"][k] + + # Pool attributes (static, normalized) + X_attr, attr_names, _ = build_pool_attributes(matched_clean) + attr_mean = np.mean(X_attr, axis=0) + attr_std = np.std(X_attr, axis=0) + attr_std[attr_std < 1e-6] = 1.0 + X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) + k_attr = X_attr_norm.shape[1] + + # Token overlap matrix + pool_tokens = {} + for i, pid in enumerate(pool_ids): + toks = _parse_tokens(matched_clean[pid]["tokens"]) + pool_tokens[i] = {_canonicalize_token(t) for t in toks[:2]} + + overlap = np.zeros((n_pools, n_pools), dtype=np.float32) + for i in range(n_pools): + for j in range(n_pools): + if i != j: + overlap[i, j] = len(pool_tokens[i] & pool_tokens[j]) + + # Peer index mapping: for pool i, peers are all j != i + n_peers = n_pools - 1 + peer_idx = np.zeros((n_pools, n_peers), dtype=np.int32) + peer_overlap = np.zeros((n_pools, n_peers), dtype=np.float32) + for i in range(n_pools): + peers = [j for j in range(n_pools) if j != i] + peer_idx[i] = peers + peer_overlap[i] = overlap[i, peers] + + # Build samples + sample_pools, sample_days = [], [] + for i in range(n_pools): + for t in range(1, n_dates): + if np.isnan(vol_matrix[t, i]) or np.isnan(vol_matrix[t - 1, i]): + continue + sample_pools.append(i) + sample_days.append(t) + sample_pools = np.array(sample_pools, dtype=np.int32) + sample_days = np.array(sample_days, dtype=np.int32) + n_samples = len(sample_pools) + + # ---- x_obs (cross-pool, 7 features) ---- + x_obs_grid = np.full((n_dates, n_pools, K_OBS_CROSS), np.nan) + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + xc = build_cross_pool_x_obs(panel, matched_clean, pid) + dates_j = panel["date"].values + for k, date in enumerate(dates_j[1:]): + x_obs_grid[date_to_idx[date], j] = xc[k] + + x_obs = np.zeros((n_samples, K_OBS_CROSS), dtype=np.float32) + for s in range(n_samples): + xval = x_obs_grid[sample_days[s], sample_pools[s]] + if np.all(np.isfinite(xval)): + x_obs[s] = xval + + # ---- Market features ---- + print(" Building market features...") + pool_feat = build_pool_market_features( + matched_clean, trend_windows=list(trend_windows)) + x_market, market_names = pool_market_features_to_matrix( + pool_feat, matched_clean, date_to_idx, pool_ids, + sample_pools, sample_days) + print(f" Market features: {len(market_names)} columns") + + # ---- Peer encoder inputs: (n_samples, n_peers, n_peer_feat) ---- + # Per peer: [peer_attrs, target_attrs, peer_vol_lag1, overlap] + # peer_vol_lag1 is the peer's volume at t-1 + vol_mean = float(np.nanmean(vol_matrix)) + vol_std = max(float(np.nanstd(vol_matrix)), 1e-6) + + # Static peer features (per pool) + peer_attrs = np.zeros((n_pools, n_peers, k_attr), dtype=np.float32) + for i in range(n_pools): + peer_attrs[i] = X_attr_norm[peer_idx[i]] + + # Per-sample peer features + peer_vol_lag1 = np.zeros((n_samples, n_peers), dtype=np.float32) + peer_mask = np.zeros((n_samples, n_peers), dtype=np.float32) + + for s in range(n_samples): + i = sample_pools[s] + t = sample_days[s] + cols = peer_idx[i] + pvols = vol_matrix[t - 1, cols] + valid = ~np.isnan(pvols) + peer_mask[s] = valid.astype(np.float32) + peer_vol_lag1[s] = np.where(valid, (pvols - vol_mean) / vol_std, 0.0) + + # Assemble peer encoder input: (n_samples, n_peers, n_peer_feat) + # [peer_attrs(k_attr), target_attrs(k_attr), vol_lag1(1), overlap(1)] + target_attrs_broad = np.broadcast_to( + X_attr_norm[sample_pools][:, None, :], + (n_samples, n_peers, k_attr)) + peer_input = np.concatenate([ + peer_attrs[sample_pools], # (n_samples, n_peers, k_attr) + target_attrs_broad, # (n_samples, n_peers, k_attr) + peer_vol_lag1[:, :, None], # (n_samples, n_peers, 1) + peer_overlap[sample_pools][:, :, None], # (n_samples, n_peers, 1) + ], axis=-1).astype(np.float32) + n_peer_feat = peer_input.shape[-1] + + # Combine linear features (x_obs + market) + x_linear = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) + linear_names = [f"xobs_{i}" for i in range(K_OBS_CROSS)] + market_names + + # Standardize linear features (except intercept) + x_mean = np.mean(x_linear, axis=0) + x_std_arr = np.std(x_linear, axis=0) + x_std_arr[x_std_arr < 1e-6] = 1.0 + x_mean[0] = 0.0 + x_std_arr[0] = 1.0 + x_linear = ((x_linear - x_mean) / x_std_arr).astype(np.float32) + + # Targets and indices + y_total = np.array([vol_matrix[sample_days[s], sample_pools[s]] + for s in range(n_samples)], dtype=np.float32) + sample_grid_days = common_to_grid[sample_pools, sample_days] + + return { + "x_linear": x_linear, # (n_samples, n_linear_feat) + "peer_input": peer_input, # (n_samples, n_peers, n_peer_feat) + "peer_mask": peer_mask, # (n_samples, n_peers) + "y_total": y_total, + "pool_idx": sample_pools, + "day_idx": sample_days, + "sample_grid_days": sample_grid_days, + "pool_coeffs": pool_coeffs, + "pool_gas": pool_gas, + "init_log_cadences": init_log_cadences, + "n_pools": n_pools, + "n_peers": n_peers, + "n_linear_feat": x_linear.shape[1], + "n_peer_feat": n_peer_feat, + "pool_ids": pool_ids, + "linear_names": linear_names, + } + + +# ---- Model ---- + +_SAMPLE_KEYS = { + "x_linear", "peer_input", "peer_mask", "y_total", + "pool_idx", "day_idx", "sample_grid_days", +} + + +def _subset(d, mask): + out = {} + for k, v in d.items(): + if k in _SAMPLE_KEYS and isinstance(v, np.ndarray): + out[k] = v[mask] + else: + out[k] = v + return out + + +def init_params(key, n_peer_feat, n_linear_feat, encoder_hidden, + n_peer_outputs, n_pools, init_log_cadences): + """Initialize all parameters. + + Encoder: peer_input → hidden → n_peer_outputs (per peer, then mean-pooled) + Linear: [x_linear, peer_outputs, peer_outputs × x_linear[1](tvl)] @ coeffs + """ + k1, k2 = jax.random.split(key) + + # Encoder: single hidden layer → n_peer_outputs + n_total_linear = n_linear_feat + n_peer_outputs + n_peer_outputs # +interactions with tvl + + params = { + "enc_W1": jax.random.normal(k1, (n_peer_feat, encoder_hidden)) * np.sqrt(2.0 / n_peer_feat), + "enc_b1": jnp.zeros(encoder_hidden), + "enc_W2": jax.random.normal(k2, (encoder_hidden, n_peer_outputs)) * 0.01, + "enc_b2": jnp.zeros(n_peer_outputs), + "noise_coeffs": jnp.zeros(n_total_linear), + "log_cadence": jnp.array(init_log_cadences), + } + return params, n_total_linear + + +def forward_encoder(params, peer_input, peer_mask): + """DeepSets encoder: per-peer MLP → masked mean → scalar(s). + + Returns (n_samples, n_peer_outputs). + """ + batch, n_peers, _ = peer_input.shape + flat = peer_input.reshape(-1, peer_input.shape[-1]) + + h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) + h = h @ params["enc_W2"] + params["enc_b2"] + h = h.reshape(batch, n_peers, -1) + + h_masked = h * peer_mask[:, :, None] + n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) + return jnp.sum(h_masked, axis=1) / n_valid # (batch, n_peer_outputs) + + +def make_loss_fn(pool_coeffs, pool_gas, n_pools): + """Loss with learnable cadence + encoder + linear noise model.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + def loss_fn(params, x_linear, peer_input, peer_mask, y_total, + sample_grid_days, pool_idx, l2_alpha, huber_delta): + + # Encoder → peer effect scalar(s) + peer_effect = forward_encoder(params, peer_input, peer_mask) + + # Build full linear input: [x_linear, peer_effect, peer_effect × tvl] + # tvl is x_linear[:, 1] (xobs_1 = log_tvl_lag1, standardized) + tvl = x_linear[:, 1:2] # keep 2D + peer_x_tvl = peer_effect * tvl # interaction + + x_full = jnp.concatenate([x_linear, peer_effect, peer_x_tvl], axis=1) + log_v_noise = x_full @ params["noise_coeffs"] + + # V_arb from PCHIP + log_cadence = params["log_cadence"] + n_samples = y_total.shape[0] + v_arb = jnp.zeros(n_samples) + for i in range(n_pools): + v_arb_all = interpolate_pool_daily( + pool_coeffs[i], log_cadence[i], pool_gas[i]) + safe_days = jnp.clip(sample_grid_days, 0, v_arb_all.shape[0] - 1) + v_arb = jnp.where(pool_idx == i, v_arb_all[safe_days], v_arb) + + log_v_arb = jnp.log(jnp.maximum(v_arb, 1e-10)) + log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) + + # Huber loss with per-pool weighting + residuals = log_v_total - y_total + abs_r = jnp.abs(residuals) + huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + pool_counts = jnp.zeros(n_pools).at[pool_idx].add( + jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) + data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active + + # L2 on encoder weights + noise coeffs + reg = l2_alpha * ( + jnp.sum(params["enc_W1"] ** 2) + + jnp.sum(params["enc_W2"] ** 2) + + jnp.sum(params["noise_coeffs"] ** 2) + ) + return data_loss + reg + + return jax.jit(jax.value_and_grad(loss_fn)) + + +def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, + verbose=True): + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + + xl = jnp.array(data["x_linear"]) + pi = jnp.array(data["peer_input"]) + pm = jnp.array(data["peer_mask"]) + yt = jnp.array(data["y_total"]) + sgd = jnp.array(data["sample_grid_days"]) + pidx = jnp.array(data["pool_idx"]) + + for epoch in range(n_epochs): + loss_val, grads = grad_fn( + params, xl, pi, pm, yt, sgd, pidx, l2_alpha, huber_delta) + loss_f = float(loss_val) + + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): + cads = np.exp(np.array(params["log_cadence"])) + pe = np.array(forward_encoder( + params, jnp.array(data["peer_input"][:100]), + jnp.array(data["peer_mask"][:100]))) + print(f" epoch {epoch:4d} loss={loss_f:.6f}" + f" cad=[{cads.min():.1f}-{np.median(cads):.1f}-{cads.max():.1f}]" + f" peer_eff=[{pe.min():.2f},{pe.mean():.2f},{pe.max():.2f}]") + + return params + + +def evaluate(params, data, label=""): + """Evaluate decomposition.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + x_linear = np.array(data["x_linear"]) + peer_input = data["peer_input"] + peer_mask = data["peer_mask"] + y_total = np.array(data["y_total"]) + pool_idx = np.array(data["pool_idx"]) + sgd = np.array(data["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + init_cads = data["init_log_cadences"] + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + + # Encoder + peer_effect = np.array(forward_encoder( + params, jnp.array(peer_input), jnp.array(peer_mask))) + + # Build full linear input + tvl = x_linear[:, 1:2] + peer_x_tvl = peer_effect * tvl + x_full = np.concatenate([x_linear, peer_effect, peer_x_tvl], axis=1) + + noise_coeffs = np.array(params["noise_coeffs"]) + log_v_noise = x_full @ noise_coeffs + v_noise = np.exp(log_v_noise) + + # V_arb + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd[mask]] + + v_obs = np.exp(y_total) + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + if label: + print(f"\n {label}:") + print(f" {'Pool'[:16]:16s} {'R²':>6s} {'Cad':>5s} → {'learn':>5s}" + f" {'Arb%':>6s} {'Noise%':>7s} {'PeerEff':>8s} {'Flag':>5s}") + print(f" {'-'*65}") + + r2s = {} + pool_diag = [] + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s[i] = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] + ci = np.exp(init_cads[i]) + cl = np.exp(log_cadence[i]) + arb_pct = np.median(v_arb[mask] / v_obs[mask]) * 100 + noise_pct = np.median(v_noise[mask] / v_obs[mask]) * 100 + pe_mean = np.mean(peer_effect[mask]) + + flags = [] + if arb_pct > 150: + flags.append("A") + if cl <= 1.01 or cl >= 59.9: + flags.append("B") + if r2s[i] < 0: + flags.append("X") + flag_str = "".join(flags) + + pool_diag.append({ + "pid": pid, "r2": r2s[i], "cad_init": ci, "cad_learned": cl, + "arb_pct": arb_pct, "noise_pct": noise_pct, + "peer_effect": pe_mean, "flags": flag_str, + }) + + print(f" {pid[:16]:16s} {r2s[i]:6.3f} {ci:5.1f} → {cl:5.1f}" + f" {arb_pct:6.0f}% {noise_pct:6.0f}% {pe_mean:+8.3f} {flag_str:>5s}") + + vals = [x for x in r2s.values() if np.isfinite(x)] + med = np.median(vals) if vals else float("nan") + healthy = [d for d in pool_diag if d["arb_pct"] <= 150 and d["r2"] > 0] + med_h = np.median([d["r2"] for d in healthy]) if healthy else float("nan") + n_path = sum(1 for d in pool_diag if d["arb_pct"] > 150) + n_bound = sum(1 for d in pool_diag + if d["cad_learned"] <= 1.01 or d["cad_learned"] >= 59.9) + + # Print coefficient analysis + nc = np.array(params["noise_coeffs"]) + n_linear = data["n_linear_feat"] + n_po = len(nc) - n_linear + n_each = n_po // 2 + + print(f"\n Median R²: {med:.4f} (healthy: {med_h:.4f})") + print(f" Healthy: {len(pool_diag) - n_path}/{len(pool_diag)}," + f" at bounds: {n_bound}") + + print(f"\n Linear coefficients:") + for j, name in enumerate(data["linear_names"]): + print(f" {name:30s} {nc[j]:+8.4f}") + for j in range(n_each): + print(f" {'peer_effect_' + str(j):30s} {nc[n_linear + j]:+8.4f}") + for j in range(n_each): + print(f" {'peer_eff_' + str(j) + '×tvl':30s} {nc[n_linear + n_each + j]:+8.4f}") + + return med, r2s, pool_diag + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--epochs", type=int, default=2000) + parser.add_argument("--lr", type=float, default=1e-3) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--encoder-hidden", type=int, default=16) + parser.add_argument("--n-peer-outputs", type=int, default=1) + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7, 14, 30]) + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Hybrid: DeepSets Peer Encoder + Linear Noise Model") + print(f" encoder_hidden={args.encoder_hidden}," + f" n_peer_outputs={args.n_peer_outputs}") + print(f" epochs={args.epochs}, lr={args.lr}, l2={args.l2_alpha}") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding data...") + t0 = time.time() + data = build_data(matched_clean, option_c_clean, + trend_windows=tuple(args.trend_windows)) + n_pools = data["n_pools"] + print(f" {len(data['pool_idx'])} samples, {n_pools} pools") + print(f" Linear features: {data['n_linear_feat']}") + print(f" Peer encoder input: {data['n_peer_feat']} per peer," + f" {data['n_peers']} peers") + print(f" Build time: {time.time() - t0:.1f}s") + + # Temporal split + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + # Init + params, n_total_linear = init_params( + jax.random.PRNGKey(42), + data["n_peer_feat"], data["n_linear_feat"], + args.encoder_hidden, args.n_peer_outputs, + n_pools, data["init_log_cadences"], + ) + + # Warm-start linear coeffs via OLS (peer_effect = 0 initially) + x_trn = data["x_linear"][train_mask] + y_trn = data["y_total"][train_mask] + # Pad with zeros for peer_effect columns + x_trn_padded = np.concatenate([ + x_trn, + np.zeros((x_trn.shape[0], args.n_peer_outputs * 2), dtype=np.float32) + ], axis=1) + sol, _, _, _ = np.linalg.lstsq(x_trn_padded, y_trn, rcond=None) + params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) + + n_enc_params = (args.encoder_hidden * data["n_peer_feat"] + + args.encoder_hidden + + args.encoder_hidden * args.n_peer_outputs + + args.n_peer_outputs) + print(f"\n Params: {n_total_linear} linear + {n_enc_params} encoder" + f" + {n_pools} cadences = {n_total_linear + n_enc_params + n_pools}") + print(f" Init cadence: {np.exp(data['init_log_cadences']).min():.1f}" + f"-{np.median(np.exp(data['init_log_cadences'])):.1f}" + f"-{np.exp(data['init_log_cadences']).max():.1f} min") + + # Train + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + + print("\n Compiling...") + t0 = time.time() + params = train(params, train_data, grad_fn, args.epochs, args.lr, + args.l2_alpha, args.huber_delta) + print(f" Training: {time.time() - t0:.1f}s") + + # Evaluate + print("\n --- Train ---") + evaluate(params, train_data) + print("\n --- Eval ---") + evaluate(params, eval_data) + + print(f"\n Baselines (eval, total volume R²):") + print(f" V_arb only: median R² = -0.33") + print(f" Linear shared: median R² = 0.39") + print(f" Linear+intercept: median R² = 0.39") + print(f" DeepSets: median R² = 0.43") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_linear_market_noise.py b/experiments/run_linear_market_noise.py new file mode 100644 index 00000000..0e51128d --- /dev/null +++ b/experiments/run_linear_market_noise.py @@ -0,0 +1,463 @@ +"""Linear noise model with market features and learnable cadence. + +V_total = V_arb(cadence) + exp(x @ coeffs) + +where x includes: + - Option C x_obs (intercept, log_tvl_lag1, dow_sin, dow_cos) + - Cross-pool lagged volumes (token-A, token-B, chain peers) + - Market features (BTC price/vol/trend, token prices/vol/trend) + +Cadence is per-pool, optimized jointly with noise coefficients via Adam +through the differentiable PCHIP grid. + +Usage: + python experiments/run_linear_market_noise.py + python experiments/run_linear_market_noise.py --trend-windows 7 14 30 + python experiments/run_linear_market_noise.py --no-market # x_obs only +""" + +import argparse +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30), + include_market=True, include_cross_pool=True): + """Build feature matrix and targets.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from quantammsim.calibration.pool_data import ( + build_x_obs, build_cross_pool_x_obs, K_OBS_REDUCED, K_OBS_CROSS, + ) + from quantammsim.calibration.market_features import ( + build_pool_market_features, pool_market_features_to_matrix, + ) + + pool_ids = sorted(matched_clean.keys()) + n_pools = len(pool_ids) + + # Common date grid + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + # Per-pool: V_arb, volumes, coeffs, gas, grid day mapping + vol_matrix = np.full((n_dates, n_pools), np.nan) + pool_coeffs = [] + pool_gas = [] + init_log_cadences = np.zeros(n_pools, dtype=np.float32) + common_to_grid = np.full((n_pools, n_dates), 0, dtype=np.int32) + + for j, pid in enumerate(pool_ids): + entry = matched_clean[pid] + oc = option_c_clean[pid] + panel = entry["panel"] + + pool_coeffs.append(entry["coeffs"]) + pool_gas.append(jnp.float64(np.exp(oc["log_gas"]))) + init_log_cadences[j] = oc["log_cadence"] + + dates = panel["date"].values + log_vols = panel["log_volume"].values.astype(float) + for k, date in enumerate(dates): + t = date_to_idx[date] + vol_matrix[t, j] = log_vols[k] + common_to_grid[j, t] = entry["day_indices"][k] + + # Build samples: require t >= 1 (for lag) + sample_pools, sample_days = [], [] + for i in range(n_pools): + for t in range(1, n_dates): + if np.isnan(vol_matrix[t, i]) or np.isnan(vol_matrix[t - 1, i]): + continue + sample_pools.append(i) + sample_days.append(t) + sample_pools = np.array(sample_pools, dtype=np.int32) + sample_days = np.array(sample_days, dtype=np.int32) + n_samples = len(sample_pools) + + # x_obs: reduced (4) or cross-pool (7) + if include_cross_pool: + k_obs = K_OBS_CROSS + x_obs_grid = np.full((n_dates, n_pools, k_obs), np.nan) + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + xc = build_cross_pool_x_obs(panel, matched_clean, pid) # (n_obs-1, 7) + dates = panel["date"].values + for k, date in enumerate(dates[1:]): + x_obs_grid[date_to_idx[date], j] = xc[k] + else: + k_obs = K_OBS_REDUCED + x_obs_grid = np.full((n_dates, n_pools, k_obs), np.nan) + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + xr = build_x_obs(panel, reduced=True) + dates = panel["date"].values + for k, date in enumerate(dates): + x_obs_grid[date_to_idx[date], j] = xr[k] + + # Per-sample x_obs + x_obs = np.zeros((n_samples, k_obs), dtype=np.float32) + for s in range(n_samples): + xval = x_obs_grid[sample_days[s], sample_pools[s]] + if np.all(np.isfinite(xval)): + x_obs[s] = xval + + # Market features + if include_market: + print(" Building market features...") + pool_feat = build_pool_market_features( + matched_clean, trend_windows=list(trend_windows)) + x_market, market_names = pool_market_features_to_matrix( + pool_feat, matched_clean, date_to_idx, pool_ids, + sample_pools, sample_days) + print(f" Market features: {len(market_names)} columns") + else: + x_market = np.zeros((n_samples, 0), dtype=np.float32) + market_names = [] + + # Combine features + x_all = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) + feat_names = [f"xobs_{i}" for i in range(k_obs)] + market_names + + # Standardize (except intercept column 0) + x_mean = np.mean(x_all, axis=0) + x_std = np.std(x_all, axis=0) + x_std[x_std < 1e-6] = 1.0 + x_mean[0] = 0.0 # don't center intercept + x_std[0] = 1.0 + x_all = ((x_all - x_mean) / x_std).astype(np.float32) + + # Targets + y_total = np.array([vol_matrix[sample_days[s], sample_pools[s]] + for s in range(n_samples)], dtype=np.float32) + sample_grid_days = common_to_grid[sample_pools, sample_days] + + return { + "x": x_all, # (n_samples, n_feat) + "y_total": y_total, # (n_samples,) + "pool_idx": sample_pools, # (n_samples,) + "day_idx": sample_days, # (n_samples,) + "sample_grid_days": sample_grid_days, # (n_samples,) + "pool_coeffs": pool_coeffs, + "pool_gas": pool_gas, + "init_log_cadences": init_log_cadences, + "n_pools": n_pools, + "n_feat": x_all.shape[1], + "pool_ids": pool_ids, + "feat_names": feat_names, + "x_mean": x_mean, + "x_std": x_std, + } + + +def make_loss_fn(pool_coeffs, pool_gas, n_pools): + """Loss function with learnable cadence + linear noise model.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + def loss_fn(params, x, y_total, sample_grid_days, pool_idx, + l2_alpha, huber_delta): + log_cadence = params["log_cadence"] + noise_coeffs = params["noise_coeffs"] + + # V_noise = exp(x @ noise_coeffs + pool_intercept) + log_v_noise = x @ noise_coeffs + if "pool_intercepts" in params: + log_v_noise = log_v_noise + params["pool_intercepts"][pool_idx] + + # V_arb from PCHIP at learned cadence + n_samples = y_total.shape[0] + v_arb = jnp.zeros(n_samples) + for i in range(n_pools): + v_arb_all = interpolate_pool_daily( + pool_coeffs[i], log_cadence[i], pool_gas[i]) + safe_days = jnp.clip(sample_grid_days, 0, v_arb_all.shape[0] - 1) + v_arb = jnp.where(pool_idx == i, v_arb_all[safe_days], v_arb) + + log_v_arb = jnp.log(jnp.maximum(v_arb, 1e-10)) + log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) + + # Huber loss with per-pool weighting + residuals = log_v_total - y_total + abs_r = jnp.abs(residuals) + huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + pool_counts = jnp.zeros(n_pools).at[pool_idx].add( + jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) + data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active + + reg = l2_alpha * jnp.sum(noise_coeffs ** 2) + return data_loss + reg + + return jax.jit(jax.value_and_grad(loss_fn)) + + +def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, + verbose=True): + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + + x = jnp.array(data["x"]) + y = jnp.array(data["y_total"]) + sgd = jnp.array(data["sample_grid_days"]) + pidx = jnp.array(data["pool_idx"]) + + for epoch in range(n_epochs): + loss_val, grads = grad_fn( + params, x, y, sgd, pidx, l2_alpha, huber_delta) + loss_f = float(loss_val) + + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): + cads = np.exp(np.array(params["log_cadence"])) + nc = np.array(params["noise_coeffs"]) + print(f" epoch {epoch:4d} loss={loss_f:.6f}" + f" cad=[{cads.min():.1f}-{np.median(cads):.1f}-{cads.max():.1f}]" + f" |coeffs|={np.mean(np.abs(nc)):.3f}") + + return params + + +def evaluate(params, data, label=""): + """Evaluate decomposition quality.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + x = np.array(data["x"]) + y_total = np.array(data["y_total"]) + pool_idx = np.array(data["pool_idx"]) + sgd = np.array(data["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + noise_coeffs = np.array(params["noise_coeffs"]) + init_cads = data["init_log_cadences"] + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + + log_v_noise = x @ noise_coeffs + if "pool_intercepts" in params: + pool_intercepts = np.array(params["pool_intercepts"]) + log_v_noise = log_v_noise + pool_intercepts[pool_idx] + v_noise = np.exp(log_v_noise) + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd[mask]] + + v_obs = np.exp(y_total) + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + if label: + print(f"\n {label}:") + print(f" {'Pool'[:16]:16s} {'R²':>6s} {'Cad':>5s} {'→':>2s} {'learn':>5s}" + f" {'Arb%':>6s} {'Noise%':>7s} {'Flag':>5s}") + print(f" {'-'*60}") + + r2s = {} + pool_diag = [] + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s[i] = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] + ci = np.exp(init_cads[i]) + cl = np.exp(log_cadence[i]) + arb_pct = np.median(v_arb[mask] / v_obs[mask]) * 100 + noise_pct = np.median(v_noise[mask] / v_obs[mask]) * 100 + + flags = [] + if arb_pct > 150: + flags.append("A") + if cl <= 1.01 or cl >= 59.9: + flags.append("B") + if r2s[i] < 0: + flags.append("X") + flag_str = "".join(flags) + + pool_diag.append({ + "pid": pid, "r2": r2s[i], "cad_init": ci, "cad_learned": cl, + "arb_pct": arb_pct, "noise_pct": noise_pct, "flags": flag_str, + }) + + print(f" {pid[:16]:16s} {r2s[i]:6.3f} {ci:5.1f} → {cl:5.1f}" + f" {arb_pct:6.0f}% {noise_pct:6.0f}% {flag_str:>5s}") + + vals = [x for x in r2s.values() if np.isfinite(x)] + med = np.median(vals) if vals else float("nan") + healthy = [d for d in pool_diag if d["arb_pct"] <= 150 and d["r2"] > 0] + med_h = np.median([d["r2"] for d in healthy]) if healthy else float("nan") + n_path = sum(1 for d in pool_diag if d["arb_pct"] > 150) + n_bound = sum(1 for d in pool_diag + if d["cad_learned"] <= 1.01 or d["cad_learned"] >= 59.9) + + print(f"\n Median R²: {med:.4f} (healthy: {med_h:.4f})") + print(f" Healthy: {len(pool_diag) - n_path}/{len(pool_diag)}," + f" at bounds: {n_bound}") + + return med, r2s, pool_diag + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--epochs", type=int, default=2000) + parser.add_argument("--lr", type=float, default=1e-3) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7, 14, 30]) + parser.add_argument("--no-market", action="store_true", + help="x_obs only, no market features") + parser.add_argument("--no-cross-pool", action="store_true", + help="Reduced x_obs (4) instead of cross-pool (7)") + parser.add_argument("--pool-intercepts", action="store_true", + help="Per-pool intercept (shared slopes + per-pool bias)") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Linear Noise Model + Learnable Cadence") + print(f" market={not args.no_market}, cross_pool={not args.no_cross_pool}" + f", pool_intercepts={args.pool_intercepts}") + print(f" trend_windows={args.trend_windows}") + print(f" epochs={args.epochs}, lr={args.lr}, l2={args.l2_alpha}") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding data...") + t0 = time.time() + data = build_data( + matched_clean, option_c_clean, + trend_windows=tuple(args.trend_windows), + include_market=not args.no_market, + include_cross_pool=not args.no_cross_pool, + ) + print(f" {len(data['pool_idx'])} samples, {data['n_pools']} pools," + f" {data['n_feat']} features, {time.time() - t0:.1f}s") + print(f" Features: {data['feat_names']}") + + # Temporal split + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + + train_data = {k: v[train_mask] if isinstance(v, np.ndarray) + and v.shape[0] == len(day_idx) else v + for k, v in data.items()} + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == len(day_idx) else v + for k, v in data.items()} + + # Init params + n_feat = data["n_feat"] + n_pools = data["n_pools"] + params = { + "log_cadence": jnp.array(data["init_log_cadences"]), + "noise_coeffs": jnp.zeros(n_feat), + } + + # Warm-start noise_coeffs via OLS on train: y_total ≈ x @ coeffs + x_trn = data["x"][train_mask] + y_trn = data["y_total"][train_mask] + sol, _, _, _ = np.linalg.lstsq(x_trn, y_trn, rcond=None) + params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) + + if args.pool_intercepts: + # Init per-pool intercepts from OLS residuals + ols_pred = x_trn @ sol + ols_resid = y_trn - ols_pred + pool_idx_trn = data["pool_idx"][train_mask] + intercepts = np.zeros(n_pools, dtype=np.float32) + for i in range(n_pools): + mask_i = pool_idx_trn == i + if mask_i.sum() > 0: + intercepts[i] = np.mean(ols_resid[mask_i]) + params["pool_intercepts"] = jnp.array(intercepts) + print(f" Per-pool intercepts: {n_pools} pools" + f" (range {intercepts.min():.2f} to {intercepts.max():.2f})") + + print(f"\n Init cadence: {np.exp(data['init_log_cadences']).min():.1f}" + f"-{np.median(np.exp(data['init_log_cadences'])):.1f}" + f"-{np.exp(data['init_log_cadences']).max():.1f} min") + print(f" OLS warm-start |coeffs|={np.mean(np.abs(sol)):.3f}") + print(f" Total params: {sum(v.size for v in params.values())}" + f" ({n_feat} coeffs + {n_pools} cadences" + f"{'+ ' + str(n_pools) + ' intercepts' if args.pool_intercepts else ''})") + + # Build loss and train + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], data["n_pools"]) + + print("\n Compiling...") + t0 = time.time() + params = train(params, train_data, grad_fn, args.epochs, args.lr, + args.l2_alpha, args.huber_delta) + print(f" Training: {time.time() - t0:.1f}s") + + # Print learned coefficients + nc = np.array(params["noise_coeffs"]) + print(f"\n Noise coefficients ({len(nc)}):") + for i, name in enumerate(data["feat_names"]): + print(f" {name:30s} {nc[i]:+8.4f}") + + # Evaluate + print("\n --- Train ---") + evaluate(params, train_data) + print("\n --- Eval ---") + evaluate(params, eval_data) + + # Baselines + print(f"\n Baselines (eval, total volume R²):") + print(f" V_arb only: median R² = -0.33") + print(f" Naive lag: median R² = 0.01") + print(f" DeepSets best: median R² = 0.43") + + +if __name__ == "__main__": + main() diff --git a/quantammsim/calibration/market_features.py b/quantammsim/calibration/market_features.py new file mode 100644 index 00000000..7a7ee59b --- /dev/null +++ b/quantammsim/calibration/market_features.py @@ -0,0 +1,297 @@ +"""Market-level and token-level features for the noise volume model. + +Derives daily features from Binance minute-level price data and pool metadata. +Features are grounded in market microstructure — what mechanistically drives +organic (non-arb) trading volume: + +Market regime: + - BTC log price, log return — crypto market regime proxy + - BTC trend (rolling mean log return) — bull/bear at various horizons + +Token-level (per pool token): + - Token log price, daily log return + - Token realized volatility — higher vol → more hedging/speculative flow + - Token Binance volume — proxy for overall token trading interest + - Token trend (rolling mean log return) + +All features are computed daily and aligned to the panel date grid. +""" + +import os +from typing import Dict, List, Optional, Tuple + +import numpy as np +import pandas as pd + +DATA_DIR = os.path.join( + os.path.dirname(os.path.dirname(os.path.abspath(__file__))), + "data", +) + +# Map wrapped/derivative tokens to their Binance underlying +TOKEN_MAP = { + "WETH": "ETH", + "WBTC": "BTC", + "wstETH": "ETH", + "waEthLidowstETH": "ETH", + "waEthLidoWETH": "ETH", + "waGnowstETH": "ETH", + "waGnoGNO": "GNO", + "waBasUSDC": "USDC", + "waBasWETH": "ETH", + "sDAI": "DAI", + "scUSD": "USDC", + "stS": "S", + "JitoSOL": "SOL", + "USDT": "USDC", # treat as $1 stablecoin +} + + +def _load_binance_daily(symbol: str) -> pd.DataFrame: + """Load Binance minute data and resample to daily OHLCV.""" + mapped = TOKEN_MAP.get(symbol, symbol) + path = os.path.join(DATA_DIR, f"{mapped}_USD.parquet") + if not os.path.exists(path): + return None + + df = pd.read_parquet(path, columns=["date", "close", "Volume USD"]) + df["date"] = pd.to_datetime(df["date"]) + df = df.set_index("date").sort_index() + + daily = df.resample("1D").agg({ + "close": "last", + "Volume USD": "sum", + }).dropna(subset=["close"]) + + daily.columns = ["close", "volume_usd"] + return daily + + +def _compute_token_features( + daily: pd.DataFrame, + trend_windows: List[int] = (7, 14, 30), + is_market: bool = False, +) -> pd.DataFrame: + """Compute daily features from a token's OHLCV. + + For market-level tokens (BTC): includes log_price as regime proxy. + For pool tokens: only returns/vol/trends (comparable across tokens). + + Returns DataFrame indexed by date. + """ + out = pd.DataFrame(index=daily.index) + log_price = np.log(daily["close"].clip(lower=1e-10)) + out["log_return"] = log_price.diff() + + if is_market: + # BTC log_price is a market regime proxy (same for all pools) + out["log_price"] = log_price + + # Realized volatility: std of log returns over trailing 7 days + out["realized_vol_7d"] = out["log_return"].rolling(7, min_periods=3).std() + + # Trend: rolling mean log return at various horizons + for w in trend_windows: + out[f"trend_{w}d"] = out["log_return"].rolling(w, min_periods=max(w // 2, 2)).mean() + + return out + + +def build_btc_daily_features( + trend_windows: List[int] = (7, 14, 30), +) -> pd.DataFrame: + """BTC daily features as market regime proxy. + + Returns DataFrame indexed by date with columns prefixed 'btc_'. + """ + daily = _load_binance_daily("BTC") + if daily is None: + raise FileNotFoundError("BTC_USD.parquet not found") + + feat = _compute_token_features(daily, trend_windows, is_market=True) + feat.columns = [f"btc_{c}" for c in feat.columns] + return feat + + +def build_token_daily_features( + symbol: str, + trend_windows: List[int] = (7, 14, 30), +) -> Optional[pd.DataFrame]: + """Daily features for a single token. Returns None if no data.""" + daily = _load_binance_daily(symbol) + if daily is None: + return None + return _compute_token_features(daily, trend_windows) + + +def _compute_pair_volatility( + symbol_a: str, + symbol_b: str, +) -> Optional[pd.DataFrame]: + """Compute realized volatility of the A/B price ratio. + + vol(log(price_A/price_B)) = vol(log(price_A) - log(price_B)) + Symmetric: A/B and B/A give identical volatility. + + Returns DataFrame indexed by date with 'pair_realized_vol_7d'. + """ + daily_a = _load_binance_daily(symbol_a) + daily_b = _load_binance_daily(symbol_b) + if daily_a is None or daily_b is None: + return None + + # Align on common dates + log_a = np.log(daily_a["close"].clip(lower=1e-10)) + log_b = np.log(daily_b["close"].clip(lower=1e-10)) + common = log_a.index.intersection(log_b.index) + if len(common) < 10: + return None + + log_ratio = log_a.loc[common] - log_b.loc[common] + log_ratio_return = log_ratio.diff() + + out = pd.DataFrame(index=common) + out["pair_realized_vol_7d"] = log_ratio_return.rolling(7, min_periods=3).std() + return out + + +def build_pool_market_features( + matched_clean: Dict[str, dict], + trend_windows: List[int] = (7, 14, 30), +) -> Dict[str, pd.DataFrame]: + """Build per-pool market feature DataFrames. + + For each pool, produces a DataFrame aligned to the pool's panel dates with: + - BTC features (market regime) + - Token A features (vs USD) + - Token B features (vs USD) + - Pair volatility (A/B ratio) + + Returns dict: pool_id -> DataFrame with all features. + """ + from quantammsim.calibration.pool_data import _parse_tokens + + # Load BTC features once + btc_feat = build_btc_daily_features(trend_windows) + + # Cache token features and pair volatilities + token_cache = {} + pair_vol_cache = {} + + pool_features = {} + for pid, entry in matched_clean.items(): + panel = entry["panel"] + dates = pd.to_datetime(panel["date"]) + + # Parse tokens + toks = _parse_tokens(entry["tokens"]) + tok_a, tok_b = toks[0], toks[1] if len(toks) > 1 else toks[0] + + # Get token features + for tok in [tok_a, tok_b]: + mapped = TOKEN_MAP.get(tok, tok) + if mapped not in token_cache: + token_cache[mapped] = build_token_daily_features(mapped, trend_windows) + + feat_a = token_cache.get(TOKEN_MAP.get(tok_a, tok_a)) + feat_b = token_cache.get(TOKEN_MAP.get(tok_b, tok_b)) + + # Pair volatility (cache by sorted token pair to avoid duplicates) + mapped_a = TOKEN_MAP.get(tok_a, tok_a) + mapped_b = TOKEN_MAP.get(tok_b, tok_b) + pair_key = tuple(sorted([mapped_a, mapped_b])) + if pair_key not in pair_vol_cache: + pair_vol_cache[pair_key] = _compute_pair_volatility(mapped_a, mapped_b) + pair_vol = pair_vol_cache[pair_key] + + # Build per-date feature vectors + rows = [] + for date in dates: + day = pd.Timestamp(date).normalize() + row = {} + + # BTC features + if day in btc_feat.index: + for col in btc_feat.columns: + row[col] = btc_feat.loc[day, col] + + # Token A features + if feat_a is not None and day in feat_a.index: + for col in feat_a.columns: + row[f"tok_a_{col}"] = feat_a.loc[day, col] + + # Token B features + if feat_b is not None and day in feat_b.index: + for col in feat_b.columns: + row[f"tok_b_{col}"] = feat_b.loc[day, col] + + # Pair volatility + if pair_vol is not None and day in pair_vol.index: + row["pair_realized_vol_7d"] = pair_vol.loc[day, "pair_realized_vol_7d"] + + rows.append(row) + + df = pd.DataFrame(rows, index=dates) + pool_features[pid] = df + + return pool_features + + +def pool_market_features_to_matrix( + pool_features: Dict[str, pd.DataFrame], + matched_clean: Dict[str, dict], + date_to_idx: Dict, + pool_ids: List[str], + sample_pools: np.ndarray, + sample_days: np.ndarray, +) -> Tuple[np.ndarray, List[str]]: + """Convert per-pool market features to a (n_samples, n_feat) matrix. + + Aligns features to the common date grid and sample indices. + NaN-fills missing values, then imputes with column mean. + + Returns (feature_matrix, feature_names). + """ + # Get feature columns from first pool + first_pid = pool_ids[0] + feat_cols = sorted(pool_features[first_pid].columns) + n_feat = len(feat_cols) + n_pools = len(pool_ids) + + # Collect all dates + n_dates = max(date_to_idx.values()) + 1 + + # Build (n_dates, n_pools, n_feat) grid + feat_grid = np.full((n_dates, n_pools, n_feat), np.nan, dtype=np.float32) + + for j, pid in enumerate(pool_ids): + if pid not in pool_features: + continue + pf = pool_features[pid] + panel_dates = matched_clean[pid]["panel"]["date"].values + for k, date in enumerate(panel_dates): + t = date_to_idx.get(date) + if t is None: + continue + for f, col in enumerate(feat_cols): + if col in pf.columns and k < len(pf): + val = pf.iloc[k][col] if col in pf.columns else np.nan + if np.isfinite(val): + feat_grid[t, j, f] = val + + # Extract per-sample + n_samples = len(sample_pools) + X = np.zeros((n_samples, n_feat), dtype=np.float32) + for s in range(n_samples): + X[s] = feat_grid[sample_days[s], sample_pools[s]] + + # Impute NaN with column mean + for f in range(n_feat): + col = X[:, f] + mask = np.isnan(col) + if mask.all(): + col[:] = 0.0 + elif mask.any(): + col[mask] = np.nanmean(col) + + return X, feat_cols From e6610316e2fa44f6659c4d5f7da7568eef127719 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 23 Mar 2026 11:46:28 +0000 Subject: [PATCH 064/115] feat: per-pool linear noise model, market_linear simulator integration, plotting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - run_linear_market_noise.py: add --per-pool (per-pool coefficients with Ridge warm-start), --no-split (train on all data), artifact saving (model.npz + meta.json). Add interaction terms and pair_realized_vol. - run_hybrid_noise.py: variable encoder depth (1-4), wider hidden search, Optuna sweep with depth/hidden in search space. - market_features.py: add pair_realized_vol_7d (A/B price ratio volatility), drop token log_price and log_volume_usd from non-BTC tokens. - noise_trades.py: new reclamm_market_linear_noise_volume() — takes precomputed noise_base and noise_tvl_coeff arrays, evaluates log(V_noise) = base + tvl_coeff * log(effective_TVL) per scan step. - reclamm_reserves.py: add "market_linear" noise model dispatch in scan step function, wire noise_base_array and noise_tvl_coeff_array through all 4 scan wrapper functions. - reclamm.py: refactor _prepare_noise_arrays to return dict, add market_linear array preparation from run_fingerprint, pass through to all calculate_reserves_* methods. - plot_calibrated_vs_real.py: new script plotting predicted vs real volume for calibrated pools using saved artifact, with stacked V_arb/V_noise decomposition. --- experiments/run_hybrid_noise.py | 324 +++++++++++++++--- experiments/run_linear_market_noise.py | 217 +++++++++--- quantammsim/pools/noise_trades.py | 45 +++ quantammsim/pools/reCLAMM/reclamm.py | 54 ++- quantammsim/pools/reCLAMM/reclamm_reserves.py | 35 ++ scripts/plot_calibrated_vs_real.py | 241 +++++++++++++ 6 files changed, 813 insertions(+), 103 deletions(-) create mode 100644 scripts/plot_calibrated_vs_real.py diff --git a/experiments/run_hybrid_noise.py b/experiments/run_hybrid_noise.py index be26b138..d815a5c5 100644 --- a/experiments/run_hybrid_noise.py +++ b/experiments/run_hybrid_noise.py @@ -186,16 +186,49 @@ def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30)): n_peer_feat = peer_input.shape[-1] # Combine linear features (x_obs + market) - x_linear = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) - linear_names = [f"xobs_{i}" for i in range(K_OBS_CROSS)] + market_names + x_base = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) + base_names = [f"xobs_{i}" for i in range(K_OBS_CROSS)] + market_names - # Standardize linear features (except intercept) - x_mean = np.mean(x_linear, axis=0) - x_std_arr = np.std(x_linear, axis=0) + # Standardize base features (except intercept) + x_mean = np.mean(x_base, axis=0) + x_std_arr = np.std(x_base, axis=0) x_std_arr[x_std_arr < 1e-6] = 1.0 x_mean[0] = 0.0 x_std_arr[0] = 1.0 - x_linear = ((x_linear - x_mean) / x_std_arr).astype(np.float32) + x_base = ((x_base - x_mean) / x_std_arr).astype(np.float32) + + # Build named column index for interaction construction + col_idx = {name: i for i, name in enumerate(base_names)} + + # Interaction terms (products of standardized features) + interactions = [] + interaction_names = [] + + def _add_interaction(name_a, name_b): + if name_a in col_idx and name_b in col_idx: + interactions.append( + x_base[:, col_idx[name_a]] * x_base[:, col_idx[name_b]]) + interaction_names.append(f"{name_a}×{name_b}") + + # TVL × volatility: deep pools respond differently to market stress + _add_interaction("xobs_1", "btc_realized_vol_7d") # tvl × btc vol + _add_interaction("xobs_1", "tok_a_realized_vol_7d") # tvl × tok_a vol + _add_interaction("xobs_1", "pair_realized_vol_7d") # tvl × pair vol + + # Cross-token volatility interaction: both tokens moving = pair activity + _add_interaction("tok_a_realized_vol_7d", "tok_b_realized_vol_7d") + + if interactions: + x_interactions = np.column_stack(interactions).astype(np.float32) + x_linear = np.concatenate([x_base, x_interactions], axis=1) + linear_names = base_names + interaction_names + else: + x_linear = x_base + linear_names = base_names + + # Track which columns are tvl and btc_vol for peer_effect interactions in loss + tvl_col = col_idx.get("xobs_1", 1) + btc_vol_col = col_idx.get("btc_realized_vol_7d") # Targets and indices y_total = np.array([vol_matrix[sample_days[s], sample_pools[s]] @@ -219,6 +252,8 @@ def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30)): "n_peer_feat": n_peer_feat, "pool_ids": pool_ids, "linear_names": linear_names, + "tvl_col": tvl_col, + "btc_vol_col": btc_vol_col, } @@ -241,46 +276,70 @@ def _subset(d, mask): def init_params(key, n_peer_feat, n_linear_feat, encoder_hidden, - n_peer_outputs, n_pools, init_log_cadences): + n_peer_outputs, n_pools, init_log_cadences, + encoder_depth=1): """Initialize all parameters. - Encoder: peer_input → hidden → n_peer_outputs (per peer, then mean-pooled) - Linear: [x_linear, peer_outputs, peer_outputs × x_linear[1](tvl)] @ coeffs + Encoder: peer_input → hidden (× depth) → n_peer_outputs (per peer, mean-pooled) + Linear: [x_linear, peer_outputs, peer×tvl, peer×btc_vol] @ coeffs + + encoder_depth: number of hidden layers (1-4). + params["enc_depth"] stores the depth as a scalar for forward_encoder. """ - k1, k2 = jax.random.split(key) - - # Encoder: single hidden layer → n_peer_outputs - n_total_linear = n_linear_feat + n_peer_outputs + n_peer_outputs # +interactions with tvl - - params = { - "enc_W1": jax.random.normal(k1, (n_peer_feat, encoder_hidden)) * np.sqrt(2.0 / n_peer_feat), - "enc_b1": jnp.zeros(encoder_hidden), - "enc_W2": jax.random.normal(k2, (encoder_hidden, n_peer_outputs)) * 0.01, - "enc_b2": jnp.zeros(n_peer_outputs), - "noise_coeffs": jnp.zeros(n_total_linear), - "log_cadence": jnp.array(init_log_cadences), - } + keys = jax.random.split(key, encoder_depth + 2) + + n_peer_linear = n_peer_outputs * 3 + n_total_linear = n_linear_feat + n_peer_linear + + params = {} + + # First layer: input → hidden + params["enc_W1"] = jax.random.normal(keys[0], (n_peer_feat, encoder_hidden)) * np.sqrt(2.0 / n_peer_feat) + params["enc_b1"] = jnp.zeros(encoder_hidden) + + # Hidden layers 2..depth: hidden → hidden + for d in range(2, encoder_depth + 1): + params[f"enc_W{d}"] = jax.random.normal(keys[d - 1], (encoder_hidden, encoder_hidden)) * np.sqrt(2.0 / encoder_hidden) + params[f"enc_b{d}"] = jnp.zeros(encoder_hidden) + + # Output layer: hidden → n_peer_outputs + out_idx = encoder_depth + 1 + params[f"enc_W{out_idx}"] = jax.random.normal(keys[-1], (encoder_hidden, n_peer_outputs)) * 0.01 + params[f"enc_b{out_idx}"] = jnp.zeros(n_peer_outputs) + + params["noise_coeffs"] = jnp.zeros(n_total_linear) + params["log_cadence"] = jnp.array(init_log_cadences) + return params, n_total_linear def forward_encoder(params, peer_input, peer_mask): """DeepSets encoder: per-peer MLP → masked mean → scalar(s). + Depth determined by counting enc_W* keys. Returns (n_samples, n_peer_outputs). """ batch, n_peers, _ = peer_input.shape flat = peer_input.reshape(-1, peer_input.shape[-1]) - h = jnp.maximum(flat @ params["enc_W1"] + params["enc_b1"], 0.0) - h = h @ params["enc_W2"] + params["enc_b2"] - h = h.reshape(batch, n_peers, -1) + # Count layers: enc_W1, enc_W2, ..., enc_W{depth+1} + n_layers = sum(1 for k in params if k.startswith("enc_W")) + # Hidden layers with ReLU + h = flat + for i in range(1, n_layers): + h = jnp.maximum(h @ params[f"enc_W{i}"] + params[f"enc_b{i}"], 0.0) + + # Output layer (no activation) + h = h @ params[f"enc_W{n_layers}"] + params[f"enc_b{n_layers}"] + + h = h.reshape(batch, n_peers, -1) h_masked = h * peer_mask[:, :, None] n_valid = jnp.maximum(jnp.sum(peer_mask, axis=1, keepdims=True), 1.0) - return jnp.sum(h_masked, axis=1) / n_valid # (batch, n_peer_outputs) + return jnp.sum(h_masked, axis=1) / n_valid -def make_loss_fn(pool_coeffs, pool_gas, n_pools): +def make_loss_fn(pool_coeffs, pool_gas, n_pools, tvl_col, btc_vol_col): """Loss with learnable cadence + encoder + linear noise model.""" from quantammsim.calibration.grid_interpolation import interpolate_pool_daily @@ -290,12 +349,14 @@ def loss_fn(params, x_linear, peer_input, peer_mask, y_total, # Encoder → peer effect scalar(s) peer_effect = forward_encoder(params, peer_input, peer_mask) - # Build full linear input: [x_linear, peer_effect, peer_effect × tvl] - # tvl is x_linear[:, 1] (xobs_1 = log_tvl_lag1, standardized) - tvl = x_linear[:, 1:2] # keep 2D - peer_x_tvl = peer_effect * tvl # interaction + # Build full linear input: [x_linear, peer, peer×tvl, peer×btc_vol] + tvl = x_linear[:, tvl_col:tvl_col + 1] + btc_vol = x_linear[:, btc_vol_col:btc_vol_col + 1] + peer_x_tvl = peer_effect * tvl + peer_x_btcvol = peer_effect * btc_vol - x_full = jnp.concatenate([x_linear, peer_effect, peer_x_tvl], axis=1) + x_full = jnp.concatenate( + [x_linear, peer_effect, peer_x_tvl, peer_x_btcvol], axis=1) log_v_noise = x_full @ params["noise_coeffs"] # V_arb from PCHIP @@ -325,10 +386,9 @@ def loss_fn(params, x_linear, peer_input, peer_mask, y_total, pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active - # L2 on encoder weights + noise coeffs + # L2 on all encoder weights + noise coeffs reg = l2_alpha * ( - jnp.sum(params["enc_W1"] ** 2) + - jnp.sum(params["enc_W2"] ** 2) + + sum(jnp.sum(v ** 2) for k, v in params.items() if k.startswith("enc_W")) + jnp.sum(params["noise_coeffs"] ** 2) ) return data_loss + reg @@ -391,10 +451,15 @@ def evaluate(params, data, label=""): peer_effect = np.array(forward_encoder( params, jnp.array(peer_input), jnp.array(peer_mask))) - # Build full linear input - tvl = x_linear[:, 1:2] + # Build full linear input (must match loss_fn construction) + tvl_col = data["tvl_col"] + btc_vol_col = data["btc_vol_col"] + tvl = x_linear[:, tvl_col:tvl_col + 1] + btc_vol = x_linear[:, btc_vol_col:btc_vol_col + 1] peer_x_tvl = peer_effect * tvl - x_full = np.concatenate([x_linear, peer_effect, peer_x_tvl], axis=1) + peer_x_btcvol = peer_effect * btc_vol + x_full = np.concatenate( + [x_linear, peer_effect, peer_x_tvl, peer_x_btcvol], axis=1) noise_coeffs = np.array(params["noise_coeffs"]) log_v_noise = x_full @ noise_coeffs @@ -469,8 +534,7 @@ def evaluate(params, data, label=""): # Print coefficient analysis nc = np.array(params["noise_coeffs"]) n_linear = data["n_linear_feat"] - n_po = len(nc) - n_linear - n_each = n_po // 2 + n_po = params["enc_W2"].shape[1] # n_peer_outputs print(f"\n Median R²: {med:.4f} (healthy: {med_h:.4f})") print(f" Healthy: {len(pool_diag) - n_path}/{len(pool_diag)}," @@ -479,23 +543,182 @@ def evaluate(params, data, label=""): print(f"\n Linear coefficients:") for j, name in enumerate(data["linear_names"]): print(f" {name:30s} {nc[j]:+8.4f}") - for j in range(n_each): + for j in range(n_po): print(f" {'peer_effect_' + str(j):30s} {nc[n_linear + j]:+8.4f}") - for j in range(n_each): - print(f" {'peer_eff_' + str(j) + '×tvl':30s} {nc[n_linear + n_each + j]:+8.4f}") + for j in range(n_po): + print(f" {'peer_eff_' + str(j) + '×tvl':30s} {nc[n_linear + n_po + j]:+8.4f}") + for j in range(n_po): + print(f" {'peer_eff_' + str(j) + '×btc_vol':30s} {nc[n_linear + 2*n_po + j]:+8.4f}") return med, r2s, pool_diag +def run_optuna(matched_clean, option_c_clean, n_trials): + """Optuna sweep over encoder + linear model hyperparameters.""" + import optuna + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + # Build data for each trend_windows config (cache to avoid rebuilding) + data_cache = {} + + def _get_data(trend_key): + if trend_key not in data_cache: + data_cache[trend_key] = build_data( + matched_clean, option_c_clean, + trend_windows=trend_key) + return data_cache[trend_key] + + # Cache grad_fn per (n_pools, tvl_col, btc_vol_col) — these are stable + grad_fn_cache = {} + + def objective(trial): + trend_w = trial.suggest_categorical("trend_window", [7, 14, 30]) + data = _get_data((trend_w,)) + n_pools = data["n_pools"] + + encoder_hidden = trial.suggest_categorical("encoder_hidden", [16, 32, 64, 128]) + encoder_depth = trial.suggest_categorical("encoder_depth", [1, 2, 3, 4]) + n_peer_outputs = trial.suggest_categorical("n_peer_outputs", [1, 2, 4]) + lr = trial.suggest_float("lr", 3e-4, 3e-3, log=True) + l2_alpha = trial.suggest_float("l2_alpha", 1e-4, 1e-2, log=True) + huber_delta = trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5]) + n_epochs = trial.suggest_categorical("n_epochs", [1000, 2000, 3000]) + + # Temporal split + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = _subset(data, train_mask) + eval_data = _subset(data, eval_mask) + + # Init + params, n_total_linear = init_params( + jax.random.PRNGKey(42), + data["n_peer_feat"], data["n_linear_feat"], + encoder_hidden, n_peer_outputs, + n_pools, data["init_log_cadences"], + encoder_depth=encoder_depth, + ) + + # OLS warm-start + x_trn = data["x_linear"][train_mask] + y_trn = data["y_total"][train_mask] + n_peer_cols = n_peer_outputs * 3 + x_trn_padded = np.concatenate([ + x_trn, np.zeros((x_trn.shape[0], n_peer_cols), dtype=np.float32) + ], axis=1) + sol, _, _, _ = np.linalg.lstsq(x_trn_padded, y_trn, rcond=None) + params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) + + # Build grad_fn (cache by config) + cache_key = (n_pools, data["tvl_col"], data["btc_vol_col"]) + if cache_key not in grad_fn_cache: + grad_fn_cache[cache_key] = make_loss_fn( + data["pool_coeffs"], data["pool_gas"], n_pools, + data["tvl_col"], data["btc_vol_col"]) + grad_fn = grad_fn_cache[cache_key] + + # Train + params = train(params, train_data, grad_fn, n_epochs, lr, + l2_alpha, huber_delta, verbose=False) + + # Eval: compute total R² + x_linear = np.array(eval_data["x_linear"]) + peer_input = eval_data["peer_input"] + peer_mask = eval_data["peer_mask"] + y_total = np.array(eval_data["y_total"]) + pool_idx = np.array(eval_data["pool_idx"]) + sgd = np.array(eval_data["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + + peer_effect = np.array(forward_encoder( + params, jnp.array(peer_input), jnp.array(peer_mask))) + + tvl_col = data["tvl_col"] + btc_vol_col = data["btc_vol_col"] + tvl = x_linear[:, tvl_col:tvl_col + 1] + btc_vol = x_linear[:, btc_vol_col:btc_vol_col + 1] + x_full = np.concatenate([ + x_linear, peer_effect, peer_effect * tvl, peer_effect * btc_vol + ], axis=1) + + noise_coeffs = np.array(params["noise_coeffs"]) + log_v_noise = x_full @ noise_coeffs + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd[mask]] + + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + r2s = [] + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s.append(1 - ss_res / max(ss_tot, 1e-10)) + + med_total = float(np.median(r2s)) if r2s else -10.0 + + cads = np.exp(log_cadence) + print(f" Trial {trial.number}: total={med_total:.4f}" + f" enc_h={encoder_hidden} d={encoder_depth} n_po={n_peer_outputs}" + f" hub={huber_delta} lr={lr:.1e} l2={l2_alpha:.1e}" + f" ep={n_epochs} tw={trend_w}" + f" cad=[{cads.min():.0f}-{np.median(cads):.0f}-{cads.max():.0f}]") + + return med_total + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print("Optuna Results (hybrid)") + print(f"{'='*70}") + print(f" Best eval total R²: {study.best_value:.4f}") + print(f" Best params:") + for k, v in sorted(study.best_params.items()): + print(f" {k}: {v}") + + print(f"\n Top 10:") + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + for t in trials[:10]: + if t.value is not None: + print(f" #{t.number}: total={t.value:.4f}" + f" enc_h={t.params['encoder_hidden']}" + f" d={t.params['encoder_depth']}" + f" n_po={t.params['n_peer_outputs']}" + f" ep={t.params['n_epochs']}" + f" tw={t.params['trend_window']}") + + return study + + def main(): parser = argparse.ArgumentParser() + parser.add_argument("--tune", type=int, default=0, + help="Number of Optuna trials (0 = single run)") parser.add_argument("--epochs", type=int, default=2000) parser.add_argument("--lr", type=float, default=1e-3) parser.add_argument("--l2-alpha", type=float, default=1e-3) parser.add_argument("--huber-delta", type=float, default=1.0) parser.add_argument("--encoder-hidden", type=int, default=16) + parser.add_argument("--encoder-depth", type=int, default=1) parser.add_argument("--n-peer-outputs", type=int, default=1) - parser.add_argument("--trend-windows", type=int, nargs="+", default=[7, 14, 30]) + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) args = parser.parse_args() os.environ.setdefault("JAX_PLATFORMS", "cpu") @@ -509,6 +732,10 @@ def main(): matched_clean, option_c_clean = load_stage1() + if args.tune > 0: + run_optuna(matched_clean, option_c_clean, args.tune) + return + print("\nBuilding data...") t0 = time.time() data = build_data(matched_clean, option_c_clean, @@ -535,15 +762,17 @@ def main(): data["n_peer_feat"], data["n_linear_feat"], args.encoder_hidden, args.n_peer_outputs, n_pools, data["init_log_cadences"], + encoder_depth=args.encoder_depth, ) # Warm-start linear coeffs via OLS (peer_effect = 0 initially) x_trn = data["x_linear"][train_mask] y_trn = data["y_total"][train_mask] - # Pad with zeros for peer_effect columns + # Pad with zeros for peer_effect columns (raw + ×tvl + ×btc_vol) + n_peer_cols = args.n_peer_outputs * 3 x_trn_padded = np.concatenate([ x_trn, - np.zeros((x_trn.shape[0], args.n_peer_outputs * 2), dtype=np.float32) + np.zeros((x_trn.shape[0], n_peer_cols), dtype=np.float32) ], axis=1) sol, _, _, _ = np.linalg.lstsq(x_trn_padded, y_trn, rcond=None) params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) @@ -559,7 +788,8 @@ def main(): f"-{np.exp(data['init_log_cadences']).max():.1f} min") # Train - grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools, + data["tvl_col"], data["btc_vol_col"]) print("\n Compiling...") t0 = time.time() diff --git a/experiments/run_linear_market_noise.py b/experiments/run_linear_market_noise.py index 0e51128d..c71dc6ac 100644 --- a/experiments/run_linear_market_noise.py +++ b/experiments/run_linear_market_noise.py @@ -140,17 +140,44 @@ def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30), x_market = np.zeros((n_samples, 0), dtype=np.float32) market_names = [] - # Combine features - x_all = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) - feat_names = [f"xobs_{i}" for i in range(k_obs)] + market_names + # Combine base features + x_base = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) + base_names = [f"xobs_{i}" for i in range(k_obs)] + market_names # Standardize (except intercept column 0) - x_mean = np.mean(x_all, axis=0) - x_std = np.std(x_all, axis=0) + x_mean = np.mean(x_base, axis=0) + x_std = np.std(x_base, axis=0) x_std[x_std < 1e-6] = 1.0 x_mean[0] = 0.0 # don't center intercept x_std[0] = 1.0 - x_all = ((x_all - x_mean) / x_std).astype(np.float32) + x_base = ((x_base - x_mean) / x_std).astype(np.float32) + + # Interaction terms (products of standardized features) + col_idx = {name: i for i, name in enumerate(base_names)} + interactions = [] + interaction_names = [] + + def _add_interaction(name_a, name_b): + if name_a in col_idx and name_b in col_idx: + interactions.append( + x_base[:, col_idx[name_a]] * x_base[:, col_idx[name_b]]) + interaction_names.append(f"{name_a}×{name_b}") + + _add_interaction("xobs_1", "btc_realized_vol_7d") # tvl × btc vol + _add_interaction("xobs_1", "tok_a_realized_vol_7d") # tvl × tok_a vol + _add_interaction("xobs_1", "pair_realized_vol_7d") # tvl × pair vol + _add_interaction("tok_a_realized_vol_7d", "tok_b_realized_vol_7d") # cross-token vol + + if interactions: + x_interactions = np.column_stack(interactions).astype(np.float32) + x_all = np.concatenate([x_base, x_interactions], axis=1) + feat_names = base_names + interaction_names + # Extend x_mean/x_std for interaction columns (already standardized → 0/1) + x_mean = np.concatenate([x_mean, np.zeros(len(interactions))]) + x_std = np.concatenate([x_std, np.ones(len(interactions))]) + else: + x_all = x_base + feat_names = base_names # Targets y_total = np.array([vol_matrix[sample_days[s], sample_pools[s]] @@ -176,7 +203,12 @@ def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30), def make_loss_fn(pool_coeffs, pool_gas, n_pools): - """Loss function with learnable cadence + linear noise model.""" + """Loss function with learnable cadence + linear noise model. + + Supports both shared coefficients (noise_coeffs shape: (n_feat,)) and + per-pool coefficients (noise_coeffs shape: (n_pools, n_feat)). + Detected at trace time from the array shape. + """ from quantammsim.calibration.grid_interpolation import interpolate_pool_daily def loss_fn(params, x, y_total, sample_grid_days, pool_idx, @@ -184,8 +216,15 @@ def loss_fn(params, x, y_total, sample_grid_days, pool_idx, log_cadence = params["log_cadence"] noise_coeffs = params["noise_coeffs"] - # V_noise = exp(x @ noise_coeffs + pool_intercept) - log_v_noise = x @ noise_coeffs + # Per-pool or shared coefficients + if noise_coeffs.ndim == 2: + # Per-pool: (n_pools, n_feat) — gather each sample's pool coeffs + per_sample_coeffs = noise_coeffs[pool_idx] # (n_samples, n_feat) + log_v_noise = jnp.sum(x * per_sample_coeffs, axis=1) + else: + # Shared: (n_feat,) + log_v_noise = x @ noise_coeffs + if "pool_intercepts" in params: log_v_noise = log_v_noise + params["pool_intercepts"][pool_idx] @@ -267,7 +306,12 @@ def evaluate(params, data, label=""): pool_ids = data["pool_ids"] n_pools = data["n_pools"] - log_v_noise = x @ noise_coeffs + if noise_coeffs.ndim == 2: + # Per-pool: (n_pools, n_feat) + per_sample_coeffs = noise_coeffs[pool_idx] + log_v_noise = np.sum(x * per_sample_coeffs, axis=1) + else: + log_v_noise = x @ noise_coeffs if "pool_intercepts" in params: pool_intercepts = np.array(params["pool_intercepts"]) log_v_noise = log_v_noise + pool_intercepts[pool_idx] @@ -349,21 +393,27 @@ def main(): parser.add_argument("--lr", type=float, default=1e-3) parser.add_argument("--l2-alpha", type=float, default=1e-3) parser.add_argument("--huber-delta", type=float, default=1.0) - parser.add_argument("--trend-windows", type=int, nargs="+", default=[7, 14, 30]) + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) parser.add_argument("--no-market", action="store_true", help="x_obs only, no market features") parser.add_argument("--no-cross-pool", action="store_true", help="Reduced x_obs (4) instead of cross-pool (7)") parser.add_argument("--pool-intercepts", action="store_true", help="Per-pool intercept (shared slopes + per-pool bias)") + parser.add_argument("--per-pool", action="store_true", + help="Per-pool noise coefficients (Option A)") + parser.add_argument("--no-split", action="store_true", + help="Train on all data (no temporal holdout)") args = parser.parse_args() os.environ.setdefault("JAX_PLATFORMS", "cpu") print("=" * 70) print("Linear Noise Model + Learnable Cadence") - print(f" market={not args.no_market}, cross_pool={not args.no_cross_pool}" - f", pool_intercepts={args.pool_intercepts}") + mode = "per-pool" if args.per_pool else ( + "shared+intercepts" if args.pool_intercepts else "shared") + print(f" mode={mode}, market={not args.no_market}," + f" cross_pool={not args.no_cross_pool}") print(f" trend_windows={args.trend_windows}") print(f" epochs={args.epochs}, lr={args.lr}, l2={args.l2_alpha}") print("=" * 70) @@ -382,34 +432,69 @@ def main(): f" {data['n_feat']} features, {time.time() - t0:.1f}s") print(f" Features: {data['feat_names']}") - # Temporal split + # Split day_idx = data["day_idx"] - split_day = int(day_idx.max() * 0.7) - train_mask = day_idx <= split_day - eval_mask = day_idx > split_day + n_samples = len(day_idx) + if args.no_split: + train_mask = np.ones(n_samples, dtype=bool) + eval_mask = None + else: + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day train_data = {k: v[train_mask] if isinstance(v, np.ndarray) - and v.shape[0] == len(day_idx) else v + and v.shape[0] == n_samples else v for k, v in data.items()} - eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) - and v.shape[0] == len(day_idx) else v - for k, v in data.items()} + if eval_mask is not None: + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + else: + eval_data = None # Init params n_feat = data["n_feat"] n_pools = data["n_pools"] - params = { - "log_cadence": jnp.array(data["init_log_cadences"]), - "noise_coeffs": jnp.zeros(n_feat), - } - - # Warm-start noise_coeffs via OLS on train: y_total ≈ x @ coeffs x_trn = data["x"][train_mask] y_trn = data["y_total"][train_mask] - sol, _, _, _ = np.linalg.lstsq(x_trn, y_trn, rcond=None) - params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) - - if args.pool_intercepts: + pool_idx_trn = data["pool_idx"][train_mask] + + if args.per_pool: + # Per-pool coefficients: (n_pools, n_feat) + # Warm-start each pool via per-pool Ridge (not OLS — avoids blowup + # on pools with few samples or near-singular features) + from sklearn.linear_model import RidgeCV + coeffs_init = np.zeros((n_pools, n_feat), dtype=np.float32) + # Shared Ridge as fallback + ridge_shared = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge_shared.fit(x_trn, y_trn) + for i in range(n_pools): + mask_i = pool_idx_trn == i + if mask_i.sum() >= 20: + ridge_i = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge_i.fit(x_trn[mask_i], y_trn[mask_i]) + coeffs_init[i] = ridge_i.coef_ + coeffs_init[i, 0] += ridge_i.intercept_ # fold intercept into xobs_0 + else: + coeffs_init[i] = ridge_shared.coef_ + coeffs_init[i, 0] += ridge_shared.intercept_ + params = { + "log_cadence": jnp.array(data["init_log_cadences"]), + "noise_coeffs": jnp.array(coeffs_init), + } + print(f"\n Per-pool coefficients: {n_pools} × {n_feat} = {n_pools * n_feat} params") + print(f" Ridge warm-start |coeffs|={np.mean(np.abs(coeffs_init)):.3f}") + else: + params = { + "log_cadence": jnp.array(data["init_log_cadences"]), + "noise_coeffs": jnp.zeros(n_feat), + } + # Warm-start noise_coeffs via OLS on train + sol, _, _, _ = np.linalg.lstsq(x_trn, y_trn, rcond=None) + params["noise_coeffs"] = jnp.array(sol.astype(np.float32)) + + if args.pool_intercepts and not args.per_pool: # Init per-pool intercepts from OLS residuals ols_pred = x_trn @ sol ols_resid = y_trn - ols_pred @@ -426,10 +511,8 @@ def main(): print(f"\n Init cadence: {np.exp(data['init_log_cadences']).min():.1f}" f"-{np.median(np.exp(data['init_log_cadences'])):.1f}" f"-{np.exp(data['init_log_cadences']).max():.1f} min") - print(f" OLS warm-start |coeffs|={np.mean(np.abs(sol)):.3f}") - print(f" Total params: {sum(v.size for v in params.values())}" - f" ({n_feat} coeffs + {n_pools} cadences" - f"{'+ ' + str(n_pools) + ' intercepts' if args.pool_intercepts else ''})") + total_params = sum(v.size for v in params.values()) + print(f" Total params: {total_params}") # Build loss and train grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], data["n_pools"]) @@ -442,15 +525,65 @@ def main(): # Print learned coefficients nc = np.array(params["noise_coeffs"]) - print(f"\n Noise coefficients ({len(nc)}):") - for i, name in enumerate(data["feat_names"]): - print(f" {name:30s} {nc[i]:+8.4f}") + if nc.ndim == 2: + # Per-pool: print median coefficient across pools + print(f"\n Per-pool noise coefficients — median across {n_pools} pools:") + for i, name in enumerate(data["feat_names"]): + vals = nc[:, i] + print(f" {name:30s} med={np.median(vals):+7.3f}" + f" [{vals.min():+7.3f}, {vals.max():+7.3f}]") + else: + print(f"\n Noise coefficients ({len(nc)}):") + for i, name in enumerate(data["feat_names"]): + print(f" {name:30s} {nc[i]:+8.4f}") # Evaluate - print("\n --- Train ---") - evaluate(params, train_data) - print("\n --- Eval ---") - evaluate(params, eval_data) + if eval_data is not None: + print("\n --- Train ---") + evaluate(params, train_data) + print("\n --- Eval ---") + evaluate(params, eval_data) + else: + print("\n --- All data ---") + evaluate(params, train_data) + + # Save artifact + artifact_dir = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", + ) + os.makedirs(artifact_dir, exist_ok=True) + artifact = { + "noise_coeffs": np.array(params["noise_coeffs"]), + "log_cadence": np.array(params["log_cadence"]), + "init_log_cadences": data["init_log_cadences"], + "feat_names": data["feat_names"], + "pool_ids": data["pool_ids"], + "n_pools": data["n_pools"], + "n_feat": data["n_feat"], + "x_mean": data["x_mean"], + "x_std": data["x_std"], + "hparams": { + "epochs": args.epochs, "lr": args.lr, + "l2_alpha": args.l2_alpha, "huber_delta": args.huber_delta, + "trend_windows": args.trend_windows, + "per_pool": args.per_pool, + "pool_intercepts": args.pool_intercepts, + }, + } + if "pool_intercepts" in params: + artifact["pool_intercepts"] = np.array(params["pool_intercepts"]) + artifact_path = os.path.join(artifact_dir, "model.npz") + np.savez(artifact_path, **{k: v for k, v in artifact.items() + if isinstance(v, np.ndarray)}) + # Save non-array metadata separately + import json + meta_path = os.path.join(artifact_dir, "meta.json") + meta = {k: v for k, v in artifact.items() if not isinstance(v, np.ndarray)} + with open(meta_path, "w") as f: + json.dump(meta, f, indent=2, default=str) + print(f"\n Saved artifact: {artifact_path}") + print(f" Saved metadata: {meta_path}") # Baselines print(f"\n Baselines (eval, total volume R²):") diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index b40154a0..08ac9596 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -347,3 +347,48 @@ def reclamm_calibrated_noise_volume( return jnp.maximum(0.0, daily_vol / 1440.0) +@jit +def reclamm_market_linear_noise_volume( + effective_value_usd, + noise_base, + noise_tvl_coeff, +): + """Market-feature linear noise model with precomputed daily coefficients. + + The full model is:: + + log(V_daily_noise) = base_t + tvl_coeff_t * log(effective_TVL) + + where ``base_t`` absorbs all non-TVL terms (intercept, market regime, + token volatility, pair volatility, day-of-week, cross-pool volumes) + and ``tvl_coeff_t`` is the effective TVL coefficient including + interaction terms (tvl×btc_vol, tvl×tok_a_vol, tvl×pair_vol). + + Both are precomputed daily from the per-pool calibrated noise model + and passed in as dynamic input arrays decimated to the arb frequency. + + Under counterfactual (varying reClAMM concentration), only + ``effective_value_usd`` changes — all market/peer features are held + at observed values via the precomputed arrays. + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + noise_base : float + Precomputed non-TVL component of log(V_daily_noise) for this step. + noise_tvl_coeff : float + Precomputed effective coefficient on log(TVL) for this step, + including base b_tvl and interaction terms. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + log_tvl = jnp.log(jnp.maximum(effective_value_usd, 1.0)) + log_daily_noise = noise_base + noise_tvl_coeff * log_tvl + daily_noise = jnp.exp(log_daily_noise) + return jnp.maximum(0.0, daily_noise / 1440.0) + + diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index f7b6d19d..58222ada 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -216,46 +216,62 @@ def _resolve_fees(params, run_fingerprint): def _prepare_noise_arrays(self, prices, run_fingerprint, start_index, bout_length, arb_freq, max_len): - """Prepare volatility and dow arrays for noise models that need them. + """Prepare dynamic input arrays for noise models. - Returns (volatility_array, dow_sin_array, dow_cos_array) — all sliced - and decimated to match arb_prices shape. Arrays are None when the - noise model does not require them. + Returns dict with keys depending on noise_model: + - "ratio": {} + - "tsoukalas_*"/"loglinear": {"volatility": array} + - "calibrated": {"volatility": array, "dow_sin": array, "dow_cos": array} + - "market_linear": {"noise_base": array, "noise_tvl_coeff": array} """ noise_model = run_fingerprint.get("noise_model", "ratio") + result = {"volatility": None, "dow_sin": None, "dow_cos": None, + "noise_base": None, "noise_tvl_coeff": None} + + if noise_model == "market_linear": + # Precomputed arrays passed via run_fingerprint + nb = run_fingerprint.get("noise_base_array") + ntc = run_fingerprint.get("noise_tvl_coeff_array") + if nb is not None: + result["noise_base"] = _prepare_dynamic_array( + jnp.array(nb), start_index, bout_length, arb_freq, max_len) + if ntc is not None: + result["noise_tvl_coeff"] = _prepare_dynamic_array( + jnp.array(ntc), start_index, bout_length, arb_freq, max_len) + return result + needs_vol = noise_model in ( "tsoukalas_sqrt", "tsoukalas_log", "loglinear", "calibrated", ) if not needs_vol: - return None, None, None + return result volatility_array = self.calculate_volatility_array( prices, run_fingerprint, ) - vol_prepared = _prepare_dynamic_array( + result["volatility"] = _prepare_dynamic_array( volatility_array, start_index, bout_length, arb_freq, max_len, ) if noise_model != "calibrated": - return vol_prepared, None, None + return result # Day-of-week sin/cos arrays for the calibrated noise model. - # Compute from startDateString + minute offsets (vectorized). import pandas as pd start_dt = pd.Timestamp(run_fingerprint["startDateString"]) n_minutes = prices.shape[0] day_indices = np.arange(n_minutes) // 1440 - start_weekday = start_dt.weekday() # Monday=0 .. Sunday=6 + start_weekday = start_dt.weekday() weekdays = ((start_weekday + day_indices) % 7).astype(np.float64) dow_sin_full = jnp.array(np.sin(2.0 * np.pi * weekdays / 7.0)) dow_cos_full = jnp.array(np.cos(2.0 * np.pi * weekdays / 7.0)) - dow_sin_prepared = _prepare_dynamic_array( + result["dow_sin"] = _prepare_dynamic_array( dow_sin_full, start_index, bout_length, arb_freq, max_len, ) - dow_cos_prepared = _prepare_dynamic_array( + result["dow_cos"] = _prepare_dynamic_array( dow_cos_full, start_index, bout_length, arb_freq, max_len, ) - return vol_prepared, dow_sin_prepared, dow_cos_prepared + return result @partial(jit, static_argnums=(2,)) def calculate_reserves_with_fees( @@ -284,10 +300,13 @@ def calculate_reserves_with_fees( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + _na = self._prepare_noise_arrays( prices, run_fingerprint, start_index, bout_length, arb_freq, s.arb_prices.shape[0], ) + arb_vol = _na["volatility"] + dow_sin = _na["dow_sin"] + dow_cos = _na["dow_cos"] if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_with_fees( @@ -312,6 +331,8 @@ def calculate_reserves_with_fees( volatility_array=arb_vol, dow_sin_array=dow_sin, dow_cos_array=dow_cos, + noise_base_array=_na["noise_base"], + noise_tvl_coeff_array=_na["noise_tvl_coeff"], ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -350,10 +371,13 @@ def calculate_reserves_and_fee_revenue_with_fees( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + _na = self._prepare_noise_arrays( prices, run_fingerprint, start_index, bout_length, arb_freq, s.arb_prices.shape[0], ) + arb_vol = _na["volatility"] + dow_sin = _na["dow_sin"] + dow_cos = _na["dow_cos"] if run_fingerprint["do_arb"]: return _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -378,6 +402,8 @@ def calculate_reserves_and_fee_revenue_with_fees( volatility_array=arb_vol, dow_sin_array=dow_sin, dow_cos_array=dow_cos, + noise_base_array=_na["noise_base"], + noise_tvl_coeff_array=_na["noise_tvl_coeff"], ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 713923c9..0953d33b 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -36,6 +36,7 @@ reclamm_tsoukalas_log_noise_volume, reclamm_loglinear_noise_volume, reclamm_calibrated_noise_volume, + reclamm_market_linear_noise_volume, ) # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) @@ -1022,6 +1023,20 @@ def _skip_schedule_state(_): arb_volume, dow_sin, dow_cos, _np, ) + noise_fee_income = (1.0 - gamma) * noise_vol + scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) + Ra_new = Ra_new * scale + Rb_new = Rb_new * scale + elif noise_model == "market_linear": + noise_base = input_list[9] + noise_tvl_coeff = input_list[10] + real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) + effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + + noise_vol = reclamm_market_linear_noise_volume( + effective_value, noise_base, noise_tvl_coeff, + ) + noise_fee_income = (1.0 - gamma) * noise_vol scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) Ra_new = Ra_new * scale @@ -1281,6 +1296,8 @@ def _jax_calc_reclamm_reserves_with_fees( volatility_array=None, dow_sin_array=None, dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, ): """Calculate reClAMM reserves over time with fees. @@ -1343,6 +1360,9 @@ def _jax_calc_reclamm_reserves_with_fees( scan_inputs.append(volatility_array) scan_inputs.append(dow_sin_array) scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) carry_init = [ initial_reserves, @@ -1386,6 +1406,8 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( volatility_array=None, dow_sin_array=None, dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" if lp_supply_array is None: @@ -1457,6 +1479,9 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( scan_inputs.append(volatility_array) scan_inputs.append(dow_sin_array) scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) carry_init = [ initial_reserves, @@ -1590,6 +1615,8 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( volatility_array=None, dow_sin_array=None, dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1654,6 +1681,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( scan_inputs.append(volatility_array) scan_inputs.append(dow_sin_array) scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) carry_init = [ initial_reserves, @@ -1697,6 +1727,8 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( volatility_array=None, dow_sin_array=None, dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1774,6 +1806,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( scan_inputs.append(volatility_array) scan_inputs.append(dow_sin_array) scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) carry_init = [ initial_reserves, diff --git a/scripts/plot_calibrated_vs_real.py b/scripts/plot_calibrated_vs_real.py new file mode 100644 index 00000000..475422b5 --- /dev/null +++ b/scripts/plot_calibrated_vs_real.py @@ -0,0 +1,241 @@ +"""Plot predicted vs real volume using saved per-pool calibrated noise model. + +Loads the artifact from experiments/run_linear_market_noise.py (--per-pool), +rebuilds the features, evaluates V_arb(learned cadence) + V_noise(x @ coeffs_i), +and generates stacked area plots showing the arb/noise decomposition per pool. + +Usage: + # First train and save: + python experiments/run_linear_market_noise.py --per-pool --no-split --epochs 2000 + + # Then plot: + python scripts/plot_calibrated_vs_real.py + python scripts/plot_calibrated_vs_real.py --artifact results/linear_market_noise/model.npz +""" + +import argparse +import json +import os +import sys +import time + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +sys.path.insert(0, os.path.dirname(os.path.dirname(__file__))) + +ARTIFACT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "calibrated_vs_real", +) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--artifact", default=os.path.join(ARTIFACT_DIR, "model.npz")) + parser.add_argument("--meta", default=os.path.join(ARTIFACT_DIR, "meta.json")) + parser.add_argument("--output-dir", default=OUTPUT_DIR) + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + os.makedirs(args.output_dir, exist_ok=True) + + # ---- Load artifact ---- + print(f"Loading artifact: {args.artifact}") + art = np.load(args.artifact, allow_pickle=True) + noise_coeffs = art["noise_coeffs"] + log_cadence = art["log_cadence"] + init_log_cadences = art["init_log_cadences"] + + with open(args.meta) as f: + meta = json.load(f) + feat_names = meta["feat_names"] + pool_ids = meta["pool_ids"] + n_pools = meta["n_pools"] + hparams = meta["hparams"] + per_pool = noise_coeffs.ndim == 2 + + print(f" {n_pools} pools, {len(feat_names)} features, per_pool={per_pool}") + print(f" hparams: {hparams}") + + # ---- Rebuild data (features only, no training) ---- + import jax.numpy as jnp + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + from experiments.run_linear_market_noise import load_stage1, build_data + + matched_clean, option_c_clean = load_stage1() + + print("\nRebuilding features...") + t0 = time.time() + data = build_data( + matched_clean, option_c_clean, + trend_windows=tuple(hparams["trend_windows"]), + include_market=True, include_cross_pool=True, + ) + print(f" {len(data['pool_idx'])} samples, {time.time() - t0:.1f}s") + + x = data["x"] + y_total = data["y_total"] + pool_idx = data["pool_idx"] + day_idx = data["day_idx"] + sgd = data["sample_grid_days"] + + # ---- Compute predictions ---- + if per_pool: + per_sample_coeffs = noise_coeffs[pool_idx] + log_v_noise = np.sum(x * per_sample_coeffs, axis=1) + else: + log_v_noise = x @ noise_coeffs + + if "pool_intercepts" in art: + log_v_noise = log_v_noise + art["pool_intercepts"][pool_idx] + + v_noise = np.exp(log_v_noise) + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd[mask]] + + v_obs = np.exp(y_total) + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total_log = np.logaddexp(log_v_arb, log_v_noise) + + # ---- Reconstruct dates ---- + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + + # ---- Plot: stacked area per pool ---- + per_page = 9 + n_pages = (n_pools + per_page - 1) // per_page + + for page in range(n_pages): + start = page * per_page + end = min(start + per_page, n_pools) + n_this = end - start + + ncols = 3 + nrows = (n_this + ncols - 1) // ncols + fig, axes = plt.subplots(nrows, ncols, figsize=(16, 4 * nrows)) + if nrows == 1: + axes = axes.reshape(1, -1) + + for idx, i in enumerate(range(start, end)): + ax = axes[idx // ncols][idx % ncols] + mask = pool_idx == i + if mask.sum() < 5: + ax.set_visible(False) + continue + + days = day_idx[mask] + dates = [pd.Timestamp(date_list[d]) for d in days] + vo = v_obs[mask] + va = v_arb[mask] + vn = v_noise[mask] + + ax.fill_between(dates, 0, va, alpha=0.3, color="steelblue", + label="V_arb") + ax.fill_between(dates, va, va + vn, alpha=0.3, color="coral", + label="V_noise") + ax.plot(dates, vo, "k-", linewidth=0.8, alpha=0.7, label="V_obs") + ax.plot(dates, va + vn, "--", color="darkred", linewidth=0.8, + alpha=0.7, label="V_pred") + + ax.set_yscale("log") + ax.set_ylabel("USD/day", fontsize=7) + ax.tick_params(labelsize=6) + ax.tick_params(axis="x", rotation=30) + + yt = y_total[mask] + pt = pred_total_log[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] + tokens = matched_clean[pid]["tokens"] + chain = matched_clean[pid]["chain"] + ci = np.exp(init_log_cadences[i]) + cl = np.exp(log_cadence[i]) + arb_share = np.median(va / vo) * 100 + noise_share = np.median(vn / vo) * 100 + b_tvl = noise_coeffs[i, 1] if per_pool else noise_coeffs[1] + + ax.set_title( + f"{tokens} ({chain})\n" + f"R\u00b2={r2:.3f} cad={ci:.0f}\u2192{cl:.0f}min " + f"arb={arb_share:.0f}% noise={noise_share:.0f}% " + f"b_tvl={b_tvl:.2f}", + fontsize=7) + ax.legend(fontsize=6, loc="upper right") + + for idx in range(n_this, nrows * ncols): + axes[idx // ncols][idx % ncols].set_visible(False) + + fig.suptitle( + f"Per-pool calibrated noise model \u2014 V_arb + V_noise " + f"(page {page+1}/{n_pages})", fontsize=10) + fig.tight_layout() + out = os.path.join(args.output_dir, f"calibrated_page{page+1}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + # ---- Summary ---- + summary = [] + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total_log[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] + va = v_arb[mask] + vn = v_noise[mask] + vo = v_obs[mask] + + summary.append({ + "pool_id": pid, + "tokens": matched_clean[pid]["tokens"], + "chain": matched_clean[pid]["chain"], + "n_obs": int(mask.sum()), + "R2": r2, + "cadence_init": float(np.exp(init_log_cadences[i])), + "cadence_learned": float(np.exp(log_cadence[i])), + "median_arb_pct": float(np.median(va / vo) * 100), + "median_noise_pct": float(np.median(vn / vo) * 100), + "b_tvl": float(noise_coeffs[i, 1] if per_pool else noise_coeffs[1]), + }) + + summary_df = pd.DataFrame(summary) + csv_path = os.path.join(args.output_dir, "summary.csv") + summary_df.to_csv(csv_path, index=False) + print(f"\n Summary: {csv_path}") + print(f" Median R\u00b2: {summary_df['R2'].median():.4f}") + print(f" Median arb: {summary_df['median_arb_pct'].median():.0f}%") + print(f" b_tvl: [{summary_df['b_tvl'].min():.2f}," + f" {summary_df['b_tvl'].max():.2f}]," + f" median={summary_df['b_tvl'].median():.2f}") + + +if __name__ == "__main__": + main() From 825c8a7f2d5973bfc379def6c576cf3d81e6902e Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 23 Mar 2026 13:08:01 +0000 Subject: [PATCH 065/115] feat: simulator integration for market_linear noise model, TVL standardization fix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - noise_model_arrays.py: new module to precompute noise_base and noise_tvl_coeff arrays from trained artifact. Decomposes per-pool coefficients into TVL-dependent and TVL-independent components. Returns tvl_mean/tvl_std for runtime standardization. - noise_trades.py: add tvl_mean/tvl_std params to reclamm_market_linear_noise_volume() — standardizes log(TVL) at runtime to match training scale. Fixes NaN blowup from raw TVL. - reclamm_reserves.py: pass noise_params (tvl_mean, tvl_std) through to market_linear dispatch. - reclamm.py: cache loaded noise arrays on pool instance to avoid repeated disk reads. Support noise_arrays_path in fingerprint. - tune_reclamm_calibrated_noise.py: add --noise-model flag (calibrated vs market_linear), --artifact-dir, --initial-pool-value. Save precomputed arrays to disk, pass path + tvl stats via fingerprint. Default dates adjusted to panel coverage period. 100 trials × 3 objectives all complete successfully. - plot_reclamm_optuna_result.py: forward noise_arrays_path in run_full_period() for market_linear re-runs. --- experiments/tune_reclamm_calibrated_noise.py | 259 ++++++++++++++++ quantammsim/calibration/noise_model_arrays.py | 285 ++++++++++++++++++ quantammsim/pools/noise_trades.py | 21 +- quantammsim/pools/reCLAMM/reclamm.py | 9 +- quantammsim/pools/reCLAMM/reclamm_reserves.py | 3 + scripts/plot_reclamm_optuna_result.py | 241 +++++++++------ 6 files changed, 716 insertions(+), 102 deletions(-) create mode 100644 experiments/tune_reclamm_calibrated_noise.py create mode 100644 quantammsim/calibration/noise_model_arrays.py diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py new file mode 100644 index 00000000..4eda1a6b --- /dev/null +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -0,0 +1,259 @@ +"""Optuna tuning of reClAMM pool parameters with calibrated noise models. + +Supports two noise model modes: + --noise-model calibrated (legacy 8-covariate model) + --noise-model market_linear (new per-pool model with market features) + +The market_linear model uses precomputed daily arrays from the per-pool +calibrated noise model artifact (results/linear_market_noise/). It evaluates: + + log(V_noise) = base_t + tvl_coeff_t * log(effective_TVL) + +where base_t absorbs all non-TVL terms (market regime, token volatility, +pair volatility, day-of-week, cross-pool volumes) and tvl_coeff_t is the +effective TVL coefficient including interaction terms. + +Pool: 0x9d1fcf346ea1b0 = AAVE/WETH Mainnet + +Usage: + cd + source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public + + # New market_linear model (default) + python experiments/tune_reclamm_calibrated_noise.py + + # Legacy 8-covariate model + python experiments/tune_reclamm_calibrated_noise.py --noise-model calibrated + + # All three objectives + python experiments/tune_reclamm_calibrated_noise.py --all-objectives + + # More trials + python experiments/tune_reclamm_calibrated_noise.py --n-trials 200 +""" + +import argparse +import json +import math +import numpy as np +from pathlib import Path +from quantammsim.runners.jax_runners import train_on_historic_data + +POOL_ID = "0x9d1fcf346ea1b0" # AAVE/WETH Mainnet + +# --- Legacy 8-covariate noise coefficients --- +NOISE_COEFFS_LEGACY = [ + -0.453, # c_0: intercept + 0.025, # c_1: log(TVL) + -0.060, # c_2: log(sigma) + 0.310, # c_3: log(TVL) * log(sigma) + -0.149, # c_4: log(TVL) * fee + 0.359, # c_5: log(sigma) * fee + 0.061, # c_6: dow_sin + 0.060, # c_7: dow_cos +] +LEGACY_LOG_CADENCE = 2.68 +LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) # ~15 min + +PARAMETER_CONFIG = { + "price_ratio": {"low": 1.01, "high": 200.0, "log_scale": True, "scalar": True}, + "centeredness_margin": {"low": 0.01, "high": 0.99, "scalar": True}, + "shift_exponent": {"low": 1e-5, "high": 125.0, "log_scale": True, "scalar": True}, +} + +OBJECTIVES = ["daily_log_sharpe", "returns_over_hodl", "fee_revenue_over_value"] + + +def _build_market_linear_arrays(args): + """Precompute noise arrays from the per-pool market noise model artifact.""" + from quantammsim.calibration.noise_model_arrays import build_simulator_arrays + + # Parse dates — strip time component for the array builder + start = args.start_date.split(" ")[0] + end = args.end_test_date.split(" ")[0] + + print(f" Building market_linear noise arrays for {POOL_ID}...") + print(f" Date range: {start} → {end}") + arrays = build_simulator_arrays( + pool_id=POOL_ID, + start_date=start, + end_date=end, + artifact_dir=args.artifact_dir, + ) + print(f" {arrays['n_days']} days, {arrays['n_minutes']} minutes") + print(f" noise_base range: [{arrays['noise_base'].min():.2f}," + f" {arrays['noise_base'].max():.2f}]") + print(f" noise_tvl_coeff range: [{arrays['noise_tvl_coeff'].min():.4f}," + f" {arrays['noise_tvl_coeff'].max():.4f}]") + + # Save arrays to disk (fingerprint can't hold numpy arrays — it gets JSON-serialized) + import os + cache_dir = os.path.join(args.artifact_dir, "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join(cache_dir, f"{POOL_ID}_{start}_{end}.npz") + np.savez(arrays_path, + noise_base=arrays["noise_base"], + noise_tvl_coeff=arrays["noise_tvl_coeff"], + tvl_mean=arrays["tvl_mean"], + tvl_std=arrays["tvl_std"]) + print(f" Saved arrays: {arrays_path}") + + # Get learned cadence from artifact + from quantammsim.calibration.noise_model_arrays import load_artifact, _find_pool_index + art, meta = load_artifact(args.artifact_dir) + pool_idx = _find_pool_index(POOL_ID, meta["pool_ids"]) + if pool_idx >= 0: + learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) + print(f" Learned cadence: {learned_cadence:.1f} min") + else: + learned_cadence = 5.0 + print(f" Pool not in calibration set, using default cadence: {learned_cadence}") + + return arrays_path, max(1, round(learned_cadence)) + + +def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): + """Build run fingerprint with calibrated noise model.""" + if args.noise_model == "market_linear" and noise_arrays_path is not None: + # Load tvl standardization stats from the saved arrays + _arr = np.load(noise_arrays_path) + noise_block = { + "noise_trader_ratio": 0.0, + "noise_model": "market_linear", + "noise_arrays_path": noise_arrays_path, + "reclamm_noise_params": { + "tvl_mean": float(_arr["tvl_mean"]), + "tvl_std": float(_arr["tvl_std"]), + }, + } + freq = arb_freq or 5 + else: + noise_block = { + "noise_trader_ratio": 0.0, + "noise_model": "calibrated", + "reclamm_noise_params": { + f"c_{i}": NOISE_COEFFS_LEGACY[i] for i in range(8) + }, + } + freq = LEGACY_ARB_FREQUENCY + + return { + "rule": "reclamm", + "tokens": ["AAVE", "ETH"], + "startDateString": args.start_date, + "endDateString": args.end_date, + "endTestDateString": args.end_test_date, + "initial_pool_value": args.initial_pool_value, + "do_arb": True, + "arb_frequency": freq, + "fees": args.fees, + "gas_cost": args.gas_cost, + "arb_fees": 0.0, + "protocol_fee_split": 0.5, + **noise_block, + "return_val": objective, + "reclamm_interpolation_method": args.interpolation, + "reclamm_centeredness_scaling": args.centeredness_scaling, + "reclamm_learn_arc_length_speed": False, + "reclamm_use_shift_exponent": True, + **({"bout_offset": args.bout_offset} if args.bout_offset is not None else {}), + "optimisation_settings": { + "method": "optuna", + "n_parameter_sets": 1, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + "optuna_settings": { + "make_scalar": True, + "expand_around": False, + "n_trials": args.n_trials, + "multi_objective": False, + "parameter_config": PARAMETER_CONFIG, + **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), + }, + }, + } + + +def run_single(objective, args, noise_arrays_path=None, arb_freq=None): + """Run Optuna tuning for a single objective.""" + print(f"\n{'='*60}") + print(f" Objective: {objective}") + print(f" Noise model: {args.noise_model}") + print(f" Pool: AAVE/WETH Mainnet ({POOL_ID})") + print(f" Train: {args.start_date} → {args.end_date}") + print(f" Test: {args.end_date} → {args.end_test_date}") + if arb_freq: + print(f" Arb frequency: {arb_freq} min (learned)") + print(f"{'='*60}\n") + + fp = build_fingerprint(objective, args, noise_arrays_path, arb_freq) + result = train_on_historic_data(fp, verbose=True) + + if result is not None: + print(f"\n=== Result ({objective}) ===") + for k, v in result.items(): + print(f" {k}: {v}") + + return result + + +def main(): + parser = argparse.ArgumentParser( + description="Tune reClAMM params with calibrated 8-covariate noise model" + ) + parser.add_argument("--n-trials", type=int, default=50) + parser.add_argument("--noise-model", default="market_linear", + choices=["calibrated", "market_linear"], + help="Noise model variant") + parser.add_argument("--artifact-dir", + default="results/linear_market_noise", + help="Artifact dir for market_linear model") + parser.add_argument("--initial-pool-value", type=float, default=20_000_000.0, + help="Initial pool TVL in USD (default: 20M)") + parser.add_argument("--fees", type=float, default=0.0025, + help="Pool fee rate (default: 0.0025 matching calibration)") + parser.add_argument("--gas-cost", type=float, default=1.0) + parser.add_argument("--objective", default="fee_revenue_over_value", + choices=OBJECTIVES) + parser.add_argument("--all-objectives", action="store_true", + help="Run all three objectives sequentially") + parser.add_argument("--interpolation", default="geometric", + choices=["geometric", "constant_arc_length"]) + parser.add_argument("--centeredness-scaling", action="store_true") + parser.add_argument("--start-date", default="2025-08-03 00:00:00") + parser.add_argument("--end-date", default="2025-12-01 00:00:00", + help="End of training / start of test") + parser.add_argument("--end-test-date", default="2026-02-18 00:00:00", + help="End of test (latest available data)") + parser.add_argument("--bout-offset", type=int, default=None) + parser.add_argument("--val-fraction", type=float, default=None) + parser.add_argument("--overfitting-penalty", type=float, default=None) + parser.add_argument("--output", type=str, default=None, + help="Save results to JSON file") + args = parser.parse_args() + + if args.all_objectives: + objectives = OBJECTIVES + else: + objectives = [args.objective] + + # Precompute noise arrays once (if using market_linear) + noise_arrays_path = None + arb_freq = None + if args.noise_model == "market_linear": + noise_arrays_path, arb_freq = _build_market_linear_arrays(args) + + all_results = {} + for obj in objectives: + result = run_single(obj, args, noise_arrays_path, arb_freq) + all_results[obj] = result + + if args.output: + out_path = Path(args.output) + out_path.parent.mkdir(parents=True, exist_ok=True) + with open(out_path, "w") as f: + json.dump(all_results, f, indent=2, default=str) + print(f"\nResults saved to {out_path}") + + +if __name__ == "__main__": + main() diff --git a/quantammsim/calibration/noise_model_arrays.py b/quantammsim/calibration/noise_model_arrays.py new file mode 100644 index 00000000..b8798c37 --- /dev/null +++ b/quantammsim/calibration/noise_model_arrays.py @@ -0,0 +1,285 @@ +"""Precompute noise_base and noise_tvl_coeff arrays for the simulator. + +Takes a trained per-pool noise model artifact and produces the two daily +arrays needed by reclamm_market_linear_noise_volume(): + + log(V_daily_noise) = noise_base_t + noise_tvl_coeff_t * log(effective_TVL) + +The arrays are at daily resolution and need to be expanded to minute-level +(by repeating each day's value 1440 times) before passing to the simulator. + +Usage: + from quantammsim.calibration.noise_model_arrays import build_simulator_arrays + + arrays = build_simulator_arrays( + pool_id="0x0b09dea16768f0", + start_date="2025-06-01", + end_date="2026-03-01", + artifact_dir="results/linear_market_noise", + ) + # arrays["noise_base"] — (n_minutes,) float64 + # arrays["noise_tvl_coeff"] — (n_minutes,) float64 +""" + +import json +import os +from datetime import date, timedelta +from typing import Dict, Optional, Tuple + +import numpy as np +import pandas as pd + + +def load_artifact(artifact_dir: str) -> Tuple[dict, dict]: + """Load model.npz + meta.json from artifact directory.""" + art = np.load(os.path.join(artifact_dir, "model.npz"), allow_pickle=True) + with open(os.path.join(artifact_dir, "meta.json")) as f: + meta = json.load(f) + return dict(art), meta + + +def _find_pool_index(pool_id: str, pool_ids: list) -> int: + """Match pool_id (full or prefix) to calibration pool list.""" + for i, cid in enumerate(pool_ids): + if pool_id.startswith(cid) or cid.startswith(pool_id): + return i + return -1 + + +def _identify_tvl_columns(feat_names: list) -> Tuple[int, list]: + """Identify which feature columns involve TVL. + + Returns: + tvl_col: index of the pure log_tvl feature (xobs_1) + tvl_interaction_cols: list of (col_idx, paired_col_idx) for + interaction terms that multiply TVL with another feature + """ + tvl_col = None + tvl_interaction_cols = [] + + for i, name in enumerate(feat_names): + if name == "xobs_1": + tvl_col = i + elif "xobs_1×" in name: + # e.g. "xobs_1×btc_realized_vol_7d" — find the paired feature + paired_name = name.split("×")[1] + for j, n2 in enumerate(feat_names): + if n2 == paired_name: + tvl_interaction_cols.append((i, j)) + break + + if tvl_col is None: + raise ValueError("xobs_1 (log_tvl) not found in feature names") + + return tvl_col, tvl_interaction_cols + + +def build_daily_features( + pool_id: str, + matched_clean: dict, + start_date: str, + end_date: str, + feat_names: list, + x_mean: np.ndarray, + x_std: np.ndarray, + trend_windows: tuple = (7,), +) -> Tuple[np.ndarray, list]: + """Build the full standardized feature matrix for a pool over a date range. + + Returns (x_daily, dates) where x_daily is (n_days, n_feat) and dates + is the list of dates. TVL column (xobs_1) is filled with the pool's + observed log_tvl_lag1 where available, 0 otherwise. + """ + from quantammsim.calibration.pool_data import ( + build_x_obs, build_cross_pool_x_obs, K_OBS_CROSS, + ) + from quantammsim.calibration.market_features import ( + build_pool_market_features, + ) + + # Find the pool + pid_match = None + for pid in matched_clean: + if pool_id.startswith(pid) or pid.startswith(pool_id): + pid_match = pid + break + if pid_match is None: + raise ValueError(f"Pool {pool_id} not found in matched_clean") + + entry = matched_clean[pid_match] + panel = entry["panel"] + + # Filter panel to date range + start = pd.Timestamp(start_date) + end = pd.Timestamp(end_date) + panel_dates = pd.to_datetime(panel["date"]) + mask = (panel_dates >= start) & (panel_dates <= end) + panel_sub = panel[mask.values].copy() + n_days = len(panel_sub) + + if n_days < 2: + raise ValueError(f"Only {n_days} days in range for pool {pool_id}") + + dates = panel_sub["date"].values + + # x_obs (cross-pool, 7 features) — need at least 1 lag + xc = build_cross_pool_x_obs(panel_sub, matched_clean, pid_match) + # xc drops first row; align + if len(xc) < n_days: + # Pad first row with zeros + xc = np.vstack([np.zeros((1, xc.shape[1])), xc]) + + # Market features + pool_feat = build_pool_market_features( + matched_clean, trend_windows=list(trend_windows)) + pf = pool_feat.get(pid_match) + if pf is None: + raise ValueError(f"No market features for {pool_id}") + + # Align market features to panel dates + n_base = K_OBS_CROSS + market_cols = [c for c in sorted(pf.columns)] + n_market = len(market_cols) + + x_base = np.zeros((n_days, n_base + n_market), dtype=np.float32) + x_base[:, :n_base] = xc[:n_days] + + for k, d in enumerate(dates): + day = pd.Timestamp(d).normalize() + if day in pf.index: + for m, col in enumerate(market_cols): + val = pf.loc[day, col] + if np.isfinite(val): + x_base[k, n_base + m] = val + + # Standardize using saved stats (base features only) + n_base_total = n_base + n_market + x_base = ((x_base - x_mean[:n_base_total]) / x_std[:n_base_total]).astype(np.float32) + + # Interaction terms + base_names = [f"xobs_{i}" for i in range(n_base)] + market_cols + col_idx = {name: i for i, name in enumerate(base_names)} + + interactions = [] + for fname in feat_names[n_base_total:]: + if "×" in fname: + parts = fname.split("×") + if parts[0] in col_idx and parts[1] in col_idx: + interactions.append( + x_base[:, col_idx[parts[0]]] * x_base[:, col_idx[parts[1]]]) + else: + interactions.append(np.zeros(n_days, dtype=np.float32)) + else: + interactions.append(np.zeros(n_days, dtype=np.float32)) + + if interactions: + x_all = np.concatenate( + [x_base, np.column_stack(interactions)], axis=1).astype(np.float32) + else: + x_all = x_base + + return x_all, dates.tolist() + + +def build_simulator_arrays( + pool_id: str, + start_date: str, + end_date: str, + artifact_dir: str = "results/linear_market_noise", + matched_clean: Optional[dict] = None, + arb_frequency: int = 1, +) -> Dict[str, np.ndarray]: + """Build noise_base and noise_tvl_coeff arrays for the simulator. + + Parameters + ---------- + pool_id : str + Pool ID (full or prefix). + start_date, end_date : str + Date range (inclusive). + artifact_dir : str + Directory containing model.npz and meta.json. + matched_clean : dict, optional + Pre-loaded matched_clean dict. If None, loads from stage1.pkl. + arb_frequency : int + Arb frequency in minutes. Arrays are at minute resolution, + repeated from daily values. + + Returns + ------- + dict with: + noise_base : (n_minutes,) array + noise_tvl_coeff : (n_minutes,) array + dates : list of dates + pool_index : int (index in calibration set, or -1) + """ + art, meta = load_artifact(artifact_dir) + noise_coeffs = art["noise_coeffs"] + feat_names = meta["feat_names"] + pool_ids = meta["pool_ids"] + x_mean = art["x_mean"] + x_std = art["x_std"] + per_pool = noise_coeffs.ndim == 2 + trend_windows = tuple(meta["hparams"]["trend_windows"]) + + # Load matched_clean if needed + if matched_clean is None: + import pickle + cache_dir = os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname( + os.path.abspath(__file__)))), + "results", "token_factored_calibration", "_cache", + ) + with open(os.path.join(cache_dir, "stage1.pkl"), "rb") as f: + data = pickle.load(f) + matched_clean = data["matched_clean"] + + # Find pool coefficients + pool_idx = _find_pool_index(pool_id, pool_ids) + if pool_idx >= 0 and per_pool: + coeffs = noise_coeffs[pool_idx] + elif per_pool: + print(f" Warning: pool {pool_id} not in calibration set, using median coeffs") + coeffs = np.median(noise_coeffs, axis=0) + else: + coeffs = noise_coeffs + + # Build daily features + x_daily, dates = build_daily_features( + pool_id, matched_clean, start_date, end_date, + feat_names, x_mean, x_std, trend_windows, + ) + n_days = len(dates) + + # Decompose into base (non-TVL) and tvl_coeff + tvl_col, tvl_interactions = _identify_tvl_columns(feat_names) + + # tvl_coeff_t = coeffs[tvl_col] + sum(coeffs[inter_col] * x[paired_col]) + tvl_coeff_daily = np.full(n_days, coeffs[tvl_col], dtype=np.float64) + for inter_col, paired_col in tvl_interactions: + tvl_coeff_daily += coeffs[inter_col] * x_daily[:, paired_col] + + # base_t = sum(coeffs[j] * x[j]) for j not in {tvl_col, interaction_cols} + tvl_related = {tvl_col} | {ic for ic, _ in tvl_interactions} + base_daily = np.zeros(n_days, dtype=np.float64) + for j in range(len(feat_names)): + if j not in tvl_related: + base_daily += coeffs[j] * x_daily[:, j] + + # Expand to minute resolution: each day's value repeats 1440 times + n_minutes = n_days * 1440 + noise_base = np.repeat(base_daily, 1440) + noise_tvl_coeff = np.repeat(tvl_coeff_daily, 1440) + + return { + "noise_base": noise_base, + "noise_tvl_coeff": noise_tvl_coeff, + "tvl_mean": float(x_mean[tvl_col]), + "tvl_std": float(x_std[tvl_col]), + "dates": dates, + "pool_index": pool_idx, + "n_days": n_days, + "n_minutes": n_minutes, + "coeffs": coeffs, + "tvl_col": tvl_col, + } diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index 08ac9596..bfef86c5 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -352,20 +352,25 @@ def reclamm_market_linear_noise_volume( effective_value_usd, noise_base, noise_tvl_coeff, + tvl_mean=0.0, + tvl_std=1.0, ): """Market-feature linear noise model with precomputed daily coefficients. The full model is:: - log(V_daily_noise) = base_t + tvl_coeff_t * log(effective_TVL) + log(V_daily_noise) = base_t + tvl_coeff_t * standardized_log_tvl where ``base_t`` absorbs all non-TVL terms (intercept, market regime, token volatility, pair volatility, day-of-week, cross-pool volumes) and ``tvl_coeff_t`` is the effective TVL coefficient including interaction terms (tvl×btc_vol, tvl×tok_a_vol, tvl×pair_vol). - Both are precomputed daily from the per-pool calibrated noise model - and passed in as dynamic input arrays decimated to the arb frequency. + The log(TVL) is standardized using the same mean/std from training + to ensure the coefficient scale matches. + + Both base_t and tvl_coeff_t are precomputed daily from the per-pool + calibrated noise model and passed in as dynamic input arrays. Under counterfactual (varying reClAMM concentration), only ``effective_value_usd`` changes — all market/peer features are held @@ -378,8 +383,11 @@ def reclamm_market_linear_noise_volume( noise_base : float Precomputed non-TVL component of log(V_daily_noise) for this step. noise_tvl_coeff : float - Precomputed effective coefficient on log(TVL) for this step, - including base b_tvl and interaction terms. + Precomputed effective coefficient on log(TVL) for this step. + tvl_mean : float + Mean of log(TVL) from training data standardization. + tvl_std : float + Std of log(TVL) from training data standardization. Returns ------- @@ -387,7 +395,8 @@ def reclamm_market_linear_noise_volume( Per-minute noise volume (USD), floored at zero. """ log_tvl = jnp.log(jnp.maximum(effective_value_usd, 1.0)) - log_daily_noise = noise_base + noise_tvl_coeff * log_tvl + standardized_log_tvl = (log_tvl - tvl_mean) / tvl_std + log_daily_noise = noise_base + noise_tvl_coeff * standardized_log_tvl daily_noise = jnp.exp(log_daily_noise) return jnp.maximum(0.0, daily_noise / 1440.0) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 58222ada..32e0c3e8 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -229,9 +229,16 @@ def _prepare_noise_arrays(self, prices, run_fingerprint, start_index, "noise_base": None, "noise_tvl_coeff": None} if noise_model == "market_linear": - # Precomputed arrays passed via run_fingerprint + # Load precomputed arrays from path (cached on instance) or direct nb = run_fingerprint.get("noise_base_array") ntc = run_fingerprint.get("noise_tvl_coeff_array") + if nb is None and "noise_arrays_path" in run_fingerprint: + path = run_fingerprint["noise_arrays_path"] + if not hasattr(self, "_market_linear_cache") or self._market_linear_cache[0] != path: + arrays = np.load(path) + self._market_linear_cache = (path, arrays["noise_base"], arrays["noise_tvl_coeff"]) + nb = self._market_linear_cache[1] + ntc = self._market_linear_cache[2] if nb is not None: result["noise_base"] = _prepare_dynamic_array( jnp.array(nb), start_index, bout_length, arb_freq, max_len) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 0953d33b..26dcd7a7 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -1033,8 +1033,11 @@ def _skip_schedule_state(_): real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + _np = noise_params if noise_params is not None else {} noise_vol = reclamm_market_linear_noise_volume( effective_value, noise_base, noise_tvl_coeff, + tvl_mean=_np.get("tvl_mean", 0.0), + tvl_std=_np.get("tvl_std", 1.0), ) noise_fee_income = (1.0 - gamma) * noise_vol diff --git a/scripts/plot_reclamm_optuna_result.py b/scripts/plot_reclamm_optuna_result.py index 35242dac..b90ba408 100644 --- a/scripts/plot_reclamm_optuna_result.py +++ b/scripts/plot_reclamm_optuna_result.py @@ -1,15 +1,20 @@ #!/usr/bin/env python3 """Plot reClAMM pool performance from Optuna tuning results. -Reads the SGD-compatible JSON output of tune_reclamm_params.py (or any Optuna -run), extracts the best trial's pool params, re-runs a forward pass over the -full train+test window, and produces a value-over-time plot with on-chain -baselines and cumulative fee revenue. +Reads SGD-compatible JSON output(s) of tune_reclamm_params.py, extracts the +best trial's pool params, re-runs a forward pass over the full train+test +window, and produces a value-over-time plot with on-chain baselines and +cumulative fee revenue. Usage: + # Single result python scripts/plot_reclamm_optuna_result.py results/run_.json - python scripts/plot_reclamm_optuna_result.py results/run_.json --output my_plot.png - python scripts/plot_reclamm_optuna_result.py results/run_.json --top-k 3 + + # Multiple results (comparison across objectives / noise models) + python scripts/plot_reclamm_optuna_result.py results/run_*.json + + # Top-3 trials from each result + python scripts/plot_reclamm_optuna_result.py results/run_*.json --top-k 3 """ import argparse @@ -36,12 +41,20 @@ BG = "#162536" TEXT_COLOR = "#E6CE97" +# Extended palette for multi-file comparison COLORS = [ - "#3498db", "#2ecc71", "#e74c3c", # top-k - "#f39c12", # on-chain launch - "#9b59b6", # on-chain current + "#3498db", "#2ecc71", "#e74c3c", "#f39c12", "#9b59b6", + "#1abc9c", "#e67e22", "#2980b9", "#c0392b", "#8e44ad", + "#27ae60", "#d35400", "#16a085", "#f1c40f", "#7f8c8d", ] +# Short labels for objectives +_OBJ_SHORT = { + "daily_log_sharpe": "sharpe", + "returns_over_hodl": "ret/hodl", + "fee_revenue_over_value": "fee_rev", +} + def _plot_order(configs): """Yield (name, meta, color_idx) with baselines first, optimized trials last.""" @@ -60,9 +73,10 @@ def parse_args(): description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter, ) - p.add_argument("results_json", help="Path to run_.json from Optuna") + p.add_argument("results_json", nargs="+", + help="Path(s) to run_.json from Optuna") p.add_argument("--top-k", type=int, default=1, - help="Plot top K trials by objective (default 1)") + help="Plot top K trials per result file (default 1)") p.add_argument("--output", default=None, help="Output PNG path (default: auto-generated)") p.add_argument("--no-onchain", action="store_true", @@ -100,6 +114,18 @@ def extract_pool_params(trial, config): return params +def _noise_model_label(config): + """Short label describing the noise model in the config.""" + nm = config.get("noise_model", "ratio") + if nm != "calibrated": + ntr = config.get("noise_trader_ratio", 0.0) + return f"{nm}(ntr={ntr})" + nc = config.get("reclamm_noise_params", {}) + n_coeffs = len(nc) + arb_freq = config.get("arb_frequency", 1) + return f"cal-{n_coeffs}cov(af={arb_freq})" + + def run_full_period(params, config, fees_override=None): """Run forward pass over the full train+test window.""" fees = fees_override if fees_override is not None else config["fees"] @@ -120,19 +146,28 @@ def run_full_period(params, config, fees_override=None): "reclamm_centeredness_scaling": config.get("reclamm_centeredness_scaling", False), "reclamm_learn_arc_length_speed": config.get("reclamm_learn_arc_length_speed", False), } + # Forward noise model settings + if "noise_model" in config: + fp["noise_model"] = config["noise_model"] + if "reclamm_noise_params" in config: + fp["reclamm_noise_params"] = config["reclamm_noise_params"] + if "noise_arrays_path" in config: + fp["noise_arrays_path"] = config["noise_arrays_path"] + if "arb_frequency" in config: + fp["arb_frequency"] = config["arb_frequency"] jax_params = {k: jnp.array(v) for k, v in params.items()} return do_run_on_historic_data(run_fingerprint=fp, params=jax_params) -def plot_results(configs, time_series, hodl_values, config, args): +def plot_results(configs, time_series, hodl_values, ref_config, args): """Two-panel plot: value-over-time + cumulative fee revenue.""" - train_end_str = config["endDateString"] + train_end_str = ref_config["endDateString"] train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") first_out = next(iter(time_series.values())) n_minutes = len(first_out["value"]) dates = pd.date_range( - start=datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S"), + start=datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S"), periods=n_minutes, freq="1min", ) step = 1440 @@ -157,7 +192,7 @@ def plot_results(configs, time_series, hodl_values, config, args): vals = np.array(out["value"][::step]) / 1e6 label = f"{name}" if "test_objective" in meta: - obj_name = config.get("return_val", "objective") + obj_name = meta.get("obj_name", "objective") label += f" (OOS {obj_name}={meta['test_objective']:.4f})" is_optimized = "On-Chain" not in name ax_val.plot(dates_daily[:len(vals)], vals, @@ -171,21 +206,19 @@ def plot_results(configs, time_series, hodl_values, config, args): ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) ylims = ax_val.get_ylim() - ax_val.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + ax_val.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", color="white", alpha=0.6, fontsize=11, ha="right", va="top") - ax_val.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + ax_val.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", color="white", alpha=0.6, fontsize=11, ha="left", va="top") _style_axis(ax_val) ax_val.set_ylabel("Pool Value ($M USD)", color=TEXT_COLOR, fontsize=12) - tokens_str = "/".join(config["tokens"]) - obj_name = config.get("return_val", "objective") - ntr = config.get("noise_trader_ratio", 0.0) + tokens_str = "/".join(ref_config["tokens"]) ax_val.set_title( - f"reClAMM Optuna-Optimized ({obj_name}, noise={ntr}) — {tokens_str}", + f"reClAMM Optuna Comparison — {tokens_str}", color=TEXT_COLOR, fontsize=13, pad=15, ) - ax_val.legend(loc="upper left", fontsize=9, facecolor=BG, + ax_val.legend(loc="upper left", fontsize=8, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) # ── Panel 2: Cumulative fee revenue ─────────────────────────────── @@ -208,7 +241,7 @@ def plot_results(configs, time_series, hodl_values, config, args): _style_axis(ax_fee) ax_fee.set_ylabel("Cumulative Fee Revenue ($K)", color=TEXT_COLOR, fontsize=12) ax_fee.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) - ax_fee.legend(loc="upper left", fontsize=9, facecolor=BG, + ax_fee.legend(loc="upper left", fontsize=8, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) else: ax_val.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) @@ -222,11 +255,11 @@ def plot_results(configs, time_series, hodl_values, config, args): plt.close() -def plot_test_only(configs, time_series, hodl_values, config, args): +def plot_test_only(configs, time_series, hodl_values, ref_config, args): """Test-period plot with all curves normalised to start at 1.0.""" - train_end_str = config["endDateString"] + train_end_str = ref_config["endDateString"] train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") - start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") + start_dt = datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S") first_out = next(iter(time_series.values())) n_minutes = len(first_out["value"]) @@ -262,14 +295,12 @@ def plot_test_only(configs, time_series, hodl_values, config, args): ax.axhline(1.0, color="white", linestyle=":", alpha=0.3, linewidth=1) _style_axis(ax) - tokens_str = "/".join(config["tokens"]) - obj_name = config.get("return_val", "objective") - ntr = config.get("noise_trader_ratio", 0.0) - ax.set_title(f"Test Period Only (normalised) — {obj_name}, noise={ntr} — {tokens_str}", + tokens_str = "/".join(ref_config["tokens"]) + ax.set_title(f"Test Period Only (normalised) — {tokens_str}", color=TEXT_COLOR, fontsize=13, pad=15) ax.set_ylabel("Normalised Value", color=TEXT_COLOR, fontsize=12) ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) - ax.legend(loc="best", fontsize=9, facecolor=BG, + ax.legend(loc="best", fontsize=8, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) fig.patch.set_facecolor(BG) @@ -281,10 +312,10 @@ def plot_test_only(configs, time_series, hodl_values, config, args): plt.close() -def plot_weights(configs, time_series, config, args): +def plot_weights(configs, time_series, ref_config, args): """Effective weight (value fraction) of token 0 over time.""" - start_dt = datetime.strptime(config["startDateString"], "%Y-%m-%d %H:%M:%S") - train_end_dt = datetime.strptime(config["endDateString"], "%Y-%m-%d %H:%M:%S") + start_dt = datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S") + train_end_dt = datetime.strptime(ref_config["endDateString"], "%Y-%m-%d %H:%M:%S") first_out = next(iter(time_series.values())) n_minutes = len(first_out["value"]) @@ -292,7 +323,7 @@ def plot_weights(configs, time_series, config, args): step = 1440 dates_daily = dates[::step] - token_name = config["tokens"][0] + token_name = ref_config["tokens"][0] fig, ax = plt.subplots(1, 1, figsize=(14, 5)) @@ -310,18 +341,18 @@ def plot_weights(configs, time_series, config, args): ax.axhline(0.5, color="white", linestyle="--", alpha=0.3, linewidth=1) ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) ylims = ax.get_ylim() - ax.text(train_end_dt - pd.Timedelta(days=10), ylims[1] * 0.97, "Train", + ax.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", color="white", alpha=0.6, fontsize=11, ha="right", va="top") - ax.text(train_end_dt + pd.Timedelta(days=10), ylims[1] * 0.97, "Test", + ax.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", color="white", alpha=0.6, fontsize=11, ha="left", va="top") _style_axis(ax) - tokens_str = "/".join(config["tokens"]) + tokens_str = "/".join(ref_config["tokens"]) ax.set_title(f"Effective {token_name} Weight — {tokens_str}", color=TEXT_COLOR, fontsize=13, pad=15) ax.set_ylabel(f"{token_name} weight (value fraction)", color=TEXT_COLOR, fontsize=12) ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) - ax.legend(loc="best", fontsize=9, facecolor=BG, + ax.legend(loc="best", fontsize=8, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) fig.patch.set_facecolor(BG) @@ -346,60 +377,79 @@ def _style_axis(ax): def main(): args = parse_args() - config, trials = load_results(args.results_json) - if args.end_test_date: - config["endTestDateString"] = args.end_test_date - if args.noise_trader_ratio is not None: - config["noise_trader_ratio"] = args.noise_trader_ratio - tokens = config["tokens"] - obj_name = config.get("return_val", "objective") - - # Sort trials by penalised objective - trials_sorted = sorted(trials, key=lambda t: t.get("objective", 0), reverse=True) - top_trials = trials_sorted[:args.top_k] - - print("=" * 80) - print(f"reClAMM Optuna Result Plotter — objective: {obj_name}") - print("=" * 80) - print(f" Results: {args.results_json}") + + # ── Load all result files ───────────────────────────────────────── + all_loaded = [] + for path in args.results_json: + config, trials = load_results(path) + if args.end_test_date: + config["endTestDateString"] = args.end_test_date + if args.noise_trader_ratio is not None: + config["noise_trader_ratio"] = args.noise_trader_ratio + all_loaded.append((path, config, trials)) + + # Use first file's config as reference for dates/tokens + ref_config = all_loaded[0][1] + tokens = ref_config["tokens"] + + print("=" * 100) + print(f"reClAMM Optuna Result Plotter — {len(all_loaded)} result file(s)") + print("=" * 100) print(f" Tokens: {'/'.join(tokens)}") - print(f" Train: {config['startDateString']} → {config['endDateString']}") - print(f" Test: {config['endDateString']} → {config['endTestDateString']}") - print(f" Fees: {config['fees']}, Gas: {config.get('gas_cost', 1.0)}") - print(f" Trials: {len(trials)} total, plotting top {len(top_trials)}") + print(f" Train: {ref_config['startDateString']} → {ref_config['endDateString']}") + print(f" Test: {ref_config['endDateString']} → {ref_config['endTestDateString']}") + # ── Build configs dict from all files ───────────────────────────── configs = {} - for i, trial in enumerate(top_trials): - params = extract_pool_params(trial, config) - name = f"#{trial.get('optuna_trial_number', i)} (rank {i+1})" - configs[name] = { - "params": params, - "objective": trial.get("objective", 0), - "train_objective": trial.get("train_objective", 0), - "test_objective": trial.get("test_objective", 0), - "train_sharpe": trial.get("train_sharpe", 0), - "validation_sharpe": trial.get("validation_sharpe", 0), - } - print(f"\n {name}:") - print(f" {obj_name}: train={trial.get('train_objective', 0):.4f} " - f"test={trial.get('test_objective', 0):.4f} " - f"penalised={trial.get('objective', 0):.4f}") - print(f" sharpe: train={trial.get('train_sharpe', 0):+.4f} " - f"val={trial.get('validation_sharpe', 0):+.4f}") - for k, v in params.items(): - print(f" {k}: {v:.6g}") + for path, config, trials in all_loaded: + obj_name = config.get("return_val", "objective") + obj_short = _OBJ_SHORT.get(obj_name, obj_name) + noise_label = _noise_model_label(config) + + trials_sorted = sorted(trials, key=lambda t: t.get("objective", 0), reverse=True) + top_trials = trials_sorted[:args.top_k] + + for i, trial in enumerate(top_trials): + params = extract_pool_params(trial, config) + rank_suffix = f" r{i+1}" if args.top_k > 1 else "" + name = f"{obj_short} {noise_label}{rank_suffix}" + configs[name] = { + "params": params, + "config": config, # per-file config for noise model + "objective": trial.get("objective", 0), + "train_objective": trial.get("train_objective", 0), + "test_objective": trial.get("test_objective", 0), + "train_sharpe": trial.get("train_sharpe", 0), + "validation_sharpe": trial.get("validation_sharpe", 0), + "obj_name": obj_name, + } + print(f"\n {name}:") + print(f" {obj_name}: train={trial.get('train_objective', 0):.4f} " + f"test={trial.get('test_objective', 0):.4f} " + f"penalised={trial.get('objective', 0):.4f}") + print(f" sharpe: train={trial.get('train_sharpe', 0):+.4f} " + f"val={trial.get('validation_sharpe', 0):+.4f}") + for k, v in params.items(): + print(f" {k}: {v:.6g}") if not args.no_onchain: - configs["On-Chain (launch)"] = {"params": dict(ONCHAIN_LAUNCH_PARAMS)} - configs["On-Chain (current)"] = {"params": dict(ONCHAIN_CURRENT_PARAMS)} + configs["On-Chain (launch)"] = { + "params": dict(ONCHAIN_LAUNCH_PARAMS), + "config": ref_config, + } + configs["On-Chain (current)"] = { + "params": dict(ONCHAIN_CURRENT_PARAMS), + "config": ref_config, + } # ── Full-period runs ────────────────────────────────────────────── - print(f"\n--- Running full-period simulations ({config['startDateString']} → " - f"{config['endTestDateString']}) ---") + print(f"\n--- Running full-period simulations ({ref_config['startDateString']} → " + f"{ref_config['endTestDateString']}) ---") time_series = {} for name, cfg in configs.items(): print(f" {name}...", end=" ", flush=True) - out = run_full_period(cfg["params"], config) + run_config = cfg.get("config", ref_config) + out = run_full_period(cfg["params"], run_config) time_series[name] = out fv = float(out["final_value"]) fr = out.get("fee_revenue") @@ -415,28 +465,29 @@ def main(): ) # ── Plots ───────────────────────────────────────────────────────── - plot_results(configs, time_series, hodl_values, config, args) - plot_test_only(configs, time_series, hodl_values, config, args) - plot_weights(configs, time_series, config, args) + plot_results(configs, time_series, hodl_values, ref_config, args) + plot_test_only(configs, time_series, hodl_values, ref_config, args) + plot_weights(configs, time_series, ref_config, args) # ── Summary table ───────────────────────────────────────────────── - print(f"\n{'=' * 120}") - print(f"SUMMARY — {'/'.join(tokens)} — {obj_name}") - print(f"{'=' * 120}") - hdr = (f"{'Config':<28s} {'Train '+obj_name:>20s} {'Test '+obj_name:>20s} " + print(f"\n{'=' * 130}") + print(f"SUMMARY — {'/'.join(tokens)}") + print(f"{'=' * 130}") + hdr = (f"{'Config':<35s} {'Objective':>12s} {'Train':>10s} {'Test':>10s} " f"{'Train SR':>10s} {'Val SR':>10s} " f"{'PR':>7s} {'Margin':>7s} {'ShiftExp':>10s} {'Full RoH':>10s}") print(hdr) - print("-" * 120) + print("-" * 130) for name, cfg in configs.items(): cp = cfg["params"] fv = float(time_series[name]["final_value"]) full_roh = fv / float(hodl_values[-1]) - 1 print( - f"{name:<28s} " - f"{cfg.get('train_objective', float('nan')):>20.4f} " - f"{cfg.get('test_objective', float('nan')):>20.4f} " + f"{name:<35s} " + f"{cfg.get('obj_name', ''):>12s} " + f"{cfg.get('train_objective', float('nan')):>10.4f} " + f"{cfg.get('test_objective', float('nan')):>10.4f} " f"{cfg.get('train_sharpe', float('nan')):>+10.4f} " f"{cfg.get('validation_sharpe', float('nan')):>+10.4f} " f"{cp.get('price_ratio', float('nan')):>7.3f} " @@ -444,7 +495,7 @@ def main(): f"{cp.get('shift_exponent', float('nan')):>10.4g} " f"{full_roh * 100:>+9.2f}%" ) - print("=" * 120) + print("=" * 130) if __name__ == "__main__": From afb8511e1b786dcf05c8305c8e5b32486e9fed38 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 26 Mar 2026 11:21:43 +0000 Subject: [PATCH 066/115] =?UTF-8?q?feat:=20causal=20TVL=20elasticity=20ana?= =?UTF-8?q?lysis=20=E2=80=94=20deconfounder=20+=20LP=20event=20study?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - run_deconfounder_noise.py: Four-strategy causal analysis of b_tvl: 1. Variance decomposition (62% between-pool, 38% within-pool) 2. Within-pool Δ regressions (median b_tvl=+0.12, daily too fast) 2b. Lagged-average TVL across windows (stable ~0.95 at all horizons) 3. TVL decomposition: price-driven vs flow-driven (IV-style) 4. Deconfounder sensitivity (Wang & Blei 2019, n_factors sweep) D'Amour critique acknowledged in docstring. Ridge warm-start, standardized Z_hat. Convergent finding: per-pool b_tvl ~1.0 is the right working estimate for counterfactuals. - scan_lp_events.py: Scan all pools for large LP deposit/withdrawal events (semi-exogenous TVL shocks). Filters pool creation events via min-age and min-tvl. Computes per-event elasticity from ±window day volume comparison. 836 events across 118 pools, median elasticity +0.84 (clean: +0.98, OLS: +0.89). Saves CSV + generates plots: elasticity histograms, deposits vs withdrawals, elasticity vs pool size, log-log scatter with OLS, boxplot by chain. No asymmetry between deposits/withdrawals, flat across pool sizes and chains. --- experiments/run_deconfounder_noise.py | 555 ++++++++++++++++++++++++++ experiments/scan_lp_events.py | 522 ++++++++++++++++++++++++ 2 files changed, 1077 insertions(+) create mode 100644 experiments/run_deconfounder_noise.py create mode 100644 experiments/scan_lp_events.py diff --git a/experiments/run_deconfounder_noise.py b/experiments/run_deconfounder_noise.py new file mode 100644 index 00000000..cd761b97 --- /dev/null +++ b/experiments/run_deconfounder_noise.py @@ -0,0 +1,555 @@ +"""Causal noise volume estimation: TVL decomposition + deconfounder sensitivity. + +Two identification strategies for the causal effect of TVL on noise volume: + +**Primary: TVL decomposition (IV-style)** + Decomposes Δlog(TVL) into: + - Price-driven: Δlog(TVL) - Δlog(shares) — market price moves, more + exogenous to pool-specific trading activity (conditional on BTC/token + market features already in the model) + - Flow-driven: Δlog(shares) — LP deposits/withdrawals, endogenous + (LPs deposit when they expect fees → correlated with noise) + If b_tvl estimated from price-driven variation ≈ observational b_tvl, + the coefficient is likely causal. + +**Secondary: Deconfounder sensitivity analysis (Wang & Blei 2019)** + Fit a factor model (PPCA) on covariates only (no outcome), extract + latent factors Z_hat, include them in the outcome model. + This is a sensitivity analysis: if b_tvl shifts substantially when + conditioning on Z_hat, there's evidence of unobserved confounding. + If it's stable, confounding through the covariate structure is small. + + NB: The deconfounder has known theoretical limitations (D'Amour 2019). + Wang & Blei (2020, arXiv:2003.04948) respond that D'Amour's + counterexamples violate the required assumptions (pinpointability). + The theory holds under its assumptions, but the key assumption (no + unobserved single-cause confounders) is domain-specific and + uncheckable. Results should be interpreted as sensitivity bounds. + +**Diagnostics:** + - Variance decomposition of log_tvl: between-pool vs within-pool + - Within-pool simple regression: Δlog(V_obs) on Δlog(TVL) per pool + - These test whether the observational b_tvl reflects cross-sectional + or temporal variation + +Usage: + python experiments/run_deconfounder_noise.py + python experiments/run_deconfounder_noise.py --n-factors 1 2 3 5 +""" + +import argparse +import os +import time + +import jax.numpy as jnp +import numpy as np + + +# ---- Factor model ---- + + +def fit_ppca(X, n_components): + """Probabilistic PCA. Returns Z_hat and the model.""" + from sklearn.decomposition import PCA + pca = PCA(n_components=n_components) + Z_hat = pca.fit_transform(X) + print(f" PPCA({n_components}): explained var = " + f"{pca.explained_variance_ratio_.sum():.3f} " + f"per-component: {np.round(pca.explained_variance_ratio_, 3)}") + return Z_hat, pca + + +def build_augmented_data(data, Z_hat): + """Augment covariate matrix with standardized substitute confounders.""" + x_orig = data["x"] + n_z = Z_hat.shape[1] + + z_mean = Z_hat.mean(axis=0) + z_std = Z_hat.std(axis=0) + z_std[z_std < 1e-6] = 1.0 + Z_std = ((Z_hat - z_mean) / z_std).astype(np.float32) + + x_aug = np.concatenate([x_orig, Z_std], axis=1) + data_aug = dict(data) + data_aug["x"] = x_aug + data_aug["n_feat"] = x_aug.shape[1] + data_aug["feat_names"] = data["feat_names"] + [f"Z_{k}" for k in range(n_z)] + data_aug["x_mean"] = np.concatenate([ + data["x_mean"], z_mean.astype(np.float32)]) + data_aug["x_std"] = np.concatenate([ + data["x_std"], z_std.astype(np.float32)]) + return data_aug + + +def _tvl_col_index(feat_names): + """Find TVL column index from feature names (robust to reordering).""" + return feat_names.index("xobs_1") + + +def _intercept_col_index(feat_names): + """Find intercept column index.""" + return feat_names.index("xobs_0") + + +# ---- TVL decomposition ---- + + +def decompose_tvl(matched_clean, pool_ids, sample_pools, sample_days, + date_to_idx, n_dates, n_pools): + """Decompose Δlog(TVL) into price-driven and flow-driven components. + + Uses log_tvl (not log_tvl_lag1) for the decomposition to avoid + mixing lags. total_shares is assumed to be contemporaneous with TVL. + + flow = Δlog(shares) — LP deposits/withdrawals + price = Δlog(tvl) - Δlog(shares) — price changes + + Returns per-sample arrays and a validity mask. + """ + log_shares = np.full((n_dates, n_pools), np.nan) + log_tvl = np.full((n_dates, n_pools), np.nan) + + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + dates = panel["date"].values + + has_shares = ("total_shares" in panel.columns and + panel["total_shares"].notna().any()) + if not has_shares: + continue + + shares = panel["total_shares"].values.astype(float) + shares = np.maximum(shares, 1e-10) + + # Use log_tvl (not lag) for contemporaneous decomposition + if "log_tvl" in panel.columns: + tvl_vals = panel["log_tvl"].values.astype(float) + else: + tvl_vals = panel["log_tvl_lag1"].values.astype(float) + + for k, date in enumerate(dates): + t = date_to_idx.get(date) + if t is not None: + log_shares[t, j] = np.log(shares[k]) + log_tvl[t, j] = tvl_vals[k] + + n_samples = len(sample_pools) + tvl_flow = np.full(n_samples, np.nan, dtype=np.float32) + tvl_price = np.full(n_samples, np.nan, dtype=np.float32) + + for s in range(n_samples): + i = sample_pools[s] + t = sample_days[s] + if t >= 1: + d_log_shares = log_shares[t, i] - log_shares[t - 1, i] + d_log_tvl = log_tvl[t, i] - log_tvl[t - 1, i] + if np.isfinite(d_log_shares) and np.isfinite(d_log_tvl): + tvl_flow[s] = d_log_shares + tvl_price[s] = d_log_tvl - d_log_shares + + valid = np.isfinite(tvl_flow) & np.isfinite(tvl_price) + return tvl_flow, tvl_price, valid + + +def run_tvl_decomposition_analysis(data, matched_clean, tvl_flow, tvl_price, valid): + """Primary identification: compare b_tvl from price-driven vs all TVL.""" + from sklearn.linear_model import RidgeCV + + y = data["y_total"] + x = data["x"] + tvl_idx = _tvl_col_index(data["feat_names"]) + + print(f"\n Valid samples (have LP shares data): {valid.sum()}/{len(valid)}") + if valid.sum() < 100: + print(" Insufficient LP shares data for TVL decomposition.") + return None + + x_valid = x[valid] + y_valid = y[valid] + tvl_price_valid = tvl_price[valid] + tvl_flow_valid = tvl_flow[valid] + + print(f" Price component: mean={tvl_price_valid.mean():.4f}," + f" std={tvl_price_valid.std():.4f}") + print(f" Flow component: mean={tvl_flow_valid.mean():.4f}," + f" std={tvl_flow_valid.std():.4f}") + + # Observational b_tvl (Ridge on all features) + ridge_obs = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge_obs.fit(x_valid, y_valid) + b_tvl_obs = ridge_obs.coef_[tvl_idx] + + # Replace TVL column with price-driven component only + x_price = x_valid.copy() + ps = tvl_price_valid.std() + x_price[:, tvl_idx] = (tvl_price_valid - tvl_price_valid.mean()) / max(ps, 1e-6) + ridge_price = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge_price.fit(x_price, y_valid) + b_tvl_price = ridge_price.coef_[tvl_idx] + + # Replace TVL column with flow-driven component only + x_flow = x_valid.copy() + fs = tvl_flow_valid.std() + x_flow[:, tvl_idx] = (tvl_flow_valid - tvl_flow_valid.mean()) / max(fs, 1e-6) + ridge_flow = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge_flow.fit(x_flow, y_valid) + b_tvl_flow = ridge_flow.coef_[tvl_idx] + + print(f"\n b_tvl estimates (Ridge, all 22 features):") + print(f" All TVL variation: {b_tvl_obs:+.4f}") + print(f" Price-driven only: {b_tvl_price:+.4f}" + f" (more exogenous)") + print(f" Flow-driven only: {b_tvl_flow:+.4f}" + f" (endogenous)") + + if abs(b_tvl_price - b_tvl_obs) < 0.3 * abs(b_tvl_obs): + print(f"\n → Price-driven ≈ observational: confounding small.") + else: + print(f"\n → Price-driven ≠ observational: potential confounding.") + + return {"obs": b_tvl_obs, "price": b_tvl_price, "flow": b_tvl_flow} + + +# ---- Variance decomposition ---- + + +def run_variance_decomposition(matched_clean, pool_ids): + """Decompose log_tvl variance into between-pool and within-pool.""" + all_tvls = [] + pool_labels = [] + + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + tvls = panel["log_tvl_lag1"].values.astype(float) + valid = np.isfinite(tvls) + all_tvls.extend(tvls[valid]) + pool_labels.extend([j] * valid.sum()) + + all_tvls = np.array(all_tvls) + pool_labels = np.array(pool_labels) + + total_var = np.var(all_tvls) + pool_means = np.array([all_tvls[pool_labels == j].mean() + for j in range(len(pool_ids))]) + between_var = np.var(pool_means) + within_vars = [np.var(all_tvls[pool_labels == j]) + for j in range(len(pool_ids))] + within_var = np.mean(within_vars) + + print(f" Total variance: {total_var:.4f}") + print(f" Between-pool: {between_var:.4f} ({between_var/total_var*100:.1f}%)") + print(f" Within-pool (avg): {within_var:.4f} ({within_var/total_var*100:.1f}%)") + print(f" Pool mean range: {pool_means.min():.1f} to {pool_means.max():.1f}") + + return {"total": total_var, "between": between_var, "within": within_var} + + +# ---- Within-pool simple regression ---- + + +def run_within_pool_regressions(matched_clean, pool_ids): + """Per-pool: Δlog(V_obs) on Δlog(TVL), no other covariates.""" + print(f"\n {'Pool':16s} {'Tokens':16s} {'b_tvl':>8s} {'R²':>6s}" + f" {'n':>5s} {'ΔTVL_std':>8s}") + print(f" {'-'*65}") + + b_tvls = [] + for pid in pool_ids: + panel = matched_clean[pid]["panel"] + log_vol = panel["log_volume"].values.astype(float) + log_tvl = panel["log_tvl_lag1"].values.astype(float) + + d_vol = np.diff(log_vol) + d_tvl = np.diff(log_tvl) + + valid = np.isfinite(d_vol) & np.isfinite(d_tvl) + if valid.sum() < 10: + continue + + dv = d_vol[valid] + dt = d_tvl[valid] + + # OLS: Δlog_vol = a + b * Δlog_tvl + X = np.column_stack([np.ones(len(dt)), dt]) + sol, _, _, _ = np.linalg.lstsq(X, dv, rcond=None) + b = sol[1] + pred = X @ sol + ss_res = np.sum((dv - pred) ** 2) + ss_tot = np.sum((dv - dv.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + b_tvls.append(b) + tokens = matched_clean[pid]["tokens"] + print(f" {pid[:16]:16s} {tokens[:16]:16s} {b:+8.3f} {r2:6.3f}" + f" {valid.sum():5d} {dt.std():8.4f}") + + if b_tvls: + print(f"\n Median within-pool b_tvl: {np.median(b_tvls):+.4f}") + print(f" Mean: {np.mean(b_tvls):+.4f}") + print(f" Std across pools: {np.std(b_tvls):.4f}") + return b_tvls + + +# ---- Lagged-average TVL analysis ---- + + +def run_lagged_average_analysis(matched_clean, pool_ids): + """Test TVL→noise at different timescales. + + If the daily Δ elasticity is ~0 but the level elasticity is ~2.5, + the effect may operate on longer timescales. Test by regressing + noise on rolling-average TVL at windows of 7, 14, 30, 60, 90 days. + If b_tvl grows with window size, the relationship is real but slow. + """ + windows = [1, 7, 14, 30, 60, 90] + + print(f"\n Window Median b_tvl Mean b_tvl Pools w/ data") + print(f" {'-'*55}") + + for w in windows: + b_tvls = [] + n_pools_used = 0 + for pid in pool_ids: + panel = matched_clean[pid]["panel"] + log_vol = panel["log_volume"].values.astype(float) + log_tvl = panel["log_tvl_lag1"].values.astype(float) + + if len(log_vol) < w + 10: + continue + + # Rolling mean TVL over window w + if w == 1: + tvl_avg = log_tvl + else: + # Simple trailing average + tvl_avg = np.full_like(log_tvl, np.nan) + for t in range(w, len(log_tvl)): + vals = log_tvl[t - w:t] + if np.all(np.isfinite(vals)): + tvl_avg[t] = np.mean(vals) + + # Within-pool: demean both series + valid = np.isfinite(log_vol) & np.isfinite(tvl_avg) + if valid.sum() < 15: + continue + + vol = log_vol[valid] + tvl = tvl_avg[valid] + vol_dm = vol - vol.mean() + tvl_dm = tvl - tvl.mean() + + # OLS: demeaned_vol = b * demeaned_tvl + if np.var(tvl_dm) < 1e-10: + continue + b = np.sum(vol_dm * tvl_dm) / np.sum(tvl_dm ** 2) + b_tvls.append(b) + n_pools_used += 1 + + if b_tvls: + print(f" {w:5d}d {np.median(b_tvls):+11.4f} {np.mean(b_tvls):+10.4f}" + f" {n_pools_used:>13d}") + + return windows + + +# ---- Deconfounder sensitivity ---- + + +def run_deconfounder(data, n_factors_list, args): + """Secondary: deconfounder sensitivity analysis across n_factors.""" + from experiments.run_linear_market_noise import make_loss_fn, train + from sklearn.linear_model import RidgeCV + + X = data["x"] + tvl_idx = _tvl_col_index(data["feat_names"]) + intercept_idx = _intercept_col_index(data["feat_names"]) + results = {} + + for n_f in n_factors_list: + print(f"\n --- n_factors={n_f} ---") + Z_hat, _ = fit_ppca(X, n_f) + data_aug = build_augmented_data(data, Z_hat) + + n_feat = data_aug["n_feat"] + n_pools = data_aug["n_pools"] + + # Ridge warm-start + ridge = RidgeCV(alphas=np.logspace(-2, 4, 50)) + ridge.fit(data_aug["x"], data_aug["y_total"]) + sol = ridge.coef_.copy() + sol[intercept_idx] += ridge.intercept_ + + params = { + "log_cadence": jnp.array(data_aug["init_log_cadences"]), + "noise_coeffs": jnp.array(sol.astype(np.float32)), + } + + grad_fn = make_loss_fn(data_aug["pool_coeffs"], data_aug["pool_gas"], n_pools) + params = train(params, data_aug, grad_fn, args.epochs, args.lr, + args.l2_alpha, args.huber_delta, verbose=False) + + nc = np.array(params["noise_coeffs"]) + b_tvl = nc[tvl_idx] + n_orig = data["n_feat"] + z_coeffs = nc[n_orig:n_orig + n_f] + + print(f" b_tvl = {b_tvl:+.4f} " + f" Z coeffs: {np.round(z_coeffs, 3)}") + + results[n_f] = { + "b_tvl": float(b_tvl), + "z_coeffs": z_coeffs.tolist(), + "explained_var": float(Z_hat.var(axis=0).sum() / X.var(axis=0).sum()), + } + + return results + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--n-factors", type=int, nargs="+", default=[1, 2, 3, 5], + help="Number of latent factors to sweep") + parser.add_argument("--epochs", type=int, default=2000) + parser.add_argument("--lr", type=float, default=3e-4) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Causal Noise Volume Estimation") + print(" 1. Variance decomposition (between vs within pool)") + print(" 2. Within-pool simple regressions") + print(" 3. TVL decomposition (price vs flow)") + print(" 4. Deconfounder sensitivity (PPCA factors)") + print(f" n_factors sweep: {args.n_factors}") + print("=" * 70) + + from experiments.run_linear_market_noise import load_stage1, build_data + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding data...") + t0 = time.time() + data = build_data( + matched_clean, option_c_clean, + trend_windows=tuple(args.trend_windows), + include_market=True, include_cross_pool=True, + ) + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + print(f" {len(data['pool_idx'])} samples, {n_pools} pools," + f" {data['n_feat']} features, {time.time() - t0:.1f}s") + + # ---- 1. Variance decomposition ---- + print(f"\n{'='*70}") + print("1. Variance Decomposition of log_tvl_lag1") + print(f"{'='*70}") + var_results = run_variance_decomposition(matched_clean, pool_ids) + + # ---- 2. Within-pool simple regressions ---- + print(f"\n{'='*70}") + print("2. Within-pool: Δlog(V_obs) ~ Δlog(TVL) (no other covariates)") + print(f"{'='*70}") + within_b_tvls = run_within_pool_regressions(matched_clean, pool_ids) + + # ---- 2b. Lagged-average TVL (timescale test) ---- + print(f"\n{'='*70}") + print("2b. Lagged-Average TVL: Does b_tvl grow with averaging window?") + print(f" (Tests whether the TVL→noise effect is slow-moving)") + print(f"{'='*70}") + run_lagged_average_analysis(matched_clean, pool_ids) + + # ---- 3. TVL decomposition ---- + print(f"\n{'='*70}") + print("3. TVL Decomposition: Price-driven vs Flow-driven") + print(f"{'='*70}") + + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + date_to_idx = {d: i for i, d in enumerate(date_list)} + + tvl_flow, tvl_price, valid = decompose_tvl( + matched_clean, pool_ids, data["pool_idx"], data["day_idx"], + date_to_idx, len(date_list), n_pools, + ) + tvl_results = run_tvl_decomposition_analysis( + data, matched_clean, tvl_flow, tvl_price, valid) + + # ---- 4. Deconfounder sensitivity ---- + print(f"\n{'='*70}") + print("4. Deconfounder Sensitivity Analysis") + print(f" (Wang & Blei 2019; D'Amour 2019; Wang & Blei 2020)") + print(f"{'='*70}") + deconf_results = run_deconfounder(data, args.n_factors, args) + + # ---- Summary ---- + print(f"\n{'='*70}") + print("SUMMARY: b_tvl across identification strategies") + print(f"{'='*70}") + + # Observational baseline from artifact + obs_path = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", "model.npz", + ) + if os.path.exists(obs_path): + obs_nc = np.load(obs_path)["noise_coeffs"] + tvl_idx = _tvl_col_index(data["feat_names"]) + if obs_nc.ndim == 2: + b_obs = float(np.median(obs_nc[:, tvl_idx])) + print(f" Observational (per-pool median): {b_obs:+.4f}") + else: + b_obs = float(obs_nc[tvl_idx]) + print(f" Observational (shared): {b_obs:+.4f}") + + between_pct = var_results["between"] / var_results["total"] * 100 + print(f"\n Variance decomposition: {between_pct:.0f}% between-pool," + f" {100-between_pct:.0f}% within-pool") + + if within_b_tvls: + print(f" Within-pool Δ regressions: " + f"median={np.median(within_b_tvls):+.4f}" + f" (mean={np.mean(within_b_tvls):+.4f})") + + if tvl_results: + print(f"\n TVL decomposition (Ridge, 22 features):") + print(f" All variation: {tvl_results['obs']:+.4f}") + print(f" Price-driven (exogenous): {tvl_results['price']:+.4f}") + print(f" Flow-driven (endogenous): {tvl_results['flow']:+.4f}") + + print(f"\n Deconfounder sensitivity (shared, learnable cadence):") + print(f" {'n_factors':>10s} {'b_tvl':>8s}") + for n_f, r in sorted(deconf_results.items()): + print(f" {n_f:>10d} {r['b_tvl']:+8.4f}") + + # Stability + b_tvls_d = [r["b_tvl"] for r in deconf_results.values()] + rng = max(b_tvls_d) - min(b_tvls_d) + mn = np.mean(b_tvls_d) + stable = rng < 0.3 * abs(mn) + print(f"\n Deconfounder: {'STABLE' if stable else 'VARIES'}" + f" (range {rng:.3f}, mean {mn:+.3f})") + + print(f"\n Interpretation:") + if tvl_results and abs(tvl_results['price']) < 0.5: + print(f" Daily b_tvl (Δ regression, price-driven) is near zero.") + print(f" This does NOT mean the long-run effect is zero:") + print(f" - Noise may respond slowly to TVL (routing updates,") + print(f" aggregator discovery, ecosystem integration)") + print(f" - The lagged-average analysis above tests this") + print(f" - The per-pool b_tvl of ~1.0 captures medium-frequency") + print(f" within-pool variation and is the best working estimate") + print(f" - Changing reClAMM concentration is a structural change") + print(f" (like being a different pool), not a daily TVL shock") + print(f" → Use per-pool b_tvl (~1.0) for counterfactuals, with") + print(f" sensitivity analysis across [0.5, 1.0, 2.0]") + + +if __name__ == "__main__": + main() diff --git a/experiments/scan_lp_events.py b/experiments/scan_lp_events.py new file mode 100644 index 00000000..7194b845 --- /dev/null +++ b/experiments/scan_lp_events.py @@ -0,0 +1,522 @@ +"""Scan all pools for large LP deposit/withdrawal events and estimate TVL→noise elasticity. + +Identifies "semi-exogenous" LP flow events — large share changes that represent +genuine deposit/withdrawal decisions, not pool creation or dust. + +Filters: + - |Δlog(shares)| > threshold (default 20%) + - Pool must have been active for at least --min-age days before the event + - Pre-event TVL must be above --min-tvl (filters out pool creation events + where initial TVL is dust) + - Enough pre/post data to estimate volume change + +For each event, computes the volume response and implied elasticity. + +Usage: + python experiments/scan_lp_events.py + python experiments/scan_lp_events.py --threshold 0.1 --window 7 + python experiments/scan_lp_events.py --use-api # fetch fresh snapshots +""" + +import argparse +import os +import time + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + + +def load_panel_data(use_api=False): + """Load pool panel data from calibration cache or API.""" + import pickle + + pools = {} + + # Stage1 calibration pools + cache_path = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", "stage1.pkl", + ) + if os.path.exists(cache_path): + with open(cache_path, "rb") as f: + data = pickle.load(f) + for pid, entry in data["matched_clean"].items(): + panel = entry["panel"].copy() + panel["pool_id"] = pid + panel["chain"] = entry["chain"] + panel["tokens"] = entry["tokens"] + pools[pid] = panel + + # Noise calibration panel (broader set) + noise_panel_path = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_calibration", "panel.parquet", + ) + if os.path.exists(noise_panel_path): + panel_all = pd.read_parquet(noise_panel_path) + for pid in panel_all["pool_id"].unique(): + if pid[:16] not in pools: # don't duplicate + pp = panel_all[panel_all["pool_id"] == pid].copy() + if len(pp) >= 30: + pools[pid[:16]] = pp + + # Top50 snapshots (even broader) + snap_dir = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "local_data", "noise_top50", "snapshots", + ) + if os.path.exists(snap_dir): + import glob + for f in glob.glob(os.path.join(snap_dir, "*.parquet")): + pid = os.path.basename(f).replace(".parquet", "") + if pid[:16] not in pools: + try: + df = pd.read_parquet(f) + if len(df) >= 30 and "total_shares" in df.columns: + df["pool_id"] = pid + pools[pid[:16]] = df + except Exception: + pass + + if use_api: + print(" Fetching fresh snapshots from Balancer API...") + from quantammsim.noise_calibration import ( + fetch_pool_snapshots, BALANCER_API_CHAINS, + ) + for pid_short, panel in list(pools.items()): + if "chain" in panel.columns: + chain = panel["chain"].iloc[0] + else: + chain = "MAINNET" + full_pid = panel["pool_id"].iloc[0] if "pool_id" in panel.columns else pid_short + try: + fresh = fetch_pool_snapshots(full_pid, chain) + if len(fresh) > len(panel): + fresh["pool_id"] = full_pid + fresh["chain"] = chain + if "tokens" in panel.columns: + fresh["tokens"] = panel["tokens"].iloc[0] + pools[pid_short] = fresh + time.sleep(0.3) + except Exception: + pass + + print(f" Loaded {len(pools)} pools") + return pools + + +def find_lp_events(panel, threshold=0.2, min_age_days=30, min_tvl=10_000): + """Find large LP deposit/withdrawal events in a single pool's panel. + + Returns list of event dicts. + """ + dates = pd.to_datetime(panel["date"]) + + # Need shares and TVL + if "total_shares" not in panel.columns: + return [] + shares = panel["total_shares"].values.astype(float) + if np.all(shares <= 0) or np.all(np.isnan(shares)): + return [] + + # TVL + if "total_liquidity_usd" in panel.columns: + tvl = panel["total_liquidity_usd"].values.astype(float) + elif "log_tvl" in panel.columns: + tvl = np.exp(panel["log_tvl"].values.astype(float)) + elif "log_tvl_lag1" in panel.columns: + tvl = np.exp(panel["log_tvl_lag1"].values.astype(float)) + else: + return [] + + # Volume + if "volume_usd" in panel.columns: + vol = panel["volume_usd"].values.astype(float) + elif "log_volume" in panel.columns: + vol = np.exp(panel["log_volume"].values.astype(float)) + else: + return [] + + log_shares = np.log(np.maximum(shares, 1e-10)) + d_log_shares = np.diff(log_shares) + + events = [] + for i in range(len(d_log_shares)): + if abs(d_log_shares[i]) < np.log(1 + threshold): + continue + + # Check min age: pool must have been active for min_age_days + days_active = (dates.iloc[i + 1] - dates.iloc[0]).days + if days_active < min_age_days: + continue + + # Check min TVL before event + if tvl[i] < min_tvl: + continue + + # Check shares aren't near-zero before (not pool creation) + if shares[i] < 1: + continue + + pct_change = (np.exp(d_log_shares[i]) - 1) * 100 + event_type = "deposit" if d_log_shares[i] > 0 else "withdrawal" + + events.append({ + "date": dates.iloc[i + 1], + "idx": i + 1, + "type": event_type, + "d_log_shares": float(d_log_shares[i]), + "pct_change": float(pct_change), + "shares_before": float(shares[i]), + "shares_after": float(shares[i + 1]), + "tvl_before": float(tvl[i]), + "tvl_after": float(tvl[i + 1]), + "vol_on_day": float(vol[i + 1]), + }) + + return events + + +def compute_event_elasticity(panel, event, window=7): + """Compute volume response around an LP event. + + Compares median volume in [event-window, event) vs [event+1, event+window+1). + """ + dates = pd.to_datetime(panel["date"]) + idx = event["idx"] + + if "volume_usd" in panel.columns: + vol = panel["volume_usd"].values.astype(float) + elif "log_volume" in panel.columns: + vol = np.exp(panel["log_volume"].values.astype(float)) + else: + return None + + if "total_liquidity_usd" in panel.columns: + tvl = panel["total_liquidity_usd"].values.astype(float) + elif "log_tvl" in panel.columns: + tvl = np.exp(panel["log_tvl"].values.astype(float)) + elif "log_tvl_lag1" in panel.columns: + tvl = np.exp(panel["log_tvl_lag1"].values.astype(float)) + else: + return None + + pre_start = max(0, idx - window) + post_end = min(len(vol), idx + 1 + window) + + if idx - pre_start < 3 or post_end - (idx + 1) < 3: + return None + + vol_pre = np.median(vol[pre_start:idx]) + vol_post = np.median(vol[idx + 1:post_end]) + tvl_pre = np.median(tvl[pre_start:idx]) + tvl_post = np.median(tvl[idx + 1:post_end]) + + if vol_pre <= 0 or tvl_pre <= 0 or tvl_post <= 0: + return None + + vol_ratio = vol_post / vol_pre + tvl_ratio = tvl_post / tvl_pre + + if abs(np.log(tvl_ratio)) < 0.05: # TVL didn't actually change much + return None + + elasticity = np.log(vol_ratio) / np.log(tvl_ratio) + + return { + "vol_pre": vol_pre, + "vol_post": vol_post, + "tvl_pre": tvl_pre, + "tvl_post": tvl_post, + "vol_ratio": vol_ratio, + "tvl_ratio": tvl_ratio, + "elasticity": elasticity, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--threshold", type=float, default=0.2, + help="Min |share change| to count as event (0.2 = 20%%)") + parser.add_argument("--window", type=int, default=7, + help="Days before/after event for volume comparison") + parser.add_argument("--min-age", type=int, default=30, + help="Min days pool must be active before event") + parser.add_argument("--min-tvl", type=float, default=10_000, + help="Min TVL before event (filters pool creation)") + parser.add_argument("--use-api", action="store_true", + help="Fetch fresh snapshots from Balancer API") + parser.add_argument("--output-dir", default="results/lp_events", + help="Output directory for CSV and plots") + args = parser.parse_args() + + print("=" * 70) + print("LP Event Scanner: Semi-Exogenous TVL Shocks") + print(f" threshold={args.threshold:.0%}, window={args.window}d," + f" min_age={args.min_age}d, min_tvl=${args.min_tvl:,.0f}") + print("=" * 70) + + pools = load_panel_data(use_api=args.use_api) + + all_events = [] + print(f"\nScanning {len(pools)} pools for LP events...") + + for pid_short, panel in pools.items(): + tokens = (panel["tokens"].iloc[0] if "tokens" in panel.columns + else "?") + chain = (panel["chain"].iloc[0] if "chain" in panel.columns + else "?") + + events = find_lp_events( + panel, threshold=args.threshold, + min_age_days=args.min_age, min_tvl=args.min_tvl) + + for ev in events: + result = compute_event_elasticity(panel, ev, window=args.window) + ev["pool_id"] = pid_short + ev["tokens"] = tokens + ev["chain"] = chain + ev["result"] = result + all_events.append(ev) + + # Sort by absolute share change + all_events.sort(key=lambda e: abs(e["d_log_shares"]), reverse=True) + + print(f"\nFound {len(all_events)} LP events across {len(pools)} pools") + events_with_elasticity = [e for e in all_events if e["result"] is not None] + print(f" {len(events_with_elasticity)} with computable elasticity") + + # Print event table + print(f"\n{'Date':12s} {'Pool':16s} {'Tokens':18s} {'Type':10s}" + f" {'Δshares':>8s} {'TVL before':>12s} {'TVL after':>12s}" + f" {'VolPre':>10s} {'VolPost':>10s} {'Elast':>7s}") + print("-" * 120) + + for ev in all_events: + r = ev["result"] + if r: + elast_str = f"{r['elasticity']:+7.2f}" + vol_pre_str = f"${r['vol_pre']:>9,.0f}" + vol_post_str = f"${r['vol_post']:>9,.0f}" + else: + elast_str = " n/a" + vol_pre_str = " n/a" + vol_post_str = " n/a" + + print(f"{str(ev['date'].date()):12s} {ev['pool_id'][:16]:16s}" + f" {str(ev['tokens'])[:18]:18s} {ev['type']:10s}" + f" {ev['pct_change']:+7.0f}%" + f" ${ev['tvl_before']:>11,.0f} ${ev['tvl_after']:>11,.0f}" + f" {vol_pre_str} {vol_post_str} {elast_str}") + + # Summary statistics + if not events_with_elasticity: + print("No events with computable elasticity.") + return + + deposits = [e for e in events_with_elasticity if e["type"] == "deposit"] + withdrawals = [e for e in events_with_elasticity if e["type"] == "withdrawal"] + + all_elast = [e["result"]["elasticity"] for e in events_with_elasticity] + dep_elast = [e["result"]["elasticity"] for e in deposits] + wth_elast = [e["result"]["elasticity"] for e in withdrawals] + clean = [e for e in events_with_elasticity + if -1 < e["result"]["elasticity"] < 5] + clean_elast = [e["result"]["elasticity"] for e in clean] + + print(f"\n{'='*70}") + print("Summary: Implied TVL→Volume Elasticity") + print(f"{'='*70}") + print(f" All events ({len(all_elast)}):" + f" median={np.median(all_elast):+.2f}" + f" mean={np.mean(all_elast):+.2f}" + f" std={np.std(all_elast):.2f}") + if dep_elast: + print(f" Deposits ({len(dep_elast)}):" + f" median={np.median(dep_elast):+.2f}" + f" mean={np.mean(dep_elast):+.2f}") + if wth_elast: + print(f" Withdrawals ({len(wth_elast)}):" + f" median={np.median(wth_elast):+.2f}" + f" mean={np.mean(wth_elast):+.2f}") + if clean_elast: + print(f"\n Clean events (elasticity in [-1, 5], n={len(clean_elast)}):") + print(f" median={np.median(clean_elast):+.2f}" + f" mean={np.mean(clean_elast):+.2f}" + f" [Q25={np.percentile(clean_elast, 25):+.2f}," + f" Q75={np.percentile(clean_elast, 75):+.2f}]") + + print(f"\n For comparison:") + print(f" Per-pool observational b_tvl: ~1.0") + print(f" Shared observational b_tvl: ~2.5") + print(f" Daily Δ within-pool: ~0.1") + + # ---- Save CSV ---- + out_dir = args.output_dir + os.makedirs(out_dir, exist_ok=True) + + rows = [] + for ev in all_events: + r = ev.get("result") or {} + rows.append({ + "date": ev["date"], + "pool_id": ev["pool_id"], + "tokens": str(ev["tokens"]), + "chain": str(ev["chain"]), + "type": ev["type"], + "pct_change": ev["pct_change"], + "tvl_before": ev["tvl_before"], + "tvl_after": ev["tvl_after"], + "shares_before": ev["shares_before"], + "shares_after": ev["shares_after"], + "vol_pre": r.get("vol_pre"), + "vol_post": r.get("vol_post"), + "tvl_ratio": r.get("tvl_ratio"), + "vol_ratio": r.get("vol_ratio"), + "elasticity": r.get("elasticity"), + }) + df = pd.DataFrame(rows) + csv_path = os.path.join(out_dir, "lp_events.csv") + df.to_csv(csv_path, index=False) + print(f"\n Saved: {csv_path} ({len(df)} events)") + + # ---- Plots ---- + # 1. Elasticity histogram (clean events, deposits vs withdrawals) + fig, axes = plt.subplots(1, 3, figsize=(16, 5)) + + ax = axes[0] + ax.hist(clean_elast, bins=40, color="steelblue", alpha=0.7, edgecolor="white") + ax.axvline(np.median(clean_elast), color="red", linestyle="--", linewidth=2, + label=f"median={np.median(clean_elast):+.2f}") + ax.axvline(1.0, color="gray", linestyle=":", alpha=0.5, label="elasticity=1") + ax.set_xlabel("Elasticity (Δlog vol / Δlog TVL)") + ax.set_ylabel("Count") + ax.set_title(f"All clean events (n={len(clean_elast)})") + ax.legend(fontsize=8) + + ax = axes[1] + dep_clean = [e["result"]["elasticity"] for e in clean if e["type"] == "deposit"] + wth_clean = [e["result"]["elasticity"] for e in clean if e["type"] == "withdrawal"] + ax.hist(dep_clean, bins=30, color="green", alpha=0.6, label=f"deposits (n={len(dep_clean)})", edgecolor="white") + ax.hist(wth_clean, bins=30, color="coral", alpha=0.6, label=f"withdrawals (n={len(wth_clean)})", edgecolor="white") + ax.axvline(1.0, color="gray", linestyle=":", alpha=0.5) + ax.set_xlabel("Elasticity") + ax.set_title("Deposits vs Withdrawals") + ax.legend(fontsize=8) + + # 2. Elasticity vs event size (|Δlog shares|) + ax = axes[2] + sizes = [abs(e["d_log_shares"]) for e in clean] + elasts = [e["result"]["elasticity"] for e in clean] + colors = ["green" if e["type"] == "deposit" else "coral" for e in clean] + ax.scatter(sizes, elasts, c=colors, alpha=0.4, s=15, edgecolors="none") + ax.axhline(1.0, color="gray", linestyle=":", alpha=0.5) + ax.axhline(np.median(elasts), color="red", linestyle="--", alpha=0.7, + label=f"median={np.median(elasts):+.2f}") + ax.set_xlabel("|Δlog(shares)| (event size)") + ax.set_ylabel("Elasticity") + ax.set_title("Elasticity vs Event Size") + ax.legend(fontsize=8) + + fig.suptitle(f"LP Event Elasticity Analysis — {len(clean)} clean events" + f" from {len(pools)} pools", fontsize=11) + fig.tight_layout() + p1 = os.path.join(out_dir, "elasticity_histograms.png") + fig.savefig(p1, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {p1}") + + # 3. Elasticity vs pre-event TVL (does pool size affect elasticity?) + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + + ax = axes[0] + tvl_pre = [e["result"]["tvl_pre"] for e in clean] + ax.scatter(tvl_pre, elasts, c=colors, alpha=0.4, s=15, edgecolors="none") + ax.set_xscale("log") + ax.axhline(1.0, color="gray", linestyle=":", alpha=0.5) + ax.set_xlabel("Pre-event TVL (USD)") + ax.set_ylabel("Elasticity") + ax.set_title("Elasticity vs Pool Size") + + # Bin by TVL decile and show median elasticity + tvl_arr = np.array(tvl_pre) + el_arr = np.array(elasts) + for q_lo, q_hi in [(0, 25), (25, 50), (50, 75), (75, 100)]: + lo = np.percentile(tvl_arr, q_lo) + hi = np.percentile(tvl_arr, q_hi) + mask = (tvl_arr >= lo) & (tvl_arr < hi + 1) + if mask.sum() > 5: + med_tvl = np.median(tvl_arr[mask]) + med_el = np.median(el_arr[mask]) + ax.plot(med_tvl, med_el, "rs", markersize=10, zorder=5) + ax.annotate(f"{med_el:.2f}", (med_tvl, med_el), + textcoords="offset points", xytext=(8, 5), fontsize=7) + + # 4. log(vol_post/vol_pre) vs log(tvl_post/tvl_pre) scatter + ax = axes[1] + log_tvl_ratio = [np.log(e["result"]["tvl_ratio"]) for e in clean] + log_vol_ratio = [np.log(e["result"]["vol_ratio"]) for e in clean] + ax.scatter(log_tvl_ratio, log_vol_ratio, c=colors, alpha=0.4, s=15, + edgecolors="none") + + # OLS fit line + x_fit = np.array(log_tvl_ratio) + y_fit = np.array(log_vol_ratio) + slope, intercept = np.polyfit(x_fit, y_fit, 1) + x_line = np.linspace(x_fit.min(), x_fit.max(), 100) + ax.plot(x_line, slope * x_line + intercept, "r-", linewidth=2, + label=f"OLS slope={slope:.2f}") + ax.plot(x_line, x_line, "k--", alpha=0.3, label="1:1 line") + ax.set_xlabel("Δlog(TVL)") + ax.set_ylabel("Δlog(Volume)") + ax.set_title("Volume Response to TVL Shocks") + ax.legend(fontsize=8) + + fig.suptitle("TVL→Volume Elasticity: Event Study", fontsize=11) + fig.tight_layout() + p2 = os.path.join(out_dir, "elasticity_scatter.png") + fig.savefig(p2, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {p2}") + + # 5. Elasticity by chain + fig, ax = plt.subplots(figsize=(10, 5)) + chain_data = {} + for e in clean: + ch = str(e["chain"]) + if ch not in chain_data: + chain_data[ch] = [] + chain_data[ch].append(e["result"]["elasticity"]) + chains_sorted = sorted(chain_data.keys(), + key=lambda c: len(chain_data[c]), reverse=True) + chains_plot = [c for c in chains_sorted if len(chain_data[c]) >= 5] + if chains_plot: + positions = range(len(chains_plot)) + bp = ax.boxplot([chain_data[c] for c in chains_plot], + positions=positions, widths=0.6, patch_artist=True) + for patch in bp["boxes"]: + patch.set_facecolor("steelblue") + patch.set_alpha(0.6) + ax.set_xticks(positions) + ax.set_xticklabels([f"{c}\n(n={len(chain_data[c])})" for c in chains_plot], + fontsize=8) + ax.axhline(1.0, color="gray", linestyle=":", alpha=0.5) + ax.axhline(np.median(clean_elast), color="red", linestyle="--", alpha=0.5, + label=f"overall median={np.median(clean_elast):.2f}") + ax.set_ylabel("Elasticity") + ax.set_title("Elasticity by Chain") + ax.set_ylim(-2, 5) + ax.legend(fontsize=8) + fig.tight_layout() + p3 = os.path.join(out_dir, "elasticity_by_chain.png") + fig.savefig(p3, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {p3}") + + +if __name__ == "__main__": + main() From 82c64c72c4c652615b8d401e2e9ae514860ce3f6 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 26 Mar 2026 11:34:01 +0000 Subject: [PATCH 067/115] feat: TVL counterfactual validation script Validates the full model (PCHIP arb + per-pool linear noise) against the AAVE/WETH natural experiment (70x TVL increase from LP deposit). Model predicts 44.3x total volume increase vs 39.2x observed (113% accuracy). V_arb carries 111x through PCHIP grid, V_noise adds 7.4x through the noise model (raw elasticity 0.42). Combined response matches observed despite individual channels having different elasticities from the event study total. Also evaluates counterfactual noise volumes at arbitrary TVL levels, using median pre-deposit market features with only TVL varying. --- experiments/validate_tvl_counterfactual.py | 236 +++++++++++++++++++++ 1 file changed, 236 insertions(+) create mode 100644 experiments/validate_tvl_counterfactual.py diff --git a/experiments/validate_tvl_counterfactual.py b/experiments/validate_tvl_counterfactual.py new file mode 100644 index 00000000..d4982b54 --- /dev/null +++ b/experiments/validate_tvl_counterfactual.py @@ -0,0 +1,236 @@ +"""Validate the noise model's TVL counterfactual predictions. + +Uses the AAVE/WETH reClAMM pool's natural experiment (70x TVL increase +from LP deposit in Jan 2026) to check whether the model's combined +V_arb + V_noise prediction matches observed volume changes. + +Tests whether the full model (PCHIP arb grid + per-pool linear noise +with b_tvl on standardized features) produces the right total volume +response, even though the noise-specific elasticity (~0.42 raw) is +lower than the event study's total elasticity (~0.9). + +Also evaluates counterfactual noise volumes at specified TVL levels. + +Usage: + python experiments/validate_tvl_counterfactual.py + python experiments/validate_tvl_counterfactual.py --pool 0x9d1fcf346ea1b0 + python experiments/validate_tvl_counterfactual.py --counterfactual-tvl 1e6 5e6 20e6 50e6 +""" + +import argparse +import json +import os +import pickle + +import numpy as np +import pandas as pd + +import jax.numpy as jnp +from quantammsim.calibration.grid_interpolation import interpolate_pool_daily +from quantammsim.calibration.noise_model_arrays import load_artifact + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +ARTIFACT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", +) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--pool", default="0x9d1fcf346ea1b0", + help="Pool ID prefix") + parser.add_argument("--artifact-dir", default=ARTIFACT_DIR) + parser.add_argument("--pre-cutoff", default="2026-01-10", + help="Date before which = pre-deposit") + parser.add_argument("--post-cutoff", default="2026-01-20", + help="Date after which = post-deposit") + parser.add_argument("--counterfactual-tvl", type=float, nargs="+", + default=[70_000, 500_000, 5_000_000, 20_000_000], + help="TVL values for counterfactual evaluation") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + # ---- Load model artifact ---- + art, meta = load_artifact(args.artifact_dir) + nc = art["noise_coeffs"] + log_cad = art["log_cadence"] + x_mean = art["x_mean"] + x_std = art["x_std"] + pool_ids = meta["pool_ids"] + feat_names = meta["feat_names"] + per_pool = nc.ndim == 2 + + # ---- Load pool data ---- + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + data = pickle.load(f) + mc = data["matched_clean"] + oc = data["option_c_clean"] + + pid = args.pool + if pid not in pool_ids: + # Try prefix match + matches = [p for p in pool_ids if p.startswith(pid) or pid.startswith(p)] + if matches: + pid = matches[0] + else: + print(f"Pool {args.pool} not found in calibration set") + return + + idx = pool_ids.index(pid) + coeffs = nc[idx] if per_pool else nc + cadence = float(np.exp(log_cad[idx])) + gas = float(np.exp(oc[pid]["log_gas"])) + tvl_col = feat_names.index("xobs_1") + + print("=" * 70) + print("TVL Counterfactual Validation") + print(f" Pool: {pid} ({mc[pid]['tokens']}, {mc[pid]['chain']})") + print(f" Learned cadence: {cadence:.1f} min") + print(f" Gas: ${gas:.2f}") + print(f" b_tvl (standardized): {coeffs[tvl_col]:.4f}") + print(f" TVL standardization: mean={x_mean[tvl_col]:.2f}," + f" std={x_std[tvl_col]:.2f}") + print(f" Raw noise elasticity: {coeffs[tvl_col]/x_std[tvl_col]:.4f}") + print("=" * 70) + + # ---- V_arb from PCHIP ---- + entry = mc[pid] + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], jnp.float64(np.log(cadence)), jnp.float64(gas))) + day_indices = entry["day_indices"] + v_arb = v_arb_all[day_indices] + + panel = entry["panel"] + log_vol = panel["log_volume"].values.astype(float) + log_tvl = panel["log_tvl_lag1"].values.astype(float) + vol_obs = np.exp(log_vol) + tvl = np.exp(log_tvl) + dates = pd.to_datetime(panel["date"]) + + pre_mask = dates < args.pre_cutoff + post_mask = dates >= args.post_cutoff + + # ---- Build full feature vectors ---- + from experiments.run_linear_market_noise import build_data + data_full = build_data(mc, oc, trend_windows=(7,), + include_market=True, include_cross_pool=True) + x_full = data_full["x"] + pool_idx_full = data_full["pool_idx"] + day_idx_full = data_full["day_idx"] + + pool_i = pool_ids.index(pid) + pool_mask = pool_idx_full == pool_i + + all_dates = set() + for p in pool_ids: + all_dates.update(mc[p]["panel"]["date"].values) + date_list = sorted(all_dates) + + sample_dates = np.array([pd.Timestamp(date_list[d]) + for d in day_idx_full[pool_mask]]) + sample_x = x_full[pool_mask] + sgd = data_full["sample_grid_days"][pool_mask] + v_arb_samples = v_arb_all[sgd] + + # Per-sample noise prediction + if per_pool: + log_v_noise = sample_x @ coeffs + else: + log_v_noise = sample_x @ coeffs + v_noise = np.exp(log_v_noise) + + sample_pre = sample_dates < pd.Timestamp(args.pre_cutoff) + sample_post = sample_dates >= pd.Timestamp(args.post_cutoff) + + # ---- Pre/post comparison ---- + print(f"\n=== Pre-deposit (before {args.pre_cutoff}) ===") + print(f" Median TVL: ${np.median(tvl[pre_mask]):>14,.0f}") + print(f" Median V_obs: ${np.median(vol_obs[pre_mask]):>14,.0f}") + print(f" Median V_arb: ${np.median(v_arb[pre_mask]):>14,.0f} (PCHIP)") + print(f" Median V_noise: ${np.median(v_noise[sample_pre]):>14,.0f} (model)") + v_total_pre = v_arb_samples[sample_pre] + v_noise[sample_pre] + print(f" Median V_total: ${np.median(v_total_pre):>14,.0f} (V_arb + V_noise)") + + print(f"\n=== Post-deposit (after {args.post_cutoff}) ===") + print(f" Median TVL: ${np.median(tvl[post_mask]):>14,.0f}") + print(f" Median V_obs: ${np.median(vol_obs[post_mask]):>14,.0f}") + print(f" Median V_arb: ${np.median(v_arb[post_mask]):>14,.0f} (PCHIP)") + print(f" Median V_noise: ${np.median(v_noise[sample_post]):>14,.0f} (model)") + v_total_post = v_arb_samples[sample_post] + v_noise[sample_post] + print(f" Median V_total: ${np.median(v_total_post):>14,.0f} (V_arb + V_noise)") + + # ---- Ratios ---- + tvl_ratio = np.median(tvl[post_mask]) / np.median(tvl[pre_mask]) + vol_ratio = np.median(vol_obs[post_mask]) / np.median(vol_obs[pre_mask]) + varb_ratio = np.median(v_arb[post_mask]) / np.median(v_arb[pre_mask]) + vnoise_ratio = np.median(v_noise[sample_post]) / np.median(v_noise[sample_pre]) + vtotal_ratio = np.median(v_total_post) / np.median(v_total_pre) + + print(f"\n=== Ratios (post / pre) ===") + print(f" TVL: {tvl_ratio:>8.1f}x") + print(f" V_obs: {vol_ratio:>8.1f}x (ground truth)") + print(f" V_arb: {varb_ratio:>8.1f}x (PCHIP grid)") + print(f" V_noise: {vnoise_ratio:>8.1f}x (noise model)") + print(f" V_total: {vtotal_ratio:>8.1f}x (V_arb + V_noise)") + print(f" Gap: {vtotal_ratio/vol_ratio:>8.2f}x (pred/obs)") + + # ---- Decomposition shares ---- + print(f"\n=== Decomposition shares ===") + arb_share_pre = np.median(v_arb[pre_mask]) / np.median(vol_obs[pre_mask]) * 100 + noise_share_pre = np.median(v_noise[sample_pre]) / np.median(vol_obs[pre_mask]) * 100 + arb_share_post = np.median(v_arb[post_mask]) / np.median(vol_obs[post_mask]) * 100 + noise_share_post = np.median(v_noise[sample_post]) / np.median(vol_obs[post_mask]) * 100 + + print(f" Pre: arb={arb_share_pre:.0f}% noise={noise_share_pre:.0f}%") + print(f" Post: arb={arb_share_post:.0f}% noise={noise_share_post:.0f}%") + + # ---- Counterfactual evaluation ---- + print(f"\n=== Counterfactual noise volumes ===") + print(f" (Using median pre-deposit market features, varying TVL only)") + print(f" {'TVL':>14s} {'V_noise/day':>12s} {'V_noise/min':>12s}" + f" {'Ratio vs 70K':>12s}") + print(f" {'-'*55}") + + x_base = np.median(sample_x[sample_pre], axis=0).copy() + baseline_tvl = 70_000 + x_baseline = x_base.copy() + x_baseline[tvl_col] = (np.log(baseline_tvl) - x_mean[tvl_col]) / x_std[tvl_col] + for i, name in enumerate(feat_names): + if name.startswith("xobs_1" + "\u00d7"): + paired_name = name.split("\u00d7")[1] + if paired_name in feat_names: + paired_idx = feat_names.index(paired_name) + x_baseline[i] = x_baseline[tvl_col] * x_base[paired_idx] + vn_baseline = np.exp(x_baseline @ coeffs) + + for cf_tvl in args.counterfactual_tvl: + x_cf = x_base.copy() + std_log_tvl = (np.log(cf_tvl) - x_mean[tvl_col]) / x_std[tvl_col] + x_cf[tvl_col] = std_log_tvl + for i, name in enumerate(feat_names): + if name.startswith("xobs_1" + "\u00d7"): + paired_name = name.split("\u00d7")[1] + if paired_name in feat_names: + paired_idx = feat_names.index(paired_name) + x_cf[i] = std_log_tvl * x_base[paired_idx] + + vn = np.exp(x_cf @ coeffs) + ratio = vn / vn_baseline + print(f" ${cf_tvl:>13,.0f} ${vn:>11,.0f} ${vn/1440:>11,.0f}" + f" {ratio:>11.1f}x") + + print(f"\n Key finding: model predicts {vtotal_ratio:.1f}x total volume" + f" increase vs {vol_ratio:.1f}x observed ({vtotal_ratio/vol_ratio:.0%} accuracy).") + print(f" V_arb ({varb_ratio:.0f}x) carries most of the response;" + f" V_noise ({vnoise_ratio:.1f}x) is secondary but adds up.") + + +if __name__ == "__main__": + main() From 75b683fb28b5f26f4d162bea5835723212851279 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 26 Mar 2026 22:36:54 +0000 Subject: [PATCH 068/115] feat: MLP noise model (Binance-only, no cross-pool DEX dependency) + volume_zscore features MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - run_mlp_noise.py: MLP noise model with learnable cadence, no panel dependency. Uses only Binance market data + pool TVL. Supports variable depth/width, per-pool bias, optax cosine LR decay, and Optuna sweep over architecture + hyperparameters. Best eval R² = 0.39 (matches linear baseline) with [16,8,4]. In-sample R² = 0.70 with [128,64,32] — overfits on temporal split. - market_features.py: add volume_zscore feature — within-token rolling z-score of daily Binance USD volume (today vs 30d trailing mean/std). Captures "unusually active day for this token" without cross-token scale issues. Added for BTC, token A, and token B. --- experiments/run_mlp_noise.py | 520 +++++++++++++++++++++ quantammsim/calibration/market_features.py | 11 + 2 files changed, 531 insertions(+) create mode 100644 experiments/run_mlp_noise.py diff --git a/experiments/run_mlp_noise.py b/experiments/run_mlp_noise.py new file mode 100644 index 00000000..bde323e1 --- /dev/null +++ b/experiments/run_mlp_noise.py @@ -0,0 +1,520 @@ +"""MLP noise model with Binance market features and learnable cadence. + +No cross-pool DEX dependency — only uses this pool's TVL + public market +data (Binance prices/volumes for BTC and the pool's tokens). + +Architecture: + log(V_noise) = MLP(x_market) + V_total = V_arb(cadence) + exp(log_v_noise) + +where x_market = [log_tvl, dow_sin, dow_cos, btc_features, tok_a_features, +tok_b_features, pair_vol, interactions]. + +Cadence is per-pool, learned jointly via Adam through PCHIP. + +Usage: + python experiments/run_mlp_noise.py + python experiments/run_mlp_noise.py --hidden 64 32 --epochs 3000 + python experiments/run_mlp_noise.py --per-pool --hidden 32 +""" + +import argparse +import os +import time + +import jax +import jax.numpy as jnp +import numpy as np + + +# ---- Model ---- + + +def init_mlp_params(key, n_input, hidden_sizes, n_pools, init_log_cadences, + per_pool=False): + """Initialize MLP parameters. + + MLP: input → hidden1 → ... → hiddenN → 1 (with ReLU activations). + If per_pool: separate output bias per pool. + """ + params = {} + keys = jax.random.split(key, len(hidden_sizes) + 2) + + # Hidden layers + in_dim = n_input + for i, h in enumerate(hidden_sizes): + params[f"W{i}"] = jax.random.normal(keys[i], (in_dim, h)) * np.sqrt(2.0 / in_dim) + params[f"b{i}"] = jnp.zeros(h) + in_dim = h + + # Output layer → scalar + params["W_out"] = jax.random.normal(keys[-2], (in_dim, 1)) * 0.01 + params["b_out"] = jnp.zeros(1) + + if per_pool: + params["pool_bias"] = jnp.zeros(n_pools) + + params["log_cadence"] = jnp.array(init_log_cadences) + return params + + +def forward_mlp(params, x, pool_idx=None): + """MLP forward pass. Returns (n_samples,) log_v_noise.""" + h = x + i = 0 + while f"W{i}" in params: + h = jnp.maximum(h @ params[f"W{i}"] + params[f"b{i}"], 0.0) + i += 1 + out = (h @ params["W_out"] + params["b_out"])[:, 0] + + if "pool_bias" in params and pool_idx is not None: + out = out + params["pool_bias"][pool_idx] + + return out + + +def make_loss_fn(pool_coeffs, pool_gas, n_pools): + """Loss with learnable cadence + MLP noise model.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + def loss_fn(params, x, y_total, sample_grid_days, pool_idx, + l2_alpha, huber_delta): + log_v_noise = forward_mlp(params, x, pool_idx) + + log_cadence = params["log_cadence"] + n_samples = y_total.shape[0] + v_arb = jnp.zeros(n_samples) + for i in range(n_pools): + v_arb_all = interpolate_pool_daily( + pool_coeffs[i], log_cadence[i], pool_gas[i]) + safe_days = jnp.clip(sample_grid_days, 0, v_arb_all.shape[0] - 1) + v_arb = jnp.where(pool_idx == i, v_arb_all[safe_days], v_arb) + + log_v_arb = jnp.log(jnp.maximum(v_arb, 1e-10)) + log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) + + residuals = log_v_total - y_total + abs_r = jnp.abs(residuals) + huber_vals = jnp.where(abs_r <= huber_delta, 0.5 * residuals ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + pool_counts = jnp.zeros(n_pools).at[pool_idx].add( + jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber_vals) + data_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active + + reg = l2_alpha * sum(jnp.sum(v ** 2) for k, v in params.items() + if k.startswith("W")) + return data_loss + reg + + return jax.jit(jax.value_and_grad(loss_fn)) + + +def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, + verbose=True, use_cosine=False, warmup_steps=100): + x = jnp.array(data["x"]) + y = jnp.array(data["y_total"]) + sgd = jnp.array(data["sample_grid_days"]) + pidx = jnp.array(data["pool_idx"]) + + if use_cosine: + import optax + schedule = optax.warmup_cosine_decay_schedule( + init_value=lr * 0.01, + peak_value=lr, + warmup_steps=warmup_steps, + decay_steps=n_epochs, + end_value=lr * 0.01, + ) + optimizer = optax.adam(learning_rate=schedule) + opt_state = optimizer.init(params) + + for epoch in range(n_epochs): + loss_val, grads = grad_fn( + params, x, y, sgd, pidx, l2_alpha, huber_delta) + updates, opt_state = optimizer.update(grads, opt_state, params) + params = optax.apply_updates(params, updates) + + if verbose and (epoch % 500 == 0 or epoch == n_epochs - 1): + cads = np.exp(np.array(params["log_cadence"])) + cur_lr = float(schedule(epoch)) + print(f" epoch {epoch:5d} loss={float(loss_val):.6f}" + f" lr={cur_lr:.2e}" + f" cad=[{cads.min():.1f}-{np.median(cads):.1f}-{cads.max():.1f}]") + else: + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + + for epoch in range(n_epochs): + loss_val, grads = grad_fn( + params, x, y, sgd, pidx, l2_alpha, huber_delta) + + for k in params: + m[k] = 0.9 * m[k] + 0.1 * grads[k] + v[k] = 0.999 * v[k] + 0.001 * grads[k] ** 2 + m_hat = m[k] / (1.0 - 0.9 ** (epoch + 1)) + v_hat = v[k] / (1.0 - 0.999 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + 1e-8) + + if verbose and (epoch % 500 == 0 or epoch == n_epochs - 1): + cads = np.exp(np.array(params["log_cadence"])) + print(f" epoch {epoch:5d} loss={float(loss_val):.6f}" + f" cad=[{cads.min():.1f}-{np.median(cads):.1f}-{cads.max():.1f}]") + + return params + + +def evaluate(params, data, label=""): + """Evaluate decomposition.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + x = np.array(data["x"]) + y_total = np.array(data["y_total"]) + pool_idx = np.array(data["pool_idx"]) + sgd = np.array(data["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + init_cads = data["init_log_cadences"] + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + + log_v_noise = np.array(forward_mlp( + params, jnp.array(x), + jnp.array(pool_idx) if "pool_bias" in params else None)) + v_noise = np.exp(log_v_noise) + + v_arb = np.zeros(len(y_total)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd[mask]] + + v_obs = np.exp(y_total) + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + if label: + print(f"\n {label}:") + print(f" {'Pool'[:16]:16s} {'R²':>6s} {'Cad':>5s} → {'learn':>5s}" + f" {'Arb%':>6s} {'Noise%':>7s} {'Flag':>5s}") + print(f" {'-'*55}") + + r2s = {} + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y_total[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s[i] = 1 - ss_res / max(ss_tot, 1e-10) + + pid = pool_ids[i] + ci = np.exp(init_cads[i]) + cl = np.exp(log_cadence[i]) + arb_pct = np.median(v_arb[mask] / v_obs[mask]) * 100 + noise_pct = np.median(v_noise[mask] / v_obs[mask]) * 100 + flags = [] + if arb_pct > 150: flags.append("A") + if cl <= 1.01 or cl >= 59.9: flags.append("B") + if r2s[i] < 0: flags.append("X") + print(f" {pid[:16]:16s} {r2s[i]:6.3f} {ci:5.1f} → {cl:5.1f}" + f" {arb_pct:6.0f}% {noise_pct:6.0f}% {''.join(flags):>5s}") + + vals = [x for x in r2s.values() if np.isfinite(x)] + med = np.median(vals) if vals else float("nan") + healthy = [r for r in r2s.values() if r > 0 and np.isfinite(r)] + med_h = np.median(healthy) if healthy else float("nan") + print(f"\n Median R²: {med:.4f} (healthy: {med_h:.4f})") + return med, r2s + + +def run_optuna(data, n_trials): + """Optuna sweep over MLP architecture and training hyperparameters.""" + import optuna + + day_idx = data["day_idx"] + n_samples = len(day_idx) + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = {k: v[train_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + + n_pools = data["n_pools"] + n_feat = data["n_feat"] + + def objective(trial): + # Architecture + n_layers = trial.suggest_int("n_layers", 1, 5) + first_hidden = trial.suggest_categorical("first_hidden", [8, 16, 32, 64]) + # Bottleneck: each layer is half the previous (min 2) + hidden = [] + h = first_hidden + for _ in range(n_layers): + hidden.append(h) + h = max(h // 2, 2) + + # Training + lr = trial.suggest_float("lr", 1e-4, 1e-2, log=True) + l2_alpha = trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True) + huber_delta = trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5, 2.0]) + n_epochs = trial.suggest_categorical("n_epochs", [2000, 5000, 10000]) + use_cosine = trial.suggest_categorical("use_cosine", [True, False]) + per_pool = trial.suggest_categorical("per_pool", [True, False]) + + params = init_mlp_params( + jax.random.PRNGKey(42), n_feat, hidden, n_pools, + data["init_log_cadences"], per_pool=per_pool) + + # OLS warm-start + x_trn = jnp.array(train_data["x"]) + y_trn = np.array(train_data["y_total"]) + h_act = np.array(x_trn) + i = 0 + while f"W{i}" in params: + h_act = np.maximum( + h_act @ np.array(params[f"W{i}"]) + np.array(params[f"b{i}"]), 0.0) + i += 1 + h_bias = np.concatenate([h_act, np.ones((h_act.shape[0], 1))], axis=1) + sol, _, _, _ = np.linalg.lstsq(h_bias, y_trn[:, None], rcond=None) + params["W_out"] = jnp.array(sol[:-1].astype(np.float32)) + params["b_out"] = jnp.array(sol[-1:].astype(np.float32)) + + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + params = train(params, train_data, grad_fn, n_epochs, lr, + l2_alpha, huber_delta, verbose=False, + use_cosine=use_cosine) + + # Eval + x_eval = np.array(eval_data["x"]) + y_eval = np.array(eval_data["y_total"]) + pool_idx_eval = np.array(eval_data["pool_idx"]) + sgd_eval = np.array(eval_data["sample_grid_days"]) + log_cadence = np.array(params["log_cadence"]) + + log_v_noise = np.array(forward_mlp( + params, jnp.array(x_eval), + jnp.array(pool_idx_eval) if per_pool else None)) + + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + v_arb = np.zeros(len(y_eval)) + for i in range(n_pools): + mask = pool_idx_eval == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + v_arb[mask] = v_arb_all[sgd_eval[mask]] + + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + pred_total = np.logaddexp(log_v_arb, log_v_noise) + + r2s = [] + for i in range(n_pools): + mask = pool_idx_eval == i + if mask.sum() < 2: + continue + yt = y_eval[mask] + pt = pred_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s.append(1 - ss_res / max(ss_tot, 1e-10)) + + med_r2 = float(np.median(r2s)) if r2s else -10.0 + arch_str = "×".join(str(h) for h in hidden) + print(f" Trial {trial.number}: eval={med_r2:.4f}" + f" arch=[{arch_str}]" + f" {'cosine' if use_cosine else 'const'}" + f" {'per_pool' if per_pool else 'shared'}" + f" lr={lr:.1e} l2={l2_alpha:.1e}" + f" hub={huber_delta} ep={n_epochs}") + return med_r2 + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print(f"Optuna Results (MLP noise)") + print(f"{'='*70}") + print(f" Best eval R²: {study.best_value:.4f}") + print(f" Best params:") + for k, v in sorted(study.best_params.items()): + print(f" {k}: {v}") + + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + print(f"\n Top 10:") + for t in trials[:10]: + if t.value is not None: + n_l = t.params["n_layers"] + fh = t.params["first_hidden"] + h = fh + arch = [] + for _ in range(n_l): + arch.append(h) + h = max(h // 2, 2) + print(f" #{t.number}: eval={t.value:.4f}" + f" arch={arch}" + f" ep={t.params['n_epochs']}" + f" {'cos' if t.params['use_cosine'] else 'cst'}" + f" {'pp' if t.params['per_pool'] else 'sh'}") + + return study + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--hidden", type=int, nargs="+", default=[32], + help="Hidden layer sizes (e.g. --hidden 64 32)") + parser.add_argument("--epochs", type=int, default=2000) + parser.add_argument("--lr", type=float, default=1e-3) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--cosine", action="store_true", + help="Use optax Adam with cosine LR decay") + parser.add_argument("--tune", type=int, default=0, + help="Optuna sweep (0 = single run)") + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) + parser.add_argument("--per-pool", action="store_true", + help="Per-pool output bias") + parser.add_argument("--pool-attrs", action="store_true", + help="Append static pool attributes to input") + parser.add_argument("--no-split", action="store_true", + help="Train on all data") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("MLP Noise Model (Binance features only, no cross-pool DEX)") + print(f" hidden={args.hidden}, per_pool={args.per_pool}") + print(f" epochs={args.epochs}, lr={args.lr}, l2={args.l2_alpha}") + print("=" * 70) + + # Build data WITHOUT cross-pool features + from experiments.run_linear_market_noise import load_stage1, build_data + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding data...") + t0 = time.time() + data = build_data( + matched_clean, option_c_clean, + trend_windows=tuple(args.trend_windows), + include_market=True, + include_cross_pool=False, # No DEX peer features + ) + n_pools = data["n_pools"] + n_feat = data["n_feat"] + print(f" {len(data['pool_idx'])} samples, {n_pools} pools," + f" {n_feat} features, {time.time() - t0:.1f}s") + + # Append pool attributes if requested + if args.pool_attrs: + from quantammsim.calibration.pool_data import build_pool_attributes + X_attr, attr_names, _ = build_pool_attributes(matched_clean) + # Standardize + attr_mean = X_attr.mean(axis=0) + attr_std = X_attr.std(axis=0) + attr_std[attr_std < 1e-6] = 1.0 + X_attr_norm = ((X_attr - attr_mean) / attr_std).astype(np.float32) + + # Broadcast to per-sample: each sample gets its pool's attributes + pool_idx = data["pool_idx"] + x_attr_samples = X_attr_norm[pool_idx] + data["x"] = np.concatenate([data["x"], x_attr_samples], axis=1) + data["n_feat"] = data["x"].shape[1] + data["feat_names"] = data["feat_names"] + attr_names + n_feat = data["n_feat"] + print(f" + {len(attr_names)} pool attributes → {n_feat} total features") + + print(f" Features: {data['feat_names']}") + + if args.tune > 0: + run_optuna(data, args.tune) + return + + # Split + if args.no_split: + train_data = data + eval_data = None + else: + day_idx = data["day_idx"] + n_samples = len(day_idx) + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = {k: v[train_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + + # Init + params = init_mlp_params( + jax.random.PRNGKey(42), n_feat, args.hidden, n_pools, + data["init_log_cadences"], per_pool=args.per_pool) + + n_params = sum(v.size for v in params.values()) + print(f"\n Total params: {n_params}" + f" (MLP: {n_params - n_pools - (n_pools if args.per_pool else 0)}," + f" cadence: {n_pools}" + f"{',' + str(n_pools) + ' pool biases' if args.per_pool else ''})") + + # Warm-start output layer via OLS through hidden activations + x_trn = jnp.array(train_data["x"]) + y_trn = np.array(train_data["y_total"]) + h = np.array(x_trn) + i = 0 + while f"W{i}" in params: + h = np.maximum(h @ np.array(params[f"W{i}"]) + np.array(params[f"b{i}"]), 0.0) + i += 1 + h_bias = np.concatenate([h, np.ones((h.shape[0], 1))], axis=1) + sol, _, _, _ = np.linalg.lstsq(h_bias, y_trn[:, None], rcond=None) + params["W_out"] = jnp.array(sol[:-1].astype(np.float32)) + params["b_out"] = jnp.array(sol[-1:].astype(np.float32)) + print(f" OLS warm-start on hidden activations") + + # Train + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + + print(f"\n Compiling + training...") + t0 = time.time() + params = train(params, train_data, grad_fn, args.epochs, args.lr, + args.l2_alpha, args.huber_delta, + use_cosine=args.cosine) + print(f" Training: {time.time() - t0:.1f}s") + + # Evaluate + if eval_data is not None: + print("\n --- Train ---") + evaluate(params, train_data) + print("\n --- Eval ---") + evaluate(params, eval_data) + else: + print("\n --- All data ---") + evaluate(params, train_data) + + print(f"\n Baselines:") + print(f" Linear (no cross-pool): median R² ≈ 0.48") + print(f" Linear (with cross-pool): median R² ≈ 0.53") + print(f" Per-pool linear: median R² ≈ 0.61") + + +if __name__ == "__main__": + main() diff --git a/quantammsim/calibration/market_features.py b/quantammsim/calibration/market_features.py index 7a7ee59b..f2dd610b 100644 --- a/quantammsim/calibration/market_features.py +++ b/quantammsim/calibration/market_features.py @@ -77,6 +77,10 @@ def _compute_token_features( For market-level tokens (BTC): includes log_price as regime proxy. For pool tokens: only returns/vol/trends (comparable across tokens). + Volume is normalised as z-score within each token: today's log-volume + relative to a 30-day trailing mean/std. This captures "is this token + unusually active today" without the cross-token scale problem. + Returns DataFrame indexed by date. """ out = pd.DataFrame(index=daily.index) @@ -90,6 +94,13 @@ def _compute_token_features( # Realized volatility: std of log returns over trailing 7 days out["realized_vol_7d"] = out["log_return"].rolling(7, min_periods=3).std() + # Volume: z-score relative to trailing 30d mean/std of log-volume + # Captures "unusually active day for this token" — comparable across tokens + log_vol = np.log(daily["volume_usd"].clip(lower=1.0)) + vol_mean_30d = log_vol.rolling(30, min_periods=10).mean() + vol_std_30d = log_vol.rolling(30, min_periods=10).std().clip(lower=0.1) + out["volume_zscore"] = (log_vol - vol_mean_30d) / vol_std_30d + # Trend: rolling mean log return at various horizons for w in trend_windows: out[f"trend_{w}d"] = out["log_return"].rolling(w, min_periods=max(w // 2, 2)).mean() From 11b8858864464bfc04da4541ba061098cdee8254 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 26 Mar 2026 23:04:28 +0000 Subject: [PATCH 069/115] feat: remove panel dependency from simulator arrays, Binance-only pipeline - noise_model_arrays.py: rewrite build_simulator_arrays to use Binance parquets directly (no panel/API dependency). Takes token_a, token_b + date range, builds all features from market data. Works for any date range covered by Binance data. Tested: 639 days for AAVE/ETH. - tune_reclamm_calibrated_noise.py: update to new build_simulator_arrays interface (token_a/token_b instead of pool_id + matched_clean). Extended date range (2024-06 to 2026-03) now works. - run_mlp_noise.py: add Optuna sweep (--tune), optax cosine LR decay (--cosine), pool attributes (--pool-attrs). --- experiments/tune_reclamm_calibrated_noise.py | 4 +- quantammsim/calibration/noise_model_arrays.py | 262 +++++++++--------- 2 files changed, 138 insertions(+), 128 deletions(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index 4eda1a6b..8a818513 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -75,10 +75,12 @@ def _build_market_linear_arrays(args): print(f" Building market_linear noise arrays for {POOL_ID}...") print(f" Date range: {start} → {end}") arrays = build_simulator_arrays( - pool_id=POOL_ID, + token_a="AAVE", + token_b="ETH", start_date=start, end_date=end, artifact_dir=args.artifact_dir, + pool_id=POOL_ID, ) print(f" {arrays['n_days']} days, {arrays['n_minutes']} minutes") print(f" noise_base range: [{arrays['noise_base'].min():.2f}," diff --git a/quantammsim/calibration/noise_model_arrays.py b/quantammsim/calibration/noise_model_arrays.py index b8798c37..0c7c904e 100644 --- a/quantammsim/calibration/noise_model_arrays.py +++ b/quantammsim/calibration/noise_model_arrays.py @@ -1,30 +1,27 @@ """Precompute noise_base and noise_tvl_coeff arrays for the simulator. -Takes a trained per-pool noise model artifact and produces the two daily -arrays needed by reclamm_market_linear_noise_volume(): +Builds daily feature vectors from Binance price data only — no panel/API +dependency. Works for any date range covered by Binance parquets. - log(V_daily_noise) = noise_base_t + noise_tvl_coeff_t * log(effective_TVL) +Produces the two arrays needed by reclamm_market_linear_noise_volume(): -The arrays are at daily resolution and need to be expanded to minute-level -(by repeating each day's value 1440 times) before passing to the simulator. + log(V_daily_noise) = noise_base_t + noise_tvl_coeff_t * log(effective_TVL) Usage: from quantammsim.calibration.noise_model_arrays import build_simulator_arrays arrays = build_simulator_arrays( - pool_id="0x0b09dea16768f0", - start_date="2025-06-01", + token_a="AAVE", token_b="ETH", + start_date="2024-06-01", end_date="2026-03-01", artifact_dir="results/linear_market_noise", + pool_id="0x9d1fcf346ea1b0", # for per-pool coeffs ) - # arrays["noise_base"] — (n_minutes,) float64 - # arrays["noise_tvl_coeff"] — (n_minutes,) float64 """ import json import os -from datetime import date, timedelta -from typing import Dict, Optional, Tuple +from typing import Dict, List, Optional, Tuple import numpy as np import pandas as pd @@ -51,8 +48,7 @@ def _identify_tvl_columns(feat_names: list) -> Tuple[int, list]: Returns: tvl_col: index of the pure log_tvl feature (xobs_1) - tvl_interaction_cols: list of (col_idx, paired_col_idx) for - interaction terms that multiply TVL with another feature + tvl_interaction_cols: list of (col_idx, paired_col_idx) """ tvl_col = None tvl_interaction_cols = [] @@ -60,9 +56,8 @@ def _identify_tvl_columns(feat_names: list) -> Tuple[int, list]: for i, name in enumerate(feat_names): if name == "xobs_1": tvl_col = i - elif "xobs_1×" in name: - # e.g. "xobs_1×btc_realized_vol_7d" — find the paired feature - paired_name = name.split("×")[1] + elif "xobs_1\u00d7" in name: + paired_name = name.split("\u00d7")[1] for j, n2 in enumerate(feat_names): if n2 == paired_name: tvl_interaction_cols.append((i, j)) @@ -74,96 +69,119 @@ def _identify_tvl_columns(feat_names: list) -> Tuple[int, list]: return tvl_col, tvl_interaction_cols -def build_daily_features( - pool_id: str, - matched_clean: dict, +def build_daily_features_from_binance( + token_a: str, + token_b: str, start_date: str, end_date: str, - feat_names: list, + feat_names: List[str], x_mean: np.ndarray, x_std: np.ndarray, trend_windows: tuple = (7,), ) -> Tuple[np.ndarray, list]: - """Build the full standardized feature matrix for a pool over a date range. - - Returns (x_daily, dates) where x_daily is (n_days, n_feat) and dates - is the list of dates. TVL column (xobs_1) is filled with the pool's - observed log_tvl_lag1 where available, 0 otherwise. + """Build daily feature matrix from Binance data only. + + No panel or API dependency. Features: + - xobs_0 (intercept), xobs_1 (log_tvl — filled with 0, handled at runtime), + xobs_2/3 (dow_sin/cos) + - BTC: log_price, log_return, realized_vol_7d, trend, volume_zscore + - Token A/B: log_return, realized_vol_7d, trend, volume_zscore + - Pair realized_vol_7d + - Interaction terms """ - from quantammsim.calibration.pool_data import ( - build_x_obs, build_cross_pool_x_obs, K_OBS_CROSS, - ) from quantammsim.calibration.market_features import ( - build_pool_market_features, + build_btc_daily_features, + build_token_daily_features, + _compute_pair_volatility, + TOKEN_MAP, ) - # Find the pool - pid_match = None - for pid in matched_clean: - if pool_id.startswith(pid) or pid.startswith(pool_id): - pid_match = pid - break - if pid_match is None: - raise ValueError(f"Pool {pool_id} not found in matched_clean") - - entry = matched_clean[pid_match] - panel = entry["panel"] - - # Filter panel to date range start = pd.Timestamp(start_date) end = pd.Timestamp(end_date) - panel_dates = pd.to_datetime(panel["date"]) - mask = (panel_dates >= start) & (panel_dates <= end) - panel_sub = panel[mask.values].copy() - n_days = len(panel_sub) - - if n_days < 2: - raise ValueError(f"Only {n_days} days in range for pool {pool_id}") - - dates = panel_sub["date"].values - - # x_obs (cross-pool, 7 features) — need at least 1 lag - xc = build_cross_pool_x_obs(panel_sub, matched_clean, pid_match) - # xc drops first row; align - if len(xc) < n_days: - # Pad first row with zeros - xc = np.vstack([np.zeros((1, xc.shape[1])), xc]) - - # Market features - pool_feat = build_pool_market_features( - matched_clean, trend_windows=list(trend_windows)) - pf = pool_feat.get(pid_match) - if pf is None: - raise ValueError(f"No market features for {pool_id}") - - # Align market features to panel dates - n_base = K_OBS_CROSS - market_cols = [c for c in sorted(pf.columns)] - n_market = len(market_cols) - - x_base = np.zeros((n_days, n_base + n_market), dtype=np.float32) - x_base[:, :n_base] = xc[:n_days] - - for k, d in enumerate(dates): - day = pd.Timestamp(d).normalize() - if day in pf.index: - for m, col in enumerate(market_cols): - val = pf.loc[day, col] - if np.isfinite(val): - x_base[k, n_base + m] = val - - # Standardize using saved stats (base features only) - n_base_total = n_base + n_market - x_base = ((x_base - x_mean[:n_base_total]) / x_std[:n_base_total]).astype(np.float32) + + # Generate complete daily date range + date_range = pd.date_range(start, end, freq="D") + n_days = len(date_range) + + # BTC features + btc_feat = build_btc_daily_features(list(trend_windows)) + + # Token features + mapped_a = TOKEN_MAP.get(token_a, token_a) + mapped_b = TOKEN_MAP.get(token_b, token_b) + feat_a = build_token_daily_features(mapped_a, list(trend_windows)) + feat_b = build_token_daily_features(mapped_b, list(trend_windows)) + + # Pair volatility + pair_vol = _compute_pair_volatility(token_a, token_b) + + # Identify which features are x_obs vs market + # x_obs features: xobs_0 (intercept), xobs_1 (tvl), xobs_2 (dow_sin), xobs_3 (dow_cos) + # Remaining xobs_4,5,6 are cross-pool — skip if not in feat_names + n_xobs = sum(1 for f in feat_names if f.startswith("xobs_")) + + # Build market feature column list (everything after x_obs, before interactions) + market_names = [f for f in feat_names + if not f.startswith("xobs_") and "\u00d7" not in f] + + # Build per-day feature vectors + x_base_cols = n_xobs + len(market_names) + x_base = np.zeros((n_days, x_base_cols), dtype=np.float32) + + for k, day in enumerate(date_range): + day_norm = day.normalize() + + # x_obs + x_base[k, 0] = 1.0 # intercept + # x_base[k, 1] = 0.0 # log_tvl — placeholder, handled at runtime + weekday = day.weekday() + if n_xobs > 2: + x_base[k, 2] = np.sin(2 * np.pi * weekday / 7) + if n_xobs > 3: + x_base[k, 3] = np.cos(2 * np.pi * weekday / 7) + # xobs_4,5,6 (cross-pool) left as 0 if present + + # Market features + col = n_xobs + for mname in market_names: + val = 0.0 + if mname.startswith("btc_") and btc_feat is not None: + bcol = mname[4:] # strip "btc_" + if day_norm in btc_feat.index and bcol in btc_feat.columns: + v = btc_feat.loc[day_norm, bcol] + if np.isfinite(v): + val = v + elif mname.startswith("tok_a_") and feat_a is not None: + acol = mname[6:] + if day_norm in feat_a.index and acol in feat_a.columns: + v = feat_a.loc[day_norm, acol] + if np.isfinite(v): + val = v + elif mname.startswith("tok_b_") and feat_b is not None: + bcol = mname[6:] + if day_norm in feat_b.index and bcol in feat_b.columns: + v = feat_b.loc[day_norm, bcol] + if np.isfinite(v): + val = v + elif mname == "pair_realized_vol_7d" and pair_vol is not None: + if day_norm in pair_vol.index: + v = pair_vol.loc[day_norm, "pair_realized_vol_7d"] + if np.isfinite(v): + val = v + x_base[k, col] = val + col += 1 + + # Standardize base features + x_base = ((x_base - x_mean[:x_base_cols]) / x_std[:x_base_cols]).astype(np.float32) # Interaction terms - base_names = [f"xobs_{i}" for i in range(n_base)] + market_cols - col_idx = {name: i for i, name in enumerate(base_names)} + base_feat_names = feat_names[:x_base_cols] + col_idx = {name: i for i, name in enumerate(base_feat_names)} interactions = [] - for fname in feat_names[n_base_total:]: - if "×" in fname: - parts = fname.split("×") + for fname in feat_names[x_base_cols:]: + if "\u00d7" in fname: + parts = fname.split("\u00d7") if parts[0] in col_idx and parts[1] in col_idx: interactions.append( x_base[:, col_idx[parts[0]]] * x_base[:, col_idx[parts[1]]]) @@ -178,40 +196,37 @@ def build_daily_features( else: x_all = x_base - return x_all, dates.tolist() + return x_all, date_range.tolist() def build_simulator_arrays( - pool_id: str, + token_a: str, + token_b: str, start_date: str, end_date: str, artifact_dir: str = "results/linear_market_noise", - matched_clean: Optional[dict] = None, - arb_frequency: int = 1, -) -> Dict[str, np.ndarray]: + pool_id: Optional[str] = None, +) -> Dict: """Build noise_base and noise_tvl_coeff arrays for the simulator. + No panel dependency — uses Binance data only. + Parameters ---------- - pool_id : str - Pool ID (full or prefix). + token_a, token_b : str + Token symbols (e.g. "AAVE", "ETH"). Mapped to Binance symbols + internally (WETH→ETH, wstETH→ETH, etc.) start_date, end_date : str Date range (inclusive). artifact_dir : str Directory containing model.npz and meta.json. - matched_clean : dict, optional - Pre-loaded matched_clean dict. If None, loads from stage1.pkl. - arb_frequency : int - Arb frequency in minutes. Arrays are at minute resolution, - repeated from daily values. + pool_id : str, optional + Pool ID for per-pool coefficients. If None or not found, + uses median coefficients. Returns ------- - dict with: - noise_base : (n_minutes,) array - noise_tvl_coeff : (n_minutes,) array - dates : list of dates - pool_index : int (index in calibration set, or -1) + dict with noise_base, noise_tvl_coeff, tvl_mean, tvl_std, dates, etc. """ art, meta = load_artifact(artifact_dir) noise_coeffs = art["noise_coeffs"] @@ -222,51 +237,44 @@ def build_simulator_arrays( per_pool = noise_coeffs.ndim == 2 trend_windows = tuple(meta["hparams"]["trend_windows"]) - # Load matched_clean if needed - if matched_clean is None: - import pickle - cache_dir = os.path.join( - os.path.dirname(os.path.dirname(os.path.dirname( - os.path.abspath(__file__)))), - "results", "token_factored_calibration", "_cache", - ) - with open(os.path.join(cache_dir, "stage1.pkl"), "rb") as f: - data = pickle.load(f) - matched_clean = data["matched_clean"] - # Find pool coefficients - pool_idx = _find_pool_index(pool_id, pool_ids) + pool_idx = -1 + if pool_id is not None: + pool_idx = _find_pool_index(pool_id, pool_ids) + if pool_idx >= 0 and per_pool: coeffs = noise_coeffs[pool_idx] + print(f" Using per-pool coefficients (pool idx {pool_idx})") elif per_pool: - print(f" Warning: pool {pool_id} not in calibration set, using median coeffs") coeffs = np.median(noise_coeffs, axis=0) + print(f" Pool not found, using median coefficients") else: coeffs = noise_coeffs - # Build daily features - x_daily, dates = build_daily_features( - pool_id, matched_clean, start_date, end_date, + # Build daily features from Binance + print(f" Building features from Binance data: {token_a}/{token_b}," + f" {start_date} → {end_date}") + x_daily, dates = build_daily_features_from_binance( + token_a, token_b, start_date, end_date, feat_names, x_mean, x_std, trend_windows, ) n_days = len(dates) + print(f" {n_days} days, {len(feat_names)} features") # Decompose into base (non-TVL) and tvl_coeff tvl_col, tvl_interactions = _identify_tvl_columns(feat_names) - # tvl_coeff_t = coeffs[tvl_col] + sum(coeffs[inter_col] * x[paired_col]) tvl_coeff_daily = np.full(n_days, coeffs[tvl_col], dtype=np.float64) for inter_col, paired_col in tvl_interactions: tvl_coeff_daily += coeffs[inter_col] * x_daily[:, paired_col] - # base_t = sum(coeffs[j] * x[j]) for j not in {tvl_col, interaction_cols} tvl_related = {tvl_col} | {ic for ic, _ in tvl_interactions} base_daily = np.zeros(n_days, dtype=np.float64) for j in range(len(feat_names)): if j not in tvl_related: base_daily += coeffs[j] * x_daily[:, j] - # Expand to minute resolution: each day's value repeats 1440 times + # Expand to minute resolution n_minutes = n_days * 1440 noise_base = np.repeat(base_daily, 1440) noise_tvl_coeff = np.repeat(tvl_coeff_daily, 1440) From 414903d59b57080af94ef1d721919095d1fd0670 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Thu, 26 Mar 2026 23:09:18 +0000 Subject: [PATCH 070/115] data: per-pool linear noise model artifact (Binance-only, 22 features, 36 pools) --- results/linear_market_noise/meta.json | 77 ++++++++++++++++++++++++++ results/linear_market_noise/model.npz | Bin 0 -> 5092 bytes 2 files changed, 77 insertions(+) create mode 100644 results/linear_market_noise/meta.json create mode 100644 results/linear_market_noise/model.npz diff --git a/results/linear_market_noise/meta.json b/results/linear_market_noise/meta.json new file mode 100644 index 00000000..35249de1 --- /dev/null +++ b/results/linear_market_noise/meta.json @@ -0,0 +1,77 @@ +{ + "feat_names": [ + "xobs_0", + "xobs_1", + "xobs_2", + "xobs_3", + "btc_log_price", + "btc_log_return", + "btc_realized_vol_7d", + "btc_trend_7d", + "btc_volume_zscore", + "pair_realized_vol_7d", + "tok_a_log_return", + "tok_a_realized_vol_7d", + "tok_a_trend_7d", + "tok_a_volume_zscore", + "tok_b_log_return", + "tok_b_realized_vol_7d", + "tok_b_trend_7d", + "tok_b_volume_zscore", + "xobs_1\u00d7btc_realized_vol_7d", + "xobs_1\u00d7tok_a_realized_vol_7d", + "xobs_1\u00d7pair_realized_vol_7d", + "tok_a_realized_vol_7d\u00d7tok_b_realized_vol_7d" + ], + "pool_ids": [ + "0x072f14b85add63", + "0x0b09dea16768f0", + "0x10f21c9bd8128a", + "0x1535d7ca00323a", + "0x21d4c792ea7e38", + "0x25ca5451cd5a50", + "0x260dbd54d87a10", + "0x272d6be442e30d", + "0x32df62dc3aed2c", + "0x36be1e97ea98ab", + "0x3de27efa2f1aa6", + "0x3e5fa9518ea95c", + "0x4683e340a80492", + "0x4cdabe9e07ca39", + "0x4fbb7870dbe7a7", + "0x571bea0e99e139", + "0x5c6ee304399dbd", + "0x5f1f4e50ba51d7", + "0x711af51a937e01", + "0x713fb5036dc700", + "0x9232a548dd9e81", + "0x92762b42a06dcd", + "0x96646936b91d6b", + "0x9d1fcf346ea1b0", + "0xa6f548df93de92", + "0xa83b8d30f61d75", + "0xb460daa847c45f", + "0xbc2acf5e821c5c", + "0xbda917a67c7d9a", + "0xcc65a812ce382a", + "0xcf354603a9aebd", + "0xcf7b51ce575551", + "0xd1d7fa8871d84d", + "0xdaba3d8ccf79ef", + "0xe99481dc77691d", + "0xf16aee6a71af1a" + ], + "n_pools": 36, + "n_feat": 22, + "hparams": { + "epochs": 2000, + "lr": 0.0003, + "l2_alpha": 0.001, + "huber_delta": 1.0, + "trend_windows": [ + 7 + ], + "per_pool": true, + "pool_intercepts": false + } +} \ No newline at end of file diff --git a/results/linear_market_noise/model.npz b/results/linear_market_noise/model.npz new file mode 100644 index 0000000000000000000000000000000000000000..c792225202ae753ace18972cf62cf5b677717a4e GIT binary patch literal 5092 zcmd5=c{o*T+uvkX<_2j%M1~@o>~%kDZAr#xPKhL zsKhyyQqg#5GBgl5r&4@-J8$20>iW)iz1MsG_`Z8z&-$(3eLwfQ_WC{dde(I>XZv9i zYMeh!i}P!ZT58L14o5VUIBFdK0N)@lch3MXZ||U~{+mO^gkDihWZy7p!BR(O7cou< zXQ$peuOQDrJqt}e%G+GeNK?-{ATT)4!{0q1aGjUPZ|&h1nQQeDOE3{DW&rN&Kj+p3cfYbSguMOHOMxyt@{!L{huDgPoe85 z>azmx^+zeDv`7p3ekozkm}CiVKhy*B3M=IO(GUg7P9TDi4YcRBq|0*W?a|mWZ7_ek z2cfD3&&Qv=myego3B7EOO6+GV*(@o{wTYf zmkw}U5jJ!@WeY#PL<;*(fK}@YDtv?-UDO%M79X<`C`S5V`vv#Gt|k}ej4>g`opbQU zUEyf4#YU84HBn&xz=2g+HjLPgQ05KO{zzwzP-RC{A7hoKrU)WmyrEvrPDKGGj<9lF zHCTQOrd{3BsE_j<$ebrHD2vCVkPfR4cM4r0n;(G#rzWE1M{>xvMoD_h0Ru8Db`QDp zPMT+7GY@pj3UIsAwIa0*ikIjlQp{Q+;y{QTavhSj;ryS6QYIWjv0--DSrRZt= zVs_bvR{ZAXDc0C~7wS=60lm88VN1^;?)ZRrl;Nxv24WLvJRd<2zZZ_y$D-#J({Sk0 z$#h4;D6}KE9MX1L;xRlWRCcBS5p{hWO7BF~!rHAXGQv}2regQ}5QLi5>4JrsNINE$ zbV@!Vy|Wt_`7Sp`+`|&KuJ>dkPJ~fYGUiivi#-L|6|vke7u{j}p;h#=dv>fRN0vzm zjm4(sy;PH3FhnoD%2u3?hNDkik(rzma#@cAv967{Vf8yanbPF_qM*SvhpT{ufepAj zUPQx_6xju4J*@K;HS*m45j1IrV)uaSY{cT6iig`0Fp4_{7Ue#yXT3VT*47I5wS=>W zK1-9zwo;tj5s1q!r&EdQ@;uhJo?cqQLE$S$V$-}XbZXODZ0XqmzpJdkTc2K_Lib3~ zwG*$hr#Cs^5pWVhtD~Vm0I zJg;)FspJvIQ@UhVjt9|t-vWkHuVN>1iAtMvlKshVGvf8QA}gyFn zv1AMQPZSp{t^WWAJ{>@uuj<^(zqg_VUCXdTaxrdLsX@m1yg{$+7m~&LlgRS85@z6Q zEIN_JPz`Uz@p=lA*kSu%E5`~pXYTTP3?r5$QyxLOt%bF^@lyDU-ra+6WGzY>j7PNC|iNkevGD^~h6iOnho*i)B* zbGKJQvP>P8@Y-CVcHIC)J=A8OMFdf^f4NhUekTg2H^smm>IiDkioo{EeMqzGCn#N> z2$yb3Nr1Y||xJHB42Yn!F7x0=0;p+f?MV_ZcqYJium_2JmxoEZugt8d9VN zAXYh%+H=ndEs~sp$F=oh>);F|*K|an{2&bk>H1hzU!Hf_$OQ$*EkSFmUg3f2c>v8!1CU4I5~9`%3Z5TAkqNkzBMOTVi%(0nQE-==fy<(%ol3ydmUuF|2*z2 z>tq|c$6$F!E7E49!yCQPpPrIm1e@E#xq|fd(053OjdIO|DDl%+x?2N1X*Xn@9wk-O zDV3wA9saCIh(4Zo+X!E6JPqko9zHp%8usRH0H2r`!3VoI@KC77X0=8<Tqx4%e9WHPaErS6+ZvQQLJu;X<%kO^gO*wrNz2;f^O3XA<%YfN z6Zv>{7S1ORZZ?9r>UB7Hw+MS_%kl)bjF`dCpQyToT<)GJSCN4CxI!=^26l{>!M49Y z2RVTTO*_`JNM|j&z4kP=&#i_`tzvGTkvOt#x`CgX--UHTU#KM>wkT-Is)|Lu zjUd5lVD+4MD34V^FjWW7U2zHVA4rn;Pe$2j%)lQ{U4)C1Eg>K`0ZDJKh0&+`G1m~ofzB$5Rw%^Ptx?!8s|0Ejx1q|y zZlXFRf!bxvp~EZZk^CMN-UT90|NK&t+Awo2Ym-_CQXi(_ql%ZQreY7=^~?>f4D3bD zeFkvL8j-NGyJ2rL5Au5IDGlSf*lnyTgkJd>8eT1I(ALY;pjkJY zv~@VKsMJH!a;vd2$tFhZKGHt3fm~TXjCVpwl6mKl%4TtjxYiBb_<6(x9QI5BrMYp4 zw5lBFDn>)gTxW<2jv>)mIw-6EIJ(6vV*hX$1bL-KoGW|>OzM0-d!D*CHm+L_jlrhm zL4^v>#omE7R{(H*PhlA^CoFek20M2>m)p?ghTn7+qGJhJ?C8hxP-u1%GJNGoSnXUG zzuK8wP_&9P6=+bUhq7S810$mUSP9kKMQFNO8x^tW3uD^97_HYnZeg<^lG@$eK{R@H z;9o6oz-?w4oQ;-3h3E4i=-1&;wRkK^QBfkXop}&$UJ6g$2H2}tqfq>vn1~jni240!Mr-HuNqwIXlj!8OaamghUE0oLPj%qiEaoR;>J0_t= z0W30YmPf^nmtl}kQRjH8iSu_IlIO;2=}Kb4?npQkg&mS#zl*T9rAfuEcN)dT5(LC+2%W2lCTRz6m-~ zD5$T$Z?OAcRQ>P1^P8OC@c&xYhXqQ^&^ikb#og-6`zgjl{N5tdAz9xf8!=U8W0^TO zHqxa}O_;UvTxO=fa4CfLVudL)GHE(f(mI2Q>u?zIiKGfpR~)E&FToYj-iiVs}*6r?_ygdBB zm8i%ovTyh|TF(Brdid$njs9tg{`qf)j+#=d3>o2BrR&n|v=B8%jc^uW<{wSEB0pc! z&uqwV$4eDqWmloW5TWsU!16oL}djCc^*KzAxdi{Q3+H3W&EL%{_%s=kJ;aE z-`{eDw;-lHg7f2M{$uj@Th(8agA_Rb^Tze#CBA Date: Fri, 27 Mar 2026 11:37:08 +0000 Subject: [PATCH 071/115] fix: noise volume cadence scaling + price preservation for 2-CLP Two bugs in noise fee income application for reClAMM pools: 1. Cadence scaling: noise model returns per-minute volume but was applied once per arb step (every arb_frequency minutes) without scaling. Now multiplies by minutes_per_step. At cadence=5, this was underestimating noise fee income by 5x. 2. Price preservation: uniform real-reserve scaling (Ra*s, Rb*s) preserves price for weighted pools but NOT for 2-CLPs where price depends on effective reserves (Ra+Va)/(Rb+Vb). Fixed by scaling effective reserves uniformly then subtracting virtuals: Ra_new = (Ra+Va)*scale - Va. Preserves quoted marginal price. Total value added still equals noise_fee_income (verified algebraically: effective_value * (scale-1) = fee_income). Both fixes applied to all CLP noise model variants (tsoukalas, loglinear, calibrated, market_linear). Also: tune script adds L-BFGS support, 25% protocol fee split, extended date range, $7M default TVL. New plotting scripts for model vs real comparison. --- experiments/tune_reclamm_calibrated_noise.py | 65 +++-- quantammsim/pools/reCLAMM/reclamm_reserves.py | 33 ++- scripts/compare_modelled_vs_real.py | 243 ++++++++++++++++ scripts/plot_model_vs_real_reclamm.py | 262 ++++++++++++++++++ 4 files changed, 571 insertions(+), 32 deletions(-) create mode 100644 scripts/compare_modelled_vs_real.py create mode 100644 scripts/plot_model_vs_real_reclamm.py diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index 8a818513..9e999bd6 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -114,6 +114,37 @@ def _build_market_linear_arrays(args): return arrays_path, max(1, round(learned_cadence)) +def _build_opt_settings(args): + """Build optimisation_settings for either optuna or bfgs.""" + if args.method == "bfgs": + return { + "method": "bfgs", + "n_parameter_sets": args.n_parameter_sets, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + "bfgs_settings": { + "maxiter": args.bfgs_maxiter, + "tol": args.bfgs_tol, + "n_evaluation_points": args.bfgs_eval_points, + "compute_dtype": "float64", + }, + } + else: + return { + "method": "optuna", + "n_parameter_sets": 1, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + "optuna_settings": { + "make_scalar": True, + "expand_around": False, + "n_trials": args.n_trials, + "multi_objective": False, + "parameter_config": PARAMETER_CONFIG, + **({"overfitting_penalty": args.overfitting_penalty} + if args.overfitting_penalty is not None else {}), + }, + } + + def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): """Build run fingerprint with calibrated noise model.""" if args.noise_model == "market_linear" and noise_arrays_path is not None: @@ -151,7 +182,7 @@ def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): "fees": args.fees, "gas_cost": args.gas_cost, "arb_fees": 0.0, - "protocol_fee_split": 0.5, + "protocol_fee_split": 0.25, **noise_block, "return_val": objective, "reclamm_interpolation_method": args.interpolation, @@ -159,19 +190,7 @@ def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): "reclamm_learn_arc_length_speed": False, "reclamm_use_shift_exponent": True, **({"bout_offset": args.bout_offset} if args.bout_offset is not None else {}), - "optimisation_settings": { - "method": "optuna", - "n_parameter_sets": 1, - **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), - "optuna_settings": { - "make_scalar": True, - "expand_around": False, - "n_trials": args.n_trials, - "multi_objective": False, - "parameter_config": PARAMETER_CONFIG, - **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), - }, - }, + "optimisation_settings": _build_opt_settings(args), } @@ -180,6 +199,7 @@ def run_single(objective, args, noise_arrays_path=None, arb_freq=None): print(f"\n{'='*60}") print(f" Objective: {objective}") print(f" Noise model: {args.noise_model}") + print(f" Method: {args.method}") print(f" Pool: AAVE/WETH Mainnet ({POOL_ID})") print(f" Train: {args.start_date} → {args.end_date}") print(f" Test: {args.end_date} → {args.end_test_date}") @@ -202,7 +222,16 @@ def main(): parser = argparse.ArgumentParser( description="Tune reClAMM params with calibrated 8-covariate noise model" ) - parser.add_argument("--n-trials", type=int, default=50) + parser.add_argument("--method", default="optuna", choices=["optuna", "bfgs"], + help="Optimisation method") + parser.add_argument("--n-trials", type=int, default=50, + help="Optuna trials (ignored for bfgs)") + parser.add_argument("--n-parameter-sets", type=int, default=1, + help="Number of parameter sets for bfgs") + parser.add_argument("--bfgs-maxiter", type=int, default=100) + parser.add_argument("--bfgs-tol", type=float, default=1e-6) + parser.add_argument("--bfgs-eval-points", type=int, default=20, + help="Number of evaluation points for bfgs") parser.add_argument("--noise-model", default="market_linear", choices=["calibrated", "market_linear"], help="Noise model variant") @@ -221,10 +250,10 @@ def main(): parser.add_argument("--interpolation", default="geometric", choices=["geometric", "constant_arc_length"]) parser.add_argument("--centeredness-scaling", action="store_true") - parser.add_argument("--start-date", default="2025-08-03 00:00:00") - parser.add_argument("--end-date", default="2025-12-01 00:00:00", + parser.add_argument("--start-date", default="2024-06-01 00:00:00") + parser.add_argument("--end-date", default="2025-06-01 00:00:00", help="End of training / start of test") - parser.add_argument("--end-test-date", default="2026-02-18 00:00:00", + parser.add_argument("--end-test-date", default="2026-03-01 00:00:00", help="End of test (latest available data)") parser.add_argument("--bout-offset", type=int, default=None) parser.add_argument("--val-fraction", type=float, default=None) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 26dcd7a7..93632551 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -1005,16 +1005,20 @@ def _skip_schedule_state(_): arb_volume, _np, ) - noise_fee_income = (1.0 - gamma) * noise_vol - scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) - Ra_new = Ra_new * scale - Rb_new = Rb_new * scale + # Scale effective reserves uniformly to preserve quoted price. + # For a 2-CLP: price ∝ (Ra+Va)/(Rb+Vb), so we must scale + # effective reserves (Ra+Va, Rb+Vb) by the same factor, then + # subtract back the fixed virtual reserves. + minutes_per_step = seconds_per_step / 60.0 + noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + Ra_new = (Ra_new + Va) * scale - Va + Rb_new = (Rb_new + Vb) * scale - Vb elif noise_model == "calibrated": volatility = input_list[9] dow_sin = input_list[10] dow_cos = input_list[11] arb_volume = 0.5 * jnp.sum(jnp.abs(applied_trade) * prices) - real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] _np = noise_params if noise_params is not None else {} @@ -1023,14 +1027,14 @@ def _skip_schedule_state(_): arb_volume, dow_sin, dow_cos, _np, ) - noise_fee_income = (1.0 - gamma) * noise_vol - scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) - Ra_new = Ra_new * scale - Rb_new = Rb_new * scale + minutes_per_step = seconds_per_step / 60.0 + noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + Ra_new = (Ra_new + Va) * scale - Va + Rb_new = (Rb_new + Vb) * scale - Vb elif noise_model == "market_linear": noise_base = input_list[9] noise_tvl_coeff = input_list[10] - real_value = jnp.sum(jnp.array([Ra_new, Rb_new]) * prices) effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] _np = noise_params if noise_params is not None else {} @@ -1040,10 +1044,11 @@ def _skip_schedule_state(_): tvl_std=_np.get("tvl_std", 1.0), ) - noise_fee_income = (1.0 - gamma) * noise_vol - scale = 1.0 + noise_fee_income / jnp.maximum(real_value, 1e-8) - Ra_new = Ra_new * scale - Rb_new = Rb_new * scale + minutes_per_step = seconds_per_step / 60.0 + noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + Ra_new = (Ra_new + Va) * scale - Va + Rb_new = (Rb_new + Vb) * scale - Vb # else: "arb_only" — no noise trades # Clamp-to-edge: if a real reserve would go negative, apply an diff --git a/scripts/compare_modelled_vs_real.py b/scripts/compare_modelled_vs_real.py new file mode 100644 index 00000000..e23c7e4e --- /dev/null +++ b/scripts/compare_modelled_vs_real.py @@ -0,0 +1,243 @@ +"""Compare modelled reClAMM noise volume against a real pool's observed volume. + +Plots the modelled noise volume for one pool (at a specified counterfactual TVL) +against the actual observed volume of another pool (or the same pool), to +sanity-check the noise model's predictions. + +Usage: + # reClAMM AAVE/ETH modelled at $7M vs weighted wstETH/AAVE real + python scripts/compare_modelled_vs_real.py \ + --model-pool 0x9d1fcf346ea1b0 --model-tvl 7e6 \ + --real-pool 0x3de27efa2f1aa6 + + # Same but at $20M + python scripts/compare_modelled_vs_real.py \ + --model-pool 0x9d1fcf346ea1b0 --model-tvl 20e6 \ + --real-pool 0x3de27efa2f1aa6 + + # Multiple TVL levels + python scripts/compare_modelled_vs_real.py \ + --model-pool 0x9d1fcf346ea1b0 --model-tvl 1e6 7e6 20e6 50e6 \ + --real-pool 0x3de27efa2f1aa6 +""" + +import argparse +import os +import pickle + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +from quantammsim.calibration.noise_model_arrays import ( + build_simulator_arrays, load_artifact, _find_pool_index, +) + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +ARTIFACT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "noise_comparison", +) + +# Token mapping for pools +POOL_TOKENS = { + "0x9d1fcf346ea1b0": ("AAVE", "ETH"), + "0x3de27efa2f1aa6": ("AAVE", "ETH"), # wstETH/AAVE ≈ same pair + "0x0b09dea16768f0": ("DAI", "ETH"), + "0xa6f548df93de92": ("BTC", "ETH"), + "0x96646936b91d6b": ("USDC", "ETH"), +} + + +def load_real_pool(pid, mc): + """Load real observed volume + TVL for a pool.""" + entry = mc[pid] + panel = entry["panel"] + dates = pd.to_datetime(panel["date"]) + vol = np.exp(panel["log_volume"].values.astype(float)) + tvl = np.exp(panel["log_tvl_lag1"].values.astype(float)) + tokens = entry["tokens"] + chain = entry["chain"] + return dates, vol, tvl, tokens, chain + + +def compute_modelled_noise(pid, tvl_value, start_date, end_date, + artifact_dir, token_a, token_b): + """Compute modelled daily noise for a pool at a given TVL.""" + arrays = build_simulator_arrays( + token_a=token_a, token_b=token_b, + start_date=start_date, end_date=end_date, + artifact_dir=artifact_dir, pool_id=pid, + ) + + n_days = arrays["n_days"] + std_lt = (np.log(tvl_value) - arrays["tvl_mean"]) / arrays["tvl_std"] + + noise_base = arrays["noise_base"][::1440][:n_days] + tvl_coeff = arrays["noise_tvl_coeff"][::1440][:n_days] + noise_daily = np.exp(noise_base + tvl_coeff * std_lt) + + dates = pd.to_datetime(arrays["dates"][:n_days]) + return dates, noise_daily + + +def main(): + parser = argparse.ArgumentParser( + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--model-pool", default="0x9d1fcf346ea1b0", + help="Pool ID for modelled noise") + parser.add_argument("--model-tvl", type=float, nargs="+", + default=[7_000_000], + help="Counterfactual TVL(s) for the modelled pool") + parser.add_argument("--model-tokens", nargs=2, default=None, + help="Token A and B for the modelled pool (auto-detected)") + parser.add_argument("--real-pool", default="0x3de27efa2f1aa6", + help="Pool ID for real observed data") + parser.add_argument("--artifact-dir", default=ARTIFACT_DIR) + parser.add_argument("--output-dir", default=OUTPUT_DIR) + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + os.makedirs(args.output_dir, exist_ok=True) + + # Load calibration data + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + data = pickle.load(f) + mc = data["matched_clean"] + + # Real pool data + real_dates, real_vol, real_tvl, real_tokens, real_chain = load_real_pool( + args.real_pool, mc) + print(f"Real pool: {args.real_pool} ({real_tokens}, {real_chain})") + print(f" {len(real_dates)} days: {real_dates.min().date()} → {real_dates.max().date()}") + print(f" TVL: ${real_tvl.min():,.0f} – ${real_tvl.max():,.0f}") + print(f" Volume: ${real_vol.min():,.0f} – ${real_vol.max():,.0f}") + + # Model tokens + if args.model_tokens: + tok_a, tok_b = args.model_tokens + elif args.model_pool[:16] in POOL_TOKENS: + tok_a, tok_b = POOL_TOKENS[args.model_pool[:16]] + else: + tok_a, tok_b = "ETH", "USDC" + print(f" Warning: unknown pool, using {tok_a}/{tok_b}") + + # Date range from real pool + start = str(real_dates.min().date()) + end = str(real_dates.max().date()) + + # Compute modelled noise at each TVL + model_results = [] + for tvl_val in args.model_tvl: + print(f"\nModelled: {args.model_pool} at ${tvl_val:,.0f} TVL") + m_dates, m_noise = compute_modelled_noise( + args.model_pool, tvl_val, start, end, + args.artifact_dir, tok_a, tok_b) + model_results.append((tvl_val, m_dates, m_noise)) + print(f" Median noise: ${np.median(m_noise):,.0f}/day" + f" ({np.median(m_noise)/tvl_val*100:.2f}% of TVL)") + + # Align dates + common_start = real_dates.min() + common_end = real_dates.max() + for _, md, _ in model_results: + common_start = max(common_start, md.min()) + common_end = min(common_end, md.max()) + + real_mask = (real_dates >= common_start) & (real_dates <= common_end) + + # Colors for different TVL levels + colors = ["#e74c3c", "#3498db", "#2ecc71", "#f39c12", "#9b59b6"] + + # Plot + fig, axes = plt.subplots(2, 1, figsize=(14, 9)) + + # 1. Volume comparison + ax = axes[0] + ax.plot(real_dates[real_mask], real_vol[real_mask] / 1e6, + "k-", linewidth=0.8, alpha=0.7, + label=f"{real_tokens} weighted (real," + f" TVL ${np.median(real_tvl[real_mask])/1e6:.0f}M)") + + for i, (tvl_val, m_dates, m_noise) in enumerate(model_results): + m_mask = (m_dates >= common_start) & (m_dates <= common_end) + c = colors[i % len(colors)] + ax.plot(m_dates[m_mask], m_noise[m_mask] / 1e6, + "-", color=c, linewidth=0.8, alpha=0.7, + label=f"reClAMM noise (modelled, TVL ${tvl_val/1e6:.0f}M)") + + ax.set_ylabel("Volume ($M/day)") + ax.set_yscale("log") + ax.set_title(f"Real weighted pool vs Modelled reClAMM noise") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # 2. Vol/TVL comparison + ax = axes[1] + real_vol_tvl = real_vol[real_mask] / real_tvl[real_mask] * 100 + ax.plot(real_dates[real_mask], real_vol_tvl, + "k-", linewidth=0.8, alpha=0.7, + label=f"{real_tokens} weighted real vol/TVL") + ax.axhline(np.median(real_vol_tvl), color="black", linestyle="--", + alpha=0.3, label=f"weighted median: {np.median(real_vol_tvl):.2f}%") + + for i, (tvl_val, m_dates, m_noise) in enumerate(model_results): + m_mask = (m_dates >= common_start) & (m_dates <= common_end) + noise_tvl = m_noise[m_mask] / tvl_val * 100 + c = colors[i % len(colors)] + ax.plot(m_dates[m_mask], noise_tvl, + "-", color=c, linewidth=0.8, alpha=0.7, + label=f"reClAMM noise/TVL (${tvl_val/1e6:.0f}M)") + ax.axhline(np.median(noise_tvl), color=c, linestyle="--", alpha=0.3, + label=f"median: {np.median(noise_tvl):.2f}%") + + ax.set_ylabel("Volume / TVL (%)") + ax.set_xlabel("Date") + ax.set_title("Volume as Fraction of TVL") + ax.legend(fontsize=7, loc="upper right") + ax.grid(True, alpha=0.3) + ymax = min( + max(np.percentile(real_vol_tvl, 95), + max(np.percentile(m_noise[m_mask] / tvl_val * 100, 95) + for tvl_val, m_dates, m_noise in model_results + for m_mask in [(m_dates >= common_start) & (m_dates <= common_end)])) * 1.5, + 50) + ax.set_ylim(0, ymax) + + fig.tight_layout() + tvl_str = "_".join(f"{t/1e6:.0f}M" for t in args.model_tvl) + out = os.path.join(args.output_dir, + f"{args.model_pool[:8]}_vs_{args.real_pool[:8]}_{tvl_str}.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"\nSaved: {out}") + + # Summary table + print(f"\n{'='*60}") + print(f"Summary") + print(f"{'='*60}") + print(f" Real {real_tokens} weighted:") + print(f" Median TVL: ${np.median(real_tvl[real_mask]):,.0f}") + print(f" Median vol: ${np.median(real_vol[real_mask]):,.0f}/day") + print(f" Median vol/TVL: {np.median(real_vol_tvl):.2f}%") + for tvl_val, m_dates, m_noise in model_results: + m_mask = (m_dates >= common_start) & (m_dates <= common_end) + med_noise = np.median(m_noise[m_mask]) + print(f" Modelled reClAMM at ${tvl_val/1e6:.0f}M:") + print(f" Median noise: ${med_noise:,.0f}/day") + print(f" Median noise/TVL: {med_noise/tvl_val*100:.2f}%") + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_model_vs_real_reclamm.py b/scripts/plot_model_vs_real_reclamm.py new file mode 100644 index 00000000..c0c3da3e --- /dev/null +++ b/scripts/plot_model_vs_real_reclamm.py @@ -0,0 +1,262 @@ +"""Plot full model (V_arb + V_noise) vs real observed volume for a pool. + +Uses the pool's actual historical TVL path, evaluates V_arb from the PCHIP +grid at the learned cadence, and V_noise from the per-pool linear model. +Compares against observed total volume. + +Usage: + python scripts/plot_model_vs_real_reclamm.py + python scripts/plot_model_vs_real_reclamm.py --pool 0x3de27efa2f1aa6 + python scripts/plot_model_vs_real_reclamm.py --pool 0x9d1fcf346ea1b0 +""" + +import argparse +import os +import pickle +import sys + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +sys.path.insert(0, os.path.dirname(os.path.dirname(__file__))) + +import jax.numpy as jnp +from quantammsim.calibration.grid_interpolation import interpolate_pool_daily +from quantammsim.calibration.noise_model_arrays import load_artifact + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +ARTIFACT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "model_vs_real", +) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--pool", default="0x9d1fcf346ea1b0", + help="Pool ID prefix") + parser.add_argument("--artifact-dir", default=ARTIFACT_DIR) + parser.add_argument("--output-dir", default=OUTPUT_DIR) + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + os.makedirs(args.output_dir, exist_ok=True) + + # Load data + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + data = pickle.load(f) + mc = data["matched_clean"] + oc = data["option_c_clean"] + + pid = args.pool + entry = mc[pid] + panel = entry["panel"] + dates = pd.to_datetime(panel["date"]) + vol_obs = np.exp(panel["log_volume"].values.astype(float)) + tvl = np.exp(panel["log_tvl_lag1"].values.astype(float)) + + # Load noise model + art, meta = load_artifact(args.artifact_dir) + pool_ids = meta["pool_ids"] + idx = pool_ids.index(pid) + coeffs = art["noise_coeffs"][idx] + cadence = float(np.exp(art["log_cadence"][idx])) + gas = float(np.exp(oc[pid]["log_gas"])) + + print(f"Pool: {pid} ({entry['tokens']}, {entry['chain']})") + print(f"Cadence: {cadence:.1f} min, Gas: ${gas}") + print(f"{len(dates)} days: {dates.min().date()} → {dates.max().date()}") + print(f"TVL: ${tvl.min():,.0f} – ${tvl.max():,.0f}") + + # V_arb from PCHIP + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], jnp.float64(np.log(cadence)), jnp.float64(gas))) + + # V_noise from model at actual TVL + from experiments.run_linear_market_noise import build_data + data_full = build_data(mc, oc, trend_windows=(7,), + include_market=True, include_cross_pool=False) + x_full = data_full["x"] + pool_idx_full = data_full["pool_idx"] + pool_mask = pool_idx_full == idx + sample_x = x_full[pool_mask] + sgd = data_full["sample_grid_days"][pool_mask] + day_idx = data_full["day_idx"][pool_mask] + + log_v_noise = sample_x @ coeffs + v_noise = np.exp(log_v_noise) + v_arb_samples = v_arb_all[sgd] + v_total_pred = v_arb_samples + v_noise + + # Align dates + all_dates = set() + for p in pool_ids: + all_dates.update(mc[p]["panel"]["date"].values) + date_list = sorted(all_dates) + sample_dates = np.array([pd.Timestamp(date_list[d]) for d in day_idx]) + + # Match TVL and obs volume + tvl_samples = np.zeros(len(sample_dates)) + vol_obs_samples = np.zeros(len(sample_dates)) + for i, sd in enumerate(sample_dates): + matches = np.where(dates == sd)[0] + if len(matches) > 0: + tvl_samples[i] = tvl[matches[0]] + vol_obs_samples[i] = vol_obs[matches[0]] + + valid = tvl_samples > 100 + sd = sample_dates[valid] + vo = vol_obs_samples[valid] + va = v_arb_samples[valid] + vn = v_noise[valid] + vt = v_total_pred[valid] + tv = tvl_samples[valid] + + # R² + log_obs = np.log(np.maximum(vo, 1)) + log_pred = np.log(np.maximum(vt, 1)) + ss_res = np.sum((log_obs - log_pred) ** 2) + ss_tot = np.sum((log_obs - log_obs.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + print(f"\nR² (log): {r2:.3f}") + print(f"Median obs: ${np.median(vo):,.0f}, pred: ${np.median(vt):,.0f}") + print(f"Median V_arb: ${np.median(va):,.0f}, V_noise: ${np.median(vn):,.0f}") + + # Fee rate + fee_rate = float(panel["swap_fee"].iloc[0]) if "swap_fee" in panel.columns else 0.003 + + # Plot + fig, axes = plt.subplots(6, 1, figsize=(14, 20), sharex=True) + + # 1. TVL + ax = axes[0] + ax.plot(sd, tv, "b-", linewidth=1) + ax.set_ylabel("TVL (USD)") + ax.set_yscale("log") + ax.set_title(f"{entry['tokens']} ({entry['chain']}) — " + f"Model (V_arb + V_noise) vs Observed " + f"[R\u00b2={r2:.3f}, cadence={cadence:.0f}min, fee={fee_rate:.4f}]") + ax.grid(True, alpha=0.3) + + # 2. Volume: stacked arb + noise vs observed + ax = axes[1] + ax.fill_between(sd, 0, va, alpha=0.3, color="steelblue", label="V_arb (PCHIP)") + ax.fill_between(sd, va, va + vn, alpha=0.3, color="coral", label="V_noise (model)") + ax.plot(sd, vo, "k-", linewidth=0.8, alpha=0.7, label="V_obs (actual)") + ax.plot(sd, vt, "r--", linewidth=0.8, alpha=0.5, label="V_pred = V_arb + V_noise") + ax.set_ylabel("Volume (USD/day)") + ax.set_yscale("log") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # 3. V_noise only + ax = axes[2] + ax.fill_between(sd, 0, vn, alpha=0.4, color="coral") + ax.plot(sd, vn, "r-", linewidth=0.8, alpha=0.7, label="V_noise (model)") + ax.axhline(np.median(vn), color="red", linestyle="--", alpha=0.5, + label=f"median: ${np.median(vn):,.0f}") + ax.set_ylabel("Noise volume (USD/day)") + ax.set_yscale("log") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # 4. Fee revenue: observed vs predicted + ax = axes[3] + fee_obs = vo * fee_rate + fee_pred = vt * fee_rate + fee_noise_only = vn * fee_rate + fee_arb = va * fee_rate + ax.fill_between(sd, 0, fee_arb, alpha=0.3, color="steelblue", label="Arb fees") + ax.fill_between(sd, fee_arb, fee_pred, alpha=0.3, color="coral", label="Noise fees") + ax.plot(sd, fee_obs, "k-", linewidth=0.8, alpha=0.7, label="Observed fees") + ax.plot(sd, fee_pred, "r--", linewidth=0.8, alpha=0.5, label="Predicted total fees") + ax.plot(sd, fee_noise_only, "m-", linewidth=0.8, alpha=0.6, + label=f"Noise fees only (med=${np.median(fee_noise_only):,.0f})") + ax.set_ylabel("Fee revenue (USD/day)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # 5. Vol/TVL + ax = axes[4] + vol_tvl_obs = vo / tv * 100 + vol_tvl_pred = vt / tv * 100 + ax.plot(sd, vol_tvl_obs, "k-", linewidth=0.8, alpha=0.7, label="Observed") + ax.plot(sd, vol_tvl_pred, "r--", linewidth=0.8, alpha=0.5, label="Predicted") + ax.axhline(np.median(vol_tvl_obs), color="black", linestyle=":", + alpha=0.3, label=f"obs median: {np.median(vol_tvl_obs):.1f}%") + ax.axhline(np.median(vol_tvl_pred), color="red", linestyle=":", + alpha=0.3, label=f"pred median: {np.median(vol_tvl_pred):.1f}%") + ax.set_ylabel("Vol / TVL (%)") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + ax.set_ylim(0, min(np.percentile(vol_tvl_obs, 95) * 2, 200)) + + # 6. Pred/Obs ratio + ax = axes[5] + ratio = vt / np.maximum(vo, 1) + ax.plot(sd, ratio, "g-", linewidth=0.8, alpha=0.7) + ax.axhline(1.0, color="black", linestyle="--", alpha=0.5, label="perfect") + ax.axhline(np.median(ratio), color="red", linestyle="--", alpha=0.5, + label=f"median: {np.median(ratio):.2f}") + ax.set_ylabel("Pred / Obs") + ax.set_xlabel("Date") + ax.set_yscale("log") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + ax.set_ylim(0.01, 100) + + fig.tight_layout() + out = os.path.join(args.output_dir, f"{pid[:16]}_model_vs_real.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"\nSaved: {out}") + + # Fee summary + fee_obs_total = np.sum(vo * fee_rate) + fee_pred_total = np.sum(vt * fee_rate) + fee_noise_total = np.sum(vn * fee_rate) + fee_arb_total = np.sum(va * fee_rate) + print(f"\nFee revenue (cumulative, fee={fee_rate:.4f}):") + print(f" Observed: ${fee_obs_total:,.0f}") + print(f" Predicted: ${fee_pred_total:,.0f}" + f" (arb: ${fee_arb_total:,.0f}, noise: ${fee_noise_total:,.0f})") + + # Pre/post deposit stats (for reClAMM AAVE/ETH) + pre = sd < pd.Timestamp("2026-01-10") + post = sd >= pd.Timestamp("2026-01-20") + if pre.sum() > 5 and post.sum() > 5: + print(f"\nPre-deposit (before Jan 10):") + print(f" TVL: ${np.median(tv[pre]):,.0f}") + print(f" V_obs: ${np.median(vo[pre]):,.0f}," + f" V_pred: ${np.median(vt[pre]):,.0f}") + print(f" V_arb: ${np.median(va[pre]):,.0f}," + f" V_noise: ${np.median(vn[pre]):,.0f}") + print(f" Fees obs: ${np.median(vo[pre])*fee_rate:,.0f}/day," + f" pred: ${np.median(vt[pre])*fee_rate:,.0f}/day") + print(f" Pred/Obs: {np.median(vt[pre] / vo[pre]):.2f}") + print(f"Post-deposit (after Jan 20):") + print(f" TVL: ${np.median(tv[post]):,.0f}") + print(f" V_obs: ${np.median(vo[post]):,.0f}," + f" V_pred: ${np.median(vt[post]):,.0f}") + print(f" V_arb: ${np.median(va[post]):,.0f}," + f" V_noise: ${np.median(vn[post]):,.0f}") + print(f" Fees obs: ${np.median(vo[post])*fee_rate:,.0f}/day," + f" pred: ${np.median(vt[post])*fee_rate:,.0f}/day") + print(f" Pred/Obs: {np.median(vt[post] / vo[post]):.2f}") + + +if __name__ == "__main__": + main() From 5d40a6556d61f78b127500e27e7ac5e3056bca78 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 27 Mar 2026 11:41:57 +0000 Subject: [PATCH 072/115] compare improvements --- scripts/compare_reclamm_thermostats.py | 1870 +++++++++++++++-- scripts/demo_run_reclamm.py | 199 +- .../reclamm/compare_reclamm_thermostats.py | 1870 +++++++++++++++-- scripts/reclamm/demo_run_reclamm.py | 199 +- .../test_compare_reclamm_thermostats.py | 284 +++ 5 files changed, 3990 insertions(+), 432 deletions(-) create mode 100644 tests/scripts/test_compare_reclamm_thermostats.py diff --git a/scripts/compare_reclamm_thermostats.py b/scripts/compare_reclamm_thermostats.py index 8a2c374c..be8ee5f0 100644 --- a/scripts/compare_reclamm_thermostats.py +++ b/scripts/compare_reclamm_thermostats.py @@ -1,19 +1,39 @@ -"""Compare geometric vs constant-arc-length thermostats on historic data. +"""Compare reCLAMM interpolation modes on historic AAVE/ETH data. -Runs AAVE/ETH reClAMM pool simulations with both interpolation methods. -Plots: pool value, cumulative LVR, price path, empirical weights, -value difference, LVR ratio, and per-step LVR distribution (∝ Δs²). +Runs the production geometric interpolation against the non-linear +constant-arc-length interpolation on: +1. The original launch-style range (price_ratio ~= 1.50) +2. A much tighter range (price_ratio = 1.10) -Usage: - cd - source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm - python scripts/compare_reclamm_thermostats.py +The aggressive case is deliberate. A local AAVE/ETH sweep showed: +price_ratio 1.15, margin 0.5, shift 0.1 -> about +$10k vs geometric +price_ratio 1.10, margin 0.5, shift 0.1 -> about +$31k vs geometric +price_ratio 1.10, margin 0.6, shift 0.1 -> about +$73k vs geometric + +So the strongest clean demo setting came from tightening the band and +slightly raising the trigger margin, while keeping the launch-style shift +speed rather than pushing shift_exponent higher. """ +import gc +import math +import os + import jax.numpy as jnp import numpy as np +import pandas as pd import matplotlib.pyplot as plt +from matplotlib.colors import Normalize, SymLogNorm, TwoSlopeNorm +from matplotlib.cm import ScalarMappable +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + calibrate_arc_length_speed, + compute_price_ratio, + initialise_reclamm_reserves, +) from quantammsim.runners.jax_runners import do_run_on_historic_data +from quantammsim.utils.data_processing.historic_data_utils import ( + get_historic_parquet_data, +) def to_daily_price_shift_base(daily_price_shift_exponent): @@ -21,57 +41,432 @@ def to_daily_price_shift_base(daily_price_shift_exponent): return 1.0 - daily_price_shift_exponent / 124649.0 +RUN_CONSTANT_ARC_LENGTH = False +INTERPOLATION_METHODS = ( + ("geometric", "constant_arc_length") + if RUN_CONSTANT_ARC_LENGTH + else ("geometric",) +) +HEATMAP_PRICE_RATIOS = np.arange(1.01, 1.50 + 1e-9, 0.025) +HEATMAP_MARGINS = np.linspace(0.05, 0.90, 20) +HEATMAP_SHIFT_EXPONENTS = np.arange(0.01, 0.50 + 1e-9, 0.025) +HEATMAP_ARC_LENGTH_SPEEDS = np.geomspace(1.0e-6, 5.0e-4, 11) +PRICE_RATIO_TICKS = np.array([1.01, 1.10, 1.20, 1.30, 1.40, 1.50]) +MARGIN_TICKS = np.array([0.05, 0.15, 0.25, 0.35, 0.45, 0.55, 0.65, 0.75, 0.85, 0.90]) +SHIFT_EXPONENT_TICKS = np.array([0.01, 0.05, 0.10, 0.20, 0.30, 0.40, 0.50]) +ARC_LENGTH_SPEED_TICKS = np.array([ + 1.0e-6, + 2.0e-6, + 5.0e-6, + 1.0e-5, + 2.0e-5, + 5.0e-5, + 1.0e-4, + 2.0e-4, + 5.0e-4, +]) +SWEEP_LINE_WIDTH = 0.45 +REFERENCE_LINE_WIDTH = 0.9 +DEFAULT_INITIAL_POOL_VALUE = 1_000_000.0 +TVL_SWEEP_VALUES = ( + 1_000_000.0, + 5_000_000.0, + 20_000_000.0, +) +CENTER_ZERO_HEATMAP_COLOR_NORM = "symlog" +CENTER_ZERO_HEATMAP_COLOR_TAG = "symlog20" +CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH = 20.0 + +AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" +DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" +DEFAULT_NOISE_MODEL = "market_linear" +DEFAULT_GAS_COST = 1.0 +DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 +LEGACY_NOISE_COEFFS = [ + -0.453, + 0.025, + -0.060, + 0.310, + -0.149, + 0.359, + 0.061, + 0.060, +] +LEGACY_LOG_CADENCE = 2.68 +LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) +AAVE_ETH_NOISE_SETTINGS = { + "enable_noise_model": True, + "noise_model": DEFAULT_NOISE_MODEL, + "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, + "noise_pool_id": AAVE_WETH_POOL_ID, + "gas_cost": DEFAULT_GAS_COST, + "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, +} + +GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS = ( + "geometric_vs_launch_geometric_pct", + "noise_geometric_final_value_musd", + "noise_vs_arb_geometric_improvement_pct", +) +CONSTANT_ARC_HEATMAP_METRIC_KEYS = ( + "efficiency_pct", + "launch_geometric_efficiency_pct", + "constant_arc_vs_launch_constant_arc_pct", + "noise_constant_arc_final_value_musd", + "noise_vs_arb_constant_arc_improvement_pct", +) +HEATMAP_METRIC_DEPENDENCIES = { + "efficiency_pct": ("noise_geometric", "noise_constant_arc"), + "launch_geometric_efficiency_pct": ("noise_constant_arc",), + "geometric_vs_launch_geometric_pct": ("noise_geometric",), + "constant_arc_vs_launch_constant_arc_pct": ("noise_constant_arc",), + "noise_geometric_final_value_musd": ("noise_geometric",), + "noise_constant_arc_final_value_musd": ("noise_constant_arc",), + "noise_vs_arb_geometric_improvement_pct": ("noise_geometric", "arb_geometric"), + "noise_vs_arb_constant_arc_improvement_pct": ( + "noise_constant_arc", + "arb_constant_arc", + ), +} + +_NOISE_SETTINGS_CACHE = {} +_WARNED_NOISE_FALLBACKS = set() + + +def get_initial_pool_value(cfg): + """Return the configured base pool TVL in USD.""" + return float(cfg.get("initial_pool_value", DEFAULT_INITIAL_POOL_VALUE)) + + +def get_tvl_millions(cfg): + """Return the configured base pool TVL in millions of USD.""" + return get_initial_pool_value(cfg) / 1_000_000.0 + + +def format_tvl_millions_slug(cfg): + """Format the TVL in millions for stable filenames.""" + tvl_millions = get_tvl_millions(cfg) + rounded = round(float(tvl_millions), 6) + if np.isclose(rounded, round(rounded)): + return f"{int(round(rounded))}m" + return f"{rounded:.6f}".rstrip("0").rstrip(".").replace(".", "p") + "m" + + +def format_tvl_millions_label(cfg): + """Format the TVL in millions for plot titles and logs.""" + return f"{get_tvl_millions(cfg):.1f}M" + + +def tvl_artifact_filename(stem, cfg, suffix=None): + """Append a TVL-in-millions suffix to a PNG artifact name.""" + parts = [stem] + if suffix: + parts.append(suffix) + parts.append(f"tvl_{format_tvl_millions_slug(cfg)}") + return "_".join(parts) + ".png" + + +def heatmap_artifact_filename(spec, cfg, suffix=None): + """Build a heatmap filename, including any colour-style tag.""" + stem = f"reclamm_heatmap_{spec['slug']}" + artifact_tag = spec.get("artifact_tag") + if artifact_tag: + stem = f"{stem}_{artifact_tag}" + return tvl_artifact_filename(stem, cfg, suffix=suffix) + + +def configs_for_tvl(base_configs, initial_pool_value): + """Attach a shared initial TVL to each compare configuration.""" + configs = [] + for cfg in base_configs: + updated = dict(cfg) + updated["initial_pool_value"] = float(initial_pool_value) + configs.append(updated) + return configs + + +def make_noise_variant_cfg(cfg, enable_noise_model): + """Return a config with either noise modelling or pure arb-only enabled.""" + updated = dict(cfg) + if enable_noise_model: + updated["enable_noise_model"] = True + return updated + + matched_noise = resolve_reclamm_noise_settings(cfg) + + updated["enable_noise_model"] = False + updated["noise_model"] = None + updated["gas_cost"] = cfg.get("gas_cost", DEFAULT_GAS_COST) + updated["protocol_fee_split"] = cfg.get( + "protocol_fee_split", DEFAULT_PROTOCOL_FEE_SPLIT + ) + updated["noise_trader_ratio"] = 0.0 + matched_arb_frequency = matched_noise.get("arb_frequency") + if matched_arb_frequency is not None: + updated["arb_frequency"] = matched_arb_frequency + for key in ( + "reclamm_noise_params", + "noise_arrays_path", + "noise_artifact_dir", + "noise_pool_id", + ): + updated.pop(key, None) + return updated + + +def _warn_noise_fallback(message): + """Print a one-time message when the preferred noise setup is unavailable.""" + if message not in _WARNED_NOISE_FALLBACKS: + print(message) + _WARNED_NOISE_FALLBACKS.add(message) + + +def _hashable_noise_params(params): + """Convert a noise-params dict into a stable cache key fragment.""" + if params is None: + return None + return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) + + +def _legacy_calibrated_noise_settings(reason=None): + """Fallback calibrated noise config used when market-linear artifacts are absent.""" + if reason: + _warn_noise_fallback( + "market_linear noise unavailable for thermostat comparison; " + f"falling back to calibrated legacy coefficients ({reason})." + ) + return { + "noise_model": "calibrated", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) + }, + "arb_frequency": LEGACY_ARB_FREQUENCY, + "noise_summary": ( + "calibrated legacy 8-covariate " + f"(arb_frequency={LEGACY_ARB_FREQUENCY})" + ), + "noise_cache_key": ( + "calibrated", + tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), + LEGACY_ARB_FREQUENCY, + ), + } + + +def resolve_reclamm_noise_settings(cfg): + """Resolve the active reCLAMM noise-model fingerprint block for a config.""" + enable_noise_model = cfg.get("enable_noise_model", False) + requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + cache_key = ( + tuple(cfg.get("tokens", [])), + cfg.get("start"), + cfg.get("end"), + enable_noise_model, + requested_mode, + cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), + cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), + cfg.get("arb_frequency"), + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + _hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + ) + if cache_key in _NOISE_SETTINGS_CACHE: + return _NOISE_SETTINGS_CACHE[cache_key] + + if not enable_noise_model: + result = { + "noise_model": None, + "noise_trader_ratio": 0.0, + "reclamm_noise_params": None, + "noise_arrays_path": None, + "arb_frequency": None, + "noise_summary": "arb-only (noise disabled)", + "noise_cache_key": ("disabled",), + } + elif requested_mode == "market_linear": + artifact_dir = cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR) + pool_id = cfg.get("noise_pool_id", AAVE_WETH_POOL_ID) + start_date = str(cfg["start"]).split(" ")[0] + end_date = str(cfg["end"]).split(" ")[0] + try: + from quantammsim.calibration.noise_model_arrays import ( + _find_pool_index, + build_simulator_arrays, + load_artifact, + ) + + model_path = os.path.join(artifact_dir, "model.npz") + meta_path = os.path.join(artifact_dir, "meta.json") + if not (os.path.exists(model_path) and os.path.exists(meta_path)): + raise FileNotFoundError( + f"expected {model_path} and {meta_path}" + ) + + cache_dir = os.path.join(artifact_dir, "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join( + cache_dir, + f"{pool_id}_{start_date}_{end_date}.npz", + ) + if not os.path.exists(arrays_path): + arrays = build_simulator_arrays( + pool_id=pool_id, + start_date=start_date, + end_date=end_date, + artifact_dir=artifact_dir, + ) + np.savez( + arrays_path, + noise_base=arrays["noise_base"], + noise_tvl_coeff=arrays["noise_tvl_coeff"], + tvl_mean=arrays["tvl_mean"], + tvl_std=arrays["tvl_std"], + ) + + with np.load(arrays_path) as arrays: + tvl_mean = float(arrays["tvl_mean"]) + tvl_std = float(arrays["tvl_std"]) + + art, meta = load_artifact(artifact_dir) + pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) + if pool_idx >= 0: + learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) + else: + learned_cadence = 5.0 + arb_frequency = max(1, round(learned_cadence)) + result = { + "noise_model": "market_linear", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, + }, + "noise_arrays_path": arrays_path, + "arb_frequency": arb_frequency, + "noise_summary": f"market_linear (arb_frequency={arb_frequency})", + "noise_cache_key": ( + "market_linear", + arrays_path, + arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), + ), + } + except Exception as exc: # pragma: no cover - fallback path depends on local artifacts + result = _legacy_calibrated_noise_settings(str(exc)) + elif requested_mode == "calibrated": + params = cfg.get("reclamm_noise_params") + if params is None: + result = _legacy_calibrated_noise_settings() + else: + arb_frequency = cfg.get("arb_frequency", LEGACY_ARB_FREQUENCY) + result = { + "noise_model": "calibrated", + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": dict(params), + "arb_frequency": arb_frequency, + "noise_summary": f"calibrated (arb_frequency={arb_frequency})", + "noise_cache_key": ( + "calibrated", + _hashable_noise_params(params), + arb_frequency, + ), + } + else: + arb_frequency = cfg.get("arb_frequency") + result = { + "noise_model": requested_mode, + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": cfg.get("reclamm_noise_params"), + "noise_arrays_path": cfg.get("noise_arrays_path"), + "arb_frequency": arb_frequency, + "noise_summary": f"{requested_mode} (arb_frequency={arb_frequency})", + "noise_cache_key": ( + requested_mode, + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + _hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + arb_frequency, + ), + } + + _NOISE_SETTINGS_CACHE[cache_key] = result + return result + + # Pool configurations to compare CONFIGS = [ { - "name": "AAVE/ETH on-chain (25bps, narrow range)", + "name": "AAVE/ETH launch-style range (25bps, reference)", "tokens": ["AAVE", "ETH"], "start": "2024-06-01 00:00:00", "end": "2025-06-01 00:00:00", "fees": 0.0025, - "price_ratio": 1.5, + "price_ratio": 1.5014, "centeredness_margin": 0.5, "daily_price_shift_exponent": 0.1, + "reason": "Original launch-style parameters.", + **AAVE_ETH_NOISE_SETTINGS, }, { - "name": "AAVE/ETH wide range (25bps)", + "name": "AAVE/ETH aggressive tight range (25bps)", "tokens": ["AAVE", "ETH"], "start": "2024-06-01 00:00:00", "end": "2025-06-01 00:00:00", "fees": 0.0025, - "price_ratio": 4.0, - "centeredness_margin": 0.2, - "daily_price_shift_exponent": 1.0, - }, - { - "name": "AAVE/ETH zero fees (narrow)", - "tokens": ["AAVE", "ETH"], - "start": "2024-06-01 00:00:00", - "end": "2025-06-01 00:00:00", - "fees": 0.0, - "price_ratio": 1.5, - "centeredness_margin": 0.5, + "price_ratio": 1.10, + "centeredness_margin": 0.60, "daily_price_shift_exponent": 0.1, + "reason": ( + "Aggressively tightened and moved to an earlier thermostat trigger. " + "At fixed price_ratio=1.10, the shift_exponent sweep still favored " + "0.1, while margin=0.60 widened the non-linear edge materially." + ), + **AAVE_ETH_NOISE_SETTINGS, }, ] -def make_fingerprint(cfg, interpolation_method, centeredness_scaling=False): +def make_fingerprint(cfg, interpolation_method): """Build run fingerprint for a given config and interpolation method.""" - return { + speed_override = ( + cfg.get("arc_length_speed") + if interpolation_method == "constant_arc_length" + else None + ) + noise_cfg = resolve_reclamm_noise_settings(cfg) + arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + fingerprint = { "tokens": cfg["tokens"], "rule": "reclamm", "startDateString": cfg["start"], "endDateString": cfg["end"], - "initial_pool_value": 1000000.0, + "initial_pool_value": get_initial_pool_value(cfg), "do_arb": True, "fees": cfg["fees"], - "gas_cost": 0.0, - "arb_fees": 0.0, + "gas_cost": cfg.get( + "gas_cost", + DEFAULT_GAS_COST if cfg.get("enable_noise_model", False) else 0.0, + ), + "arb_fees": cfg.get("arb_fees", 0.0), + "protocol_fee_split": cfg.get( + "protocol_fee_split", + DEFAULT_PROTOCOL_FEE_SPLIT if cfg.get("enable_noise_model", False) else 0.0, + ), + "noise_trader_ratio": noise_cfg.get("noise_trader_ratio", 0.0), "reclamm_interpolation_method": interpolation_method, - "reclamm_arc_length_speed": None, # auto-calibrate - "reclamm_centeredness_scaling": centeredness_scaling, + "reclamm_arc_length_speed": speed_override, } + if noise_cfg.get("noise_model") is not None: + fingerprint["noise_model"] = noise_cfg["noise_model"] + if noise_cfg.get("reclamm_noise_params") is not None: + fingerprint["reclamm_noise_params"] = noise_cfg["reclamm_noise_params"] + if noise_cfg.get("noise_arrays_path") is not None: + fingerprint["noise_arrays_path"] = noise_cfg["noise_arrays_path"] + if arb_frequency is not None: + fingerprint["arb_frequency"] = arb_frequency + return fingerprint def make_params(cfg): @@ -85,40 +480,1129 @@ def make_params(cfg): } -def run_comparison(cfg): - """Run all thermostat variants, return results dict.""" +def load_shared_price_data(configs, root=None): + """Load the shared historic price panel once for all compare runs.""" + tokens = sorted({token for cfg in configs for token in cfg["tokens"]}) + return get_historic_parquet_data(tokens, cols=["close"], root=root) + + +def run_comparison(cfg, price_data=None, low_data_mode=False): + """Run both interpolation variants, return results dict.""" params = make_params(cfg) results = {} - for method in ["geometric", "constant_arc_length"]: + for method in INTERPOLATION_METHODS: fp = make_fingerprint(cfg, method) results[method] = do_run_on_historic_data( - run_fingerprint=fp, params=params + run_fingerprint=fp, + params=params, + price_data=price_data, + low_data_mode=low_data_mode, + ) + + return results + + +def _set_padded_ylim(ax, series_list, pad_ratio=0.04): + """Fit the y-axis tightly around the plotted series.""" + flat = [ + np.asarray(series, dtype=float).ravel() + for series in series_list + if np.asarray(series).size > 0 + ] + if not flat: + return + + values = np.concatenate(flat) + values = values[np.isfinite(values)] + if values.size == 0: + return + + ymin = float(values.min()) + ymax = float(values.max()) + if np.isclose(ymin, ymax): + pad = max(abs(ymin) * pad_ratio, 1e-6) + else: + pad = (ymax - ymin) * pad_ratio + ax.set_ylim(ymin - pad, ymax + pad) + + +def _cache_size(cache): + """Count memoized final-value runs.""" + return len(cache.get("_final_value_cache", {})) + + +def _comparison_cache_size(cache): + """Count memoized scalar comparison bundles.""" + return len(cache.get("_comparison_cache", {})) + + +def make_sweep_cache(price_data): + """Create a shared cache for heatmap and line sweeps.""" + return { + "_shared_price_data": price_data, + "_final_value_cache": {}, + "_comparison_cache": {}, + } + + +def _missing_artifacts(progress_label, filenames): + """Report which plot artifacts still need to be generated.""" + missing = [filename for filename in filenames if not os.path.exists(filename)] + if not missing: + print(f"[{progress_label}] skipping sweep: all artifacts already exist.") + return set() + + existing_count = len(filenames) - len(missing) + if existing_count: + print( + f"[{progress_label}] reusing {existing_count}/{len(filenames)} " + "existing artifacts; generating the missing outputs." ) + return set(missing) + - # Geometric + centeredness-proportional scaling (scales decay duration) - fp_geo_scaled = make_fingerprint(cfg, "geometric", centeredness_scaling=True) - results["geometric_scaled"] = do_run_on_historic_data( - run_fingerprint=fp_geo_scaled, params=params +def _speed_cache_key(speed): + """Stable cache token for optional arc-length speed.""" + if speed is None: + return None + return round(float(speed), 12) + + +def _make_method_cache_key(cfg, method): + """Cache key for a single-method final-value run.""" + noise_cfg = resolve_reclamm_noise_settings(cfg) + arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + key = ( + method, + bool(cfg.get("enable_noise_model", False)), + round(float(cfg["price_ratio"]), 6), + round(float(cfg["centeredness_margin"]), 6), + round(float(cfg["daily_price_shift_exponent"]), 6), + round(get_initial_pool_value(cfg), 2), + noise_cfg.get("noise_cache_key"), + None if arb_frequency is None else int(arb_frequency), + round( + float( + cfg.get( + "gas_cost", + DEFAULT_GAS_COST if cfg.get("enable_noise_model", False) else 0.0, + ) + ), + 6, + ), + round( + float( + cfg.get( + "protocol_fee_split", + DEFAULT_PROTOCOL_FEE_SPLIT if cfg.get("enable_noise_model", False) else 0.0, + ) + ), + 6, + ), ) + if method == "constant_arc_length": + key += (_speed_cache_key(cfg.get("arc_length_speed")),) + return key - # Arc-length + centeredness-proportional scaling (scales speed) - fp_cal_scaled = make_fingerprint(cfg, "constant_arc_length", centeredness_scaling=True) - results["cal_scaled"] = do_run_on_historic_data( - run_fingerprint=fp_cal_scaled, params=params + +def _make_comparison_cache_key(cfg, launch_final_values): + """Cache key for scalar heatmap metrics at a single parameter point.""" + noise_cfg = make_noise_variant_cfg(cfg, True) + arb_only_cfg = make_noise_variant_cfg(cfg, False) + key = [ + _make_method_cache_key(noise_cfg, "geometric"), + _make_method_cache_key(arb_only_cfg, "geometric"), + round(float(launch_final_values["geometric"]), 6), + ] + if RUN_CONSTANT_ARC_LENGTH: + key.extend( + [ + _make_method_cache_key(noise_cfg, "constant_arc_length"), + _make_method_cache_key(arb_only_cfg, "constant_arc_length"), + round(float(launch_final_values["constant_arc_length"]), 6), + ] + ) + return tuple(key) + + +def _run_method_final_value_cached(cfg, method, cache): + """Memoize final value for a single interpolation method.""" + final_value_cache = cache.setdefault("_final_value_cache", {}) + key = _make_method_cache_key(cfg, method) + if key not in final_value_cache: + result = do_run_on_historic_data( + run_fingerprint=make_fingerprint(cfg, method), + params=make_params(cfg), + price_data=cache["_shared_price_data"], + low_data_mode=True, + ) + final_value_cache[key] = float(result["final_value"]) + del result + gc.collect() + return final_value_cache[key] + + +def extract_comparison_metrics_from_final_values( + geo_final, arc_final, launch_final_values +): + """Summarize scalar comparison metrics from final values only.""" + return { + "efficiency_pct": (arc_final / max(abs(geo_final), 1e-12) - 1.0) * 100.0, + "launch_geometric_efficiency_pct": ( + arc_final / max(abs(launch_final_values["geometric"]), 1e-12) - 1.0 + ) + * 100.0, + "geometric_vs_launch_geometric_pct": ( + geo_final / max(abs(launch_final_values["geometric"]), 1e-12) - 1.0 + ) + * 100.0, + "constant_arc_vs_launch_constant_arc_pct": ( + arc_final + / max(abs(launch_final_values["constant_arc_length"]), 1e-12) + - 1.0 + ) + * 100.0, + } + + +def _load_required_heatmap_final_values(cfg, cache, metric_keys): + """Load only the cached final values needed for the requested heatmap metrics.""" + required_sources = set() + for metric_key in metric_keys: + required_sources.update(HEATMAP_METRIC_DEPENDENCIES[metric_key]) + + if not RUN_CONSTANT_ARC_LENGTH and any( + source.endswith("constant_arc") for source in required_sources + ): + raise ValueError( + "Constant-arc heatmap metric requested while RUN_CONSTANT_ARC_LENGTH=False" + ) + + final_values = {} + noise_cfg = None + arb_only_cfg = None + + if any(source.startswith("noise_") for source in required_sources): + noise_cfg = make_noise_variant_cfg(cfg, True) + if any(source.startswith("arb_") for source in required_sources): + arb_only_cfg = make_noise_variant_cfg(cfg, False) + + if "noise_geometric" in required_sources: + final_values["noise_geometric"] = _run_method_final_value_cached( + noise_cfg, + "geometric", + cache, + ) + if "noise_constant_arc" in required_sources: + final_values["noise_constant_arc"] = _run_method_final_value_cached( + noise_cfg, + "constant_arc_length", + cache, + ) + if "arb_geometric" in required_sources: + final_values["arb_geometric"] = _run_method_final_value_cached( + arb_only_cfg, + "geometric", + cache, + ) + if "arb_constant_arc" in required_sources: + final_values["arb_constant_arc"] = _run_method_final_value_cached( + arb_only_cfg, + "constant_arc_length", + cache, + ) + return final_values + + +def extract_heatmap_metrics_from_mode_final_values( + metric_keys, + final_values, + launch_final_values, +): + """Collect the requested scalar heatmap metrics from cached final values.""" + metrics = {} + + if "efficiency_pct" in metric_keys: + metrics["efficiency_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(final_values["noise_geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "launch_geometric_efficiency_pct" in metric_keys: + metrics["launch_geometric_efficiency_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(launch_final_values["geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "geometric_vs_launch_geometric_pct" in metric_keys: + metrics["geometric_vs_launch_geometric_pct"] = ( + final_values["noise_geometric"] + / max(abs(launch_final_values["geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "constant_arc_vs_launch_constant_arc_pct" in metric_keys: + metrics["constant_arc_vs_launch_constant_arc_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(launch_final_values["constant_arc_length"]), 1e-12) + - 1.0 + ) * 100.0 + + if "noise_geometric_final_value_musd" in metric_keys: + metrics["noise_geometric_final_value_musd"] = ( + final_values["noise_geometric"] / 1e6 + ) + + if "noise_constant_arc_final_value_musd" in metric_keys: + metrics["noise_constant_arc_final_value_musd"] = ( + final_values["noise_constant_arc"] / 1e6 + ) + + if "noise_vs_arb_geometric_improvement_pct" in metric_keys: + metrics["noise_vs_arb_geometric_improvement_pct"] = ( + final_values["noise_geometric"] + / max(abs(final_values["arb_geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "noise_vs_arb_constant_arc_improvement_pct" in metric_keys: + metrics["noise_vs_arb_constant_arc_improvement_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(final_values["arb_constant_arc"]), 1e-12) + - 1.0 + ) * 100.0 + + return metrics + + +def extract_comparison_metrics(results, launch_final_values): + """Summarize scalar heatmap metrics for a pair of runs.""" + geo = results["geometric"] + arc = results["constant_arc_length"] + + geo_final = float(geo["final_value"]) + arc_final = float(arc["final_value"]) + + return extract_comparison_metrics_from_final_values( + geo_final, + arc_final, + launch_final_values=launch_final_values, ) - return results + +def run_comparison_cached(cfg, cache, launch_final_values, metric_keys): + """Memoize scalar heatmap metrics across heatmap sweeps.""" + requested_metric_keys = tuple(dict.fromkeys(metric_keys)) + comparison_cache = cache.setdefault("_comparison_cache", {}) + cache_key = _make_comparison_cache_key(cfg, launch_final_values) + cached_metrics = comparison_cache.setdefault(cache_key, {}) + missing_metric_keys = [ + metric_key for metric_key in requested_metric_keys if metric_key not in cached_metrics + ] + if missing_metric_keys: + final_values = _load_required_heatmap_final_values( + cfg, + cache, + missing_metric_keys, + ) + cached_metrics.update( + extract_heatmap_metrics_from_mode_final_values( + missing_metric_keys, + final_values, + launch_final_values=launch_final_values, + ) + ) + return { + metric_key: cached_metrics[metric_key] for metric_key in requested_metric_keys + } + + +def build_heatmap_matrices( + x_values, + y_values, + x_key, + y_key, + base_cfg, + metric_keys, + cache, + progress_label, + launch_final_values, +): + """Evaluate multiple metrics over a 2D parameter grid in one pass.""" + data = { + metric_key: np.zeros((len(y_values), len(x_values)), dtype=float) + for metric_key in metric_keys + } + total_points = len(y_values) * len(x_values) + + print( + f"[{progress_label}] start: {len(y_values)} rows x {len(x_values)} cols " + f"= {total_points} parameter points" + ) + + for yi, y_value in enumerate(y_values): + final_cache_before_row = _cache_size(cache) + comparison_cache_before_row = _comparison_cache_size(cache) + for xi, x_value in enumerate(x_values): + cfg = dict(base_cfg) + cfg[x_key] = float(x_value) + cfg[y_key] = float(y_value) + metrics = run_comparison_cached( + cfg, + cache, + launch_final_values=launch_final_values, + metric_keys=metric_keys, + ) + for metric_key in metric_keys: + data[metric_key][yi, xi] = metrics[metric_key] + + completed_points = (yi + 1) * len(x_values) + row_new_final_runs = _cache_size(cache) - final_cache_before_row + row_new_comparisons = ( + _comparison_cache_size(cache) - comparison_cache_before_row + ) + row_pct = completed_points / total_points * 100.0 + print( + f"[{progress_label}] row {yi + 1}/{len(y_values)} complete " + f"({y_key}={float(y_value):.4f}, {completed_points}/{total_points} " + f"points, {row_pct:.1f}%, {row_new_final_runs} new final-value runs, " + f"{row_new_comparisons} new comparison bundles)" + ) + + print( + f"[{progress_label}] done: " + + ", ".join( + ( + f"{metric_key} min={float(np.nanmin(data[metric_key])):.4f}, " + f"max={float(np.nanmax(data[metric_key])):.4f}" + ) + for metric_key in metric_keys + ) + + ( + f", final_value_cache_size={_cache_size(cache)}, " + f"comparison_cache_size={_comparison_cache_size(cache)}" + ) + ) + + return data + + +def build_metric_curve( + x_values, + x_key, + base_cfg, + metric_key, + cache, + launch_final_values, +): + """Evaluate one metric over a 1D sweep.""" + data = np.zeros(len(x_values), dtype=float) + for xi, x_value in enumerate(x_values): + cfg = dict(base_cfg) + cfg[x_key] = float(x_value) + metrics = run_comparison_cached( + cfg, + cache, + launch_final_values=launch_final_values, + metric_keys=(metric_key,), + ) + data[xi] = metrics[metric_key] + return data + + +def _compute_axis_edges(values, scale="linear"): + """Convert axis centers to cell edges for pcolormesh.""" + values = np.asarray(values, dtype=float) + if values.size == 1: + if scale == "log": + return np.array([values[0] / np.sqrt(10.0), values[0] * np.sqrt(10.0)]) + pad = max(abs(values[0]) * 0.5, 1.0) + return np.array([values[0] - pad, values[0] + pad]) + + if scale == "log": + log_values = np.log10(values) + edges = np.empty(values.size + 1, dtype=float) + edges[1:-1] = 0.5 * (log_values[:-1] + log_values[1:]) + edges[0] = log_values[0] - 0.5 * (log_values[1] - log_values[0]) + edges[-1] = log_values[-1] + 0.5 * (log_values[-1] - log_values[-2]) + return 10.0 ** edges + + edges = np.empty(values.size + 1, dtype=float) + edges[1:-1] = 0.5 * (values[:-1] + values[1:]) + edges[0] = values[0] - 0.5 * (values[1] - values[0]) + edges[-1] = values[-1] + 0.5 * (values[-1] - values[-2]) + return edges + + +def plot_heatmap( + data, + x_values, + y_values, + x_label, + y_label, + title, + colorbar_label, + filename, + xticks=None, + yticks=None, + xscale="linear", + center_zero=True, + cmap=None, + color_norm=None, + symlog_linthresh=None, +): + """Render and save a single heatmap.""" + finite = np.asarray(data, dtype=float) + finite = finite[np.isfinite(finite)] + + if center_zero: + vmax = max(abs(float(np.nanmin(data))), abs(float(np.nanmax(data))), 1e-9) + if color_norm == "symlog" and symlog_linthresh is not None and vmax > symlog_linthresh: + norm = SymLogNorm( + linthresh=symlog_linthresh, + linscale=1.0, + vmin=-vmax, + vmax=vmax, + base=10.0, + ) + else: + norm = TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) + cmap_name = cmap or "RdYlGn" + else: + if finite.size == 0: + vmin, vmax = 0.0, 1.0 + else: + vmin = float(finite.min()) + vmax = float(finite.max()) + if np.isclose(vmin, vmax): + pad = max(abs(vmin) * 0.01, 1e-9) + vmin -= pad + vmax += pad + norm = Normalize(vmin=vmin, vmax=vmax) + cmap_name = cmap or "viridis" + + x_edges = _compute_axis_edges(x_values, scale=xscale) + y_edges = _compute_axis_edges(y_values, scale="linear") + + fig, ax = plt.subplots(figsize=(8.5, 6.0)) + im = ax.pcolormesh( + x_edges, + y_edges, + data, + cmap=cmap_name, + norm=norm, + shading="auto", + ) + + ax.set_xlabel(x_label) + ax.set_ylabel(y_label) + ax.set_title(title) + if xscale == "log": + ax.set_xscale("log") + ax.set_xticks(np.asarray(xticks if xticks is not None else x_values, dtype=float)) + ax.set_yticks(np.asarray(yticks if yticks is not None else y_values, dtype=float)) + ax.grid(False) + + cbar = fig.colorbar(im, ax=ax) + cbar.set_label(colorbar_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + +def plot_arc_speed_line_chart( + data, + x_values, + y_values, + y_label, + title, + filename, + launch_curve, + launch_auto_speed=None, +): + """Plot thin multi-series efficiency lines over the arc-speed sweep.""" + fig, ax = plt.subplots(figsize=(10.5, 5.75)) + cmap = plt.cm.viridis + colors = cmap(np.linspace(0.0, 1.0, len(y_values))) + plotted_series = [] + + for yi, (y_value, color) in enumerate(zip(y_values, colors)): + series = np.asarray(data[yi], dtype=float) + plotted_series.append(series) + ax.plot( + x_values, + series, + color=color, + linewidth=SWEEP_LINE_WIDTH, + alpha=0.8, + ) + + launch_curve = np.asarray(launch_curve, dtype=float) + plotted_series.append(launch_curve) + ax.plot( + x_values, + launch_curve, + color="black", + linewidth=REFERENCE_LINE_WIDTH, + alpha=0.9, + label="Current launch config", + ) + if launch_auto_speed is not None: + ax.axvline( + float(launch_auto_speed), + color="black", + ls=":", + linewidth=0.8, + alpha=0.7, + label="Launch auto-cal speed", + ) + + ax.axhline(0.0, color="gray", ls="--", linewidth=0.8, alpha=0.5) + ax.set_xscale("log") + ax.set_xticks(ARC_LENGTH_SPEED_TICKS) + ax.set_xlabel("Arc-length speed") + ax.set_ylabel("Efficiency vs geometric (%)") + ax.set_title(title) + _set_padded_ylim(ax, plotted_series, pad_ratio=0.08) + ax.grid(True, alpha=0.25) + ax.legend(fontsize=8) + + sm = ScalarMappable( + norm=Normalize(vmin=float(np.min(y_values)), vmax=float(np.max(y_values))), + cmap=cmap, + ) + sm.set_array([]) + cbar = fig.colorbar(sm, ax=ax) + cbar.set_label(y_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + +def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): + """Generate pairwise heatmaps for thermostat tuning and noise-vs-arb effects.""" + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data) + metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs heatmap geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "launch_geometric_efficiency_pct", + "title": "Efficiency vs launch-style geometric", + "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", + "slug": "launch_geometric_efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "geometric_vs_launch_geometric_pct", + "title": "Geometric tuning vs launch-style geometric", + "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", + "slug": "geometric_vs_launch_geometric", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "constant_arc_vs_launch_constant_arc_pct", + "title": "Const arc tuning vs launch-style const arc", + "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", + "slug": "constant_arc_vs_launch_constant_arc", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_geometric_final_value_musd", + "title": "Geometric final value with noise model", + "colorbar_label": "Geometric final value with noise model ($M)", + "slug": "noise_geometric_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_geometric_improvement_pct", + "title": "Noise-model improvement over arb-only (geometric)", + "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", + "slug": "noise_vs_arb_geometric_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + if not RUN_CONSTANT_ARC_LENGTH: + metric_specs = [ + spec + for spec in metric_specs + if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS + ] + pair_specs = [ + { + "slug": "price_ratio_vs_margin", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_MARGINS, + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "x_label": "Price ratio", + "y_label": "Centeredness margin", + "title_suffix": ( + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": PRICE_RATIO_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "shift_exp_vs_margin", + "x_values": HEATMAP_SHIFT_EXPONENTS, + "y_values": HEATMAP_MARGINS, + "x_key": "daily_price_shift_exponent", + "y_key": "centeredness_margin", + "x_label": "Shift exponent", + "y_label": "Centeredness margin", + "title_suffix": f"price_ratio fixed at {base_cfg['price_ratio']:.2f}", + "xticks": SHIFT_EXPONENT_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "price_ratio_vs_shift_exp", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "price_ratio", + "y_key": "daily_price_shift_exponent", + "x_label": "Price ratio", + "y_label": "Shift exponent", + "title_suffix": ( + f"margin fixed at {base_cfg['centeredness_margin']:.2f}" + ), + "xticks": PRICE_RATIO_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + }, + ] + + metric_spec_map = {spec["key"]: spec for spec in metric_specs} + + if RUN_CONSTANT_ARC_LENGTH: + print( + "Using launch-style benchmarks " + f"Geo=${launch_final_values['geometric']:,.0f}, " + f"Const Arc=${launch_final_values['constant_arc_length']:,.0f}, " + f"TVL={format_tvl_millions_label(base_cfg)}." + ) + print( + "Running {count} heatmap pair sweeps sequentially " + "(current outputs use cached noise-model runs; improvement heatmaps " + "reuse those values and add cached arb-only runs).".format( + count=len(pair_specs) + ) + ) + else: + print( + "Using launch-style geometric benchmark " + f"Geo=${launch_final_values['geometric']:,.0f}, " + f"TVL={format_tvl_millions_label(base_cfg)}." + ) + print( + "RUN_CONSTANT_ARC_LENGTH=False, so only geometric heatmaps will be generated " + "and only geometric/arb-only geometric runs will be scheduled." + ) + + for pair in pair_specs: + output_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair["slug"], + ) + for spec in metric_specs + } + missing_files = _missing_artifacts( + pair["slug"], + list(output_files.values()), + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in metric_specs + if output_files[spec["key"]] in missing_files + ] + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=base_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair["slug"], + launch_final_values=launch_final_values, + ) + print(f"[{pair['slug']}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['title_suffix']} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=output_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + del data_by_metric + gc.collect() + + if owns_cache: + cache.clear() + gc.collect() + print("Released heatmap metric cache.") + + +def compute_auto_calibrated_arc_length_speed(cfg, price_data): + """Compute the launch/reference auto-calibrated speed for a config.""" + start_ts = pd.Timestamp(cfg["start"]) + + if isinstance(price_data.index, pd.DatetimeIndex): + row = price_data.loc[start_ts] + else: + start_unix_ms = int(start_ts.timestamp() * 1000.0) + index_values = price_data.index.to_numpy(dtype=np.int64) + row_idx = int(np.searchsorted(index_values, start_unix_ms, side="left")) + if row_idx >= len(index_values): + row_idx = len(index_values) - 1 + if row_idx > 0 and index_values[row_idx] != start_unix_ms: + prev_idx = row_idx - 1 + if abs(index_values[prev_idx] - start_unix_ms) <= abs( + index_values[row_idx] - start_unix_ms + ): + row_idx = prev_idx + row = price_data.iloc[row_idx] + + if isinstance(row, pd.DataFrame): + row = row.iloc[0] + + if isinstance(price_data.columns, pd.MultiIndex): + initial_price_values = [ + float(row[(token, "close")]) + for token in cfg["tokens"] + ] + else: + initial_price_values = [ + float(row[f"close_{token}"]) + for token in cfg["tokens"] + ] + + initial_prices = jnp.array(initial_price_values, dtype=jnp.float64) + initial_reserves, Va, Vb = initialise_reclamm_reserves( + get_initial_pool_value(cfg), + initial_prices, + float(cfg["price_ratio"]), + ) + market_price_0 = float(initial_prices[0] / initial_prices[1]) + sqrt_Q = jnp.sqrt( + compute_price_ratio( + initial_reserves[0], + initial_reserves[1], + Va, + Vb, + ) + ) + return float( + calibrate_arc_length_speed( + initial_reserves[0], + initial_reserves[1], + Va, + Vb, + to_daily_price_shift_base(float(cfg["daily_price_shift_exponent"])), + 60.0, + sqrt_Q, + market_price_0, + centeredness_margin=float(cfg["centeredness_margin"]), + ) + ) + + +def generate_arc_speed_efficiency_artifacts( + base_cfg, + launch_cfg, + price_data, + launch_final_values, + cache=None, +): + """Generate arc-speed heatmaps plus the existing efficiency line charts.""" + if not RUN_CONSTANT_ARC_LENGTH: + print("\nSkipping arc-speed heatmaps because RUN_CONSTANT_ARC_LENGTH=False.") + return + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data) + launch_auto_speed = compute_auto_calibrated_arc_length_speed(launch_cfg, price_data) + heatmap_metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in heatmap_metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + pair_specs = [ + { + "slug": "arc_speed_vs_price_ratio", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_PRICE_RATIOS, + "x_key": "arc_length_speed", + "y_key": "price_ratio", + "x_label": "Arc-length speed", + "y_label": "Price ratio", + "title_suffix": ( + f"margin fixed at {base_cfg['centeredness_margin']:.2f}, " + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": PRICE_RATIO_TICKS, + }, + { + "slug": "arc_speed_vs_margin", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_MARGINS, + "x_key": "arc_length_speed", + "y_key": "centeredness_margin", + "x_label": "Arc-length speed", + "y_label": "Centeredness margin", + "title_suffix": ( + f"price_ratio fixed at {base_cfg['price_ratio']:.2f}, " + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "arc_speed_vs_shift_exp", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "arc_length_speed", + "y_key": "daily_price_shift_exponent", + "x_label": "Arc-length speed", + "y_label": "Shift exponent", + "title_suffix": ( + f"price_ratio fixed at {base_cfg['price_ratio']:.2f}, " + f"margin fixed at {base_cfg['centeredness_margin']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + }, + ] + metric_spec_map = {spec["key"]: spec for spec in heatmap_metric_specs} + + print( + "\nGenerating arc-speed heatmaps and line charts " + f"(launch auto-cal speed={launch_auto_speed:.3e}, TVL={format_tvl_millions_label(base_cfg)})..." + ) + + for pair in pair_specs: + heatmap_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair["slug"], + ) + for spec in heatmap_metric_specs + } + line_filename = tvl_artifact_filename( + "reclamm_line_efficiency", + base_cfg, + suffix=pair["slug"], + ) + missing_files = _missing_artifacts( + pair["slug"], + list(heatmap_files.values()) + [line_filename], + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in heatmap_metric_specs + if heatmap_files[spec["key"]] in missing_files + ] + if line_filename in missing_files and "efficiency_pct" not in missing_metric_keys: + missing_metric_keys.append("efficiency_pct") + + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=base_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair["slug"], + launch_final_values=launch_final_values, + ) + for metric_key in missing_metric_keys: + if metric_key not in heatmap_files: + continue + if heatmap_files[metric_key] not in missing_files: + continue + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['title_suffix']} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=heatmap_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + xscale="log", + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + + if line_filename in missing_files: + efficiency_data = data_by_metric["efficiency_pct"] + launch_curve = build_metric_curve( + x_values=pair["x_values"], + x_key=pair["x_key"], + base_cfg=launch_cfg, + metric_key="efficiency_pct", + cache=cache, + launch_final_values=launch_final_values, + ) + plot_arc_speed_line_chart( + data=efficiency_data, + x_values=pair["x_values"], + y_values=pair["y_values"], + y_label=pair["y_label"], + title=( + "Arc-speed efficiency sweep: " + f"{pair['title_suffix']} | TVL {format_tvl_millions_label(base_cfg)}" + ), + filename=line_filename, + launch_curve=launch_curve, + launch_auto_speed=launch_auto_speed, + ) + del data_by_metric + gc.collect() + + if owns_cache: + cache.clear() + gc.collect() + print("Released arc-speed sweep cache.") + + +def get_launch_final_values(all_results, launch_cfg, price_data): + """Reuse launch-style runs when available; otherwise run them once.""" + for cfg, results in all_results: + if cfg["name"] == launch_cfg["name"]: + launch_final_values = { + "geometric": float(results["geometric"]["final_value"]), + } + if "constant_arc_length" in results: + launch_final_values["constant_arc_length"] = float( + results["constant_arc_length"]["final_value"] + ) + return launch_final_values + + print("\nRunning launch-style benchmarks for heatmaps...") + launch_results = run_comparison( + launch_cfg, + price_data=price_data, + low_data_mode=True, + ) + launch_final_values = { + "geometric": float(launch_results["geometric"]["final_value"]), + } + if "constant_arc_length" in launch_results: + launch_final_values["constant_arc_length"] = float( + launch_results["constant_arc_length"]["final_value"] + ) + del launch_results + gc.collect() + return launch_final_values + def print_comparison(cfg, results): """Print text summary table.""" - methods = [ - ("Geometric", results["geometric"]), - ("Geo+Scaled", results["geometric_scaled"]), - ("Const Arc", results["constant_arc_length"]), - ("Arc+Scaled", results["cal_scaled"]), - ] + methods = [("Geometric", results["geometric"])] + has_constant_arc = "constant_arc_length" in results + if has_constant_arc: + methods.append(("Const Arc", results["constant_arc_length"])) + noise_cfg = resolve_reclamm_noise_settings(cfg) hodl_value = float((methods[0][1]["reserves"][0] * methods[0][1]["prices"][-1]).sum()) @@ -128,6 +1612,18 @@ def print_comparison(cfg, results): f"margin={cfg['centeredness_margin']}, " f"shift_exp={cfg['daily_price_shift_exponent']}, " f"fees={cfg['fees']}") + print( + f" base_tvl=${get_initial_pool_value(cfg):,.0f} " + f"(TVL {format_tvl_millions_label(cfg)})" + ) + print(f" note={cfg['reason']}") + print( + f" noise={noise_cfg['noise_summary']}, " + f"gas={cfg.get('gas_cost', 0.0)}, " + f"protocol_fee_split={cfg.get('protocol_fee_split', 0.0)}" + ) + if not has_constant_arc: + print(" constant_arc=disabled") print("-" * 105) header = " {:20s}".format("") for name, _ in methods: @@ -158,18 +1654,27 @@ def print_comparison(cfg, results): vs = (float(r["final_value"]) / hodl_value - 1) * 100 row += f" {vs:>13.2f}%" print(row) + + if has_constant_arc: + geo_final = float(results["geometric"]["final_value"]) + arc_final = float(results["constant_arc_length"]["final_value"]) + geo_lvr = hodl_value - geo_final + arc_lvr = hodl_value - arc_final + print(f" {'Const Arc - Geo':20s} ${arc_final - geo_final:>13,.0f}") + print(f" {'LVR saved vs Geo':20s} ${geo_lvr - arc_lvr:>13,.0f}") print("=" * 105) + def plot_comparison(cfg, results, fig_idx): - """Plot 4-panel comparison for one config.""" - # Method name → (result dict, color, linestyle) + """Plot comparison diagnostics for one config.""" + tvl_label = format_tvl_millions_label(cfg) variants = { "Geometric": (results["geometric"], "C0", "-"), - "Geo+Scaled": (results["geometric_scaled"], "C1", "-"), - "Const arc-len": (results["constant_arc_length"], "C2", "--"), - "Arc+Scaled": (results["cal_scaled"], "C3", "--"), } + has_constant_arc = "constant_arc_length" in results + if has_constant_arc: + variants["Const arc-len"] = (results["constant_arc_length"], "C2", "--") geo = results["geometric"] geo_prices = np.array(geo["prices"]) @@ -181,22 +1686,21 @@ def plot_comparison(cfg, results, fig_idx): price_ratio_traj = geo_prices[:n_steps, 0] / geo_prices[:n_steps, 1] fig, axes = plt.subplots(2, 2, figsize=(14, 10)) - fig.suptitle(cfg["name"], fontsize=13, fontweight="bold") + fig.suptitle(f"{cfg['name']} — TVL {tvl_label}", fontsize=13, fontweight="bold") - # (0,0) Pool value over time ax = axes[0, 0] + plotted_values = [] for name, (r, color, ls) in variants.items(): vals = np.array(r["value"]) + plotted_values.append(vals / 1e6) ax.plot(t_days, vals / 1e6, color=color, ls=ls, label=name, alpha=0.9) - ax.plot(t_days, np.array(hodl_traj) / 1e6, color="gray", ls=":", - alpha=0.5, label="HODL") + _set_padded_ylim(ax, plotted_values, pad_ratio=0.03) ax.set_xlabel("Days") ax.set_ylabel("Pool value ($M)") ax.set_title("Pool value") ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (0,1) Cumulative LVR ax = axes[0, 1] for name, (r, color, ls) in variants.items(): vals = np.array(r["value"]) @@ -208,7 +1712,6 @@ def plot_comparison(cfg, results, fig_idx): ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (1,0) Price ratio ax = axes[1, 0] ax.plot(t_days, price_ratio_traj, color="C4", alpha=0.7) ax.set_xlabel("Days") @@ -216,7 +1719,6 @@ def plot_comparison(cfg, results, fig_idx): ax.set_title("Price path") ax.grid(True, alpha=0.3) - # (1,1) Empirical weights ax = axes[1, 1] for name, (r, color, ls) in variants.items(): w = np.array(r["weights"]) @@ -230,19 +1732,21 @@ def plot_comparison(cfg, results, fig_idx): ax.grid(True, alpha=0.3) plt.tight_layout() - fname = f"reclamm_thermostat_comparison_{fig_idx}.png" + fname = tvl_artifact_filename("reclamm_thermostat_comparison", cfg, suffix=str(fig_idx)) plt.savefig(fname, dpi=150) print(f"Saved {fname}") plt.close(fig) - # Second figure: diagnostics + if not has_constant_arc: + print("Skipping constant-arc comparison diagnostics because RUN_CONSTANT_ARC_LENGTH=False.") + return + geo_values = np.array(geo["value"]) geo_lvr = np.array(hodl_traj) - geo_values fig2, axes2 = plt.subplots(1, 3, figsize=(18, 5)) - fig2.suptitle(f"{cfg['name']} — diagnostics", fontsize=13, fontweight="bold") + fig2.suptitle(f"{cfg['name']} — diagnostics — TVL {tvl_label}", fontsize=13, fontweight="bold") - # (left) Value difference vs geometric ax = axes2[0] for name, (r, color, ls) in variants.items(): if name == "Geometric": @@ -257,7 +1761,6 @@ def plot_comparison(cfg, results, fig_idx): ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (middle) LVR ratio over time ax = axes2[1] mask = np.abs(geo_lvr) > 100 if mask.any(): @@ -279,7 +1782,6 @@ def plot_comparison(cfg, results, fig_idx): ax.set_title("Relative LVR") ax.grid(True, alpha=0.3) - # (right) Per-step LVR histogram ax = axes2[2] all_pos = [] for name, (r, color, ls) in variants.items(): @@ -306,74 +1808,180 @@ def plot_comparison(cfg, results, fig_idx): ax.grid(True, alpha=0.3) plt.tight_layout() - fname2 = f"reclamm_thermostat_diff_{fig_idx}.png" + fname2 = tvl_artifact_filename("reclamm_thermostat_diff", cfg, suffix=str(fig_idx)) plt.savefig(fname2, dpi=150) print(f"Saved {fname2}") plt.close(fig2) + arc_values = np.array(results["constant_arc_length"]["value"]) + n_eff = min(len(geo_values), len(arc_values)) + t_eff = np.arange(n_eff) / (60 * 24) + efficiency_pct = ( + (arc_values[:n_eff] - geo_values[:n_eff]) + / np.maximum(np.abs(geo_values[:n_eff]), 1e-12) + * 100.0 + ) + + fig3, ax3 = plt.subplots(1, 1, figsize=(10, 4.5)) + fig3.suptitle(f"{cfg['name']} — efficiency — TVL {tvl_label}", fontsize=13, fontweight="bold") + ax3.plot( + t_eff, + efficiency_pct, + color="C2", + linewidth=1.8, + label="(Const Arc - Geo) / Geo", + ) + ax3.axhline(0.0, color="gray", ls="--", alpha=0.6) + _set_padded_ylim(ax3, [efficiency_pct], pad_ratio=0.08) + ax3.set_xlabel("Days") + ax3.set_ylabel("Efficiency vs geometric (%)") + ax3.set_title("Efficiency") + ax3.legend(fontsize=8) + ax3.grid(True, alpha=0.3) + + plt.tight_layout() + fname3 = tvl_artifact_filename("reclamm_thermostat_efficiency", cfg, suffix=str(fig_idx)) + plt.savefig(fname3, dpi=150) + print(f"Saved {fname3}") + plt.close(fig3) + + if __name__ == "__main__": - all_results = [] - for i, cfg in enumerate(CONFIGS): - print(f"\n>>> Running {cfg['name']}...") - try: - results = run_comparison(cfg) - print_comparison(cfg, results) - plot_comparison(cfg, results, i) - all_results.append((cfg, results)) - except Exception as e: - print(f" FAILED: {e}") - import traceback - traceback.print_exc() - - # Summary overlay: all configs on one figure (pool value normalised) - if len(all_results) > 1: - fig, axes = plt.subplots(1, 2, figsize=(16, 5)) - fig.suptitle("Cross-config comparison (normalised)", fontsize=13, - fontweight="bold") - - method_keys = [ - ("geometric", "geo", "-"), - ("geometric_scaled", "geo+s", "-."), - ("constant_arc_length", "arc", "--"), - ("cal_scaled", "arc+s", ":"), - ] + shared_price_data = load_shared_price_data(CONFIGS) + + for initial_pool_value in TVL_SWEEP_VALUES: + tvl_configs = configs_for_tvl(CONFIGS, initial_pool_value) + tvl_label = format_tvl_millions_label(tvl_configs[0]) + print(f"\n=== TVL sweep: {tvl_label} ===") + + all_results = [] + for i, cfg in enumerate(tvl_configs): + print(f"\n>>> Running {cfg['name']} at TVL {tvl_label}...") + try: + results = run_comparison(cfg, price_data=shared_price_data) + print_comparison(cfg, results) + plot_comparison(cfg, results, i) + all_results.append((cfg, results)) + except Exception as e: + print(f" FAILED: {e}") + import traceback - for i, (cfg, results) in enumerate(all_results): - geo_v = np.array(results["geometric"]["value"]) - t = np.arange(len(geo_v)) / (60 * 24) - short_name = cfg["name"].split("(")[0].strip() - - for j, (key, suffix, ls) in enumerate(method_keys): - v = np.array(results[key]["value"]) - color_idx = i * len(method_keys) + j - - # (left) Normalised pool value - axes[0].plot(t, v / v[0], ls=ls, alpha=0.8, - label=f"{short_name} {suffix}", - color=f"C{color_idx % 10}") - - # (right) Value difference vs geometric (skip geo itself) - if key != "geometric": - pct_diff = (v - geo_v) / geo_v * 100 - axes[1].plot(t, pct_diff, ls=ls, alpha=0.8, - label=f"{short_name} {suffix}", - color=f"C{color_idx % 10}") - - axes[0].set_xlabel("Days") - axes[0].set_ylabel("Normalised pool value") - axes[0].set_title("Pool value (V/V0)") - axes[0].legend(fontsize=6, ncol=2) - axes[0].grid(True, alpha=0.3) - - axes[1].set_xlabel("Days") - axes[1].set_ylabel("(Method - Geo) / Geo (%)") - axes[1].set_title("Relative value difference vs Geometric") - axes[1].axhline(0, color="gray", ls="--", alpha=0.5) - axes[1].legend(fontsize=6, ncol=2) - axes[1].grid(True, alpha=0.3) - - plt.tight_layout() - plt.savefig("reclamm_thermostat_summary.png", dpi=150) - print("\nSaved reclamm_thermostat_summary.png") - plt.close(fig) + traceback.print_exc() + + if len(all_results) > 1: + if RUN_CONSTANT_ARC_LENGTH: + fig, axes = plt.subplots(1, 2, figsize=(16, 5)) + fig.suptitle( + f"Cross-config comparison (normalised) — TVL {tvl_label}", + fontsize=13, + fontweight="bold", + ) + + method_keys = [ + ("geometric", "geo", "-"), + ("constant_arc_length", "arc", "--"), + ] + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + + for j, (key, suffix, ls) in enumerate(method_keys): + v = np.array(results[key]["value"]) + color_idx = i * len(method_keys) + j + + axes[0].plot( + t, + v / v[0], + ls=ls, + alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}", + ) + + if key != "geometric": + pct_diff = (v - geo_v) / geo_v * 100 + axes[1].plot( + t, + pct_diff, + ls=ls, + alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}", + ) + + axes[0].set_xlabel("Days") + axes[0].set_ylabel("Normalised pool value") + axes[0].set_title("Pool value (V/V0)") + axes[0].legend(fontsize=6, ncol=2) + axes[0].grid(True, alpha=0.3) + + axes[1].set_xlabel("Days") + axes[1].set_ylabel("Efficiency vs geometric (%)") + axes[1].set_title("Efficiency vs Geometric") + axes[1].axhline(0, color="gray", ls="--", alpha=0.5) + axes[1].legend(fontsize=6, ncol=2) + axes[1].grid(True, alpha=0.3) + else: + fig, ax = plt.subplots(1, 1, figsize=(9, 5)) + fig.suptitle( + f"Cross-config comparison (normalised geometric) — TVL {tvl_label}", + fontsize=13, + fontweight="bold", + ) + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + ax.plot( + t, + geo_v / geo_v[0], + ls="-", + alpha=0.8, + label=f"{short_name} geo", + color=f"C{i % 10}", + ) + + ax.set_xlabel("Days") + ax.set_ylabel("Normalised pool value") + ax.set_title("Geometric pool value (V/V0)") + ax.legend(fontsize=6, ncol=2) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + summary_name = tvl_artifact_filename( + "reclamm_thermostat_summary", + tvl_configs[0], + ) + plt.savefig(summary_name, dpi=150) + print(f"\nSaved {summary_name}") + plt.close(fig) + + launch_final_values = get_launch_final_values( + all_results, + launch_cfg=tvl_configs[0], + price_data=shared_price_data, + ) + shared_sweep_cache = make_sweep_cache(shared_price_data) + + print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") + generate_heatmaps( + dict(tvl_configs[1]), + shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + + generate_arc_speed_efficiency_artifacts( + dict(tvl_configs[1]), + launch_cfg=dict(tvl_configs[0]), + price_data=shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + shared_sweep_cache.clear() + gc.collect() + print(f"Released shared sweep cache for TVL {tvl_label}.") diff --git a/scripts/demo_run_reclamm.py b/scripts/demo_run_reclamm.py index 3ea21ec6..132f5122 100644 --- a/scripts/demo_run_reclamm.py +++ b/scripts/demo_run_reclamm.py @@ -36,105 +36,127 @@ def balancer_fingerprint(tokens, start, end, fees): } +def reclamm_fingerprint(tokens, start, end, fees, interpolation_method="geometric"): + """Build a reCLAMM fingerprint for a demo scenario.""" + return { + "tokens": tokens, + "rule": "reclamm", + "startDateString": start, + "endDateString": end, + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": fees, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + "reclamm_interpolation_method": interpolation_method, + "reclamm_arc_length_speed": None, + } + + +def reclamm_params(price_ratio, centeredness_margin, daily_price_shift_exponent): + """Build reCLAMM params from a concise config.""" + return { + "price_ratio": jnp.array(price_ratio), + "centeredness_margin": jnp.array(centeredness_margin), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(daily_price_shift_exponent) + ), + } + + +def _apply_active_noise_settings(fp): + """Enable the active AAVE/ETH reCLAMM noise model for demo runs.""" + if fp.get("rule") != "reclamm" or list(fp.get("tokens", [])) != ["AAVE", "ETH"]: + return fp, "disabled" + + from compare_reclamm_thermostats import ( + AAVE_ETH_NOISE_SETTINGS, + resolve_reclamm_noise_settings, + ) + + cfg = { + "tokens": fp["tokens"], + "start": fp["startDateString"], + "end": fp["endDateString"], + "enable_noise_model": True, + "noise_model": AAVE_ETH_NOISE_SETTINGS["noise_model"], + "noise_artifact_dir": AAVE_ETH_NOISE_SETTINGS["noise_artifact_dir"], + "noise_pool_id": AAVE_ETH_NOISE_SETTINGS["noise_pool_id"], + "gas_cost": fp.get("gas_cost", AAVE_ETH_NOISE_SETTINGS["gas_cost"]), + "protocol_fee_split": fp.get( + "protocol_fee_split", + AAVE_ETH_NOISE_SETTINGS["protocol_fee_split"], + ), + "arb_frequency": fp.get("arb_frequency"), + "noise_trader_ratio": fp.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": fp.get("reclamm_noise_params"), + "noise_arrays_path": fp.get("noise_arrays_path"), + } + noise_cfg = resolve_reclamm_noise_settings(cfg) + + updated = dict(fp) + updated["gas_cost"] = cfg["gas_cost"] + updated["protocol_fee_split"] = cfg["protocol_fee_split"] + updated["noise_trader_ratio"] = noise_cfg.get("noise_trader_ratio", 0.0) + for key in ("noise_model", "reclamm_noise_params", "noise_arrays_path", "arb_frequency"): + if noise_cfg.get(key) is not None: + updated[key] = noise_cfg[key] + return updated, noise_cfg["noise_summary"] + + SCENARIOS = [ { - "name": "AAVE/ETH on-chain (25bps)", + "name": "AAVE/ETH launch-style range (25bps, geometric)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0025, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(1.5), - "centeredness_margin": jnp.array(0.5), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.1) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="geometric", + ), + "params": reclamm_params(1.5014, 0.5, 0.1), }, }, { - "name": "AAVE/ETH zero fees", + "name": "AAVE/ETH tighter launch-style range (25bps, geometric)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(1.5), - "centeredness_margin": jnp.array(0.5), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.1) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="geometric", + ), + "params": reclamm_params(1.15, 0.5, 0.1), }, }, { - "name": "AAVE/ETH wide range (25bps)", + "name": "AAVE/ETH tighter launch-style range (25bps, constant arc)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0025, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(4.0), - "centeredness_margin": jnp.array(0.2), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(1.0) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="constant_arc_length", + ), + "params": reclamm_params(1.15, 0.5, 0.1), }, }, { "name": "BTC/ETH (10bps)", "reclamm": { - "fingerprint": { - "tokens": ["BTC", "ETH"], - "rule": "reclamm", - "startDateString": "2024-01-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.001, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(2.0), - "centeredness_margin": jnp.array(0.3), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.5) - ), - }, + "fingerprint": reclamm_fingerprint( + ["BTC", "ETH"], + "2024-01-01 00:00:00", + "2025-06-01 00:00:00", + 0.001, + interpolation_method="geometric", + ), + "params": reclamm_params(2.0, 0.3, 0.5), }, }, ] @@ -143,7 +165,7 @@ def balancer_fingerprint(tokens, start, end, fees): def run_scenario(scenario): """Run a reClAMM config and its Balancer 50/50 baseline, print comparison.""" rc = scenario["reclamm"] - fp = rc["fingerprint"] + fp, noise_summary = _apply_active_noise_settings(dict(rc["fingerprint"])) # Run reClAMM reclamm_result = do_run_on_historic_data( @@ -173,7 +195,14 @@ def run_scenario(scenario): print("=" * 80) print(f" {scenario['name']}") - print(f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']}") + print( + f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']} | " + f"Interpolation: {fp.get('reclamm_interpolation_method', 'geometric')}" + ) + print( + f" Noise: {noise_summary} | Gas: {fp.get('gas_cost', 0.0)} | " + f"Protocol fee split: {fp.get('protocol_fee_split', 0.0)}" + ) print("-" * 80) print(f" {'':30s} {'reClAMM':>14s} {'Balancer 50/50':>14s}") print(f" {'Initial value':30s} ${rc_init:>13,.0f} ${bal_init:>13,.0f}") diff --git a/scripts/reclamm/compare_reclamm_thermostats.py b/scripts/reclamm/compare_reclamm_thermostats.py index 8a2c374c..be8ee5f0 100644 --- a/scripts/reclamm/compare_reclamm_thermostats.py +++ b/scripts/reclamm/compare_reclamm_thermostats.py @@ -1,19 +1,39 @@ -"""Compare geometric vs constant-arc-length thermostats on historic data. +"""Compare reCLAMM interpolation modes on historic AAVE/ETH data. -Runs AAVE/ETH reClAMM pool simulations with both interpolation methods. -Plots: pool value, cumulative LVR, price path, empirical weights, -value difference, LVR ratio, and per-step LVR distribution (∝ Δs²). +Runs the production geometric interpolation against the non-linear +constant-arc-length interpolation on: +1. The original launch-style range (price_ratio ~= 1.50) +2. A much tighter range (price_ratio = 1.10) -Usage: - cd - source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim-reclamm - python scripts/compare_reclamm_thermostats.py +The aggressive case is deliberate. A local AAVE/ETH sweep showed: +price_ratio 1.15, margin 0.5, shift 0.1 -> about +$10k vs geometric +price_ratio 1.10, margin 0.5, shift 0.1 -> about +$31k vs geometric +price_ratio 1.10, margin 0.6, shift 0.1 -> about +$73k vs geometric + +So the strongest clean demo setting came from tightening the band and +slightly raising the trigger margin, while keeping the launch-style shift +speed rather than pushing shift_exponent higher. """ +import gc +import math +import os + import jax.numpy as jnp import numpy as np +import pandas as pd import matplotlib.pyplot as plt +from matplotlib.colors import Normalize, SymLogNorm, TwoSlopeNorm +from matplotlib.cm import ScalarMappable +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + calibrate_arc_length_speed, + compute_price_ratio, + initialise_reclamm_reserves, +) from quantammsim.runners.jax_runners import do_run_on_historic_data +from quantammsim.utils.data_processing.historic_data_utils import ( + get_historic_parquet_data, +) def to_daily_price_shift_base(daily_price_shift_exponent): @@ -21,57 +41,432 @@ def to_daily_price_shift_base(daily_price_shift_exponent): return 1.0 - daily_price_shift_exponent / 124649.0 +RUN_CONSTANT_ARC_LENGTH = False +INTERPOLATION_METHODS = ( + ("geometric", "constant_arc_length") + if RUN_CONSTANT_ARC_LENGTH + else ("geometric",) +) +HEATMAP_PRICE_RATIOS = np.arange(1.01, 1.50 + 1e-9, 0.025) +HEATMAP_MARGINS = np.linspace(0.05, 0.90, 20) +HEATMAP_SHIFT_EXPONENTS = np.arange(0.01, 0.50 + 1e-9, 0.025) +HEATMAP_ARC_LENGTH_SPEEDS = np.geomspace(1.0e-6, 5.0e-4, 11) +PRICE_RATIO_TICKS = np.array([1.01, 1.10, 1.20, 1.30, 1.40, 1.50]) +MARGIN_TICKS = np.array([0.05, 0.15, 0.25, 0.35, 0.45, 0.55, 0.65, 0.75, 0.85, 0.90]) +SHIFT_EXPONENT_TICKS = np.array([0.01, 0.05, 0.10, 0.20, 0.30, 0.40, 0.50]) +ARC_LENGTH_SPEED_TICKS = np.array([ + 1.0e-6, + 2.0e-6, + 5.0e-6, + 1.0e-5, + 2.0e-5, + 5.0e-5, + 1.0e-4, + 2.0e-4, + 5.0e-4, +]) +SWEEP_LINE_WIDTH = 0.45 +REFERENCE_LINE_WIDTH = 0.9 +DEFAULT_INITIAL_POOL_VALUE = 1_000_000.0 +TVL_SWEEP_VALUES = ( + 1_000_000.0, + 5_000_000.0, + 20_000_000.0, +) +CENTER_ZERO_HEATMAP_COLOR_NORM = "symlog" +CENTER_ZERO_HEATMAP_COLOR_TAG = "symlog20" +CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH = 20.0 + +AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" +DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" +DEFAULT_NOISE_MODEL = "market_linear" +DEFAULT_GAS_COST = 1.0 +DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 +LEGACY_NOISE_COEFFS = [ + -0.453, + 0.025, + -0.060, + 0.310, + -0.149, + 0.359, + 0.061, + 0.060, +] +LEGACY_LOG_CADENCE = 2.68 +LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) +AAVE_ETH_NOISE_SETTINGS = { + "enable_noise_model": True, + "noise_model": DEFAULT_NOISE_MODEL, + "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, + "noise_pool_id": AAVE_WETH_POOL_ID, + "gas_cost": DEFAULT_GAS_COST, + "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, +} + +GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS = ( + "geometric_vs_launch_geometric_pct", + "noise_geometric_final_value_musd", + "noise_vs_arb_geometric_improvement_pct", +) +CONSTANT_ARC_HEATMAP_METRIC_KEYS = ( + "efficiency_pct", + "launch_geometric_efficiency_pct", + "constant_arc_vs_launch_constant_arc_pct", + "noise_constant_arc_final_value_musd", + "noise_vs_arb_constant_arc_improvement_pct", +) +HEATMAP_METRIC_DEPENDENCIES = { + "efficiency_pct": ("noise_geometric", "noise_constant_arc"), + "launch_geometric_efficiency_pct": ("noise_constant_arc",), + "geometric_vs_launch_geometric_pct": ("noise_geometric",), + "constant_arc_vs_launch_constant_arc_pct": ("noise_constant_arc",), + "noise_geometric_final_value_musd": ("noise_geometric",), + "noise_constant_arc_final_value_musd": ("noise_constant_arc",), + "noise_vs_arb_geometric_improvement_pct": ("noise_geometric", "arb_geometric"), + "noise_vs_arb_constant_arc_improvement_pct": ( + "noise_constant_arc", + "arb_constant_arc", + ), +} + +_NOISE_SETTINGS_CACHE = {} +_WARNED_NOISE_FALLBACKS = set() + + +def get_initial_pool_value(cfg): + """Return the configured base pool TVL in USD.""" + return float(cfg.get("initial_pool_value", DEFAULT_INITIAL_POOL_VALUE)) + + +def get_tvl_millions(cfg): + """Return the configured base pool TVL in millions of USD.""" + return get_initial_pool_value(cfg) / 1_000_000.0 + + +def format_tvl_millions_slug(cfg): + """Format the TVL in millions for stable filenames.""" + tvl_millions = get_tvl_millions(cfg) + rounded = round(float(tvl_millions), 6) + if np.isclose(rounded, round(rounded)): + return f"{int(round(rounded))}m" + return f"{rounded:.6f}".rstrip("0").rstrip(".").replace(".", "p") + "m" + + +def format_tvl_millions_label(cfg): + """Format the TVL in millions for plot titles and logs.""" + return f"{get_tvl_millions(cfg):.1f}M" + + +def tvl_artifact_filename(stem, cfg, suffix=None): + """Append a TVL-in-millions suffix to a PNG artifact name.""" + parts = [stem] + if suffix: + parts.append(suffix) + parts.append(f"tvl_{format_tvl_millions_slug(cfg)}") + return "_".join(parts) + ".png" + + +def heatmap_artifact_filename(spec, cfg, suffix=None): + """Build a heatmap filename, including any colour-style tag.""" + stem = f"reclamm_heatmap_{spec['slug']}" + artifact_tag = spec.get("artifact_tag") + if artifact_tag: + stem = f"{stem}_{artifact_tag}" + return tvl_artifact_filename(stem, cfg, suffix=suffix) + + +def configs_for_tvl(base_configs, initial_pool_value): + """Attach a shared initial TVL to each compare configuration.""" + configs = [] + for cfg in base_configs: + updated = dict(cfg) + updated["initial_pool_value"] = float(initial_pool_value) + configs.append(updated) + return configs + + +def make_noise_variant_cfg(cfg, enable_noise_model): + """Return a config with either noise modelling or pure arb-only enabled.""" + updated = dict(cfg) + if enable_noise_model: + updated["enable_noise_model"] = True + return updated + + matched_noise = resolve_reclamm_noise_settings(cfg) + + updated["enable_noise_model"] = False + updated["noise_model"] = None + updated["gas_cost"] = cfg.get("gas_cost", DEFAULT_GAS_COST) + updated["protocol_fee_split"] = cfg.get( + "protocol_fee_split", DEFAULT_PROTOCOL_FEE_SPLIT + ) + updated["noise_trader_ratio"] = 0.0 + matched_arb_frequency = matched_noise.get("arb_frequency") + if matched_arb_frequency is not None: + updated["arb_frequency"] = matched_arb_frequency + for key in ( + "reclamm_noise_params", + "noise_arrays_path", + "noise_artifact_dir", + "noise_pool_id", + ): + updated.pop(key, None) + return updated + + +def _warn_noise_fallback(message): + """Print a one-time message when the preferred noise setup is unavailable.""" + if message not in _WARNED_NOISE_FALLBACKS: + print(message) + _WARNED_NOISE_FALLBACKS.add(message) + + +def _hashable_noise_params(params): + """Convert a noise-params dict into a stable cache key fragment.""" + if params is None: + return None + return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) + + +def _legacy_calibrated_noise_settings(reason=None): + """Fallback calibrated noise config used when market-linear artifacts are absent.""" + if reason: + _warn_noise_fallback( + "market_linear noise unavailable for thermostat comparison; " + f"falling back to calibrated legacy coefficients ({reason})." + ) + return { + "noise_model": "calibrated", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) + }, + "arb_frequency": LEGACY_ARB_FREQUENCY, + "noise_summary": ( + "calibrated legacy 8-covariate " + f"(arb_frequency={LEGACY_ARB_FREQUENCY})" + ), + "noise_cache_key": ( + "calibrated", + tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), + LEGACY_ARB_FREQUENCY, + ), + } + + +def resolve_reclamm_noise_settings(cfg): + """Resolve the active reCLAMM noise-model fingerprint block for a config.""" + enable_noise_model = cfg.get("enable_noise_model", False) + requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + cache_key = ( + tuple(cfg.get("tokens", [])), + cfg.get("start"), + cfg.get("end"), + enable_noise_model, + requested_mode, + cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), + cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), + cfg.get("arb_frequency"), + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + _hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + ) + if cache_key in _NOISE_SETTINGS_CACHE: + return _NOISE_SETTINGS_CACHE[cache_key] + + if not enable_noise_model: + result = { + "noise_model": None, + "noise_trader_ratio": 0.0, + "reclamm_noise_params": None, + "noise_arrays_path": None, + "arb_frequency": None, + "noise_summary": "arb-only (noise disabled)", + "noise_cache_key": ("disabled",), + } + elif requested_mode == "market_linear": + artifact_dir = cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR) + pool_id = cfg.get("noise_pool_id", AAVE_WETH_POOL_ID) + start_date = str(cfg["start"]).split(" ")[0] + end_date = str(cfg["end"]).split(" ")[0] + try: + from quantammsim.calibration.noise_model_arrays import ( + _find_pool_index, + build_simulator_arrays, + load_artifact, + ) + + model_path = os.path.join(artifact_dir, "model.npz") + meta_path = os.path.join(artifact_dir, "meta.json") + if not (os.path.exists(model_path) and os.path.exists(meta_path)): + raise FileNotFoundError( + f"expected {model_path} and {meta_path}" + ) + + cache_dir = os.path.join(artifact_dir, "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join( + cache_dir, + f"{pool_id}_{start_date}_{end_date}.npz", + ) + if not os.path.exists(arrays_path): + arrays = build_simulator_arrays( + pool_id=pool_id, + start_date=start_date, + end_date=end_date, + artifact_dir=artifact_dir, + ) + np.savez( + arrays_path, + noise_base=arrays["noise_base"], + noise_tvl_coeff=arrays["noise_tvl_coeff"], + tvl_mean=arrays["tvl_mean"], + tvl_std=arrays["tvl_std"], + ) + + with np.load(arrays_path) as arrays: + tvl_mean = float(arrays["tvl_mean"]) + tvl_std = float(arrays["tvl_std"]) + + art, meta = load_artifact(artifact_dir) + pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) + if pool_idx >= 0: + learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) + else: + learned_cadence = 5.0 + arb_frequency = max(1, round(learned_cadence)) + result = { + "noise_model": "market_linear", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, + }, + "noise_arrays_path": arrays_path, + "arb_frequency": arb_frequency, + "noise_summary": f"market_linear (arb_frequency={arb_frequency})", + "noise_cache_key": ( + "market_linear", + arrays_path, + arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), + ), + } + except Exception as exc: # pragma: no cover - fallback path depends on local artifacts + result = _legacy_calibrated_noise_settings(str(exc)) + elif requested_mode == "calibrated": + params = cfg.get("reclamm_noise_params") + if params is None: + result = _legacy_calibrated_noise_settings() + else: + arb_frequency = cfg.get("arb_frequency", LEGACY_ARB_FREQUENCY) + result = { + "noise_model": "calibrated", + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": dict(params), + "arb_frequency": arb_frequency, + "noise_summary": f"calibrated (arb_frequency={arb_frequency})", + "noise_cache_key": ( + "calibrated", + _hashable_noise_params(params), + arb_frequency, + ), + } + else: + arb_frequency = cfg.get("arb_frequency") + result = { + "noise_model": requested_mode, + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": cfg.get("reclamm_noise_params"), + "noise_arrays_path": cfg.get("noise_arrays_path"), + "arb_frequency": arb_frequency, + "noise_summary": f"{requested_mode} (arb_frequency={arb_frequency})", + "noise_cache_key": ( + requested_mode, + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + _hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + arb_frequency, + ), + } + + _NOISE_SETTINGS_CACHE[cache_key] = result + return result + + # Pool configurations to compare CONFIGS = [ { - "name": "AAVE/ETH on-chain (25bps, narrow range)", + "name": "AAVE/ETH launch-style range (25bps, reference)", "tokens": ["AAVE", "ETH"], "start": "2024-06-01 00:00:00", "end": "2025-06-01 00:00:00", "fees": 0.0025, - "price_ratio": 1.5, + "price_ratio": 1.5014, "centeredness_margin": 0.5, "daily_price_shift_exponent": 0.1, + "reason": "Original launch-style parameters.", + **AAVE_ETH_NOISE_SETTINGS, }, { - "name": "AAVE/ETH wide range (25bps)", + "name": "AAVE/ETH aggressive tight range (25bps)", "tokens": ["AAVE", "ETH"], "start": "2024-06-01 00:00:00", "end": "2025-06-01 00:00:00", "fees": 0.0025, - "price_ratio": 4.0, - "centeredness_margin": 0.2, - "daily_price_shift_exponent": 1.0, - }, - { - "name": "AAVE/ETH zero fees (narrow)", - "tokens": ["AAVE", "ETH"], - "start": "2024-06-01 00:00:00", - "end": "2025-06-01 00:00:00", - "fees": 0.0, - "price_ratio": 1.5, - "centeredness_margin": 0.5, + "price_ratio": 1.10, + "centeredness_margin": 0.60, "daily_price_shift_exponent": 0.1, + "reason": ( + "Aggressively tightened and moved to an earlier thermostat trigger. " + "At fixed price_ratio=1.10, the shift_exponent sweep still favored " + "0.1, while margin=0.60 widened the non-linear edge materially." + ), + **AAVE_ETH_NOISE_SETTINGS, }, ] -def make_fingerprint(cfg, interpolation_method, centeredness_scaling=False): +def make_fingerprint(cfg, interpolation_method): """Build run fingerprint for a given config and interpolation method.""" - return { + speed_override = ( + cfg.get("arc_length_speed") + if interpolation_method == "constant_arc_length" + else None + ) + noise_cfg = resolve_reclamm_noise_settings(cfg) + arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + fingerprint = { "tokens": cfg["tokens"], "rule": "reclamm", "startDateString": cfg["start"], "endDateString": cfg["end"], - "initial_pool_value": 1000000.0, + "initial_pool_value": get_initial_pool_value(cfg), "do_arb": True, "fees": cfg["fees"], - "gas_cost": 0.0, - "arb_fees": 0.0, + "gas_cost": cfg.get( + "gas_cost", + DEFAULT_GAS_COST if cfg.get("enable_noise_model", False) else 0.0, + ), + "arb_fees": cfg.get("arb_fees", 0.0), + "protocol_fee_split": cfg.get( + "protocol_fee_split", + DEFAULT_PROTOCOL_FEE_SPLIT if cfg.get("enable_noise_model", False) else 0.0, + ), + "noise_trader_ratio": noise_cfg.get("noise_trader_ratio", 0.0), "reclamm_interpolation_method": interpolation_method, - "reclamm_arc_length_speed": None, # auto-calibrate - "reclamm_centeredness_scaling": centeredness_scaling, + "reclamm_arc_length_speed": speed_override, } + if noise_cfg.get("noise_model") is not None: + fingerprint["noise_model"] = noise_cfg["noise_model"] + if noise_cfg.get("reclamm_noise_params") is not None: + fingerprint["reclamm_noise_params"] = noise_cfg["reclamm_noise_params"] + if noise_cfg.get("noise_arrays_path") is not None: + fingerprint["noise_arrays_path"] = noise_cfg["noise_arrays_path"] + if arb_frequency is not None: + fingerprint["arb_frequency"] = arb_frequency + return fingerprint def make_params(cfg): @@ -85,40 +480,1129 @@ def make_params(cfg): } -def run_comparison(cfg): - """Run all thermostat variants, return results dict.""" +def load_shared_price_data(configs, root=None): + """Load the shared historic price panel once for all compare runs.""" + tokens = sorted({token for cfg in configs for token in cfg["tokens"]}) + return get_historic_parquet_data(tokens, cols=["close"], root=root) + + +def run_comparison(cfg, price_data=None, low_data_mode=False): + """Run both interpolation variants, return results dict.""" params = make_params(cfg) results = {} - for method in ["geometric", "constant_arc_length"]: + for method in INTERPOLATION_METHODS: fp = make_fingerprint(cfg, method) results[method] = do_run_on_historic_data( - run_fingerprint=fp, params=params + run_fingerprint=fp, + params=params, + price_data=price_data, + low_data_mode=low_data_mode, + ) + + return results + + +def _set_padded_ylim(ax, series_list, pad_ratio=0.04): + """Fit the y-axis tightly around the plotted series.""" + flat = [ + np.asarray(series, dtype=float).ravel() + for series in series_list + if np.asarray(series).size > 0 + ] + if not flat: + return + + values = np.concatenate(flat) + values = values[np.isfinite(values)] + if values.size == 0: + return + + ymin = float(values.min()) + ymax = float(values.max()) + if np.isclose(ymin, ymax): + pad = max(abs(ymin) * pad_ratio, 1e-6) + else: + pad = (ymax - ymin) * pad_ratio + ax.set_ylim(ymin - pad, ymax + pad) + + +def _cache_size(cache): + """Count memoized final-value runs.""" + return len(cache.get("_final_value_cache", {})) + + +def _comparison_cache_size(cache): + """Count memoized scalar comparison bundles.""" + return len(cache.get("_comparison_cache", {})) + + +def make_sweep_cache(price_data): + """Create a shared cache for heatmap and line sweeps.""" + return { + "_shared_price_data": price_data, + "_final_value_cache": {}, + "_comparison_cache": {}, + } + + +def _missing_artifacts(progress_label, filenames): + """Report which plot artifacts still need to be generated.""" + missing = [filename for filename in filenames if not os.path.exists(filename)] + if not missing: + print(f"[{progress_label}] skipping sweep: all artifacts already exist.") + return set() + + existing_count = len(filenames) - len(missing) + if existing_count: + print( + f"[{progress_label}] reusing {existing_count}/{len(filenames)} " + "existing artifacts; generating the missing outputs." ) + return set(missing) + - # Geometric + centeredness-proportional scaling (scales decay duration) - fp_geo_scaled = make_fingerprint(cfg, "geometric", centeredness_scaling=True) - results["geometric_scaled"] = do_run_on_historic_data( - run_fingerprint=fp_geo_scaled, params=params +def _speed_cache_key(speed): + """Stable cache token for optional arc-length speed.""" + if speed is None: + return None + return round(float(speed), 12) + + +def _make_method_cache_key(cfg, method): + """Cache key for a single-method final-value run.""" + noise_cfg = resolve_reclamm_noise_settings(cfg) + arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + key = ( + method, + bool(cfg.get("enable_noise_model", False)), + round(float(cfg["price_ratio"]), 6), + round(float(cfg["centeredness_margin"]), 6), + round(float(cfg["daily_price_shift_exponent"]), 6), + round(get_initial_pool_value(cfg), 2), + noise_cfg.get("noise_cache_key"), + None if arb_frequency is None else int(arb_frequency), + round( + float( + cfg.get( + "gas_cost", + DEFAULT_GAS_COST if cfg.get("enable_noise_model", False) else 0.0, + ) + ), + 6, + ), + round( + float( + cfg.get( + "protocol_fee_split", + DEFAULT_PROTOCOL_FEE_SPLIT if cfg.get("enable_noise_model", False) else 0.0, + ) + ), + 6, + ), ) + if method == "constant_arc_length": + key += (_speed_cache_key(cfg.get("arc_length_speed")),) + return key - # Arc-length + centeredness-proportional scaling (scales speed) - fp_cal_scaled = make_fingerprint(cfg, "constant_arc_length", centeredness_scaling=True) - results["cal_scaled"] = do_run_on_historic_data( - run_fingerprint=fp_cal_scaled, params=params + +def _make_comparison_cache_key(cfg, launch_final_values): + """Cache key for scalar heatmap metrics at a single parameter point.""" + noise_cfg = make_noise_variant_cfg(cfg, True) + arb_only_cfg = make_noise_variant_cfg(cfg, False) + key = [ + _make_method_cache_key(noise_cfg, "geometric"), + _make_method_cache_key(arb_only_cfg, "geometric"), + round(float(launch_final_values["geometric"]), 6), + ] + if RUN_CONSTANT_ARC_LENGTH: + key.extend( + [ + _make_method_cache_key(noise_cfg, "constant_arc_length"), + _make_method_cache_key(arb_only_cfg, "constant_arc_length"), + round(float(launch_final_values["constant_arc_length"]), 6), + ] + ) + return tuple(key) + + +def _run_method_final_value_cached(cfg, method, cache): + """Memoize final value for a single interpolation method.""" + final_value_cache = cache.setdefault("_final_value_cache", {}) + key = _make_method_cache_key(cfg, method) + if key not in final_value_cache: + result = do_run_on_historic_data( + run_fingerprint=make_fingerprint(cfg, method), + params=make_params(cfg), + price_data=cache["_shared_price_data"], + low_data_mode=True, + ) + final_value_cache[key] = float(result["final_value"]) + del result + gc.collect() + return final_value_cache[key] + + +def extract_comparison_metrics_from_final_values( + geo_final, arc_final, launch_final_values +): + """Summarize scalar comparison metrics from final values only.""" + return { + "efficiency_pct": (arc_final / max(abs(geo_final), 1e-12) - 1.0) * 100.0, + "launch_geometric_efficiency_pct": ( + arc_final / max(abs(launch_final_values["geometric"]), 1e-12) - 1.0 + ) + * 100.0, + "geometric_vs_launch_geometric_pct": ( + geo_final / max(abs(launch_final_values["geometric"]), 1e-12) - 1.0 + ) + * 100.0, + "constant_arc_vs_launch_constant_arc_pct": ( + arc_final + / max(abs(launch_final_values["constant_arc_length"]), 1e-12) + - 1.0 + ) + * 100.0, + } + + +def _load_required_heatmap_final_values(cfg, cache, metric_keys): + """Load only the cached final values needed for the requested heatmap metrics.""" + required_sources = set() + for metric_key in metric_keys: + required_sources.update(HEATMAP_METRIC_DEPENDENCIES[metric_key]) + + if not RUN_CONSTANT_ARC_LENGTH and any( + source.endswith("constant_arc") for source in required_sources + ): + raise ValueError( + "Constant-arc heatmap metric requested while RUN_CONSTANT_ARC_LENGTH=False" + ) + + final_values = {} + noise_cfg = None + arb_only_cfg = None + + if any(source.startswith("noise_") for source in required_sources): + noise_cfg = make_noise_variant_cfg(cfg, True) + if any(source.startswith("arb_") for source in required_sources): + arb_only_cfg = make_noise_variant_cfg(cfg, False) + + if "noise_geometric" in required_sources: + final_values["noise_geometric"] = _run_method_final_value_cached( + noise_cfg, + "geometric", + cache, + ) + if "noise_constant_arc" in required_sources: + final_values["noise_constant_arc"] = _run_method_final_value_cached( + noise_cfg, + "constant_arc_length", + cache, + ) + if "arb_geometric" in required_sources: + final_values["arb_geometric"] = _run_method_final_value_cached( + arb_only_cfg, + "geometric", + cache, + ) + if "arb_constant_arc" in required_sources: + final_values["arb_constant_arc"] = _run_method_final_value_cached( + arb_only_cfg, + "constant_arc_length", + cache, + ) + return final_values + + +def extract_heatmap_metrics_from_mode_final_values( + metric_keys, + final_values, + launch_final_values, +): + """Collect the requested scalar heatmap metrics from cached final values.""" + metrics = {} + + if "efficiency_pct" in metric_keys: + metrics["efficiency_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(final_values["noise_geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "launch_geometric_efficiency_pct" in metric_keys: + metrics["launch_geometric_efficiency_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(launch_final_values["geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "geometric_vs_launch_geometric_pct" in metric_keys: + metrics["geometric_vs_launch_geometric_pct"] = ( + final_values["noise_geometric"] + / max(abs(launch_final_values["geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "constant_arc_vs_launch_constant_arc_pct" in metric_keys: + metrics["constant_arc_vs_launch_constant_arc_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(launch_final_values["constant_arc_length"]), 1e-12) + - 1.0 + ) * 100.0 + + if "noise_geometric_final_value_musd" in metric_keys: + metrics["noise_geometric_final_value_musd"] = ( + final_values["noise_geometric"] / 1e6 + ) + + if "noise_constant_arc_final_value_musd" in metric_keys: + metrics["noise_constant_arc_final_value_musd"] = ( + final_values["noise_constant_arc"] / 1e6 + ) + + if "noise_vs_arb_geometric_improvement_pct" in metric_keys: + metrics["noise_vs_arb_geometric_improvement_pct"] = ( + final_values["noise_geometric"] + / max(abs(final_values["arb_geometric"]), 1e-12) + - 1.0 + ) * 100.0 + + if "noise_vs_arb_constant_arc_improvement_pct" in metric_keys: + metrics["noise_vs_arb_constant_arc_improvement_pct"] = ( + final_values["noise_constant_arc"] + / max(abs(final_values["arb_constant_arc"]), 1e-12) + - 1.0 + ) * 100.0 + + return metrics + + +def extract_comparison_metrics(results, launch_final_values): + """Summarize scalar heatmap metrics for a pair of runs.""" + geo = results["geometric"] + arc = results["constant_arc_length"] + + geo_final = float(geo["final_value"]) + arc_final = float(arc["final_value"]) + + return extract_comparison_metrics_from_final_values( + geo_final, + arc_final, + launch_final_values=launch_final_values, ) - return results + +def run_comparison_cached(cfg, cache, launch_final_values, metric_keys): + """Memoize scalar heatmap metrics across heatmap sweeps.""" + requested_metric_keys = tuple(dict.fromkeys(metric_keys)) + comparison_cache = cache.setdefault("_comparison_cache", {}) + cache_key = _make_comparison_cache_key(cfg, launch_final_values) + cached_metrics = comparison_cache.setdefault(cache_key, {}) + missing_metric_keys = [ + metric_key for metric_key in requested_metric_keys if metric_key not in cached_metrics + ] + if missing_metric_keys: + final_values = _load_required_heatmap_final_values( + cfg, + cache, + missing_metric_keys, + ) + cached_metrics.update( + extract_heatmap_metrics_from_mode_final_values( + missing_metric_keys, + final_values, + launch_final_values=launch_final_values, + ) + ) + return { + metric_key: cached_metrics[metric_key] for metric_key in requested_metric_keys + } + + +def build_heatmap_matrices( + x_values, + y_values, + x_key, + y_key, + base_cfg, + metric_keys, + cache, + progress_label, + launch_final_values, +): + """Evaluate multiple metrics over a 2D parameter grid in one pass.""" + data = { + metric_key: np.zeros((len(y_values), len(x_values)), dtype=float) + for metric_key in metric_keys + } + total_points = len(y_values) * len(x_values) + + print( + f"[{progress_label}] start: {len(y_values)} rows x {len(x_values)} cols " + f"= {total_points} parameter points" + ) + + for yi, y_value in enumerate(y_values): + final_cache_before_row = _cache_size(cache) + comparison_cache_before_row = _comparison_cache_size(cache) + for xi, x_value in enumerate(x_values): + cfg = dict(base_cfg) + cfg[x_key] = float(x_value) + cfg[y_key] = float(y_value) + metrics = run_comparison_cached( + cfg, + cache, + launch_final_values=launch_final_values, + metric_keys=metric_keys, + ) + for metric_key in metric_keys: + data[metric_key][yi, xi] = metrics[metric_key] + + completed_points = (yi + 1) * len(x_values) + row_new_final_runs = _cache_size(cache) - final_cache_before_row + row_new_comparisons = ( + _comparison_cache_size(cache) - comparison_cache_before_row + ) + row_pct = completed_points / total_points * 100.0 + print( + f"[{progress_label}] row {yi + 1}/{len(y_values)} complete " + f"({y_key}={float(y_value):.4f}, {completed_points}/{total_points} " + f"points, {row_pct:.1f}%, {row_new_final_runs} new final-value runs, " + f"{row_new_comparisons} new comparison bundles)" + ) + + print( + f"[{progress_label}] done: " + + ", ".join( + ( + f"{metric_key} min={float(np.nanmin(data[metric_key])):.4f}, " + f"max={float(np.nanmax(data[metric_key])):.4f}" + ) + for metric_key in metric_keys + ) + + ( + f", final_value_cache_size={_cache_size(cache)}, " + f"comparison_cache_size={_comparison_cache_size(cache)}" + ) + ) + + return data + + +def build_metric_curve( + x_values, + x_key, + base_cfg, + metric_key, + cache, + launch_final_values, +): + """Evaluate one metric over a 1D sweep.""" + data = np.zeros(len(x_values), dtype=float) + for xi, x_value in enumerate(x_values): + cfg = dict(base_cfg) + cfg[x_key] = float(x_value) + metrics = run_comparison_cached( + cfg, + cache, + launch_final_values=launch_final_values, + metric_keys=(metric_key,), + ) + data[xi] = metrics[metric_key] + return data + + +def _compute_axis_edges(values, scale="linear"): + """Convert axis centers to cell edges for pcolormesh.""" + values = np.asarray(values, dtype=float) + if values.size == 1: + if scale == "log": + return np.array([values[0] / np.sqrt(10.0), values[0] * np.sqrt(10.0)]) + pad = max(abs(values[0]) * 0.5, 1.0) + return np.array([values[0] - pad, values[0] + pad]) + + if scale == "log": + log_values = np.log10(values) + edges = np.empty(values.size + 1, dtype=float) + edges[1:-1] = 0.5 * (log_values[:-1] + log_values[1:]) + edges[0] = log_values[0] - 0.5 * (log_values[1] - log_values[0]) + edges[-1] = log_values[-1] + 0.5 * (log_values[-1] - log_values[-2]) + return 10.0 ** edges + + edges = np.empty(values.size + 1, dtype=float) + edges[1:-1] = 0.5 * (values[:-1] + values[1:]) + edges[0] = values[0] - 0.5 * (values[1] - values[0]) + edges[-1] = values[-1] + 0.5 * (values[-1] - values[-2]) + return edges + + +def plot_heatmap( + data, + x_values, + y_values, + x_label, + y_label, + title, + colorbar_label, + filename, + xticks=None, + yticks=None, + xscale="linear", + center_zero=True, + cmap=None, + color_norm=None, + symlog_linthresh=None, +): + """Render and save a single heatmap.""" + finite = np.asarray(data, dtype=float) + finite = finite[np.isfinite(finite)] + + if center_zero: + vmax = max(abs(float(np.nanmin(data))), abs(float(np.nanmax(data))), 1e-9) + if color_norm == "symlog" and symlog_linthresh is not None and vmax > symlog_linthresh: + norm = SymLogNorm( + linthresh=symlog_linthresh, + linscale=1.0, + vmin=-vmax, + vmax=vmax, + base=10.0, + ) + else: + norm = TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) + cmap_name = cmap or "RdYlGn" + else: + if finite.size == 0: + vmin, vmax = 0.0, 1.0 + else: + vmin = float(finite.min()) + vmax = float(finite.max()) + if np.isclose(vmin, vmax): + pad = max(abs(vmin) * 0.01, 1e-9) + vmin -= pad + vmax += pad + norm = Normalize(vmin=vmin, vmax=vmax) + cmap_name = cmap or "viridis" + + x_edges = _compute_axis_edges(x_values, scale=xscale) + y_edges = _compute_axis_edges(y_values, scale="linear") + + fig, ax = plt.subplots(figsize=(8.5, 6.0)) + im = ax.pcolormesh( + x_edges, + y_edges, + data, + cmap=cmap_name, + norm=norm, + shading="auto", + ) + + ax.set_xlabel(x_label) + ax.set_ylabel(y_label) + ax.set_title(title) + if xscale == "log": + ax.set_xscale("log") + ax.set_xticks(np.asarray(xticks if xticks is not None else x_values, dtype=float)) + ax.set_yticks(np.asarray(yticks if yticks is not None else y_values, dtype=float)) + ax.grid(False) + + cbar = fig.colorbar(im, ax=ax) + cbar.set_label(colorbar_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + +def plot_arc_speed_line_chart( + data, + x_values, + y_values, + y_label, + title, + filename, + launch_curve, + launch_auto_speed=None, +): + """Plot thin multi-series efficiency lines over the arc-speed sweep.""" + fig, ax = plt.subplots(figsize=(10.5, 5.75)) + cmap = plt.cm.viridis + colors = cmap(np.linspace(0.0, 1.0, len(y_values))) + plotted_series = [] + + for yi, (y_value, color) in enumerate(zip(y_values, colors)): + series = np.asarray(data[yi], dtype=float) + plotted_series.append(series) + ax.plot( + x_values, + series, + color=color, + linewidth=SWEEP_LINE_WIDTH, + alpha=0.8, + ) + + launch_curve = np.asarray(launch_curve, dtype=float) + plotted_series.append(launch_curve) + ax.plot( + x_values, + launch_curve, + color="black", + linewidth=REFERENCE_LINE_WIDTH, + alpha=0.9, + label="Current launch config", + ) + if launch_auto_speed is not None: + ax.axvline( + float(launch_auto_speed), + color="black", + ls=":", + linewidth=0.8, + alpha=0.7, + label="Launch auto-cal speed", + ) + + ax.axhline(0.0, color="gray", ls="--", linewidth=0.8, alpha=0.5) + ax.set_xscale("log") + ax.set_xticks(ARC_LENGTH_SPEED_TICKS) + ax.set_xlabel("Arc-length speed") + ax.set_ylabel("Efficiency vs geometric (%)") + ax.set_title(title) + _set_padded_ylim(ax, plotted_series, pad_ratio=0.08) + ax.grid(True, alpha=0.25) + ax.legend(fontsize=8) + + sm = ScalarMappable( + norm=Normalize(vmin=float(np.min(y_values)), vmax=float(np.max(y_values))), + cmap=cmap, + ) + sm.set_array([]) + cbar = fig.colorbar(sm, ax=ax) + cbar.set_label(y_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + +def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): + """Generate pairwise heatmaps for thermostat tuning and noise-vs-arb effects.""" + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data) + metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs heatmap geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "launch_geometric_efficiency_pct", + "title": "Efficiency vs launch-style geometric", + "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", + "slug": "launch_geometric_efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "geometric_vs_launch_geometric_pct", + "title": "Geometric tuning vs launch-style geometric", + "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", + "slug": "geometric_vs_launch_geometric", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "constant_arc_vs_launch_constant_arc_pct", + "title": "Const arc tuning vs launch-style const arc", + "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", + "slug": "constant_arc_vs_launch_constant_arc", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_geometric_final_value_musd", + "title": "Geometric final value with noise model", + "colorbar_label": "Geometric final value with noise model ($M)", + "slug": "noise_geometric_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_geometric_improvement_pct", + "title": "Noise-model improvement over arb-only (geometric)", + "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", + "slug": "noise_vs_arb_geometric_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + if not RUN_CONSTANT_ARC_LENGTH: + metric_specs = [ + spec + for spec in metric_specs + if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS + ] + pair_specs = [ + { + "slug": "price_ratio_vs_margin", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_MARGINS, + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "x_label": "Price ratio", + "y_label": "Centeredness margin", + "title_suffix": ( + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": PRICE_RATIO_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "shift_exp_vs_margin", + "x_values": HEATMAP_SHIFT_EXPONENTS, + "y_values": HEATMAP_MARGINS, + "x_key": "daily_price_shift_exponent", + "y_key": "centeredness_margin", + "x_label": "Shift exponent", + "y_label": "Centeredness margin", + "title_suffix": f"price_ratio fixed at {base_cfg['price_ratio']:.2f}", + "xticks": SHIFT_EXPONENT_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "price_ratio_vs_shift_exp", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "price_ratio", + "y_key": "daily_price_shift_exponent", + "x_label": "Price ratio", + "y_label": "Shift exponent", + "title_suffix": ( + f"margin fixed at {base_cfg['centeredness_margin']:.2f}" + ), + "xticks": PRICE_RATIO_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + }, + ] + + metric_spec_map = {spec["key"]: spec for spec in metric_specs} + + if RUN_CONSTANT_ARC_LENGTH: + print( + "Using launch-style benchmarks " + f"Geo=${launch_final_values['geometric']:,.0f}, " + f"Const Arc=${launch_final_values['constant_arc_length']:,.0f}, " + f"TVL={format_tvl_millions_label(base_cfg)}." + ) + print( + "Running {count} heatmap pair sweeps sequentially " + "(current outputs use cached noise-model runs; improvement heatmaps " + "reuse those values and add cached arb-only runs).".format( + count=len(pair_specs) + ) + ) + else: + print( + "Using launch-style geometric benchmark " + f"Geo=${launch_final_values['geometric']:,.0f}, " + f"TVL={format_tvl_millions_label(base_cfg)}." + ) + print( + "RUN_CONSTANT_ARC_LENGTH=False, so only geometric heatmaps will be generated " + "and only geometric/arb-only geometric runs will be scheduled." + ) + + for pair in pair_specs: + output_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair["slug"], + ) + for spec in metric_specs + } + missing_files = _missing_artifacts( + pair["slug"], + list(output_files.values()), + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in metric_specs + if output_files[spec["key"]] in missing_files + ] + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=base_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair["slug"], + launch_final_values=launch_final_values, + ) + print(f"[{pair['slug']}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['title_suffix']} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=output_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + del data_by_metric + gc.collect() + + if owns_cache: + cache.clear() + gc.collect() + print("Released heatmap metric cache.") + + +def compute_auto_calibrated_arc_length_speed(cfg, price_data): + """Compute the launch/reference auto-calibrated speed for a config.""" + start_ts = pd.Timestamp(cfg["start"]) + + if isinstance(price_data.index, pd.DatetimeIndex): + row = price_data.loc[start_ts] + else: + start_unix_ms = int(start_ts.timestamp() * 1000.0) + index_values = price_data.index.to_numpy(dtype=np.int64) + row_idx = int(np.searchsorted(index_values, start_unix_ms, side="left")) + if row_idx >= len(index_values): + row_idx = len(index_values) - 1 + if row_idx > 0 and index_values[row_idx] != start_unix_ms: + prev_idx = row_idx - 1 + if abs(index_values[prev_idx] - start_unix_ms) <= abs( + index_values[row_idx] - start_unix_ms + ): + row_idx = prev_idx + row = price_data.iloc[row_idx] + + if isinstance(row, pd.DataFrame): + row = row.iloc[0] + + if isinstance(price_data.columns, pd.MultiIndex): + initial_price_values = [ + float(row[(token, "close")]) + for token in cfg["tokens"] + ] + else: + initial_price_values = [ + float(row[f"close_{token}"]) + for token in cfg["tokens"] + ] + + initial_prices = jnp.array(initial_price_values, dtype=jnp.float64) + initial_reserves, Va, Vb = initialise_reclamm_reserves( + get_initial_pool_value(cfg), + initial_prices, + float(cfg["price_ratio"]), + ) + market_price_0 = float(initial_prices[0] / initial_prices[1]) + sqrt_Q = jnp.sqrt( + compute_price_ratio( + initial_reserves[0], + initial_reserves[1], + Va, + Vb, + ) + ) + return float( + calibrate_arc_length_speed( + initial_reserves[0], + initial_reserves[1], + Va, + Vb, + to_daily_price_shift_base(float(cfg["daily_price_shift_exponent"])), + 60.0, + sqrt_Q, + market_price_0, + centeredness_margin=float(cfg["centeredness_margin"]), + ) + ) + + +def generate_arc_speed_efficiency_artifacts( + base_cfg, + launch_cfg, + price_data, + launch_final_values, + cache=None, +): + """Generate arc-speed heatmaps plus the existing efficiency line charts.""" + if not RUN_CONSTANT_ARC_LENGTH: + print("\nSkipping arc-speed heatmaps because RUN_CONSTANT_ARC_LENGTH=False.") + return + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data) + launch_auto_speed = compute_auto_calibrated_arc_length_speed(launch_cfg, price_data) + heatmap_metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in heatmap_metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + pair_specs = [ + { + "slug": "arc_speed_vs_price_ratio", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_PRICE_RATIOS, + "x_key": "arc_length_speed", + "y_key": "price_ratio", + "x_label": "Arc-length speed", + "y_label": "Price ratio", + "title_suffix": ( + f"margin fixed at {base_cfg['centeredness_margin']:.2f}, " + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": PRICE_RATIO_TICKS, + }, + { + "slug": "arc_speed_vs_margin", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_MARGINS, + "x_key": "arc_length_speed", + "y_key": "centeredness_margin", + "x_label": "Arc-length speed", + "y_label": "Centeredness margin", + "title_suffix": ( + f"price_ratio fixed at {base_cfg['price_ratio']:.2f}, " + f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": MARGIN_TICKS + }, + { + "slug": "arc_speed_vs_shift_exp", + "x_values": HEATMAP_ARC_LENGTH_SPEEDS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "arc_length_speed", + "y_key": "daily_price_shift_exponent", + "x_label": "Arc-length speed", + "y_label": "Shift exponent", + "title_suffix": ( + f"price_ratio fixed at {base_cfg['price_ratio']:.2f}, " + f"margin fixed at {base_cfg['centeredness_margin']:.2f}" + ), + "xticks": ARC_LENGTH_SPEED_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + }, + ] + metric_spec_map = {spec["key"]: spec for spec in heatmap_metric_specs} + + print( + "\nGenerating arc-speed heatmaps and line charts " + f"(launch auto-cal speed={launch_auto_speed:.3e}, TVL={format_tvl_millions_label(base_cfg)})..." + ) + + for pair in pair_specs: + heatmap_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair["slug"], + ) + for spec in heatmap_metric_specs + } + line_filename = tvl_artifact_filename( + "reclamm_line_efficiency", + base_cfg, + suffix=pair["slug"], + ) + missing_files = _missing_artifacts( + pair["slug"], + list(heatmap_files.values()) + [line_filename], + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in heatmap_metric_specs + if heatmap_files[spec["key"]] in missing_files + ] + if line_filename in missing_files and "efficiency_pct" not in missing_metric_keys: + missing_metric_keys.append("efficiency_pct") + + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=base_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair["slug"], + launch_final_values=launch_final_values, + ) + for metric_key in missing_metric_keys: + if metric_key not in heatmap_files: + continue + if heatmap_files[metric_key] not in missing_files: + continue + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['title_suffix']} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=heatmap_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + xscale="log", + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + + if line_filename in missing_files: + efficiency_data = data_by_metric["efficiency_pct"] + launch_curve = build_metric_curve( + x_values=pair["x_values"], + x_key=pair["x_key"], + base_cfg=launch_cfg, + metric_key="efficiency_pct", + cache=cache, + launch_final_values=launch_final_values, + ) + plot_arc_speed_line_chart( + data=efficiency_data, + x_values=pair["x_values"], + y_values=pair["y_values"], + y_label=pair["y_label"], + title=( + "Arc-speed efficiency sweep: " + f"{pair['title_suffix']} | TVL {format_tvl_millions_label(base_cfg)}" + ), + filename=line_filename, + launch_curve=launch_curve, + launch_auto_speed=launch_auto_speed, + ) + del data_by_metric + gc.collect() + + if owns_cache: + cache.clear() + gc.collect() + print("Released arc-speed sweep cache.") + + +def get_launch_final_values(all_results, launch_cfg, price_data): + """Reuse launch-style runs when available; otherwise run them once.""" + for cfg, results in all_results: + if cfg["name"] == launch_cfg["name"]: + launch_final_values = { + "geometric": float(results["geometric"]["final_value"]), + } + if "constant_arc_length" in results: + launch_final_values["constant_arc_length"] = float( + results["constant_arc_length"]["final_value"] + ) + return launch_final_values + + print("\nRunning launch-style benchmarks for heatmaps...") + launch_results = run_comparison( + launch_cfg, + price_data=price_data, + low_data_mode=True, + ) + launch_final_values = { + "geometric": float(launch_results["geometric"]["final_value"]), + } + if "constant_arc_length" in launch_results: + launch_final_values["constant_arc_length"] = float( + launch_results["constant_arc_length"]["final_value"] + ) + del launch_results + gc.collect() + return launch_final_values + def print_comparison(cfg, results): """Print text summary table.""" - methods = [ - ("Geometric", results["geometric"]), - ("Geo+Scaled", results["geometric_scaled"]), - ("Const Arc", results["constant_arc_length"]), - ("Arc+Scaled", results["cal_scaled"]), - ] + methods = [("Geometric", results["geometric"])] + has_constant_arc = "constant_arc_length" in results + if has_constant_arc: + methods.append(("Const Arc", results["constant_arc_length"])) + noise_cfg = resolve_reclamm_noise_settings(cfg) hodl_value = float((methods[0][1]["reserves"][0] * methods[0][1]["prices"][-1]).sum()) @@ -128,6 +1612,18 @@ def print_comparison(cfg, results): f"margin={cfg['centeredness_margin']}, " f"shift_exp={cfg['daily_price_shift_exponent']}, " f"fees={cfg['fees']}") + print( + f" base_tvl=${get_initial_pool_value(cfg):,.0f} " + f"(TVL {format_tvl_millions_label(cfg)})" + ) + print(f" note={cfg['reason']}") + print( + f" noise={noise_cfg['noise_summary']}, " + f"gas={cfg.get('gas_cost', 0.0)}, " + f"protocol_fee_split={cfg.get('protocol_fee_split', 0.0)}" + ) + if not has_constant_arc: + print(" constant_arc=disabled") print("-" * 105) header = " {:20s}".format("") for name, _ in methods: @@ -158,18 +1654,27 @@ def print_comparison(cfg, results): vs = (float(r["final_value"]) / hodl_value - 1) * 100 row += f" {vs:>13.2f}%" print(row) + + if has_constant_arc: + geo_final = float(results["geometric"]["final_value"]) + arc_final = float(results["constant_arc_length"]["final_value"]) + geo_lvr = hodl_value - geo_final + arc_lvr = hodl_value - arc_final + print(f" {'Const Arc - Geo':20s} ${arc_final - geo_final:>13,.0f}") + print(f" {'LVR saved vs Geo':20s} ${geo_lvr - arc_lvr:>13,.0f}") print("=" * 105) + def plot_comparison(cfg, results, fig_idx): - """Plot 4-panel comparison for one config.""" - # Method name → (result dict, color, linestyle) + """Plot comparison diagnostics for one config.""" + tvl_label = format_tvl_millions_label(cfg) variants = { "Geometric": (results["geometric"], "C0", "-"), - "Geo+Scaled": (results["geometric_scaled"], "C1", "-"), - "Const arc-len": (results["constant_arc_length"], "C2", "--"), - "Arc+Scaled": (results["cal_scaled"], "C3", "--"), } + has_constant_arc = "constant_arc_length" in results + if has_constant_arc: + variants["Const arc-len"] = (results["constant_arc_length"], "C2", "--") geo = results["geometric"] geo_prices = np.array(geo["prices"]) @@ -181,22 +1686,21 @@ def plot_comparison(cfg, results, fig_idx): price_ratio_traj = geo_prices[:n_steps, 0] / geo_prices[:n_steps, 1] fig, axes = plt.subplots(2, 2, figsize=(14, 10)) - fig.suptitle(cfg["name"], fontsize=13, fontweight="bold") + fig.suptitle(f"{cfg['name']} — TVL {tvl_label}", fontsize=13, fontweight="bold") - # (0,0) Pool value over time ax = axes[0, 0] + plotted_values = [] for name, (r, color, ls) in variants.items(): vals = np.array(r["value"]) + plotted_values.append(vals / 1e6) ax.plot(t_days, vals / 1e6, color=color, ls=ls, label=name, alpha=0.9) - ax.plot(t_days, np.array(hodl_traj) / 1e6, color="gray", ls=":", - alpha=0.5, label="HODL") + _set_padded_ylim(ax, plotted_values, pad_ratio=0.03) ax.set_xlabel("Days") ax.set_ylabel("Pool value ($M)") ax.set_title("Pool value") ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (0,1) Cumulative LVR ax = axes[0, 1] for name, (r, color, ls) in variants.items(): vals = np.array(r["value"]) @@ -208,7 +1712,6 @@ def plot_comparison(cfg, results, fig_idx): ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (1,0) Price ratio ax = axes[1, 0] ax.plot(t_days, price_ratio_traj, color="C4", alpha=0.7) ax.set_xlabel("Days") @@ -216,7 +1719,6 @@ def plot_comparison(cfg, results, fig_idx): ax.set_title("Price path") ax.grid(True, alpha=0.3) - # (1,1) Empirical weights ax = axes[1, 1] for name, (r, color, ls) in variants.items(): w = np.array(r["weights"]) @@ -230,19 +1732,21 @@ def plot_comparison(cfg, results, fig_idx): ax.grid(True, alpha=0.3) plt.tight_layout() - fname = f"reclamm_thermostat_comparison_{fig_idx}.png" + fname = tvl_artifact_filename("reclamm_thermostat_comparison", cfg, suffix=str(fig_idx)) plt.savefig(fname, dpi=150) print(f"Saved {fname}") plt.close(fig) - # Second figure: diagnostics + if not has_constant_arc: + print("Skipping constant-arc comparison diagnostics because RUN_CONSTANT_ARC_LENGTH=False.") + return + geo_values = np.array(geo["value"]) geo_lvr = np.array(hodl_traj) - geo_values fig2, axes2 = plt.subplots(1, 3, figsize=(18, 5)) - fig2.suptitle(f"{cfg['name']} — diagnostics", fontsize=13, fontweight="bold") + fig2.suptitle(f"{cfg['name']} — diagnostics — TVL {tvl_label}", fontsize=13, fontweight="bold") - # (left) Value difference vs geometric ax = axes2[0] for name, (r, color, ls) in variants.items(): if name == "Geometric": @@ -257,7 +1761,6 @@ def plot_comparison(cfg, results, fig_idx): ax.legend(fontsize=8) ax.grid(True, alpha=0.3) - # (middle) LVR ratio over time ax = axes2[1] mask = np.abs(geo_lvr) > 100 if mask.any(): @@ -279,7 +1782,6 @@ def plot_comparison(cfg, results, fig_idx): ax.set_title("Relative LVR") ax.grid(True, alpha=0.3) - # (right) Per-step LVR histogram ax = axes2[2] all_pos = [] for name, (r, color, ls) in variants.items(): @@ -306,74 +1808,180 @@ def plot_comparison(cfg, results, fig_idx): ax.grid(True, alpha=0.3) plt.tight_layout() - fname2 = f"reclamm_thermostat_diff_{fig_idx}.png" + fname2 = tvl_artifact_filename("reclamm_thermostat_diff", cfg, suffix=str(fig_idx)) plt.savefig(fname2, dpi=150) print(f"Saved {fname2}") plt.close(fig2) + arc_values = np.array(results["constant_arc_length"]["value"]) + n_eff = min(len(geo_values), len(arc_values)) + t_eff = np.arange(n_eff) / (60 * 24) + efficiency_pct = ( + (arc_values[:n_eff] - geo_values[:n_eff]) + / np.maximum(np.abs(geo_values[:n_eff]), 1e-12) + * 100.0 + ) + + fig3, ax3 = plt.subplots(1, 1, figsize=(10, 4.5)) + fig3.suptitle(f"{cfg['name']} — efficiency — TVL {tvl_label}", fontsize=13, fontweight="bold") + ax3.plot( + t_eff, + efficiency_pct, + color="C2", + linewidth=1.8, + label="(Const Arc - Geo) / Geo", + ) + ax3.axhline(0.0, color="gray", ls="--", alpha=0.6) + _set_padded_ylim(ax3, [efficiency_pct], pad_ratio=0.08) + ax3.set_xlabel("Days") + ax3.set_ylabel("Efficiency vs geometric (%)") + ax3.set_title("Efficiency") + ax3.legend(fontsize=8) + ax3.grid(True, alpha=0.3) + + plt.tight_layout() + fname3 = tvl_artifact_filename("reclamm_thermostat_efficiency", cfg, suffix=str(fig_idx)) + plt.savefig(fname3, dpi=150) + print(f"Saved {fname3}") + plt.close(fig3) + + if __name__ == "__main__": - all_results = [] - for i, cfg in enumerate(CONFIGS): - print(f"\n>>> Running {cfg['name']}...") - try: - results = run_comparison(cfg) - print_comparison(cfg, results) - plot_comparison(cfg, results, i) - all_results.append((cfg, results)) - except Exception as e: - print(f" FAILED: {e}") - import traceback - traceback.print_exc() - - # Summary overlay: all configs on one figure (pool value normalised) - if len(all_results) > 1: - fig, axes = plt.subplots(1, 2, figsize=(16, 5)) - fig.suptitle("Cross-config comparison (normalised)", fontsize=13, - fontweight="bold") - - method_keys = [ - ("geometric", "geo", "-"), - ("geometric_scaled", "geo+s", "-."), - ("constant_arc_length", "arc", "--"), - ("cal_scaled", "arc+s", ":"), - ] + shared_price_data = load_shared_price_data(CONFIGS) + + for initial_pool_value in TVL_SWEEP_VALUES: + tvl_configs = configs_for_tvl(CONFIGS, initial_pool_value) + tvl_label = format_tvl_millions_label(tvl_configs[0]) + print(f"\n=== TVL sweep: {tvl_label} ===") + + all_results = [] + for i, cfg in enumerate(tvl_configs): + print(f"\n>>> Running {cfg['name']} at TVL {tvl_label}...") + try: + results = run_comparison(cfg, price_data=shared_price_data) + print_comparison(cfg, results) + plot_comparison(cfg, results, i) + all_results.append((cfg, results)) + except Exception as e: + print(f" FAILED: {e}") + import traceback - for i, (cfg, results) in enumerate(all_results): - geo_v = np.array(results["geometric"]["value"]) - t = np.arange(len(geo_v)) / (60 * 24) - short_name = cfg["name"].split("(")[0].strip() - - for j, (key, suffix, ls) in enumerate(method_keys): - v = np.array(results[key]["value"]) - color_idx = i * len(method_keys) + j - - # (left) Normalised pool value - axes[0].plot(t, v / v[0], ls=ls, alpha=0.8, - label=f"{short_name} {suffix}", - color=f"C{color_idx % 10}") - - # (right) Value difference vs geometric (skip geo itself) - if key != "geometric": - pct_diff = (v - geo_v) / geo_v * 100 - axes[1].plot(t, pct_diff, ls=ls, alpha=0.8, - label=f"{short_name} {suffix}", - color=f"C{color_idx % 10}") - - axes[0].set_xlabel("Days") - axes[0].set_ylabel("Normalised pool value") - axes[0].set_title("Pool value (V/V0)") - axes[0].legend(fontsize=6, ncol=2) - axes[0].grid(True, alpha=0.3) - - axes[1].set_xlabel("Days") - axes[1].set_ylabel("(Method - Geo) / Geo (%)") - axes[1].set_title("Relative value difference vs Geometric") - axes[1].axhline(0, color="gray", ls="--", alpha=0.5) - axes[1].legend(fontsize=6, ncol=2) - axes[1].grid(True, alpha=0.3) - - plt.tight_layout() - plt.savefig("reclamm_thermostat_summary.png", dpi=150) - print("\nSaved reclamm_thermostat_summary.png") - plt.close(fig) + traceback.print_exc() + + if len(all_results) > 1: + if RUN_CONSTANT_ARC_LENGTH: + fig, axes = plt.subplots(1, 2, figsize=(16, 5)) + fig.suptitle( + f"Cross-config comparison (normalised) — TVL {tvl_label}", + fontsize=13, + fontweight="bold", + ) + + method_keys = [ + ("geometric", "geo", "-"), + ("constant_arc_length", "arc", "--"), + ] + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + + for j, (key, suffix, ls) in enumerate(method_keys): + v = np.array(results[key]["value"]) + color_idx = i * len(method_keys) + j + + axes[0].plot( + t, + v / v[0], + ls=ls, + alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}", + ) + + if key != "geometric": + pct_diff = (v - geo_v) / geo_v * 100 + axes[1].plot( + t, + pct_diff, + ls=ls, + alpha=0.8, + label=f"{short_name} {suffix}", + color=f"C{color_idx % 10}", + ) + + axes[0].set_xlabel("Days") + axes[0].set_ylabel("Normalised pool value") + axes[0].set_title("Pool value (V/V0)") + axes[0].legend(fontsize=6, ncol=2) + axes[0].grid(True, alpha=0.3) + + axes[1].set_xlabel("Days") + axes[1].set_ylabel("Efficiency vs geometric (%)") + axes[1].set_title("Efficiency vs Geometric") + axes[1].axhline(0, color="gray", ls="--", alpha=0.5) + axes[1].legend(fontsize=6, ncol=2) + axes[1].grid(True, alpha=0.3) + else: + fig, ax = plt.subplots(1, 1, figsize=(9, 5)) + fig.suptitle( + f"Cross-config comparison (normalised geometric) — TVL {tvl_label}", + fontsize=13, + fontweight="bold", + ) + + for i, (cfg, results) in enumerate(all_results): + geo_v = np.array(results["geometric"]["value"]) + t = np.arange(len(geo_v)) / (60 * 24) + short_name = cfg["name"].split("(")[0].strip() + ax.plot( + t, + geo_v / geo_v[0], + ls="-", + alpha=0.8, + label=f"{short_name} geo", + color=f"C{i % 10}", + ) + + ax.set_xlabel("Days") + ax.set_ylabel("Normalised pool value") + ax.set_title("Geometric pool value (V/V0)") + ax.legend(fontsize=6, ncol=2) + ax.grid(True, alpha=0.3) + + plt.tight_layout() + summary_name = tvl_artifact_filename( + "reclamm_thermostat_summary", + tvl_configs[0], + ) + plt.savefig(summary_name, dpi=150) + print(f"\nSaved {summary_name}") + plt.close(fig) + + launch_final_values = get_launch_final_values( + all_results, + launch_cfg=tvl_configs[0], + price_data=shared_price_data, + ) + shared_sweep_cache = make_sweep_cache(shared_price_data) + + print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") + generate_heatmaps( + dict(tvl_configs[1]), + shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + + generate_arc_speed_efficiency_artifacts( + dict(tvl_configs[1]), + launch_cfg=dict(tvl_configs[0]), + price_data=shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + shared_sweep_cache.clear() + gc.collect() + print(f"Released shared sweep cache for TVL {tvl_label}.") diff --git a/scripts/reclamm/demo_run_reclamm.py b/scripts/reclamm/demo_run_reclamm.py index 3ea21ec6..132f5122 100644 --- a/scripts/reclamm/demo_run_reclamm.py +++ b/scripts/reclamm/demo_run_reclamm.py @@ -36,105 +36,127 @@ def balancer_fingerprint(tokens, start, end, fees): } +def reclamm_fingerprint(tokens, start, end, fees, interpolation_method="geometric"): + """Build a reCLAMM fingerprint for a demo scenario.""" + return { + "tokens": tokens, + "rule": "reclamm", + "startDateString": start, + "endDateString": end, + "initial_pool_value": 1000000.0, + "do_arb": True, + "fees": fees, + "gas_cost": 0.0, + "arb_fees": 0.0, + "chunk_period": 60, + "weight_interpolation_period": 60, + "reclamm_interpolation_method": interpolation_method, + "reclamm_arc_length_speed": None, + } + + +def reclamm_params(price_ratio, centeredness_margin, daily_price_shift_exponent): + """Build reCLAMM params from a concise config.""" + return { + "price_ratio": jnp.array(price_ratio), + "centeredness_margin": jnp.array(centeredness_margin), + "daily_price_shift_base": jnp.array( + to_daily_price_shift_base(daily_price_shift_exponent) + ), + } + + +def _apply_active_noise_settings(fp): + """Enable the active AAVE/ETH reCLAMM noise model for demo runs.""" + if fp.get("rule") != "reclamm" or list(fp.get("tokens", [])) != ["AAVE", "ETH"]: + return fp, "disabled" + + from compare_reclamm_thermostats import ( + AAVE_ETH_NOISE_SETTINGS, + resolve_reclamm_noise_settings, + ) + + cfg = { + "tokens": fp["tokens"], + "start": fp["startDateString"], + "end": fp["endDateString"], + "enable_noise_model": True, + "noise_model": AAVE_ETH_NOISE_SETTINGS["noise_model"], + "noise_artifact_dir": AAVE_ETH_NOISE_SETTINGS["noise_artifact_dir"], + "noise_pool_id": AAVE_ETH_NOISE_SETTINGS["noise_pool_id"], + "gas_cost": fp.get("gas_cost", AAVE_ETH_NOISE_SETTINGS["gas_cost"]), + "protocol_fee_split": fp.get( + "protocol_fee_split", + AAVE_ETH_NOISE_SETTINGS["protocol_fee_split"], + ), + "arb_frequency": fp.get("arb_frequency"), + "noise_trader_ratio": fp.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": fp.get("reclamm_noise_params"), + "noise_arrays_path": fp.get("noise_arrays_path"), + } + noise_cfg = resolve_reclamm_noise_settings(cfg) + + updated = dict(fp) + updated["gas_cost"] = cfg["gas_cost"] + updated["protocol_fee_split"] = cfg["protocol_fee_split"] + updated["noise_trader_ratio"] = noise_cfg.get("noise_trader_ratio", 0.0) + for key in ("noise_model", "reclamm_noise_params", "noise_arrays_path", "arb_frequency"): + if noise_cfg.get(key) is not None: + updated[key] = noise_cfg[key] + return updated, noise_cfg["noise_summary"] + + SCENARIOS = [ { - "name": "AAVE/ETH on-chain (25bps)", + "name": "AAVE/ETH launch-style range (25bps, geometric)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0025, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(1.5), - "centeredness_margin": jnp.array(0.5), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.1) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="geometric", + ), + "params": reclamm_params(1.5014, 0.5, 0.1), }, }, { - "name": "AAVE/ETH zero fees", + "name": "AAVE/ETH tighter launch-style range (25bps, geometric)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(1.5), - "centeredness_margin": jnp.array(0.5), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.1) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="geometric", + ), + "params": reclamm_params(1.15, 0.5, 0.1), }, }, { - "name": "AAVE/ETH wide range (25bps)", + "name": "AAVE/ETH tighter launch-style range (25bps, constant arc)", "reclamm": { - "fingerprint": { - "tokens": ["AAVE", "ETH"], - "rule": "reclamm", - "startDateString": "2024-06-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.0025, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(4.0), - "centeredness_margin": jnp.array(0.2), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(1.0) - ), - }, + "fingerprint": reclamm_fingerprint( + ["AAVE", "ETH"], + "2024-06-01 00:00:00", + "2025-06-01 00:00:00", + 0.0025, + interpolation_method="constant_arc_length", + ), + "params": reclamm_params(1.15, 0.5, 0.1), }, }, { "name": "BTC/ETH (10bps)", "reclamm": { - "fingerprint": { - "tokens": ["BTC", "ETH"], - "rule": "reclamm", - "startDateString": "2024-01-01 00:00:00", - "endDateString": "2025-06-01 00:00:00", - "initial_pool_value": 1000000.0, - "do_arb": True, - "fees": 0.001, - "gas_cost": 0.0, - "arb_fees": 0.0, - "chunk_period": 60, - "weight_interpolation_period": 60, - }, - "params": { - "price_ratio": jnp.array(2.0), - "centeredness_margin": jnp.array(0.3), - "daily_price_shift_base": jnp.array( - to_daily_price_shift_base(0.5) - ), - }, + "fingerprint": reclamm_fingerprint( + ["BTC", "ETH"], + "2024-01-01 00:00:00", + "2025-06-01 00:00:00", + 0.001, + interpolation_method="geometric", + ), + "params": reclamm_params(2.0, 0.3, 0.5), }, }, ] @@ -143,7 +165,7 @@ def balancer_fingerprint(tokens, start, end, fees): def run_scenario(scenario): """Run a reClAMM config and its Balancer 50/50 baseline, print comparison.""" rc = scenario["reclamm"] - fp = rc["fingerprint"] + fp, noise_summary = _apply_active_noise_settings(dict(rc["fingerprint"])) # Run reClAMM reclamm_result = do_run_on_historic_data( @@ -173,7 +195,14 @@ def run_scenario(scenario): print("=" * 80) print(f" {scenario['name']}") - print(f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']}") + print( + f" Tokens: {', '.join(fp['tokens'])} | Fees: {fp['fees']} | " + f"Interpolation: {fp.get('reclamm_interpolation_method', 'geometric')}" + ) + print( + f" Noise: {noise_summary} | Gas: {fp.get('gas_cost', 0.0)} | " + f"Protocol fee split: {fp.get('protocol_fee_split', 0.0)}" + ) print("-" * 80) print(f" {'':30s} {'reClAMM':>14s} {'Balancer 50/50':>14s}") print(f" {'Initial value':30s} ${rc_init:>13,.0f} ${bal_init:>13,.0f}") diff --git a/tests/scripts/test_compare_reclamm_thermostats.py b/tests/scripts/test_compare_reclamm_thermostats.py new file mode 100644 index 00000000..7072f73c --- /dev/null +++ b/tests/scripts/test_compare_reclamm_thermostats.py @@ -0,0 +1,284 @@ +"""Tests for heatmap skip logic in compare_reclamm_thermostats.py.""" + +import importlib.util +from pathlib import Path +import sys +import types + +import numpy as np +import pytest + + +SCRIPT_PATH = ( + Path(__file__).resolve().parents[2] + / "scripts" + / "compare_reclamm_thermostats.py" +) + + +def _load_script_module(): + injected_modules = {} + + def inject_module(name, module): + injected_modules[name] = sys.modules.get(name) + sys.modules[name] = module + + # Minimal stubs so the script can be imported without the full runtime + # stack present in this test environment. + jax_module = types.ModuleType("jax") + jax_module.numpy = np + inject_module("jax", jax_module) + inject_module("jax.numpy", np) + + pandas_module = types.ModuleType("pandas") + pandas_module.Timestamp = lambda value: value + pandas_module.DatetimeIndex = tuple + pandas_module.DataFrame = type("DataFrame", (), {}) + inject_module("pandas", pandas_module) + + matplotlib_module = types.ModuleType("matplotlib") + pyplot_module = types.ModuleType("matplotlib.pyplot") + pyplot_module.cm = types.SimpleNamespace(viridis=lambda values: values) + colors_module = types.ModuleType("matplotlib.colors") + colors_module.TwoSlopeNorm = object + colors_module.Normalize = object + cm_module = types.ModuleType("matplotlib.cm") + cm_module.ScalarMappable = object + inject_module("matplotlib", matplotlib_module) + inject_module("matplotlib.pyplot", pyplot_module) + inject_module("matplotlib.colors", colors_module) + inject_module("matplotlib.cm", cm_module) + + quantammsim_module = types.ModuleType("quantammsim") + runners_module = types.ModuleType("quantammsim.runners") + jax_runners_module = types.ModuleType("quantammsim.runners.jax_runners") + jax_runners_module.do_run_on_historic_data = lambda **kwargs: { + "final_value": 0.0 + } + runners_module.jax_runners = jax_runners_module + + pools_module = types.ModuleType("quantammsim.pools") + reclamm_pkg_module = types.ModuleType("quantammsim.pools.reCLAMM") + reserves_module = types.ModuleType( + "quantammsim.pools.reCLAMM.reclamm_reserves" + ) + reserves_module.calibrate_arc_length_speed = lambda *args, **kwargs: 0.0 + reserves_module.compute_price_ratio = lambda *args, **kwargs: 1.0 + reserves_module.initialise_reclamm_reserves = ( + lambda *args, **kwargs: (np.array([1.0, 1.0]), 1.0, 1.0) + ) + reclamm_pkg_module.reclamm_reserves = reserves_module + pools_module.reCLAMM = reclamm_pkg_module + + utils_module = types.ModuleType("quantammsim.utils") + data_processing_module = types.ModuleType("quantammsim.utils.data_processing") + historic_utils_module = types.ModuleType( + "quantammsim.utils.data_processing.historic_data_utils" + ) + historic_utils_module.get_historic_parquet_data = lambda *args, **kwargs: None + data_processing_module.historic_data_utils = historic_utils_module + utils_module.data_processing = data_processing_module + + quantammsim_module.runners = runners_module + quantammsim_module.pools = pools_module + quantammsim_module.utils = utils_module + + inject_module("quantammsim", quantammsim_module) + inject_module("quantammsim.runners", runners_module) + inject_module("quantammsim.runners.jax_runners", jax_runners_module) + inject_module("quantammsim.pools", pools_module) + inject_module("quantammsim.pools.reCLAMM", reclamm_pkg_module) + inject_module("quantammsim.pools.reCLAMM.reclamm_reserves", reserves_module) + inject_module("quantammsim.utils", utils_module) + inject_module("quantammsim.utils.data_processing", data_processing_module) + inject_module( + "quantammsim.utils.data_processing.historic_data_utils", + historic_utils_module, + ) + + spec = importlib.util.spec_from_file_location( + "test_compare_reclamm_thermostats_module", + SCRIPT_PATH, + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + try: + spec.loader.exec_module(module) + return module + finally: + for name, original in injected_modules.items(): + if original is None: + sys.modules.pop(name, None) + else: + sys.modules[name] = original + + +@pytest.fixture +def script_module(): + return _load_script_module() + + +@pytest.fixture +def base_cfg(): + return { + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "price_ratio": 1.10, + "centeredness_margin": 0.60, + "daily_price_shift_exponent": 0.1, + } + + +@pytest.fixture +def launch_final_values(): + return { + "geometric": 1_000_000.0, + "constant_arc_length": 1_010_000.0, + } + + +def test_generate_heatmaps_skips_existing_pairs( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + monkeypatch.setattr(script_module.os.path, "exists", lambda filename: True) + monkeypatch.setattr( + script_module, + "build_heatmap_matrices", + lambda **kwargs: pytest.fail("heatmap sweep should have been skipped"), + ) + monkeypatch.setattr( + script_module, + "plot_heatmap", + lambda **kwargs: pytest.fail("plotting should have been skipped"), + ) + + script_module.generate_heatmaps( + base_cfg, + price_data=None, + launch_final_values=launch_final_values, + cache={}, + ) + + +def test_generate_heatmaps_only_renders_missing_artifacts( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + missing_file = "reclamm_heatmap_efficiency_price_ratio_vs_margin.png" + + def fake_exists(filename): + if filename == missing_file: + return False + return filename.startswith("reclamm_heatmap_") + + build_calls = [] + plotted_files = [] + + def fake_build_heatmap_matrices(**kwargs): + build_calls.append(kwargs) + return { + "efficiency_pct": np.zeros( + (len(kwargs["y_values"]), len(kwargs["x_values"])), + dtype=float, + ) + } + + def fake_plot_heatmap(**kwargs): + plotted_files.append(kwargs["filename"]) + + monkeypatch.setattr(script_module.os.path, "exists", fake_exists) + monkeypatch.setattr( + script_module, + "build_heatmap_matrices", + fake_build_heatmap_matrices, + ) + monkeypatch.setattr(script_module, "plot_heatmap", fake_plot_heatmap) + + script_module.generate_heatmaps( + base_cfg, + price_data=None, + launch_final_values=launch_final_values, + cache={}, + ) + + assert len(build_calls) == 1 + assert build_calls[0]["progress_label"] == "price_ratio_vs_margin" + assert build_calls[0]["metric_keys"] == ["efficiency_pct"] + assert plotted_files == [missing_file] + + +def test_arc_speed_artifacts_only_build_missing_line_output( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + missing_line = "reclamm_line_efficiency_arc_speed_vs_price_ratio.png" + + def fake_exists(filename): + if filename == missing_line: + return False + return filename.startswith("reclamm_") + + build_calls = [] + curve_calls = [] + plotted_lines = [] + + def fake_build_heatmap_matrices(**kwargs): + build_calls.append(kwargs) + return { + "efficiency_pct": np.zeros( + (len(kwargs["y_values"]), len(kwargs["x_values"])), + dtype=float, + ) + } + + def fake_build_metric_curve(**kwargs): + curve_calls.append(kwargs) + return np.zeros(len(kwargs["x_values"]), dtype=float) + + monkeypatch.setattr(script_module.os.path, "exists", fake_exists) + monkeypatch.setattr( + script_module, + "compute_auto_calibrated_arc_length_speed", + lambda cfg, price_data: 1.23e-4, + ) + monkeypatch.setattr( + script_module, + "build_heatmap_matrices", + fake_build_heatmap_matrices, + ) + monkeypatch.setattr( + script_module, + "build_metric_curve", + fake_build_metric_curve, + ) + monkeypatch.setattr( + script_module, + "plot_heatmap", + lambda **kwargs: pytest.fail("existing heatmap should not be redrawn"), + ) + monkeypatch.setattr( + script_module, + "plot_arc_speed_line_chart", + lambda **kwargs: plotted_lines.append(kwargs["filename"]), + ) + + script_module.generate_arc_speed_efficiency_artifacts( + base_cfg=base_cfg, + launch_cfg=dict(base_cfg), + price_data=None, + launch_final_values=launch_final_values, + cache={}, + ) + + assert len(build_calls) == 1 + assert build_calls[0]["progress_label"] == "arc_speed_vs_price_ratio" + assert len(curve_calls) == 1 + assert curve_calls[0]["x_key"] == "arc_length_speed" + assert plotted_lines == [missing_line] From 29af7df58f75b40cc20dcafea5da3ba25a6bfa2c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Sun, 29 Mar 2026 16:33:48 +0100 Subject: [PATCH 073/115] data: add sim arrays for 0x9d1fcf346ea1b0 --- ...0x9d1fcf346ea1b0_2024-06-01_2026-03-01.npz | Bin 0 -> 14723606 bytes ...0x9d1fcf346ea1b0_2025-08-03_2026-02-18.npz | Bin 0 -> 4609046 bytes 2 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2024-06-01_2026-03-01.npz create mode 100644 results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2025-08-03_2026-02-18.npz diff --git a/results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2024-06-01_2026-03-01.npz b/results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2024-06-01_2026-03-01.npz new file mode 100644 index 0000000000000000000000000000000000000000..368146ce740397c0ad452f244099d4def90e33e0 GIT binary patch literal 14723606 zcmeF(>ARL?xd!lCRB#H#AycGq$y)EY5KY4&@(Cg^OT@$xMIOPeh>B1SiI%7-qUD5H zV#cyEFm2~-rM5BWdDxPgijRT2NKDf9RH6^FI6Th7;nVq^*YDHqbHIM< zZ?JXxzdt*sQ;yzz+_%0uD|P-Fn6^$QoqYV9=~K$7bEXeJY1SF*r7iyTZ#w^V=fCLu zxnloGUpe5g_0p;7%-Xc+bB>u^+iPg;v&ZgHYY(j*d-CkB&7OMFl#^#qo8I~7Uz|E~ z&h-4x=Nvb6*7W@M^*whTvCC&hv>WZAXAS-T{n;c{hu3C|X%|utcHj;;zyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00-hbFkw>b zqGo>oEcIXq?tlXv-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$Ah-iFj$HZjK6$574|d=VIKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJ5av3?fji&;2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`290>2g-c#3p(8|A))Po(k0}gP2103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii;12YS`OBwH$lrgK zdawg`zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%e_-GQ!479V?Ctw>T2cHj;;zyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwP`>x;pAX9KpQRq`z#VXa103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14n%k0)KxRa ze$-PWsRui72OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`2(H$uNxZ>(R*Yi7dsRui72OQu42ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2(H+=h^YMqa^ZRG12Rm>F9N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW5B z9jK~V3%}PYlGKA8xC0JwfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h0f#?n#_{y3`KUpi1)Po(k0}gP2103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nnM#s5tJ5{Qg<$!4BL3 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N<872ddkiI&)|}f2UjO!4BL32ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N<832j)$MUr~319!jy4sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|u9@pE zUENb8sRui72OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`2(H$sf-nZ_({Qg<$!4BL32ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<8B2Zq=FzDFbfPErqc;0`#z0S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ZB3L zJ$nB+=hliO^F9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW5R9VjoFKVifC{b#8MJ8%aa-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCJGTc;T|SOK-~G zf0lZ%19!jy4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4I1t@|a`ICjPHg3O>QWDO;0`#z0S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ZB4$bGyaDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7}% zJn_`JPd4)JB=uki?tlXv-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$Ah-kN!?U(pkasHeUFZ+vw{tw>T2 zcHj;;zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00*Kwu&l4PdS1OqQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4qB~Gd+2l(rn?;g(umg9%0S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPkfsZ|L&Qb0B zJ4rp*fji&;2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29Ek3~{ueF&b#H$EEcIXq?tlXv-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ah-k7fQcs_*(#FMgB`d74sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKYAE4%{~D zhTr_OQ6#AcJ8%aa-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCJGTC?DVY+asGrl6tTMcfbJ-aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W5Z!^U4=&tkLAyv&4|d=VIKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1r zJJ8qcS-WUhk)$5%z#VXa103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14n%jL{L80@jO)qY>6Uu119!jy4sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|$+wrgH}d;usRui7 z2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`2(H%J8oF9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW5R9cVUI9J!#oNKy}W;0`#z0S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ckPrrKR`J%)4|d=VIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJFxnHu3p*MsnmlV zxC0JwfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h0f#?oA{p{nfb#^NCUENB%; z>cI}&0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K$bO&Z1+Wq7v`TeuhgB`d74sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKYAE4wQ2a`ugLI{7zl!!4BL32ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<832f7{} zI%jFUNKy}W;0`#z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ckQ$Y{oBco15Q1OFh_uJKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0(l7-dXRQS1Xd#gB`d74sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKYAM z4h(C&wM+i~b1C&;2kw9a9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8a3Hz^V~(6TqM6@6OFh_uJKz8ZIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0;58%U{2>vs0-DJ8%aa z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCJGT*nZQ8F7AA%T1tk)$5% zz#VXa103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14n%jLY!BY~^Nk`&J=lRe-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC(e?m#v1-s@J^JMYw`9_+v!aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS`#cA)Rh2S0kJ zStO|kJ8%aa-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCJGT=$bg|n9sC}B=uki?tlXv-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4v8-fGhw#GyaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0Wz=7xvl>4u{cvQVeQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4qC3!c%#eY1G>atlUGyaDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7xv96soWA}a0eXV00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<701Hm1*eA2?*4z3qT>cI}&0S7q10S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO*|QrXr(Ej6 z4%`6;IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G;6Qi>POV*cWF!AhQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4f;-SP_ts-iZRhWFOFh_uJKz8ZIKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G2<|}NjJxi< zHow!Cdawg`zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%e_-GS=xfeUtN7D?*C4%`6;IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6Qi>jz06PkMj4QE2#%Na0eXV00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<701Hm0A zuNn7=L-O~Zr5^0S9dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8M0cS2-d8q$t@BP_>cI}&0S7q10S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO&0iCO^~4@1La}?7$sx zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S-iW;KOF9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW5R9jJDBF9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW5R9q9Yh3mdGc6-nyB4%`6;IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-%4e59xo5pdQV(|E z4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4!aLBM{M&`Y^Y0|}Udpf{{8+O{ zQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4!aJ~b;PIQc^Y0|}Ufji&;2ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ek3~MeqG)Of$cKmU^%QcfbJ- zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W z5Zr;TNq@g;V1B1B^7OFh_uJKz8ZIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0(l7(v7dblE42f^GyaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0Wz=7xvjQ;+Vd*7%RN$SB4+yMtTzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKy(MH>)#!cI}&0S7q10S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K$a0kknpWALkqexN@cHj;;zyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwP~DVz z*XH-nQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4qB~HoSby=z&i9|C9_+v!aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_Kci^bg4_jK#@1La}?7$sxfCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S*Lr zpzoMHjz7OuB&i2Ga0eXV00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<701JNBQf7X5W>FxaeXQ>A}a0eXV00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<701JND$$rjI?(#Y?hr5^0S z9dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8M0eo1!OyR$=l9Q24|d=VIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b1LJ5a5tEkC}eNKy}W;0`#z0S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02f{nBXtzmU>dL>9 z)Po(k0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROii=ngzNddg+P^80702Rm>F9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW5R9r*1b_21X>`)8>KJ8%aa-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCJGTSUCNf z37wrvJ=lRe-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC(e?!Zz1JbOlGr&14g;0`#z0S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ZB3Le&K?@?Nu+5)Po(k0}gP2103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nnL2 zAN$%XJw=jwumg9%0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02RIPjfn_6aT)l68|19-j2kw9a9N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Hz^RoC`cJ)7S@OFh_uJKz8Z zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G zi0(l7;FU`s?JknkgB`d74sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKYAM4qSEq3wPD>?Uh0#a0eXV00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<701F;?W zZSU1%t9HGRdawg`zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%e_-GNk{{=4pJop<_D4|d=VIKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1LJ5WBca;JwHMUr~319!jy z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zI1t`}B|8l0*UG<>)Po(k0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROii;0{zz4SC?G{QYOC2Rm>F9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW5R9q2m$vh`;)izM}6 z2kw9a9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8a3Hz^)w&Ve4eiPA)TJKmz#VXa103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14g`0g@A${wzbfxk>cI}&0S7q10S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO)+G4&3OH z&O3dn2Rm>F9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW5B9q8Kist;$@izM}62kw9a9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Hz^)$>Eg|E-q4|19-j2kw9a9N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Ht? zeS>B{Hmg}AsRui72OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`2(H$rcI{jnI+eMOkumg9%0S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPjf$F($9kjHkNKy}W;0`#z z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ckQ$<@NKoAKG0csRui72OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`2(H*GHd-(2Ewfs(9>cI}&0S7q10S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO+|$xMuzQ{pU*R z!4BL32ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N<832g*Tr-nLtQr!Vzj2kw9a9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8a3H(`H@vud%Xt?mwK=R zcfbJ-aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0W5Z-}vM!o%){7%1;dawg`zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GTK+o;t5Lzkil`umg9%0S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPifpWyF-<;4c zlGKA8xC0JwfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h0f#?oYQ_i~i#m;xir5^0S9dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl81b1NBy3>~*(JYeGgB`d74sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKYAE4qSd_ z&--WQ_s>!fcHj;;zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00*KwP~JZMjsI>GN$SB4+yMtTzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKzIl4nbCjG;rVxxdawg`zyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_ z+<~tDS$Oapy+x9Gumg9%0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02RIPjf$H(ANA_zJN$SB4+yMtTzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKy(MncMg32v{sR%9_+v! zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_KccA*o==Zm6=6C8+4|d=VIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b1LJJ9vr&Axtc{{FMngB`d74sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKYAE4pf7e+_6V~ z|19-j2kw9a9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8a3Hz^UE@ys#3POTPF?E34%`6;IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QK(n*aXCK__cI}&0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K$bO)-ZUpqDR6iMpA4%`6;IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-?mceP{^J@&l6tTMcfbJ- zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W z5Z!@l&goaLsONX;QV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4f;&*&e8ZZZ^7o&m9_+v!aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_Kcc9vS&((9=`TNgO z4|d=VIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b1LJMjFmcl@~X{#ok54%`6;IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-Mt|cUr?l!tl6tTMcfbJ-aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W5Z!^<^BzC< z;zp6A9_+v!aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_Kcc6OrgOfg$-#<$|*nvCX00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<700~`qNz;@}+n>6$9B=uki?tlXv-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4wl z-}TPrwfz2B>cI}&0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K$a0kjOPIzK?ey1<>UUs7teT1Aq2umg9%0S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPj zf%2vDTSvExB=uki?tlXv-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$Ai4viXI#47;^9S-dawg`zyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GROhcerIve*Y}>U!fcHj;;zyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00)9QP`-QJKIhkqB=uki?tlXv-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$AiM)( z|MAQfJ^6Q%dawg`zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%e_+<|J<1w}J||5@t64%`6;IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-=AH1sU6(eCB=uki?tlXv z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zAi4vWZ#>}PVfp*dQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4qB~G+_8%vm)XLxKmU^%QcfbJ-aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W5Z!@hal!8S`_Gls zgB`d74sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKYA64wSbqPo3|SOFh_uJKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3Gi0(l3os(yt(v!dcEcIXq?tlXv-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4vS)?Bl= zmft^1J=lRe-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC%|?m*X`pXom`ztfj`umg9%0S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPjfqP%Pf9#d{oxaqA9k>GyaDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7xv zR3oM>dbm*}sRui72OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`2;T;%!#N3DK`FE0fumg9%0S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPifpYL}%cj+eB=uki?tlXv z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zAi4w95r5eKxt{#}XQ>A}a0eXV00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<701Hm0A`_?QR)cO9i)Po(k0}gP2103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii;0~-kY^%+d)Qcqb zUKl{brB1t{ifji&;2ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ek3~=k|NB z-&ySy_efa$RopPxMJ8%aa-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCJGT=o`E30}ti*&r%O| z;0`#z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ckPrT|asL-+PNB^F9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW5R9VoY1bjjlU{b#8MJ8%aa-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCJGTm^Xj=n$h|Fv($qfxC0Jw zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 zf#?pDKe(vh!L9sGUFyLO+yMtTzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)hKy(L|rbS2P??0DP4|d=VIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1LJ5Zf?(TTszJC%B{ z19!jy4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4I1t`}Ge+#YLp}daQV(|E4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4f;&*Yx9s-)JKrgndawg`zyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_+=0H82c+4J zB1t{ifji&;2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29Ek2fIq-&RWpDoev($qfxC0JwfCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#42Q=RC0N{(6z59_+v!aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_KcVOO| z2h;|47fI^D4%`6;IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G;6QW-x>oFa+LE3kNj=zsJKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0(jj+_$HEsg}RfE%jgr?tlXv z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zAi4uP-n7fe&Q7Hs?7$sxfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S-iW;DRyL<@NmjS?a+K+`<2{bEj`tmgO42pI#F!r!-SB z4}^uh?;2hP1qX28j?9S=gA6ppe9Z0E(m*{l$CN^IphQg#abT5!hzPkIfUOeIq&srJ zsT?1ToY3U{IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJFw!ZR#!8>f0pi`2X?>#4sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t={ zw>CL;;VZeP(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4qB~GcJZ9Cwop<`u9rVBsIKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1LJ1}A5-TVBsl?v$&dSC|} z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCJGTsI6Ew?xy@sU%G=H*Z~JPzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)hKy(Mn$#;)`vNsjd9rVBsIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJ5asw$0g&2q(Zub z9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G;6QW-%KZoIesF&OEZsp5?0^Ft-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4ux^FR4Q|NQ=0x`Q6r0S7q10S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO&m8r9q3E zsgUlV2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4I1t@|YRZw%Z{A3SbO$}K0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROii@DBX;{4c%Tlm92_4tih*9N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW5B9q4-J ziWA<=J(cdD2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4I1t@|a^MAbO$}K0}gP2103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii@DAMfWZx5;`G1n`pa*uq0S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPi zfojG95B#c;3h54dU9U+bOkKTCJe13TaV2ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ej~e?d$#f_vY`EOLx!% zJKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3Gi0(l3xnlf%jr^T%=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`2!5t{q-8uKRdj3wibO$}K0}gP2103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii;0{#(ntajP zX8!)ObO$}K0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROii=nkB9{+&}hJ(cdD2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4I1t={a>4v1+qF_5-9ZoRfCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<5=x&zf? z%NJbI$lrgK?w|*DzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%e_+=1G4Mc=J+Po+EPfgNyw103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14up4L%4@rC-g*Bl-9ZoRfCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<5= zxC7OkiKp}qN`-U>J+K1~aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0Wz=7}%eE#6$ALzV)mhPYjcEAA+aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W5Z!@MeYXB?Pk#R_-9ZoR zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<5=xC6gDY2xVP8>x`)pa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02RIPjf!YxlUGjW)Dx^E;fgNyw103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14n%jLdjCI8JExus z=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`2(H-dh_VS5WH&P+pK@aSJ103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sam40~7ALKTCJe13TaV2ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ek2f z^|hO~+_yUw(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4qC3#uvi8c42c|;0gC5uc2ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<872ddY*E}zl){S-OKB*Z~JPzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKy(KdE@&?5^i;Zo9@qf~IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-jy`+B*hYT; zEZsp5?0^Ft-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$Ah-kNCbzHtK{FN79rVBsIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJ22*${mvelzf&&VK@aSJ103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sam41MO3X zoHi}Lf0pi`2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4I1t`}Kh4^9aU=gv(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4f;&*PzA*EDnyHZPpa*uq0S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPj zfpXZ?Bc96dpQStKfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14n%jLy6V1d=XSpTEZsp5?0^Ft-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4wRe*FI0dVc>b-9ZoR zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<5=x&xC|J+YzFQ|S(RUzF-WZ{_#T(jD}`4miL84sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4qC0T#;OCznn%_T5 zchCbn-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC(e?!XlzmhRN)sdNWDumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#435?H@k!#bzp`JLrKOaDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_Kcc3Z#4sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|_V|9IrZw~T zpQStKfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14g_~#;p(f8?deH{bO$}K0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nj;_9z5#F?);r@=?;2e2OQu42ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2(H$7q z=b@+T`TetW2R*O@4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKYA64%BA-*MQTTsgUlV2X?>#4sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|^2rfbj&0=cKTCJe13TaV z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9Ek3~s)gUWJAePVl25j~6NuBr4(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4f;&(yx^t&na!;i@=z$$@fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S-iWpn7@k zX@Bg^-+z|wpa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02RIPjfz?Oe_;NG9f0pi`2X?>#4sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t={_I00c)^bm!JLrKOaDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_K zcc7g7kKf&q-#<%t&;vW*00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70100C%Ky}ZVv%ghOg>(l!umcWofCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f$$Do|H;g2d-DGz-9ZoR zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<5=xC7<+n_YTI?x}PKJ+K1~aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0Wz=7xv)DB5k4R5AGx`Q6r0S7q10S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$bO)+=zus&2{Qg_&9@qf~IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW- zZn}5n-&*-qh&bO$}K0}gP2103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii;0~0_r%k#&_f)!r9@qf~ zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z;6QW-s%Nh`bkm;v{b%V8dSC|}-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCIrDsJ;L6%(uEzA>Ba_?0^Ft-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4w9fy<69>-qc7 z(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4f;&*2u=B^a=AKG-&;vW*00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<700~`qNKsxNo;e+%4B;7#|?0^Ft-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ah-k73-3?7 zsg(-p4tih*9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW5R9q797-ov+Rrb4=d9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-s_kF=`H0TxmhPYjcEAA+aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0W5Z!@MTOD=YmHGX%bO$}K0}gP2103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nj+v`fPDiJ-<_z?w|*D zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%e_+=1pTr%ZaFHx<$y^uP`{zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00*KwP+rzp{ojN0_n)OZ=z$$@fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S-iW;OOxWyxz+1 zpQStKfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14g`0g{qf!(-kp0Y-9ZoRfCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<5=x&ze@mmm37Zz`la=z$$@fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<(BVA30# zel~yqxsvXn2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4I1t@|lb^Wx4~_i(S-OKB*Z~JPzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKy(NC^xJOhW`6%H-9ZoRfCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<5= zxC8AaUE8e8J(cdD2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4I1t@|YM0F)J-(j5|18}>5A1*g9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Hz^UwZV&%^LarvvdbN zumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h0f#?ogzR~4VJMZ+RJLrKOaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_Kci`dW6X$g2_s`ND^uP`{zyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00)9Q@Ybt0jD9!2 z)0gg`2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4I1t@|cH_q@C-#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|_3Q4a*Yo>l=?;2e2OQu42ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2!5#SY zU(f#Rf%%=jbO$}K0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROii=nho>oc5Xh2j%ZSOLx!%JKz8ZIKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0;7Pi|0N*Fu#A6?w|*D zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%e_+=1G~WB1y=o(ky>dSC|}-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCJGTD0dt%WmQiqq&w(=9dLjH9N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8M0cS2^YRlv$nT$} zJLrKOaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_Kcc82NyOnp>Qz6|!5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8a3Hz^<#vaCb!C44EZsp5?0^Ft-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$Ai4vmTzJR! zT|KFg?w|*DzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%e_-GOS(vq(fB#v!gC5uc2ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<832g+&x*m%?YopR|8dSC|} z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCJ$jxOCO=UHP4UDcwO2?0^Ft-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$Ai4v?&p6}3PEVyf=z$$@fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S*LrpnB*>J2r==Lb`(< z*Z~JPzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)hKzIi>tm{)7lK&^^4tih*9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW5R9r*h<&gp68_s`ND^uP`{zyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*Kwu|EZsp5?0^Ft-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$Ah-i#`j6W0f>tV|JLrKOaDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_~cVO@COWy0ef0pi`2X?>#4sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t={ z8K>TJ)ymvc=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`2(H*FrHhs;e%~VKt&;vW*00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70100C%K(+oy=dNj_Lb`(<*Z~JP zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h zKy(Lc9dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl81b5)sZD(xpTK@jCbO$}K0}gP2103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nj#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4I1t@|YNL#4sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@| zWm_IFv!35SOLx!%JKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G2<||8qgyYWH8_8#Te^cD*Z~JPzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)hKyU}j#a)fpTB(rkpa*uq z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2RIPkfeT*x{T|KyKS_7c13TaV2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29Ek3~n2gt6SXqLM#7I(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4f;&*oT(#+k|Nc(7bcd9`*RBV8 zpa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K$TnDOhw{5>u%in*N?w|*DzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%e_+=1OrnsxoVxu?<{^uP`{zyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwP`x^K$7{Ru z_n)OZ=z$$@fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S*Lrp!{&&uYOWbg>(l!umcWofCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#?o&m#ZgNJ*klHpa*uq0S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPjfv)4W zyL$IVDx^E;fgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14n%k0ibodhKc$`u=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2(H-cz;Mqst9+cmyOLx!%JKz8Z zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z2<||&==S2SRw|@B=z$$@fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S-iWpxpo4xBR}53h54dU-+z|wpa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02RIPjff@IIaqpk?q(Zub9@qf~IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-%7dz*eS1?O z-9ZoRfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<5=x&u}3QzuSqr9!%c9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-${&6Gpq~xN@6@F`=z$$@fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S*Lrpn84& zZI^Yv|18}>5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8a3Hz^!$*JP&z+u1chCbn-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC%|?m+wWv+rNpOoem@J+K1~aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7xv z)E3Oy=?nS&vvdbNumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h0f#?nl>9hHZFV|Bc-9ZoRfCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<5=x&!6fg(l!umcWofCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#?pLvdh##->auW zx`Q6r0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K$bO*}uBR{-5fB#v!gC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N<872j05vv7sO3_s`ND^uP`{zyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwQ2q44 zYDqm6(jD}`4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4qB}6^y>s?CswWlF9rVBsIKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJ5UZg#4sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t}~ zIZw{{_xoq*4xPQ~fgb3A9dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4I1t!@Z*Td^4bA-fq&w(=9dLjH9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8M0eoMF>C9+`TetW2R*O@ z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKYA64piOq-rPC&RJwy6*Z~JPzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)hKzIixz5n`%&iiNS4tih*9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW5R9oW6!qlXX9@1LbR z=z$$@fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S*LrV9ZxWJy8x#g>(l!umcWofCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#?pjXMMbUcq%m^Zr@7gC5uc2ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<832g;`& zeqmg1{!X`a2R*O@4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKYA64vg7$>_7kBN`-U>J+K1~aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7xvlwV!nb>@)#{b%V8dSC|} z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCIrDs2-~gey*7c=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`2;T^c|z3<-9dH*ckK@aSJ103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sam41B)*H{<3<0|18}> z5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8a3Hz^)BD_WP^YKT9rVBsIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b1LJ8Os zed!K*UTiFiIXD&49rVBsIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJFx0gk8k?D-c(3;&;vW*00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70100C%z$5*4 ze&dLGDx^E;fgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14n%jL9C^moBbup@?w|*DzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GORJ-z~;8QX$_&9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G;6QK(YU{^5K6PL!q&w(=9dLjH9N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8M0cP({gqG7?8)Eh zmhPYjcEAA+aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0W5ZrH?@(! z|18}>5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8a3Hz^*FRr_hYSpQStKfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14up3g)&KLZA^Crj?w|*DzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_+<~(1$bF{wrb4=d9@qf~ zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z;6Qi>-W#{gR~q?$lJ1}fcEAA+aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0W5Zr;P|C)^_)l(teK@aSJ103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sam61JA7Z=Fy$^&(a2gM{8burj`FE=?;2e2OQu4 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z!5wH{{Mep{H&Y?qK@aSJ103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sam60~_yq$*x25|0LZ(5A1*g9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Ht?)zUSiKB=cdx`Q6r z0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K$bO*|Jk85qzOoem@J+K1~aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0Wz=7xvRNq=YZeHj6&(afYh;UGw{A=?;2e2OQu42ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`2(H&Uzvp)0pZKOiFgC5uc2ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<872igzMUb0z! z|18}>5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8a3Hz^<(eB0@0Z^{OLx!%JKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3Gi0(jjO*M2#GZoSu^uP`{zyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwP_B4$ z%p0BWKTCJe13TaV2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29Ek3~MdQCZH-G=Rl9dLjH z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zM0eoO>o0t)mES*0chCbn-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC%|?!b(H%|EtXPla>`J+K1~aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=7xvw2vC`^N;d7ed!K* zUz&?w|*DzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GTNwb1r(mk>9CHchCbn-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC%|?m&6LDSQ4u zyY|o0MjXZg{FfL9oeBo$nB_DcMN0`*gl3Q~#S)OBLLj7i!9qyViwGh@mrflWx=07# zK>Vq7Z0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROii!aGoI{)<<$|D@`J9GC+RaDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0Wz=480Fguve zs!=Pb`XC49fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<7W=nm9>RIYBbefp|C$bmWF00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<700~{#217GfUJAWFjpz4Dhm;(-QfCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0fucK5 z{D9Z zy}=~Ou1`*T<0Ly@efMT#v$+wk#kH&2%hqD{3Gw{f$M>o}$bmWF00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f&cB)!D`e%7y#hJwtLW` z1@{4>msl)A|>{B*OsooD;myRskZ-AQ@3DDMsKlp94k z7)_>=YSl?-U;-AXJ_z6Yp+06&dEIxewc>6SqOD9>> z)BgXQ#Fy<>Od`L~UOHQhH#=7z&7O?2R4*U@?s+rIQ$HNNszdKZRfm+=hshT zuQpoT)VKDgH1kIgrHp49c^zEpzFBV_j{QKu4JKyJ#Cu`B|thCbY<#_y# K{o0SC*!}@*8kyz* literal 0 HcmV?d00001 diff --git a/results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2025-08-03_2026-02-18.npz b/results/linear_market_noise/_sim_arrays/0x9d1fcf346ea1b0_2025-08-03_2026-02-18.npz new file mode 100644 index 0000000000000000000000000000000000000000..03413c971dd7d22862ae47280993d2c7cf9ffe60 GIT binary patch literal 4609046 zcmeF(`?J?|od@tY2O;xT8L3!99I)>`~KR0&`+-)9?Uqe(QL+f zexJ|BPw)B6m^N^)A?dGwhNp|(Tkz+Lsv}bCuLIJM)Vpv&U(cN8+`gXCy^GEtkUstK z|I+%mTmMGu&yP-;Ieo_I1Jbf|L1kV~-Qm4hDG z0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K$cn3~Bp zeq-NSe*f&sK@aSJ103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sam416K{aYIf_LzAFbkumcWofCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#?oAy!q61o%#Kt{n8h z4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4!aMLx@u`2R<=@GbgC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N<872TnM>c%s!)R}Ok$2OQu42ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2(H+?J;$78he*f&s zK@aSJ103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sam21AG1Pf_;w9@AO?c=z$$@fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S-iWpnPM|CyyPKzyIvYK@aSJ103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sam41Kq2y zdwfKG|Ln>^5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8a3Ht?9jm8JIJjCQR}Ok$2OQu42ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2(H&?WI{b|LD@AhUpa*uq0S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPj zf%3u4XTO`@Kf7|!13TaV2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29Ek2fTmNsj-din_D+fKW0}gP2103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=ngawnsMUR{QlXMgC5uc z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N<872bQe5>)^IZkz6_GfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14n%jLTzA#;zpod`m4hDG0S7q10S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$cn2z*N3cEAA+aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W5Zr}k13KX2R*O@4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKYAM4!rl`;EvJxcXH*R2X?>#4sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t@|KYjheN+Z92cIBW4cEAA+ zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W z5Z!@IZ%=-qmft_Sa?k@i-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC%|?m%-*d+%xWBDr$V13TaV2ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ek2fxpvCCFX#8qt{n8h z4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4!aMNFLr-1YdjIUoK@aSJ103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sam215?{49dmJ`NUj|8zz#UT0S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02f{mWROe;ywBA3v za?k@i-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC%|?!d~v`+mGN_tcex9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G;6QW-I+iZpzIVMyt{n8h4miL84sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4qB~ITx@5)n zT9I5i=z$$@fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S-iWpyT9SZ+@@!{byGWdSC|}-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCIrD*g0h7{;%fmKf7|!13TaV2ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29Ek2f z^E+#I_jVV_m4hDG0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K$bO$P2$ppa*uq0S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02RIPjfpWpo8#j;2-+y-H zpa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02RIPjflUV-cWE`he|F`d2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4I1t={{(-lyy0SZer`wf-9@qf~IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QK(ns>C- zUTqY~m4hDG0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K$cn2Dx^mD1JKz8ZIKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G2<||6 z%Hr!@trp3ZgC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N<8B2kxG+d~NIfvnvNZumcWofCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0f#41_Z@=-B$0|i~<)8<4zyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_ z-GOq|s3`|^62%x^mD1JKz8Z zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G zi0(l7%Ctk*cINLtyK>M2JKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G2<||0&rQGh=W3B$Ip~2MaDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_Kcc8PmZB|(+k}C&2 zumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h0f#?pj^&Iq*FV>3W%0Un8fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<5=x&v1~a@(J$2=v2X?>#4sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t={{$&^ZdSksv zt{n8h4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4qC3!W^SUp5F28?v<)8<4zyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GR}A_j~@OYLQ$y=z$$@fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S-iWpt<9g zwolcHko_m4hDG0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K$bO*}!zTCN{^-kZFgC5uc2ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<832Rfz=U4O)={7&7K zgC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N<832g=Heum49afB)H)gC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N<872liig<$2Zo{@Im-9@qf~IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G;6QK(I_~+| z0hi@>`mP-Gzz#UT0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02f{nB?ef>BwcbCwa?k@i-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC%|?m+)7M=of~J$2=v2X?>#4sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4I1t`} zEeCIUU}XNCTsi209dLjH9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl81b3iZF@DS1Mv+`O=z$$@fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S-iW;Fbq=Y`?8mBv%f4Ukpa*(j2OQu42ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14g_{!-T^P)P|yE;t{n8h4miL84sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4qC0T+ z)E(8X{QlXMgC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N<832g=Tyb|02|>dHY6?0^Ft-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$AiM*o{pGbITJN7-Ip~2MaDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_K zci@Y|9-Y~p-#@!@&;vW*00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<700~`qMz|^lydZIa|NUj|8zz#UT0S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ckRBf8~dp$2W@P%0Un8 zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<5=x&zHYQ@*sKRwP#rdSC|}-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCJGTC?9$FH^a_gQ6yImdSC|}-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCJGTXr8#M z?flXC`_HZ%^uP`{zyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00)9QP(Iet{cOERt{n8h4miL84sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4!aMM*ch0@5_5RtFgC5uc2ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<87 z2X0&aolVvJ{@Im-9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G;6QW-77xDfgjP>oIp~2MaDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS^fcVJa-dHCG?opM(WdSC|} z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCJGT=-;?z)@k{jzAFbkumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h0f#?pD)wj;6cNfW(gC5uc2ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<872ksy6_}&Y3XU}is-^rDO z9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G;6QK(%CplQW1KR&l#cmADRIp~2MaDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS^fccA%yLl3(= z_tcex9@qf~IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G;6Qi>hK(3{OfCOTt{n8h4miL84sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4qC0T$`#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4I1t@|=EzrfT-%+$|Ln>^5A1*g9N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8a3Ht?<>cYd53c5S>aHC0zz#UT z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ZB4${9t^~>{|Z*vnvNZumcWofCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h0f#?og|3Bq78~OdSD+fKW0}gP2103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROii=nnj0oi^uP`{zyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00*KwP+tGbU-UNe_n%!k=z$$@ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S-iWVAYyGKhx@|D+fKW0}gP2103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROii;0|;wKlSP#)be+_T{-B19dLjH9N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl81b5)TjW>0_QZ14z z2R*O@4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKYAE4wN5`oxVH2e|F`d2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4I1t@|=JzlC&t;V&xpL40JKz8ZIKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G2=BoA+n;}? zk$)#w4tih*9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW5B9q3>E*pV~qMRMh!2X?>#4sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4I1t`}L6a{#d`$kGTsi209dLjH9N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl81b3j^ zxP8(`)grla&;vW*00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70100C%K=bXM#=v@!Tsi209dLjH9N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8M0cS4ht1P(YJLCNm4hDG0S7q1 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K$ za0mLIJ?Yw)Dn)YTpa*uq0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02RIPjf#xA!yYSu4{GD!B4tih*9N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW5B9Vic4dDnlBDv~P) zJ+K1~aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0Wz=7}%4D9^PO(XK}V@Grs%K$oxCGa?k@i-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC(e?!by!d(Nrm_s^~z^uP`{zyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00)9Q z(BHG~>^=1&xpL40JKz8ZIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3Gi0(kydC233=l9R99Q42rIKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b1rJ8=IsgV#^470H!@9@qf~ zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z;6QW-`X64iVc-1z*_DGH*Z~JPzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)hKy(M1+i#pXEWdwt<)8<4zyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%e_-GTDja?I#@kz6_G zfgNyw103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14n%jLxn=j%*IVCzcIBW4cEAA+aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0W5Z!_ApM2tV`TNgJR}Ok$2OQu42ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2!5t`f9dUT0 zJAbF!m4hDG0S7q10S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K$bO%nle$``*{QlXMgC5uc2ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N<832UdQt{>=Vrkz6_GfgNyw103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14g_~# z#8eR2`98e|mkKI8NOX<53UGOwra%q5jc?UkeFPpEXYSLQEVvUJJZ-Z=}G%r(zK-jx{%jf#fdS z9sv>Y5CnUz@>8t*0~Y=fvA1z1mk>8tX=CxsGCRAo`_Aq=GdvvN00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh1 z4sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4 zIKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G z-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$ zfCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h0 z0S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K0 z2ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`2 z9N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8 zaDW3G-~b0WzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0W zzyS_$fCC)h00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h z00%h00S<70103K02ROh14sd`29N+*4IKTl8aDW3G-~b0WzyS_$fCC)h00%h00S<70 z103K02ROh14sd`29N+*4IKTl8aDW3G;K1KF(CaQPwd3YV<#zV%L0J9}uTs64wX^Bz zQU5qpqpi`&t2|ph?@w>0=c#<$d-kZepJ%7p>v9n4gGsqtl=p`_ZlH}zkXSrgm}Mkce8j?{8PCRf1zkSzxS}2#oM!Y+m~6aTx3yC2mf;tpY~fZ ziF~2Gc(oX_pNHRP8+Wo)Z=Fx~{4~o`I~pI>q5q<)L(1!V?E1>r&zG@ljo2@7w0Ue= z+=Fo#4*xVjb0=JlFw5UR?lktr^-niv|M$r5lX>LDTJ*Zht@L;)9zSBc`m__zzW^&1 B@I3$k literal 0 HcmV?d00001 From 8331b7f0ec284da37e61773926abcffef96bd554 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Sun, 29 Mar 2026 16:39:18 +0100 Subject: [PATCH 074/115] diagnostics --- .codex | 0 .../compare_reclamm_geometric_noise_runs.py | 13 + scripts/compare_reclamm_thermostats.py | 1075 ++++++++++---- .../compare_reclamm_geometric_noise_runs.py | 674 +++++++++ .../reclamm/compare_reclamm_thermostats.py | 1075 ++++++++++---- .../reclamm/find_adjacent_heatmap_pairs.py | 1264 +++++++++++++++++ ...st_compare_reclamm_geometric_noise_runs.py | 143 ++ .../test_compare_reclamm_thermostats.py | 404 +++++- .../test_find_adjacent_heatmap_pairs.py | 313 ++++ 9 files changed, 4430 insertions(+), 531 deletions(-) create mode 100644 .codex create mode 100644 scripts/compare_reclamm_geometric_noise_runs.py create mode 100644 scripts/reclamm/compare_reclamm_geometric_noise_runs.py create mode 100644 scripts/reclamm/find_adjacent_heatmap_pairs.py create mode 100644 tests/scripts/test_compare_reclamm_geometric_noise_runs.py create mode 100644 tests/scripts/test_find_adjacent_heatmap_pairs.py diff --git a/.codex b/.codex new file mode 100644 index 00000000..e69de29b diff --git a/scripts/compare_reclamm_geometric_noise_runs.py b/scripts/compare_reclamm_geometric_noise_runs.py new file mode 100644 index 00000000..96230ed4 --- /dev/null +++ b/scripts/compare_reclamm_geometric_noise_runs.py @@ -0,0 +1,13 @@ +"""Wrapper for the canonical reCLAMM geometric noise comparison script.""" + +from __future__ import annotations + +import runpy +from pathlib import Path + + +if __name__ == "__main__": + runpy.run_path( + str(Path(__file__).with_name("reclamm") / "compare_reclamm_geometric_noise_runs.py"), + run_name="__main__", + ) diff --git a/scripts/compare_reclamm_thermostats.py b/scripts/compare_reclamm_thermostats.py index be8ee5f0..21352a4b 100644 --- a/scripts/compare_reclamm_thermostats.py +++ b/scripts/compare_reclamm_thermostats.py @@ -16,6 +16,7 @@ """ import gc +import hashlib import math import os @@ -41,17 +42,25 @@ def to_daily_price_shift_base(daily_price_shift_exponent): return 1.0 - daily_price_shift_exponent / 124649.0 -RUN_CONSTANT_ARC_LENGTH = False +def build_inclusive_sweep(start, stop, step): + """Build a sweep that keeps the requested step and explicitly includes the stop.""" + values = np.arange(start, stop + 1.0e-12, step, dtype=float) + if values.size == 0 or not np.isclose(values[-1], stop): + values = np.append(values, float(stop)) + return values + + +RUN_CONSTANT_ARC_LENGTH = True INTERPOLATION_METHODS = ( ("geometric", "constant_arc_length") if RUN_CONSTANT_ARC_LENGTH else ("geometric",) ) -HEATMAP_PRICE_RATIOS = np.arange(1.01, 1.50 + 1e-9, 0.025) -HEATMAP_MARGINS = np.linspace(0.05, 0.90, 20) -HEATMAP_SHIFT_EXPONENTS = np.arange(0.01, 0.50 + 1e-9, 0.025) +HEATMAP_PRICE_RATIOS = build_inclusive_sweep(1.01, 3.00, 0.025) +HEATMAP_MARGINS = np.linspace(0.05, 0.90, 39) +HEATMAP_SHIFT_EXPONENTS = build_inclusive_sweep(0.01, 0.50, 0.0125) HEATMAP_ARC_LENGTH_SPEEDS = np.geomspace(1.0e-6, 5.0e-4, 11) -PRICE_RATIO_TICKS = np.array([1.01, 1.10, 1.20, 1.30, 1.40, 1.50]) +PRICE_RATIO_TICKS = np.array([1.01, 1.25, 1.50, 2.00, 2.50, 3.00]) MARGIN_TICKS = np.array([0.05, 0.15, 0.25, 0.35, 0.45, 0.55, 0.65, 0.75, 0.85, 0.90]) SHIFT_EXPONENT_TICKS = np.array([0.01, 0.05, 0.10, 0.20, 0.30, 0.40, 0.50]) ARC_LENGTH_SPEED_TICKS = np.array([ @@ -76,6 +85,17 @@ def to_daily_price_shift_base(daily_price_shift_exponent): CENTER_ZERO_HEATMAP_COLOR_NORM = "symlog" CENTER_ZERO_HEATMAP_COLOR_TAG = "symlog20" CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH = 20.0 +FIXED_SLICE_FRACTIONS = (0.125, 0.375, 0.625, 0.875) +FIXED_SLICE_LABELS = ("Q1", "Q2", "Q3", "Q4") +THREE_D_VIEW_ELEVATION = 22.0 +THREE_D_VIEW_AZIMUTH = 140.0 +HEATMAP_FORWARD_CACHE_ENABLED = True +HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" +HEATMAP_FORWARD_CACHE_ROOT = os.path.join( + "results", + "reclamm_heatmap_forward_cache", +) +HEATMAP_FORWARD_CACHE_FLUSH_EVERY = 32 AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" @@ -94,14 +114,28 @@ def to_daily_price_shift_base(daily_price_shift_exponent): ] LEGACY_LOG_CADENCE = 2.68 LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) +FIXED_COMPARE_ARB_FREQUENCY = LEGACY_ARB_FREQUENCY AAVE_ETH_NOISE_SETTINGS = { "enable_noise_model": True, "noise_model": DEFAULT_NOISE_MODEL, "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, "noise_pool_id": AAVE_WETH_POOL_ID, + "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, "gas_cost": DEFAULT_GAS_COST, "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, } +PERSISTED_FORWARD_VALUE_COLUMNS = ( + "cache_key_hash", + "final_value", + "method", + "enable_noise_model", + "noise_model", + "price_ratio", + "centeredness_margin", + "daily_price_shift_exponent", + "initial_pool_value", + "arb_frequency", +) GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS = ( "geometric_vs_launch_geometric_pct", @@ -175,6 +209,23 @@ def heatmap_artifact_filename(spec, cfg, suffix=None): return tvl_artifact_filename(stem, cfg, suffix=suffix) +def three_d_heatmap_artifact_filename(spec, cfg, suffix=None): + """Build a 3D heatmap filename, including any colour-style tag.""" + stem = f"reclamm_heatmap_3d_{spec['slug']}" + artifact_tag = spec.get("artifact_tag") + if artifact_tag: + stem = f"{stem}_{artifact_tag}" + return tvl_artifact_filename(stem, cfg, suffix=suffix) + + +def format_heatmap_param_value(value): + """Format a sweep parameter compactly for titles and logs.""" + value = float(value) + if abs(value) >= 1.0: + return f"{value:.2f}".rstrip("0").rstrip(".") + return f"{value:.3f}".rstrip("0").rstrip(".") + + def configs_for_tvl(base_configs, initial_pool_value): """Attach a shared initial TVL to each compare configuration.""" configs = [] @@ -185,35 +236,76 @@ def configs_for_tvl(base_configs, initial_pool_value): return configs -def make_noise_variant_cfg(cfg, enable_noise_model): - """Return a config with either noise modelling or pure arb-only enabled.""" +def _normalize_arb_frequency(value, default=FIXED_COMPARE_ARB_FREQUENCY): + """Return a stable integer arb cadence for thermostat comparisons.""" + if value is None: + if default is None: + return None + value = default + return max(int(round(float(value))), 1) + + +def get_effective_arb_frequency(cfg, noise_cfg=None): + """Resolve the arb cadence used by a thermostat comparison run.""" + del noise_cfg + return _normalize_arb_frequency(FIXED_COMPARE_ARB_FREQUENCY) + + +def normalize_compare_run_cfg(cfg, enable_noise_model=None): + """Canonicalize the compare-run config so non-axis inputs stay fixed.""" updated = dict(cfg) - if enable_noise_model: - updated["enable_noise_model"] = True - return updated + updated["price_ratio"] = float(cfg["price_ratio"]) + updated["centeredness_margin"] = float(cfg["centeredness_margin"]) + updated["daily_price_shift_exponent"] = float(cfg["daily_price_shift_exponent"]) + updated["initial_pool_value"] = float(get_initial_pool_value(cfg)) + updated["gas_cost"] = DEFAULT_GAS_COST + updated["protocol_fee_split"] = DEFAULT_PROTOCOL_FEE_SPLIT + updated["arb_fees"] = 0.0 + updated["arb_frequency"] = get_effective_arb_frequency(cfg) + updated["noise_trader_ratio"] = 0.0 - matched_noise = resolve_reclamm_noise_settings(cfg) + arc_length_speed = cfg.get("arc_length_speed") + if arc_length_speed is None: + updated.pop("arc_length_speed", None) + else: + updated["arc_length_speed"] = float(arc_length_speed) - updated["enable_noise_model"] = False - updated["noise_model"] = None - updated["gas_cost"] = cfg.get("gas_cost", DEFAULT_GAS_COST) - updated["protocol_fee_split"] = cfg.get( - "protocol_fee_split", DEFAULT_PROTOCOL_FEE_SPLIT + use_noise = ( + bool(cfg.get("enable_noise_model", False)) + if enable_noise_model is None + else bool(enable_noise_model) ) - updated["noise_trader_ratio"] = 0.0 - matched_arb_frequency = matched_noise.get("arb_frequency") - if matched_arb_frequency is not None: - updated["arb_frequency"] = matched_arb_frequency - for key in ( - "reclamm_noise_params", - "noise_arrays_path", - "noise_artifact_dir", - "noise_pool_id", - ): - updated.pop(key, None) + updated["enable_noise_model"] = use_noise + + requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + if use_noise: + updated["noise_model"] = requested_mode + if requested_mode == "market_linear": + updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + updated["noise_pool_id"] = AAVE_WETH_POOL_ID + else: + updated.pop("noise_artifact_dir", None) + updated.pop("noise_pool_id", None) + updated.pop("reclamm_noise_params", None) + updated.pop("noise_arrays_path", None) + else: + updated["noise_model"] = None + for key in ( + "reclamm_noise_params", + "noise_arrays_path", + "noise_artifact_dir", + "noise_pool_id", + ): + updated.pop(key, None) + return updated +def make_noise_variant_cfg(cfg, enable_noise_model): + """Return a config with either noise modelling or pure arb-only enabled.""" + return normalize_compare_run_cfg(cfg, enable_noise_model=enable_noise_model) + + def _warn_noise_fallback(message): """Print a one-time message when the preferred noise setup is unavailable.""" if message not in _WARNED_NOISE_FALLBACKS: @@ -228,36 +320,39 @@ def _hashable_noise_params(params): return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) -def _legacy_calibrated_noise_settings(reason=None): +def _legacy_calibrated_noise_settings(reason=None, arb_frequency=None): """Fallback calibrated noise config used when market-linear artifacts are absent.""" if reason: _warn_noise_fallback( "market_linear noise unavailable for thermostat comparison; " f"falling back to calibrated legacy coefficients ({reason})." ) + arb_frequency = _normalize_arb_frequency(arb_frequency) return { "noise_model": "calibrated", "noise_trader_ratio": 0.0, "reclamm_noise_params": { f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) }, - "arb_frequency": LEGACY_ARB_FREQUENCY, + "arb_frequency": arb_frequency, "noise_summary": ( "calibrated legacy 8-covariate " - f"(arb_frequency={LEGACY_ARB_FREQUENCY})" + f"(arb_frequency={arb_frequency})" ), "noise_cache_key": ( "calibrated", tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), - LEGACY_ARB_FREQUENCY, + arb_frequency, ), } def resolve_reclamm_noise_settings(cfg): """Resolve the active reCLAMM noise-model fingerprint block for a config.""" + cfg = normalize_compare_run_cfg(cfg) enable_noise_model = cfg.get("enable_noise_model", False) requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + requested_arb_frequency = get_effective_arb_frequency(cfg) cache_key = ( tuple(cfg.get("tokens", [])), cfg.get("start"), @@ -266,7 +361,7 @@ def resolve_reclamm_noise_settings(cfg): requested_mode, cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), - cfg.get("arb_frequency"), + requested_arb_frequency, round(float(cfg.get("noise_trader_ratio", 0.0)), 12), _hashable_noise_params(cfg.get("reclamm_noise_params")), cfg.get("noise_arrays_path"), @@ -280,7 +375,7 @@ def resolve_reclamm_noise_settings(cfg): "noise_trader_ratio": 0.0, "reclamm_noise_params": None, "noise_arrays_path": None, - "arb_frequency": None, + "arb_frequency": requested_arb_frequency, "noise_summary": "arb-only (noise disabled)", "noise_cache_key": ("disabled",), } @@ -290,11 +385,7 @@ def resolve_reclamm_noise_settings(cfg): start_date = str(cfg["start"]).split(" ")[0] end_date = str(cfg["end"]).split(" ")[0] try: - from quantammsim.calibration.noise_model_arrays import ( - _find_pool_index, - build_simulator_arrays, - load_artifact, - ) + from quantammsim.calibration.noise_model_arrays import build_simulator_arrays model_path = os.path.join(artifact_dir, "model.npz") meta_path = os.path.join(artifact_dir, "meta.json") @@ -311,6 +402,8 @@ def resolve_reclamm_noise_settings(cfg): ) if not os.path.exists(arrays_path): arrays = build_simulator_arrays( + token_a=cfg["tokens"][0], + token_b=cfg["tokens"][1], pool_id=pool_id, start_date=start_date, end_date=end_date, @@ -328,13 +421,7 @@ def resolve_reclamm_noise_settings(cfg): tvl_mean = float(arrays["tvl_mean"]) tvl_std = float(arrays["tvl_std"]) - art, meta = load_artifact(artifact_dir) - pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) - if pool_idx >= 0: - learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) - else: - learned_cadence = 5.0 - arb_frequency = max(1, round(learned_cadence)) + arb_frequency = requested_arb_frequency result = { "noise_model": "market_linear", "noise_trader_ratio": 0.0, @@ -351,30 +438,19 @@ def resolve_reclamm_noise_settings(cfg): arb_frequency, round(tvl_mean, 12), round(tvl_std, 12), - ), - } + ), + } except Exception as exc: # pragma: no cover - fallback path depends on local artifacts - result = _legacy_calibrated_noise_settings(str(exc)) + result = _legacy_calibrated_noise_settings( + str(exc), + arb_frequency=requested_arb_frequency, + ) elif requested_mode == "calibrated": - params = cfg.get("reclamm_noise_params") - if params is None: - result = _legacy_calibrated_noise_settings() - else: - arb_frequency = cfg.get("arb_frequency", LEGACY_ARB_FREQUENCY) - result = { - "noise_model": "calibrated", - "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), - "reclamm_noise_params": dict(params), - "arb_frequency": arb_frequency, - "noise_summary": f"calibrated (arb_frequency={arb_frequency})", - "noise_cache_key": ( - "calibrated", - _hashable_noise_params(params), - arb_frequency, - ), - } + result = _legacy_calibrated_noise_settings( + arb_frequency=requested_arb_frequency + ) else: - arb_frequency = cfg.get("arb_frequency") + arb_frequency = requested_arb_frequency result = { "noise_model": requested_mode, "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), @@ -430,13 +506,14 @@ def resolve_reclamm_noise_settings(cfg): def make_fingerprint(cfg, interpolation_method): """Build run fingerprint for a given config and interpolation method.""" + cfg = normalize_compare_run_cfg(cfg) speed_override = ( cfg.get("arc_length_speed") if interpolation_method == "constant_arc_length" else None ) noise_cfg = resolve_reclamm_noise_settings(cfg) - arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) fingerprint = { "tokens": cfg["tokens"], "rule": "reclamm", @@ -471,6 +548,7 @@ def make_fingerprint(cfg, interpolation_method): def make_params(cfg): """Build pool params from config.""" + cfg = normalize_compare_run_cfg(cfg) return { "price_ratio": jnp.array(cfg["price_ratio"]), "centeredness_margin": jnp.array(cfg["centeredness_margin"]), @@ -528,7 +606,7 @@ def _set_padded_ylim(ax, series_list, pad_ratio=0.04): def _cache_size(cache): - """Count memoized final-value runs.""" + """Count memoized final-value cache entries materialised in memory.""" return len(cache.get("_final_value_cache", {})) @@ -537,13 +615,140 @@ def _comparison_cache_size(cache): return len(cache.get("_comparison_cache", {})) -def make_sweep_cache(price_data): - """Create a shared cache for heatmap and line sweeps.""" +def _heatmap_forward_cache_scope_slug(cfg): + """Build a compact cache scope slug for a shared-TVL heatmap run.""" + if cfg is None: + return "unspecified_tvl" + return f"tvl_{format_tvl_millions_slug(cfg)}" + + +def _heatmap_forward_cache_path(cfg): + """Return the parquet path for persisted scalar forward values.""" + if not HEATMAP_FORWARD_CACHE_ENABLED: + return None + return os.path.join( + HEATMAP_FORWARD_CACHE_ROOT, + HEATMAP_FORWARD_CACHE_RUN_NAME, + f"forward_values_{_heatmap_forward_cache_scope_slug(cfg)}.parquet", + ) + + +def _make_method_cache_hash(key): + """Build a compact stable digest for a method cache key.""" + return hashlib.sha256(repr(key).encode("utf-8")).hexdigest() + + +def _build_persistent_final_value_record(cfg, method, cache_key_hash, final_value): + """Build one self-describing parquet row for a cached scalar run result.""" + cfg = normalize_compare_run_cfg(cfg) + noise_cfg = resolve_reclamm_noise_settings(cfg) return { + "cache_key_hash": str(cache_key_hash), + "final_value": float(final_value), + "method": str(method), + "enable_noise_model": bool(cfg.get("enable_noise_model", False)), + "noise_model": noise_cfg.get("noise_model"), + "price_ratio": float(cfg["price_ratio"]), + "centeredness_margin": float(cfg["centeredness_margin"]), + "daily_price_shift_exponent": float(cfg["daily_price_shift_exponent"]), + "initial_pool_value": float(get_initial_pool_value(cfg)), + "arb_frequency": get_effective_arb_frequency(cfg, noise_cfg), + } + + +def _load_persistent_final_value_cache(cache): + """Load persisted scalar forward values from parquet once per sweep cache.""" + if cache.get("_persistent_final_value_cache_loaded"): + return + + disk_cache = {} + disk_records = {} + cache_path = cache.get("_persistent_final_value_cache_path") + if cache_path and os.path.exists(cache_path): + frame = pd.read_parquet(cache_path) + if not frame.empty: + for row in frame.itertuples(index=False): + cache_key_hash = str(row.cache_key_hash) + final_value = float(row.final_value) + disk_cache[cache_key_hash] = final_value + record = { + "cache_key_hash": cache_key_hash, + "final_value": final_value, + } + for column in PERSISTED_FORWARD_VALUE_COLUMNS: + if column in {"cache_key_hash", "final_value"}: + continue + record[column] = getattr(row, column, None) + disk_records[cache_key_hash] = record + print( + f"Loaded {len(disk_cache)} persisted heatmap forward values from {cache_path}" + ) + + cache["_persistent_final_value_cache"] = disk_cache + cache["_persistent_final_value_records"] = disk_records + cache["_persistent_final_value_cache_loaded"] = True + + +def flush_sweep_cache(cache, force=False): + """Persist newly computed scalar forward values to parquet.""" + if not HEATMAP_FORWARD_CACHE_ENABLED: + return + + pending = cache.get("_pending_persistent_final_values") + if not pending: + return + if not force and len(pending) < HEATMAP_FORWARD_CACHE_FLUSH_EVERY: + return + + _load_persistent_final_value_cache(cache) + disk_cache = cache.setdefault("_persistent_final_value_cache", {}) + disk_records = cache.setdefault("_persistent_final_value_records", {}) + for cache_key_hash, record in pending.items(): + merged = dict(disk_records.get(cache_key_hash, {})) + merged.update(record) + merged["cache_key_hash"] = str(cache_key_hash) + merged["final_value"] = float(merged["final_value"]) + disk_records[cache_key_hash] = merged + disk_cache[cache_key_hash] = merged["final_value"] + + cache_path = cache.get("_persistent_final_value_cache_path") + if cache_path is None: + pending.clear() + return + + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + sorted_records = [disk_records[key] for key in sorted(disk_records)] + payload = { + column: [record.get(column) for record in sorted_records] + for column in PERSISTED_FORWARD_VALUE_COLUMNS + } + payload["final_value"] = np.asarray(payload["final_value"], dtype=np.float64) + frame = pd.DataFrame(payload) + frame.sort_values("cache_key_hash", inplace=True, ignore_index=True) + frame.to_parquet(cache_path, index=False, compression="zstd") + print( + f"Persisted {len(pending)} new heatmap forward values to {cache_path} " + f"({len(disk_cache)} total cached values)." + ) + pending.clear() + + +def make_sweep_cache(price_data, cache_scope_cfg=None): + """Create a shared cache for heatmap and line sweeps.""" + cache = { "_shared_price_data": price_data, "_final_value_cache": {}, "_comparison_cache": {}, + "_pending_persistent_final_values": {}, + "_persistent_final_value_cache": {}, + "_persistent_final_value_records": {}, + "_persistent_final_value_cache_loaded": False, + "_persistent_final_value_cache_path": _heatmap_forward_cache_path( + cache_scope_cfg + ), } + _load_persistent_final_value_cache(cache) + return cache def _missing_artifacts(progress_label, filenames): @@ -571,8 +776,9 @@ def _speed_cache_key(speed): def _make_method_cache_key(cfg, method): """Cache key for a single-method final-value run.""" + cfg = normalize_compare_run_cfg(cfg) noise_cfg = resolve_reclamm_noise_settings(cfg) - arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) key = ( method, bool(cfg.get("enable_noise_model", False)), @@ -630,16 +836,34 @@ def _run_method_final_value_cached(cfg, method, cache): """Memoize final value for a single interpolation method.""" final_value_cache = cache.setdefault("_final_value_cache", {}) key = _make_method_cache_key(cfg, method) - if key not in final_value_cache: - result = do_run_on_historic_data( - run_fingerprint=make_fingerprint(cfg, method), - params=make_params(cfg), - price_data=cache["_shared_price_data"], - low_data_mode=True, + if key in final_value_cache: + return final_value_cache[key] + + _load_persistent_final_value_cache(cache) + key_hash = _make_method_cache_hash(key) + persisted_cache = cache.setdefault("_persistent_final_value_cache", {}) + if key_hash in persisted_cache: + final_value_cache[key] = persisted_cache[key_hash] + return final_value_cache[key] + + result = do_run_on_historic_data( + run_fingerprint=make_fingerprint(cfg, method), + params=make_params(cfg), + price_data=cache["_shared_price_data"], + low_data_mode=True, + ) + final_value_cache[key] = float(result["final_value"]) + cache.setdefault("_pending_persistent_final_values", {})[key_hash] = ( + _build_persistent_final_value_record( + cfg=cfg, + method=method, + cache_key_hash=key_hash, + final_value=final_value_cache[key], ) - final_value_cache[key] = float(result["final_value"]) - del result - gc.collect() + ) + flush_sweep_cache(cache, force=False) + del result + gc.collect() return final_value_cache[key] @@ -860,15 +1084,16 @@ def build_heatmap_matrices( data[metric_key][yi, xi] = metrics[metric_key] completed_points = (yi + 1) * len(x_values) - row_new_final_runs = _cache_size(cache) - final_cache_before_row + row_new_final_entries = _cache_size(cache) - final_cache_before_row row_new_comparisons = ( _comparison_cache_size(cache) - comparison_cache_before_row ) row_pct = completed_points / total_points * 100.0 + flush_sweep_cache(cache, force=True) print( f"[{progress_label}] row {yi + 1}/{len(y_values)} complete " f"({y_key}={float(y_value):.4f}, {completed_points}/{total_points} " - f"points, {row_pct:.1f}%, {row_new_final_runs} new final-value runs, " + f"points, {row_pct:.1f}%, {row_new_final_entries} new final-value cache entries, " f"{row_new_comparisons} new comparison bundles)" ) @@ -910,6 +1135,7 @@ def build_metric_curve( metric_keys=(metric_key,), ) data[xi] = metrics[metric_key] + flush_sweep_cache(cache, force=True) return data @@ -937,6 +1163,226 @@ def _compute_axis_edges(values, scale="linear"): return edges +def build_fixed_slice_variants(values): + """Pick four representative quarter-range slices from a sweep grid.""" + values = np.asarray(values, dtype=float) + if values.size < len(FIXED_SLICE_FRACTIONS): + raise ValueError("Need at least four grid points to build fixed slices") + + variants = [] + used_indices = set() + for idx, fraction in enumerate(FIXED_SLICE_FRACTIONS): + target_index = int(round(fraction * (values.size - 1))) + while target_index in used_indices and target_index + 1 < values.size: + target_index += 1 + while target_index in used_indices and target_index - 1 >= 0: + target_index -= 1 + if target_index in used_indices: + raise ValueError("Could not build four unique fixed slices from sweep grid") + used_indices.add(target_index) + variants.append( + { + "index": target_index, + "fraction": fraction, + "label": FIXED_SLICE_LABELS[idx], + "slug": f"q{idx + 1}", + "value": float(values[target_index]), + } + ) + return variants + + +def _pair_slice_suffix(pair, slice_variant): + """Build a stable artifact suffix for a pairwise fixed-variable slice.""" + return f"{pair['slug']}_{pair['fixed_slug']}_{slice_variant['slug']}" + + +def _build_heatmap_norm( + data_arrays, + center_zero, + color_norm=None, + symlog_linthresh=None, +): + """Build a color normalizer shared by 2D and 3D heatmaps.""" + finite_parts = [] + for data in data_arrays: + finite = np.asarray(data, dtype=float) + finite = finite[np.isfinite(finite)] + if finite.size: + finite_parts.append(finite) + finite = np.concatenate(finite_parts) if finite_parts else np.array([], dtype=float) + + if center_zero: + if finite.size == 0: + vmax = 1.0 + else: + vmax = max(abs(float(finite.min())), abs(float(finite.max())), 1e-9) + if ( + color_norm == "symlog" + and symlog_linthresh is not None + and vmax > symlog_linthresh + ): + return SymLogNorm( + linthresh=symlog_linthresh, + linscale=1.0, + vmin=-vmax, + vmax=vmax, + base=10.0, + ) + return TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) + + if finite.size == 0: + vmin, vmax = 0.0, 1.0 + else: + vmin = float(finite.min()) + vmax = float(finite.max()) + if np.isclose(vmin, vmax): + pad = max(abs(vmin) * 0.01, 1e-9) + vmin -= pad + vmax += pad + return Normalize(vmin=vmin, vmax=vmax) + + +def get_pair_heatmap_metric_specs(): + """Return the standard thermostat pairwise heatmap metrics.""" + metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs heatmap geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "launch_geometric_efficiency_pct", + "title": "Efficiency vs launch-style geometric", + "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", + "slug": "launch_geometric_efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "geometric_vs_launch_geometric_pct", + "title": "Geometric tuning vs launch-style geometric", + "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", + "slug": "geometric_vs_launch_geometric", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "constant_arc_vs_launch_constant_arc_pct", + "title": "Const arc tuning vs launch-style const arc", + "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", + "slug": "constant_arc_vs_launch_constant_arc", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_geometric_final_value_musd", + "title": "Geometric final value with noise model", + "colorbar_label": "Geometric final value with noise model ($M)", + "slug": "noise_geometric_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_geometric_improvement_pct", + "title": "Noise-model improvement over arb-only (geometric)", + "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", + "slug": "noise_vs_arb_geometric_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + if not RUN_CONSTANT_ARC_LENGTH: + metric_specs = [ + spec + for spec in metric_specs + if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS + ] + return metric_specs + + +def get_pair_heatmap_specs(base_cfg): + """Return the three pairwise thermostat heatmap families plus slice settings.""" + fixed_slice_variants = { + "price_ratio": build_fixed_slice_variants(HEATMAP_PRICE_RATIOS), + "centeredness_margin": build_fixed_slice_variants(HEATMAP_MARGINS), + "daily_price_shift_exponent": build_fixed_slice_variants( + HEATMAP_SHIFT_EXPONENTS + ), + } + return [ + { + "slug": "price_ratio_vs_margin", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_MARGINS, + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "x_label": "Price ratio", + "y_label": "Centeredness margin", + "xticks": PRICE_RATIO_TICKS, + "yticks": MARGIN_TICKS, + "fixed_key": "daily_price_shift_exponent", + "fixed_label": "Shift exponent", + "fixed_slug": "shift_exp", + "fixed_slices": fixed_slice_variants["daily_price_shift_exponent"], + }, + { + "slug": "shift_exp_vs_margin", + "x_values": HEATMAP_SHIFT_EXPONENTS, + "y_values": HEATMAP_MARGINS, + "x_key": "daily_price_shift_exponent", + "y_key": "centeredness_margin", + "x_label": "Shift exponent", + "y_label": "Centeredness margin", + "xticks": SHIFT_EXPONENT_TICKS, + "yticks": MARGIN_TICKS, + "fixed_key": "price_ratio", + "fixed_label": "Price ratio", + "fixed_slug": "price_ratio", + "fixed_slices": fixed_slice_variants["price_ratio"], + }, + { + "slug": "price_ratio_vs_shift_exp", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "price_ratio", + "y_key": "daily_price_shift_exponent", + "x_label": "Price ratio", + "y_label": "Shift exponent", + "xticks": PRICE_RATIO_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + "fixed_key": "centeredness_margin", + "fixed_label": "Centeredness margin", + "fixed_slug": "margin", + "fixed_slices": fixed_slice_variants["centeredness_margin"], + }, + ] + + def plot_heatmap( data, x_values, @@ -955,34 +1401,13 @@ def plot_heatmap( symlog_linthresh=None, ): """Render and save a single heatmap.""" - finite = np.asarray(data, dtype=float) - finite = finite[np.isfinite(finite)] - - if center_zero: - vmax = max(abs(float(np.nanmin(data))), abs(float(np.nanmax(data))), 1e-9) - if color_norm == "symlog" and symlog_linthresh is not None and vmax > symlog_linthresh: - norm = SymLogNorm( - linthresh=symlog_linthresh, - linscale=1.0, - vmin=-vmax, - vmax=vmax, - base=10.0, - ) - else: - norm = TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) - cmap_name = cmap or "RdYlGn" - else: - if finite.size == 0: - vmin, vmax = 0.0, 1.0 - else: - vmin = float(finite.min()) - vmax = float(finite.max()) - if np.isclose(vmin, vmax): - pad = max(abs(vmin) * 0.01, 1e-9) - vmin -= pad - vmax += pad - norm = Normalize(vmin=vmin, vmax=vmax) - cmap_name = cmap or "viridis" + norm = _build_heatmap_norm( + [data], + center_zero=center_zero, + color_norm=color_norm, + symlog_linthresh=symlog_linthresh, + ) + cmap_name = cmap or ("RdYlGn" if center_zero else "viridis") x_edges = _compute_axis_edges(x_values, scale=xscale) y_edges = _compute_axis_edges(y_values, scale="linear") @@ -1015,6 +1440,112 @@ def plot_heatmap( plt.close(fig) +def plot_three_variable_heatmap_3d( + price_margin_data, + shift_margin_data, + price_shift_data, + fixed_price_ratio, + fixed_margin, + fixed_shift_exponent, + title, + colorbar_label, + filename, + center_zero=True, + cmap=None, + color_norm=None, + symlog_linthresh=None, +): + """Render orthogonal 3D heatmap surfaces across the three thermostat variables.""" + norm = _build_heatmap_norm( + [price_margin_data, shift_margin_data, price_shift_data], + center_zero=center_zero, + color_norm=color_norm, + symlog_linthresh=symlog_linthresh, + ) + cmap_name = cmap or ("RdYlGn" if center_zero else "viridis") + cmap_obj = plt.get_cmap(cmap_name) + + price_margin_x, price_margin_y = np.meshgrid(HEATMAP_PRICE_RATIOS, HEATMAP_MARGINS) + price_margin_z = np.full_like(price_margin_x, fixed_shift_exponent, dtype=float) + + shift_margin_z, shift_margin_y = np.meshgrid( + HEATMAP_SHIFT_EXPONENTS, + HEATMAP_MARGINS, + ) + shift_margin_x = np.full_like(shift_margin_z, fixed_price_ratio, dtype=float) + + price_shift_x, price_shift_z = np.meshgrid( + HEATMAP_PRICE_RATIOS, + HEATMAP_SHIFT_EXPONENTS, + ) + price_shift_y = np.full_like(price_shift_x, fixed_margin, dtype=float) + + fig = plt.figure(figsize=(10.5, 7.2)) + ax = fig.add_subplot(111, projection="3d") + ax.set_facecolor("white") + fig.patch.set_facecolor("white") + + ax.plot_surface( + price_margin_x, + price_margin_y, + price_margin_z, + facecolors=cmap_obj(norm(np.asarray(price_margin_data, dtype=float))), + shade=False, + ) + ax.plot_surface( + shift_margin_x, + shift_margin_y, + shift_margin_z, + facecolors=cmap_obj(norm(np.asarray(shift_margin_data, dtype=float))), + shade=False, + ) + ax.plot_surface( + price_shift_x, + price_shift_y, + price_shift_z, + facecolors=cmap_obj(norm(np.asarray(price_shift_data, dtype=float))), + shade=False, + ) + + ax.set_xlim(float(HEATMAP_PRICE_RATIOS.min()), float(HEATMAP_PRICE_RATIOS.max())) + ax.set_ylim(float(HEATMAP_MARGINS.min()), float(HEATMAP_MARGINS.max())) + ax.set_zlim( + float(HEATMAP_SHIFT_EXPONENTS.min()), + float(HEATMAP_SHIFT_EXPONENTS.max()), + ) + ax.set_xlabel("Price ratio") + ax.set_ylabel("Centeredness margin") + ax.set_zlabel("Shift exponent") + ax.set_xticks(PRICE_RATIO_TICKS) + ax.set_yticks(MARGIN_TICKS[::2]) + ax.set_zticks(SHIFT_EXPONENT_TICKS) + ax.set_title(title) + ax.grid(False) + ax.view_init(elev=THREE_D_VIEW_ELEVATION, azim=THREE_D_VIEW_AZIMUTH) + try: + ax.set_box_aspect( + ( + float(HEATMAP_PRICE_RATIOS.max() - HEATMAP_PRICE_RATIOS.min()), + float(HEATMAP_MARGINS.max() - HEATMAP_MARGINS.min()), + float( + HEATMAP_SHIFT_EXPONENTS.max() - HEATMAP_SHIFT_EXPONENTS.min() + ), + ) + ) + except AttributeError: + pass + + sm = ScalarMappable(norm=norm, cmap=cmap_obj) + sm.set_array([]) + cbar = fig.colorbar(sm, ax=ax, fraction=0.03, pad=0.1, shrink=0.82) + cbar.set_label(colorbar_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + def plot_arc_speed_line_chart( data, x_values, @@ -1090,128 +1621,11 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): """Generate pairwise heatmaps for thermostat tuning and noise-vs-arb effects.""" owns_cache = cache is None if cache is None: - cache = make_sweep_cache(price_data) - metric_specs = [ - { - "key": "efficiency_pct", - "title": "Efficiency vs heatmap geometric", - "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", - "slug": "efficiency", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "launch_geometric_efficiency_pct", - "title": "Efficiency vs launch-style geometric", - "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", - "slug": "launch_geometric_efficiency", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "geometric_vs_launch_geometric_pct", - "title": "Geometric tuning vs launch-style geometric", - "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", - "slug": "geometric_vs_launch_geometric", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "constant_arc_vs_launch_constant_arc_pct", - "title": "Const arc tuning vs launch-style const arc", - "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", - "slug": "constant_arc_vs_launch_constant_arc", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "noise_geometric_final_value_musd", - "title": "Geometric final value with noise model", - "colorbar_label": "Geometric final value with noise model ($M)", - "slug": "noise_geometric_final_value", - "center_zero": False, - "cmap": "viridis", - }, - { - "key": "noise_constant_arc_final_value_musd", - "title": "Const arc final value with noise model", - "colorbar_label": "Const Arc final value with noise model ($M)", - "slug": "noise_constant_arc_final_value", - "center_zero": False, - "cmap": "viridis", - }, - { - "key": "noise_vs_arb_geometric_improvement_pct", - "title": "Noise-model improvement over arb-only (geometric)", - "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", - "slug": "noise_vs_arb_geometric_improvement", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "noise_vs_arb_constant_arc_improvement_pct", - "title": "Noise-model improvement over arb-only (const arc)", - "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", - "slug": "noise_vs_arb_constant_arc_improvement", - "center_zero": True, - "cmap": "RdYlGn", - }, - ] - for spec in metric_specs: - if spec["center_zero"]: - spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM - spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH - spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG - if not RUN_CONSTANT_ARC_LENGTH: - metric_specs = [ - spec - for spec in metric_specs - if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS - ] - pair_specs = [ - { - "slug": "price_ratio_vs_margin", - "x_values": HEATMAP_PRICE_RATIOS, - "y_values": HEATMAP_MARGINS, - "x_key": "price_ratio", - "y_key": "centeredness_margin", - "x_label": "Price ratio", - "y_label": "Centeredness margin", - "title_suffix": ( - f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" - ), - "xticks": PRICE_RATIO_TICKS, - "yticks": MARGIN_TICKS - }, - { - "slug": "shift_exp_vs_margin", - "x_values": HEATMAP_SHIFT_EXPONENTS, - "y_values": HEATMAP_MARGINS, - "x_key": "daily_price_shift_exponent", - "y_key": "centeredness_margin", - "x_label": "Shift exponent", - "y_label": "Centeredness margin", - "title_suffix": f"price_ratio fixed at {base_cfg['price_ratio']:.2f}", - "xticks": SHIFT_EXPONENT_TICKS, - "yticks": MARGIN_TICKS - }, - { - "slug": "price_ratio_vs_shift_exp", - "x_values": HEATMAP_PRICE_RATIOS, - "y_values": HEATMAP_SHIFT_EXPONENTS, - "x_key": "price_ratio", - "y_key": "daily_price_shift_exponent", - "x_label": "Price ratio", - "y_label": "Shift exponent", - "title_suffix": ( - f"margin fixed at {base_cfg['centeredness_margin']:.2f}" - ), - "xticks": PRICE_RATIO_TICKS, - "yticks": SHIFT_EXPONENT_TICKS, - }, - ] - + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) + metric_specs = get_pair_heatmap_metric_specs() + pair_specs = get_pair_heatmap_specs(base_cfg) metric_spec_map = {spec["key"]: spec for spec in metric_specs} + slice_count = len(pair_specs[0]["fixed_slices"]) if pair_specs else 0 if RUN_CONSTANT_ARC_LENGTH: print( @@ -1222,9 +1636,11 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): ) print( "Running {count} heatmap pair sweeps sequentially " - "(current outputs use cached noise-model runs; improvement heatmaps " - "reuse those values and add cached arb-only runs).".format( - count=len(pair_specs) + "(3 pair grids x {slice_count} fixed-variable quarter slices; " + "cached noise-model runs are reused across the absolute, launch, " + "and arb-only comparison outputs).".format( + count=len(pair_specs) * slice_count, + slice_count=slice_count, ) ) else: @@ -1235,20 +1651,135 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): ) print( "RUN_CONSTANT_ARC_LENGTH=False, so only geometric heatmaps will be generated " - "and only geometric/arb-only geometric runs will be scheduled." + f"across {len(pair_specs) * slice_count} fixed-variable pair sweeps." ) for pair in pair_specs: + for slice_variant in pair["fixed_slices"]: + pair_suffix = _pair_slice_suffix(pair, slice_variant) + slice_cfg = dict(base_cfg) + slice_cfg[pair["fixed_key"]] = float(slice_variant["value"]) + output_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair_suffix, + ) + for spec in metric_specs + } + missing_files = _missing_artifacts( + pair_suffix, + list(output_files.values()), + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in metric_specs + if output_files[spec["key"]] in missing_files + ] + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=slice_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair_suffix, + launch_final_values=launch_final_values, + ) + print(f"[{pair_suffix}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['fixed_label']} {slice_variant['label']} " + f"slice fixed at {format_heatmap_param_value(slice_variant['value'])} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=output_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + del data_by_metric + gc.collect() + + if owns_cache: + flush_sweep_cache(cache, force=True) + cache.clear() + gc.collect() + print("Released heatmap metric cache.") + + +def generate_three_variable_3d_heatmaps( + base_cfg, + price_data, + launch_final_values, + cache=None, +): + """Render 3D thermostat heatmaps from the three pairwise quarter slices.""" + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) + + metric_specs = get_pair_heatmap_metric_specs() + metric_spec_map = {spec["key"]: spec for spec in metric_specs} + pair_specs = get_pair_heatmap_specs(base_cfg) + pair_by_fixed_key = {pair["fixed_key"]: pair for pair in pair_specs} + price_margin_pair = pair_by_fixed_key["daily_price_shift_exponent"] + shift_margin_pair = pair_by_fixed_key["price_ratio"] + price_shift_pair = pair_by_fixed_key["centeredness_margin"] + slice_count = len(price_margin_pair["fixed_slices"]) + + def build_pair_slice_data(pair, slice_variant, metric_keys): + pair_cfg = dict(base_cfg) + pair_cfg[pair["fixed_key"]] = float(slice_variant["value"]) + return build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=pair_cfg, + metric_keys=metric_keys, + cache=cache, + progress_label=f"3d_{_pair_slice_suffix(pair, slice_variant)}", + launch_final_values=launch_final_values, + ) + + print( + "\nGenerating 3D thermostat heatmaps " + f"({slice_count} quarter-slice variants, TVL={format_tvl_millions_label(base_cfg)})..." + ) + + for slice_idx in range(slice_count): + shift_slice = price_margin_pair["fixed_slices"][slice_idx] + price_slice = shift_margin_pair["fixed_slices"][slice_idx] + margin_slice = price_shift_pair["fixed_slices"][slice_idx] + slice_slug = shift_slice["slug"] + slice_label = shift_slice["label"] + output_files = { - spec["key"]: heatmap_artifact_filename( + spec["key"]: three_d_heatmap_artifact_filename( spec, base_cfg, - suffix=pair["slug"], + suffix=f"slice_{slice_slug}", ) for spec in metric_specs } missing_files = _missing_artifacts( - pair["slug"], + f"3d_slice_{slice_slug}", list(output_files.values()), ) if not missing_files: @@ -1259,46 +1790,53 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): for spec in metric_specs if output_files[spec["key"]] in missing_files ] - data_by_metric = build_heatmap_matrices( - x_values=pair["x_values"], - y_values=pair["y_values"], - x_key=pair["x_key"], - y_key=pair["y_key"], - base_cfg=base_cfg, - metric_keys=missing_metric_keys, - cache=cache, - progress_label=pair["slug"], - launch_final_values=launch_final_values, + price_margin_data = build_pair_slice_data( + price_margin_pair, + shift_slice, + missing_metric_keys, + ) + shift_margin_data = build_pair_slice_data( + shift_margin_pair, + price_slice, + missing_metric_keys, + ) + price_shift_data = build_pair_slice_data( + price_shift_pair, + margin_slice, + missing_metric_keys, ) - print(f"[{pair['slug']}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: spec = metric_spec_map[metric_key] - plot_heatmap( - data=data_by_metric[metric_key], - x_values=pair["x_values"], - y_values=pair["y_values"], - x_label=pair["x_label"], - y_label=pair["y_label"], + plot_three_variable_heatmap_3d( + price_margin_data=price_margin_data[metric_key], + shift_margin_data=shift_margin_data[metric_key], + price_shift_data=price_shift_data[metric_key], + fixed_price_ratio=float(price_slice["value"]), + fixed_margin=float(margin_slice["value"]), + fixed_shift_exponent=float(shift_slice["value"]), title=( - f"{spec['title']}: {pair['title_suffix']} | " - f"TVL {format_tvl_millions_label(base_cfg)}" + f"{spec['title']} 3D {slice_label} slice | TVL {format_tvl_millions_label(base_cfg)}\n" + f"price_ratio={format_heatmap_param_value(price_slice['value'])}, " + f"margin={format_heatmap_param_value(margin_slice['value'])}, " + f"shift_exp={format_heatmap_param_value(shift_slice['value'])}" ), colorbar_label=spec["colorbar_label"], filename=output_files[metric_key], - xticks=pair["xticks"], - yticks=pair["yticks"], center_zero=spec["center_zero"], cmap=spec["cmap"], color_norm=spec.get("color_norm"), symlog_linthresh=spec.get("symlog_linthresh"), ) - del data_by_metric + + del price_margin_data, shift_margin_data, price_shift_data gc.collect() if owns_cache: + flush_sweep_cache(cache, force=True) cache.clear() gc.collect() - print("Released heatmap metric cache.") + print("Released 3D heatmap cache.") def compute_auto_calibrated_arc_length_speed(cfg, price_data): @@ -1378,7 +1916,7 @@ def generate_arc_speed_efficiency_artifacts( return owns_cache = cache is None if cache is None: - cache = make_sweep_cache(price_data) + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) launch_auto_speed = compute_auto_calibrated_arc_length_speed(launch_cfg, price_data) heatmap_metric_specs = [ { @@ -1559,6 +2097,7 @@ def generate_arc_speed_efficiency_artifacts( gc.collect() if owns_cache: + flush_sweep_cache(cache, force=True) cache.clear() gc.collect() print("Released arc-speed sweep cache.") @@ -1965,7 +2504,10 @@ def plot_comparison(cfg, results, fig_idx): launch_cfg=tvl_configs[0], price_data=shared_price_data, ) - shared_sweep_cache = make_sweep_cache(shared_price_data) + shared_sweep_cache = make_sweep_cache( + shared_price_data, + cache_scope_cfg=tvl_configs[1], + ) print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") generate_heatmaps( @@ -1982,6 +2524,13 @@ def plot_comparison(cfg, results, fig_idx): launch_final_values=launch_final_values, cache=shared_sweep_cache, ) + generate_three_variable_3d_heatmaps( + dict(tvl_configs[1]), + price_data=shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + flush_sweep_cache(shared_sweep_cache, force=True) shared_sweep_cache.clear() gc.collect() print(f"Released shared sweep cache for TVL {tvl_label}.") diff --git a/scripts/reclamm/compare_reclamm_geometric_noise_runs.py b/scripts/reclamm/compare_reclamm_geometric_noise_runs.py new file mode 100644 index 00000000..3e385dd0 --- /dev/null +++ b/scripts/reclamm/compare_reclamm_geometric_noise_runs.py @@ -0,0 +1,674 @@ +"""Compare two geometric reCLAMM runs against matched arb-only baselines. + +This script reuses the same AAVE/ETH reCLAMM fingerprint and parameter wiring as +``compare_reclamm_thermostats.py``, but runs only the geometric interpolation +mode and plots: +1. Share price / TVL over time in absolute USD terms +2. Pool weights over time +3. Estimated gross swap volume over time +4. Noise-model improvement over arb-only over time + +Because these runs use no LP supply changes, share price and TVL are the same +series here. + +Usage: + python scripts/reclamm/compare_reclamm_geometric_noise_runs.py + python scripts/reclamm/compare_reclamm_geometric_noise_runs.py \ + --adjacent-csv scripts/results/...csv --adjacent-row-index 0 +""" + +from __future__ import annotations + +import argparse +import importlib.util +import json +from pathlib import Path +from typing import Mapping, Optional, Sequence + +import numpy as np +import pandas as pd + + +DEFAULT_SOURCE_HEATMAP_DESCRIPTION = ( + "legacy unsuffixed price_ratio_vs_margin heatmap with shift_exp fixed at 0.10" +) +DEFAULT_RUN_SPECS = [ + { + "name": "Green cell near price_ratio 1.31", + "price_ratio": 1.31, + "centeredness_margin": 0.6763157894736842, + "daily_price_shift_exponent": 0.10, + "tvl_usd": 1_000_000.0, + "color": "C0", + "reason": ( + "Geometric noise-model run taken from the positive cell in the " + "legacy price_ratio-vs-margin heatmap at price_ratio=1.31 and the " + "lower adjacent centeredness row." + ), + }, + { + "name": "Red cell near price_ratio 1.31", + "price_ratio": 1.31, + "centeredness_margin": 0.7210526315789474, + "daily_price_shift_exponent": 0.10, + "tvl_usd": 1_000_000.0, + "color": "C1", + "reason": ( + "Geometric noise-model run taken from the negative cell directly " + "above the green cell in the legacy price_ratio-vs-margin heatmap " + "at price_ratio=1.31." + ), + }, +] +DEFAULT_OUTPUT_FILE = "reclamm_geometric_noise_pair_compare.png" +VARIANT_STYLES = { + "noise": {"linestyle": "-", "alpha": 0.95, "linewidth": 2.2}, + "arb": {"linestyle": "--", "alpha": 0.85, "linewidth": 2.0}, +} + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Compare two geometric reCLAMM runs plus matched arb-only baselines. " + "Optionally source the run pair from an adjacent-heatmap CSV row." + ) + ) + parser.add_argument( + "--adjacent-csv", + default=None, + help="Optional adjacent-pairs CSV generated by find_adjacent_heatmap_pairs.py.", + ) + parser.add_argument( + "--adjacent-row-index", + type=int, + default=0, + help="Which row from --adjacent-csv to use. Defaults to the largest-diff row.", + ) + parser.add_argument( + "--output-file", + default=None, + help="Optional PNG output path override.", + ) + return parser.parse_args() + + +def load_runtime_dependencies(): + """Load heavy runtime dependencies only when an actual simulation is needed.""" + thermostat_path = Path(__file__).with_name("compare_reclamm_thermostats.py") + spec = importlib.util.spec_from_file_location( + "reclamm_compare_reclamm_thermostats_runtime", + thermostat_path, + ) + if spec is None or spec.loader is None: + raise RuntimeError(f"Could not load compare module from {thermostat_path}") + thermostat_compare = importlib.util.module_from_spec(spec) + try: + spec.loader.exec_module(thermostat_compare) + except ModuleNotFoundError as exc: # pragma: no cover - depends on local runtime deps + if exc.name == "jax": + raise RuntimeError( + "compare_reclamm_geometric_noise_runs.py requires JAX to execute " + "the simulator runs, but JAX is not available in this environment." + ) from exc + raise + + from quantammsim.runners.jax_runners import do_run_on_historic_data + + return thermostat_compare, do_run_on_historic_data + + +def build_default_base_config(thermostat_compare): + """Return the aggressive AAVE/ETH geometric base config used by the compare script.""" + return dict(thermostat_compare.CONFIGS[1]) + + +def load_adjacent_csv_row(csv_path: Path, row_index: int = 0) -> Mapping[str, object]: + """Load one row from an adjacent-pairs CSV.""" + frame = pd.read_csv(csv_path) + if frame.empty: + raise ValueError(f"Adjacent CSV is empty: {csv_path}") + if not 0 <= row_index < len(frame): + raise IndexError( + f"adjacent-row-index {row_index} is out of range for {len(frame)} rows" + ) + return frame.iloc[row_index].to_dict() + + +def build_run_specs_from_adjacent_row( + row: Mapping[str, object], + csv_path: Optional[Path] = None, + row_index: int = 0, +): + """Convert one adjacent-pairs CSV row into the two run specs this script compares.""" + metric_key = str(row.get("metric_key", "unknown_metric")) + metric_unit = str(row.get("metric_unit", "value")) + pair_slug = str(row.get("pair_slug", "unknown_pair")) + slice_slug = str(row.get("slice_slug", "unknown_slice")) + adjacency_axis = str(row.get("adjacency_axis", "unknown_axis")) + diff_abs = float(row["heatmap_value_diff_abs"]) + source_noise_profile = str(row.get("source_noise_profile", "unknown")) + + source_description = ( + f"{pair_slug} {slice_slug} adjacent pair from {metric_key} " + f"({adjacency_axis}, abs diff={diff_abs:.6f} {metric_unit}, " + f"noise_profile={source_noise_profile})" + ) + if csv_path is not None: + source_description = ( + f"{csv_path.name} row {row_index} | {source_description}" + ) + + run_specs = [] + for prefix, color in (("1", "C0"), ("2", "C1")): + heatmap_value = float(row[f"{prefix}_heatmap_value"]) + spec = { + "name": f"Top diff row cell {prefix}", + "price_ratio": float(row[f"{prefix}_price_ratio"]), + "centeredness_margin": float(row[f"{prefix}_centeredness_margin"]), + "daily_price_shift_exponent": float(row[f"{prefix}_daily_price_shift_exponent"]), + "tvl_usd": float(row[f"{prefix}_tvl_usd"]), + "color": color, + "source_noise_profile": source_noise_profile, + "reason": ( + f"Run derived from adjacent heatmap CSV row {row_index}, cell {prefix}. " + f"Source={pair_slug}/{slice_slug}, adjacency={adjacency_axis}, " + f"heatmap_value={heatmap_value:.6f} {metric_unit}, " + f"noise_profile={source_noise_profile}." + ), + } + run_specs.append(spec) + return source_description, run_specs + + +def default_output_file_for_adjacent_csv(csv_path: Path, row_index: int = 0) -> Path: + """Build a deterministic output PNG path for an adjacent-pairs CSV selection.""" + stem = f"{csv_path.stem}_row_{row_index}_geometric_noise_compare" + return csv_path.with_name(stem + ".png") + + +def build_run_config(spec, base_config): + """Build the base geometric reCLAMM config for one highlighted heatmap cell.""" + cfg = dict(base_config) + cfg.update( + { + "name": spec["name"], + "price_ratio": float(spec["price_ratio"]), + "centeredness_margin": float(spec["centeredness_margin"]), + "daily_price_shift_exponent": float(spec["daily_price_shift_exponent"]), + "initial_pool_value": float(spec["tvl_usd"]), + "reason": spec.get( + "reason", + "Standalone geometric noise-model comparison run.", + ), + "enable_noise_model": True, + } + ) + source_noise_profile = spec.get("source_noise_profile") + if source_noise_profile == "legacy_calibrated": + cfg["noise_model"] = "calibrated" + cfg.pop("reclamm_noise_params", None) + cfg.pop("noise_arrays_path", None) + elif source_noise_profile == "market_linear": + cfg["noise_model"] = "market_linear" + return cfg + + +def build_run_variants(spec, base_config, thermostat_compare): + """Build matched noise-model and arb-only variants for one highlighted cell.""" + noise_cfg = build_run_config(spec, base_config=base_config) + noise_cfg["variant_key"] = "noise" + noise_cfg["variant_label"] = "noise-model" + + arb_cfg = thermostat_compare.make_noise_variant_cfg(noise_cfg, False) + arb_cfg["variant_key"] = "arb" + arb_cfg["variant_label"] = "arb-only" + arb_cfg["name"] = noise_cfg["name"] + arb_cfg["reason"] = ( + f"{noise_cfg['reason']} Matched arb-only baseline with noise disabled." + ) + return {"spec": spec, "noise": noise_cfg, "arb": arb_cfg} + + +def build_run_label(cfg, thermostat_compare): + """Build a compact legend label for a run.""" + return ( + f"{cfg['name']} ({cfg.get('variant_label', 'run')}) | PR {cfg['price_ratio']:.4g}, " + f"M {cfg['centeredness_margin']:.3g}, " + f"Shift {cfg['daily_price_shift_exponent']:.3g}, " + f"TVL {thermostat_compare.format_tvl_millions_label(cfg)}" + ) + + +def build_time_index(run_fingerprint, periods, step_minutes=1): + """Return a DatetimeIndex for a result series with the given cadence.""" + return pd.date_range( + start=pd.Timestamp(run_fingerprint["startDateString"]), + periods=int(periods), + freq=f"{max(int(step_minutes), 1)}min", + ) + + +def infer_series_step_minutes(run_fingerprint, series_length, minute_length): + """Infer the cadence of a result series from its length.""" + if minute_length and series_length and series_length != minute_length: + ratio = float(minute_length) / float(series_length) + rounded_ratio = max(int(round(ratio)), 1) + if abs(ratio - rounded_ratio) < 1.0e-9: + return rounded_ratio + return max(int(run_fingerprint.get("arb_frequency", 1)), 1) + + +def build_daily_weight_series( + weights, + tokens, + run_fingerprint, + minute_length, +): + """Build a daily weight frame from a weight array at inferred cadence.""" + weights = np.asarray(weights, dtype=float) + weight_step_minutes = infer_series_step_minutes( + run_fingerprint, + len(weights), + minute_length, + ) + weight_dates = build_time_index( + run_fingerprint, + len(weights), + step_minutes=weight_step_minutes, + ) + weight_frame = pd.DataFrame(weights, index=weight_dates, columns=tokens) + return weight_frame.resample("1D").last() + + +def estimate_gross_volume_usd(cfg, result, default_protocol_fee_split=0.25): + """Recover gross traded USD volume from LP fee revenue.""" + fee_revenue = result.get("fee_revenue") + if fee_revenue is None: + return np.zeros(len(np.asarray(result["value"])), dtype=float) + + fee_revenue = np.asarray(fee_revenue, dtype=float) + lp_fee_rate_share = float(cfg["fees"]) * ( + 1.0 - float(cfg.get("protocol_fee_split", default_protocol_fee_split)) + ) + if lp_fee_rate_share <= 0.0: + return np.zeros_like(fee_revenue) + return fee_revenue / lp_fee_rate_share + + +def build_daily_run_series(cfg, run_fingerprint, result, default_protocol_fee_split=0.25): + """Build daily TVL/share-price, weight, and volume series for plotting.""" + value = np.asarray(result["value"], dtype=float) + value_dates = build_time_index(run_fingerprint, len(value), step_minutes=1) + value_series = pd.Series(value, index=value_dates) + daily_value = value_series.resample("1D").last() + + zero_fee_weights = np.asarray(result["weights"], dtype=float) + daily_zero_fee_weights = build_daily_weight_series( + zero_fee_weights, + cfg["tokens"], + run_fingerprint, + len(value), + ) + + reserves = np.asarray(result["reserves"], dtype=float) + prices = np.asarray(result["prices"], dtype=float) + reserve_value = reserves * prices + reserve_value_totals = np.maximum( + reserve_value.sum(axis=1, keepdims=True), + 1.0e-12, + ) + actual_reserve_value_weights = reserve_value / reserve_value_totals + daily_actual_reserve_value_weights = build_daily_weight_series( + actual_reserve_value_weights, + cfg["tokens"], + run_fingerprint, + len(value), + ) + + gross_volume_usd = estimate_gross_volume_usd( + cfg, + result, + default_protocol_fee_split=default_protocol_fee_split, + ) + volume_step_minutes = ( + 1 + if len(gross_volume_usd) == len(value) + else infer_series_step_minutes(run_fingerprint, len(gross_volume_usd), len(value)) + ) + volume_dates = build_time_index( + run_fingerprint, + len(gross_volume_usd), + step_minutes=volume_step_minutes, + ) + daily_volume = pd.Series(gross_volume_usd, index=volume_dates).resample("1D").sum() + + return { + "daily_value": daily_value, + "daily_zero_fee_weights": daily_zero_fee_weights, + "daily_actual_reserve_value_weights": daily_actual_reserve_value_weights, + "daily_volume": daily_volume, + } + + +def _terminal_json_default(value): + """Serialize NumPy/JAX-backed values for readable terminal logging.""" + if isinstance(value, Path): + return str(value) + if hasattr(value, "tolist"): + return value.tolist() + return str(value) + + +def print_run_inputs_to_terminal(cfg, run_fingerprint, update_params): + """Print the full run fingerprint and update params for a triggered run.""" + print( + f"Run inputs for {cfg['name']} ({cfg.get('variant_label', 'run')}):" + ) + print( + json.dumps( + { + "run_fingerprint": run_fingerprint, + "update_params": update_params, + }, + indent=2, + sort_keys=True, + default=_terminal_json_default, + ) + ) + + +def run_single_config(cfg, price_data, thermostat_compare, do_run_on_historic_data): + """Run one geometric noise-model configuration.""" + run_fingerprint = thermostat_compare.make_fingerprint(cfg, "geometric") + update_params = thermostat_compare.make_params(cfg) + print_run_inputs_to_terminal(cfg, run_fingerprint, update_params) + result = do_run_on_historic_data( + run_fingerprint=run_fingerprint, + params=update_params, + price_data=price_data, + ) + return { + "config": cfg, + "fingerprint": run_fingerprint, + "noise_summary": thermostat_compare.resolve_reclamm_noise_settings(cfg)[ + "noise_summary" + ], + "result": result, + "series": build_daily_run_series( + cfg, + run_fingerprint, + result, + default_protocol_fee_split=thermostat_compare.DEFAULT_PROTOCOL_FEE_SPLIT, + ), + } + + +def plot_pair_results( + run_pairs, + thermostat_compare, + source_heatmap_description, + output_file, +): + """Plot paired noise-model/arb-only outputs for the two highlighted cells.""" + import matplotlib.pyplot as plt + + fig, axes = plt.subplots( + 5, + 1, + figsize=(14, 16), + sharex=True, + gridspec_kw={"height_ratios": [2.2, 1.5, 1.5, 1.4, 1.6]}, + ) + ( + ax_value, + ax_zero_fee_weights, + ax_actual_reserve_weights, + ax_volume, + ax_improvement, + ) = axes + + for pair in run_pairs: + spec = pair["spec"] + color = spec["color"] + noise_output = pair["noise"] + arb_output = pair["arb"] + + for variant_key, output in (("noise", noise_output), ("arb", arb_output)): + cfg = output["config"] + label = build_run_label(cfg, thermostat_compare=thermostat_compare) + series = output["series"] + style = VARIANT_STYLES[variant_key] + + ax_value.plot( + series["daily_value"].index, + series["daily_value"].to_numpy(dtype=float) / 1e6, + color=color, + linestyle=style["linestyle"], + linewidth=style["linewidth"], + alpha=style["alpha"], + label=label, + ) + + for token_idx, token in enumerate(cfg["tokens"]): + token_linestyle = style["linestyle"] if token_idx == 0 else ":" + ax_zero_fee_weights.plot( + series["daily_zero_fee_weights"].index, + series["daily_zero_fee_weights"][token].to_numpy(dtype=float), + color=color, + linestyle=token_linestyle, + linewidth=1.8 if token_idx == 0 else 1.6, + alpha=style["alpha"], + label=f"{cfg['name']} {cfg['variant_label']} {token}", + ) + ax_actual_reserve_weights.plot( + series["daily_actual_reserve_value_weights"].index, + series["daily_actual_reserve_value_weights"][token].to_numpy( + dtype=float + ), + color=color, + linestyle=token_linestyle, + linewidth=1.8 if token_idx == 0 else 1.6, + alpha=style["alpha"], + label=f"{cfg['name']} {cfg['variant_label']} {token}", + ) + + ax_volume.plot( + series["daily_volume"].index, + series["daily_volume"].to_numpy(dtype=float) / 1e6, + color=color, + linestyle=style["linestyle"], + linewidth=style["linewidth"], + alpha=style["alpha"], + label=f"{cfg['name']} {cfg['variant_label']}", + ) + + noise_value, arb_value = noise_output["series"]["daily_value"].align( + arb_output["series"]["daily_value"], + join="inner", + ) + improvement_pct = (noise_value - arb_value) / arb_value * 100.0 + ax_improvement.plot( + improvement_pct.index, + improvement_pct.to_numpy(dtype=float), + color=color, + linewidth=2.2, + label=noise_output["config"]["name"], + ) + + shared_noise_summary = run_pairs[0]["noise"]["noise_summary"] + fig.suptitle( + "reCLAMM geometric noise-model vs arb-only comparison", + fontsize=14, + fontweight="bold", + ) + fig.text( + 0.5, + 0.965, + ( + "Compare-script AAVE/ETH fingerprint | " + f"Cell source: {source_heatmap_description} | " + f"Noise: {shared_noise_summary} | " + "Share price equals TVL here because LP supply is fixed at 1.0" + ), + ha="center", + va="top", + fontsize=10, + ) + + ax_value.set_ylabel("Share price / TVL ($M)") + ax_value.set_title("Absolute share price / TVL") + ax_value.grid(True, alpha=0.3) + ax_value.legend(fontsize=8) + + ax_zero_fee_weights.set_ylabel("Weight") + ax_zero_fee_weights.set_title("Reported zero-fee empirical weights") + ax_zero_fee_weights.set_ylim(-0.02, 1.02) + ax_zero_fee_weights.grid(True, alpha=0.3) + ax_zero_fee_weights.legend(fontsize=8, ncol=2) + + ax_actual_reserve_weights.set_ylabel("Weight") + ax_actual_reserve_weights.set_title("Actual reserve value weights") + ax_actual_reserve_weights.set_ylim(-0.02, 1.02) + ax_actual_reserve_weights.grid(True, alpha=0.3) + ax_actual_reserve_weights.legend(fontsize=8, ncol=2) + + ax_volume.set_ylabel("Daily volume ($M)") + ax_volume.set_title("Estimated gross swap volume") + ax_volume.grid(True, alpha=0.3) + ax_volume.legend(fontsize=8) + + ax_improvement.axhline(0.0, color="black", linewidth=0.9, alpha=0.55) + ax_improvement.set_ylabel("Noise vs arb (%)") + ax_improvement.set_title("Daily TVL improvement: (noise - arb) / arb") + ax_improvement.set_xlabel("Date") + ax_improvement.grid(True, alpha=0.3) + ax_improvement.legend(fontsize=8) + + output_file = Path(output_file) + output_file.parent.mkdir(parents=True, exist_ok=True) + plt.tight_layout(rect=(0.0, 0.0, 1.0, 0.945)) + plt.savefig(output_file, dpi=180) + print(f"Saved {output_file}") + plt.close(fig) + + +def print_final_heatmap_summary(run_pairs, source_heatmap_description): + """Print the final values and heatmap-equivalent improvement for each pair.""" + print(f"\nSource heatmap selection: {source_heatmap_description}") + print("Final values represented by the heatmap metric:") + for pair in run_pairs: + spec = pair["spec"] + noise_final = float(pair["noise"]["series"]["daily_value"].iloc[-1]) + arb_final = float(pair["arb"]["series"]["daily_value"].iloc[-1]) + improvement_pct = (noise_final - arb_final) / arb_final * 100.0 + print( + f" {spec['name']}: " + f"noise=${noise_final:,.2f}, " + f"arb=${arb_final:,.2f}, " + f"heatmap_improvement={improvement_pct:.6f}%" + ) + + +def run_pair_comparison( + run_specs: Sequence[Mapping[str, object]], + source_heatmap_description: str, + output_file, +): + """Run both configs and render the comparison figure.""" + thermostat_compare, do_run_on_historic_data = load_runtime_dependencies() + base_config = build_default_base_config(thermostat_compare) + run_pairs = [ + build_run_variants( + spec, + base_config=base_config, + thermostat_compare=thermostat_compare, + ) + for spec in run_specs + ] + run_configs = [ + cfg + for pair in run_pairs + for variant_key, cfg in pair.items() + if variant_key in ("noise", "arb") + ] + price_data = thermostat_compare.load_shared_price_data(run_configs) + + completed_pairs = [] + for pair in run_pairs: + completed_pair = {"spec": pair["spec"]} + for variant_key in ("noise", "arb"): + cfg = pair[variant_key] + print( + f"Running {cfg['name']} ({cfg['variant_label']}) | " + f"price_ratio={cfg['price_ratio']}, " + f"margin={cfg['centeredness_margin']}, " + f"shift_exp={cfg['daily_price_shift_exponent']}, " + f"TVL={thermostat_compare.format_tvl_millions_label(cfg)}" + ) + completed_pair[variant_key] = run_single_config( + cfg, + price_data, + thermostat_compare=thermostat_compare, + do_run_on_historic_data=do_run_on_historic_data, + ) + completed_pairs.append(completed_pair) + + plot_pair_results( + completed_pairs, + thermostat_compare=thermostat_compare, + source_heatmap_description=source_heatmap_description, + output_file=output_file, + ) + print_final_heatmap_summary( + completed_pairs, + source_heatmap_description=source_heatmap_description, + ) + return output_file + + +def run_adjacent_csv_row_comparison( + csv_path, + row_index: int = 0, + output_file=None, +): + """Run the standard geometric comparison using one adjacent-pairs CSV row.""" + csv_path = Path(csv_path) + row = load_adjacent_csv_row(csv_path, row_index=row_index) + source_heatmap_description, run_specs = build_run_specs_from_adjacent_row( + row, + csv_path=csv_path, + row_index=row_index, + ) + resolved_output_file = ( + Path(output_file) + if output_file is not None + else default_output_file_for_adjacent_csv(csv_path, row_index=row_index) + ) + return run_pair_comparison( + run_specs=run_specs, + source_heatmap_description=source_heatmap_description, + output_file=resolved_output_file, + ) + + +def main(cli_args: Optional[argparse.Namespace] = None): + """Entry point for CLI execution.""" + args = cli_args or parse_args() + if args.adjacent_csv: + return run_adjacent_csv_row_comparison( + args.adjacent_csv, + row_index=args.adjacent_row_index, + output_file=args.output_file, + ) + + output_file = Path(args.output_file) if args.output_file else Path(DEFAULT_OUTPUT_FILE) + return run_pair_comparison( + run_specs=DEFAULT_RUN_SPECS, + source_heatmap_description=DEFAULT_SOURCE_HEATMAP_DESCRIPTION, + output_file=output_file, + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/reclamm/compare_reclamm_thermostats.py b/scripts/reclamm/compare_reclamm_thermostats.py index be8ee5f0..21352a4b 100644 --- a/scripts/reclamm/compare_reclamm_thermostats.py +++ b/scripts/reclamm/compare_reclamm_thermostats.py @@ -16,6 +16,7 @@ """ import gc +import hashlib import math import os @@ -41,17 +42,25 @@ def to_daily_price_shift_base(daily_price_shift_exponent): return 1.0 - daily_price_shift_exponent / 124649.0 -RUN_CONSTANT_ARC_LENGTH = False +def build_inclusive_sweep(start, stop, step): + """Build a sweep that keeps the requested step and explicitly includes the stop.""" + values = np.arange(start, stop + 1.0e-12, step, dtype=float) + if values.size == 0 or not np.isclose(values[-1], stop): + values = np.append(values, float(stop)) + return values + + +RUN_CONSTANT_ARC_LENGTH = True INTERPOLATION_METHODS = ( ("geometric", "constant_arc_length") if RUN_CONSTANT_ARC_LENGTH else ("geometric",) ) -HEATMAP_PRICE_RATIOS = np.arange(1.01, 1.50 + 1e-9, 0.025) -HEATMAP_MARGINS = np.linspace(0.05, 0.90, 20) -HEATMAP_SHIFT_EXPONENTS = np.arange(0.01, 0.50 + 1e-9, 0.025) +HEATMAP_PRICE_RATIOS = build_inclusive_sweep(1.01, 3.00, 0.025) +HEATMAP_MARGINS = np.linspace(0.05, 0.90, 39) +HEATMAP_SHIFT_EXPONENTS = build_inclusive_sweep(0.01, 0.50, 0.0125) HEATMAP_ARC_LENGTH_SPEEDS = np.geomspace(1.0e-6, 5.0e-4, 11) -PRICE_RATIO_TICKS = np.array([1.01, 1.10, 1.20, 1.30, 1.40, 1.50]) +PRICE_RATIO_TICKS = np.array([1.01, 1.25, 1.50, 2.00, 2.50, 3.00]) MARGIN_TICKS = np.array([0.05, 0.15, 0.25, 0.35, 0.45, 0.55, 0.65, 0.75, 0.85, 0.90]) SHIFT_EXPONENT_TICKS = np.array([0.01, 0.05, 0.10, 0.20, 0.30, 0.40, 0.50]) ARC_LENGTH_SPEED_TICKS = np.array([ @@ -76,6 +85,17 @@ def to_daily_price_shift_base(daily_price_shift_exponent): CENTER_ZERO_HEATMAP_COLOR_NORM = "symlog" CENTER_ZERO_HEATMAP_COLOR_TAG = "symlog20" CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH = 20.0 +FIXED_SLICE_FRACTIONS = (0.125, 0.375, 0.625, 0.875) +FIXED_SLICE_LABELS = ("Q1", "Q2", "Q3", "Q4") +THREE_D_VIEW_ELEVATION = 22.0 +THREE_D_VIEW_AZIMUTH = 140.0 +HEATMAP_FORWARD_CACHE_ENABLED = True +HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" +HEATMAP_FORWARD_CACHE_ROOT = os.path.join( + "results", + "reclamm_heatmap_forward_cache", +) +HEATMAP_FORWARD_CACHE_FLUSH_EVERY = 32 AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" @@ -94,14 +114,28 @@ def to_daily_price_shift_base(daily_price_shift_exponent): ] LEGACY_LOG_CADENCE = 2.68 LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) +FIXED_COMPARE_ARB_FREQUENCY = LEGACY_ARB_FREQUENCY AAVE_ETH_NOISE_SETTINGS = { "enable_noise_model": True, "noise_model": DEFAULT_NOISE_MODEL, "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, "noise_pool_id": AAVE_WETH_POOL_ID, + "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, "gas_cost": DEFAULT_GAS_COST, "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, } +PERSISTED_FORWARD_VALUE_COLUMNS = ( + "cache_key_hash", + "final_value", + "method", + "enable_noise_model", + "noise_model", + "price_ratio", + "centeredness_margin", + "daily_price_shift_exponent", + "initial_pool_value", + "arb_frequency", +) GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS = ( "geometric_vs_launch_geometric_pct", @@ -175,6 +209,23 @@ def heatmap_artifact_filename(spec, cfg, suffix=None): return tvl_artifact_filename(stem, cfg, suffix=suffix) +def three_d_heatmap_artifact_filename(spec, cfg, suffix=None): + """Build a 3D heatmap filename, including any colour-style tag.""" + stem = f"reclamm_heatmap_3d_{spec['slug']}" + artifact_tag = spec.get("artifact_tag") + if artifact_tag: + stem = f"{stem}_{artifact_tag}" + return tvl_artifact_filename(stem, cfg, suffix=suffix) + + +def format_heatmap_param_value(value): + """Format a sweep parameter compactly for titles and logs.""" + value = float(value) + if abs(value) >= 1.0: + return f"{value:.2f}".rstrip("0").rstrip(".") + return f"{value:.3f}".rstrip("0").rstrip(".") + + def configs_for_tvl(base_configs, initial_pool_value): """Attach a shared initial TVL to each compare configuration.""" configs = [] @@ -185,35 +236,76 @@ def configs_for_tvl(base_configs, initial_pool_value): return configs -def make_noise_variant_cfg(cfg, enable_noise_model): - """Return a config with either noise modelling or pure arb-only enabled.""" +def _normalize_arb_frequency(value, default=FIXED_COMPARE_ARB_FREQUENCY): + """Return a stable integer arb cadence for thermostat comparisons.""" + if value is None: + if default is None: + return None + value = default + return max(int(round(float(value))), 1) + + +def get_effective_arb_frequency(cfg, noise_cfg=None): + """Resolve the arb cadence used by a thermostat comparison run.""" + del noise_cfg + return _normalize_arb_frequency(FIXED_COMPARE_ARB_FREQUENCY) + + +def normalize_compare_run_cfg(cfg, enable_noise_model=None): + """Canonicalize the compare-run config so non-axis inputs stay fixed.""" updated = dict(cfg) - if enable_noise_model: - updated["enable_noise_model"] = True - return updated + updated["price_ratio"] = float(cfg["price_ratio"]) + updated["centeredness_margin"] = float(cfg["centeredness_margin"]) + updated["daily_price_shift_exponent"] = float(cfg["daily_price_shift_exponent"]) + updated["initial_pool_value"] = float(get_initial_pool_value(cfg)) + updated["gas_cost"] = DEFAULT_GAS_COST + updated["protocol_fee_split"] = DEFAULT_PROTOCOL_FEE_SPLIT + updated["arb_fees"] = 0.0 + updated["arb_frequency"] = get_effective_arb_frequency(cfg) + updated["noise_trader_ratio"] = 0.0 - matched_noise = resolve_reclamm_noise_settings(cfg) + arc_length_speed = cfg.get("arc_length_speed") + if arc_length_speed is None: + updated.pop("arc_length_speed", None) + else: + updated["arc_length_speed"] = float(arc_length_speed) - updated["enable_noise_model"] = False - updated["noise_model"] = None - updated["gas_cost"] = cfg.get("gas_cost", DEFAULT_GAS_COST) - updated["protocol_fee_split"] = cfg.get( - "protocol_fee_split", DEFAULT_PROTOCOL_FEE_SPLIT + use_noise = ( + bool(cfg.get("enable_noise_model", False)) + if enable_noise_model is None + else bool(enable_noise_model) ) - updated["noise_trader_ratio"] = 0.0 - matched_arb_frequency = matched_noise.get("arb_frequency") - if matched_arb_frequency is not None: - updated["arb_frequency"] = matched_arb_frequency - for key in ( - "reclamm_noise_params", - "noise_arrays_path", - "noise_artifact_dir", - "noise_pool_id", - ): - updated.pop(key, None) + updated["enable_noise_model"] = use_noise + + requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + if use_noise: + updated["noise_model"] = requested_mode + if requested_mode == "market_linear": + updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + updated["noise_pool_id"] = AAVE_WETH_POOL_ID + else: + updated.pop("noise_artifact_dir", None) + updated.pop("noise_pool_id", None) + updated.pop("reclamm_noise_params", None) + updated.pop("noise_arrays_path", None) + else: + updated["noise_model"] = None + for key in ( + "reclamm_noise_params", + "noise_arrays_path", + "noise_artifact_dir", + "noise_pool_id", + ): + updated.pop(key, None) + return updated +def make_noise_variant_cfg(cfg, enable_noise_model): + """Return a config with either noise modelling or pure arb-only enabled.""" + return normalize_compare_run_cfg(cfg, enable_noise_model=enable_noise_model) + + def _warn_noise_fallback(message): """Print a one-time message when the preferred noise setup is unavailable.""" if message not in _WARNED_NOISE_FALLBACKS: @@ -228,36 +320,39 @@ def _hashable_noise_params(params): return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) -def _legacy_calibrated_noise_settings(reason=None): +def _legacy_calibrated_noise_settings(reason=None, arb_frequency=None): """Fallback calibrated noise config used when market-linear artifacts are absent.""" if reason: _warn_noise_fallback( "market_linear noise unavailable for thermostat comparison; " f"falling back to calibrated legacy coefficients ({reason})." ) + arb_frequency = _normalize_arb_frequency(arb_frequency) return { "noise_model": "calibrated", "noise_trader_ratio": 0.0, "reclamm_noise_params": { f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) }, - "arb_frequency": LEGACY_ARB_FREQUENCY, + "arb_frequency": arb_frequency, "noise_summary": ( "calibrated legacy 8-covariate " - f"(arb_frequency={LEGACY_ARB_FREQUENCY})" + f"(arb_frequency={arb_frequency})" ), "noise_cache_key": ( "calibrated", tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), - LEGACY_ARB_FREQUENCY, + arb_frequency, ), } def resolve_reclamm_noise_settings(cfg): """Resolve the active reCLAMM noise-model fingerprint block for a config.""" + cfg = normalize_compare_run_cfg(cfg) enable_noise_model = cfg.get("enable_noise_model", False) requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + requested_arb_frequency = get_effective_arb_frequency(cfg) cache_key = ( tuple(cfg.get("tokens", [])), cfg.get("start"), @@ -266,7 +361,7 @@ def resolve_reclamm_noise_settings(cfg): requested_mode, cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), - cfg.get("arb_frequency"), + requested_arb_frequency, round(float(cfg.get("noise_trader_ratio", 0.0)), 12), _hashable_noise_params(cfg.get("reclamm_noise_params")), cfg.get("noise_arrays_path"), @@ -280,7 +375,7 @@ def resolve_reclamm_noise_settings(cfg): "noise_trader_ratio": 0.0, "reclamm_noise_params": None, "noise_arrays_path": None, - "arb_frequency": None, + "arb_frequency": requested_arb_frequency, "noise_summary": "arb-only (noise disabled)", "noise_cache_key": ("disabled",), } @@ -290,11 +385,7 @@ def resolve_reclamm_noise_settings(cfg): start_date = str(cfg["start"]).split(" ")[0] end_date = str(cfg["end"]).split(" ")[0] try: - from quantammsim.calibration.noise_model_arrays import ( - _find_pool_index, - build_simulator_arrays, - load_artifact, - ) + from quantammsim.calibration.noise_model_arrays import build_simulator_arrays model_path = os.path.join(artifact_dir, "model.npz") meta_path = os.path.join(artifact_dir, "meta.json") @@ -311,6 +402,8 @@ def resolve_reclamm_noise_settings(cfg): ) if not os.path.exists(arrays_path): arrays = build_simulator_arrays( + token_a=cfg["tokens"][0], + token_b=cfg["tokens"][1], pool_id=pool_id, start_date=start_date, end_date=end_date, @@ -328,13 +421,7 @@ def resolve_reclamm_noise_settings(cfg): tvl_mean = float(arrays["tvl_mean"]) tvl_std = float(arrays["tvl_std"]) - art, meta = load_artifact(artifact_dir) - pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) - if pool_idx >= 0: - learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) - else: - learned_cadence = 5.0 - arb_frequency = max(1, round(learned_cadence)) + arb_frequency = requested_arb_frequency result = { "noise_model": "market_linear", "noise_trader_ratio": 0.0, @@ -351,30 +438,19 @@ def resolve_reclamm_noise_settings(cfg): arb_frequency, round(tvl_mean, 12), round(tvl_std, 12), - ), - } + ), + } except Exception as exc: # pragma: no cover - fallback path depends on local artifacts - result = _legacy_calibrated_noise_settings(str(exc)) + result = _legacy_calibrated_noise_settings( + str(exc), + arb_frequency=requested_arb_frequency, + ) elif requested_mode == "calibrated": - params = cfg.get("reclamm_noise_params") - if params is None: - result = _legacy_calibrated_noise_settings() - else: - arb_frequency = cfg.get("arb_frequency", LEGACY_ARB_FREQUENCY) - result = { - "noise_model": "calibrated", - "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), - "reclamm_noise_params": dict(params), - "arb_frequency": arb_frequency, - "noise_summary": f"calibrated (arb_frequency={arb_frequency})", - "noise_cache_key": ( - "calibrated", - _hashable_noise_params(params), - arb_frequency, - ), - } + result = _legacy_calibrated_noise_settings( + arb_frequency=requested_arb_frequency + ) else: - arb_frequency = cfg.get("arb_frequency") + arb_frequency = requested_arb_frequency result = { "noise_model": requested_mode, "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), @@ -430,13 +506,14 @@ def resolve_reclamm_noise_settings(cfg): def make_fingerprint(cfg, interpolation_method): """Build run fingerprint for a given config and interpolation method.""" + cfg = normalize_compare_run_cfg(cfg) speed_override = ( cfg.get("arc_length_speed") if interpolation_method == "constant_arc_length" else None ) noise_cfg = resolve_reclamm_noise_settings(cfg) - arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) fingerprint = { "tokens": cfg["tokens"], "rule": "reclamm", @@ -471,6 +548,7 @@ def make_fingerprint(cfg, interpolation_method): def make_params(cfg): """Build pool params from config.""" + cfg = normalize_compare_run_cfg(cfg) return { "price_ratio": jnp.array(cfg["price_ratio"]), "centeredness_margin": jnp.array(cfg["centeredness_margin"]), @@ -528,7 +606,7 @@ def _set_padded_ylim(ax, series_list, pad_ratio=0.04): def _cache_size(cache): - """Count memoized final-value runs.""" + """Count memoized final-value cache entries materialised in memory.""" return len(cache.get("_final_value_cache", {})) @@ -537,13 +615,140 @@ def _comparison_cache_size(cache): return len(cache.get("_comparison_cache", {})) -def make_sweep_cache(price_data): - """Create a shared cache for heatmap and line sweeps.""" +def _heatmap_forward_cache_scope_slug(cfg): + """Build a compact cache scope slug for a shared-TVL heatmap run.""" + if cfg is None: + return "unspecified_tvl" + return f"tvl_{format_tvl_millions_slug(cfg)}" + + +def _heatmap_forward_cache_path(cfg): + """Return the parquet path for persisted scalar forward values.""" + if not HEATMAP_FORWARD_CACHE_ENABLED: + return None + return os.path.join( + HEATMAP_FORWARD_CACHE_ROOT, + HEATMAP_FORWARD_CACHE_RUN_NAME, + f"forward_values_{_heatmap_forward_cache_scope_slug(cfg)}.parquet", + ) + + +def _make_method_cache_hash(key): + """Build a compact stable digest for a method cache key.""" + return hashlib.sha256(repr(key).encode("utf-8")).hexdigest() + + +def _build_persistent_final_value_record(cfg, method, cache_key_hash, final_value): + """Build one self-describing parquet row for a cached scalar run result.""" + cfg = normalize_compare_run_cfg(cfg) + noise_cfg = resolve_reclamm_noise_settings(cfg) return { + "cache_key_hash": str(cache_key_hash), + "final_value": float(final_value), + "method": str(method), + "enable_noise_model": bool(cfg.get("enable_noise_model", False)), + "noise_model": noise_cfg.get("noise_model"), + "price_ratio": float(cfg["price_ratio"]), + "centeredness_margin": float(cfg["centeredness_margin"]), + "daily_price_shift_exponent": float(cfg["daily_price_shift_exponent"]), + "initial_pool_value": float(get_initial_pool_value(cfg)), + "arb_frequency": get_effective_arb_frequency(cfg, noise_cfg), + } + + +def _load_persistent_final_value_cache(cache): + """Load persisted scalar forward values from parquet once per sweep cache.""" + if cache.get("_persistent_final_value_cache_loaded"): + return + + disk_cache = {} + disk_records = {} + cache_path = cache.get("_persistent_final_value_cache_path") + if cache_path and os.path.exists(cache_path): + frame = pd.read_parquet(cache_path) + if not frame.empty: + for row in frame.itertuples(index=False): + cache_key_hash = str(row.cache_key_hash) + final_value = float(row.final_value) + disk_cache[cache_key_hash] = final_value + record = { + "cache_key_hash": cache_key_hash, + "final_value": final_value, + } + for column in PERSISTED_FORWARD_VALUE_COLUMNS: + if column in {"cache_key_hash", "final_value"}: + continue + record[column] = getattr(row, column, None) + disk_records[cache_key_hash] = record + print( + f"Loaded {len(disk_cache)} persisted heatmap forward values from {cache_path}" + ) + + cache["_persistent_final_value_cache"] = disk_cache + cache["_persistent_final_value_records"] = disk_records + cache["_persistent_final_value_cache_loaded"] = True + + +def flush_sweep_cache(cache, force=False): + """Persist newly computed scalar forward values to parquet.""" + if not HEATMAP_FORWARD_CACHE_ENABLED: + return + + pending = cache.get("_pending_persistent_final_values") + if not pending: + return + if not force and len(pending) < HEATMAP_FORWARD_CACHE_FLUSH_EVERY: + return + + _load_persistent_final_value_cache(cache) + disk_cache = cache.setdefault("_persistent_final_value_cache", {}) + disk_records = cache.setdefault("_persistent_final_value_records", {}) + for cache_key_hash, record in pending.items(): + merged = dict(disk_records.get(cache_key_hash, {})) + merged.update(record) + merged["cache_key_hash"] = str(cache_key_hash) + merged["final_value"] = float(merged["final_value"]) + disk_records[cache_key_hash] = merged + disk_cache[cache_key_hash] = merged["final_value"] + + cache_path = cache.get("_persistent_final_value_cache_path") + if cache_path is None: + pending.clear() + return + + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + sorted_records = [disk_records[key] for key in sorted(disk_records)] + payload = { + column: [record.get(column) for record in sorted_records] + for column in PERSISTED_FORWARD_VALUE_COLUMNS + } + payload["final_value"] = np.asarray(payload["final_value"], dtype=np.float64) + frame = pd.DataFrame(payload) + frame.sort_values("cache_key_hash", inplace=True, ignore_index=True) + frame.to_parquet(cache_path, index=False, compression="zstd") + print( + f"Persisted {len(pending)} new heatmap forward values to {cache_path} " + f"({len(disk_cache)} total cached values)." + ) + pending.clear() + + +def make_sweep_cache(price_data, cache_scope_cfg=None): + """Create a shared cache for heatmap and line sweeps.""" + cache = { "_shared_price_data": price_data, "_final_value_cache": {}, "_comparison_cache": {}, + "_pending_persistent_final_values": {}, + "_persistent_final_value_cache": {}, + "_persistent_final_value_records": {}, + "_persistent_final_value_cache_loaded": False, + "_persistent_final_value_cache_path": _heatmap_forward_cache_path( + cache_scope_cfg + ), } + _load_persistent_final_value_cache(cache) + return cache def _missing_artifacts(progress_label, filenames): @@ -571,8 +776,9 @@ def _speed_cache_key(speed): def _make_method_cache_key(cfg, method): """Cache key for a single-method final-value run.""" + cfg = normalize_compare_run_cfg(cfg) noise_cfg = resolve_reclamm_noise_settings(cfg) - arb_frequency = cfg.get("arb_frequency", noise_cfg.get("arb_frequency")) + arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) key = ( method, bool(cfg.get("enable_noise_model", False)), @@ -630,16 +836,34 @@ def _run_method_final_value_cached(cfg, method, cache): """Memoize final value for a single interpolation method.""" final_value_cache = cache.setdefault("_final_value_cache", {}) key = _make_method_cache_key(cfg, method) - if key not in final_value_cache: - result = do_run_on_historic_data( - run_fingerprint=make_fingerprint(cfg, method), - params=make_params(cfg), - price_data=cache["_shared_price_data"], - low_data_mode=True, + if key in final_value_cache: + return final_value_cache[key] + + _load_persistent_final_value_cache(cache) + key_hash = _make_method_cache_hash(key) + persisted_cache = cache.setdefault("_persistent_final_value_cache", {}) + if key_hash in persisted_cache: + final_value_cache[key] = persisted_cache[key_hash] + return final_value_cache[key] + + result = do_run_on_historic_data( + run_fingerprint=make_fingerprint(cfg, method), + params=make_params(cfg), + price_data=cache["_shared_price_data"], + low_data_mode=True, + ) + final_value_cache[key] = float(result["final_value"]) + cache.setdefault("_pending_persistent_final_values", {})[key_hash] = ( + _build_persistent_final_value_record( + cfg=cfg, + method=method, + cache_key_hash=key_hash, + final_value=final_value_cache[key], ) - final_value_cache[key] = float(result["final_value"]) - del result - gc.collect() + ) + flush_sweep_cache(cache, force=False) + del result + gc.collect() return final_value_cache[key] @@ -860,15 +1084,16 @@ def build_heatmap_matrices( data[metric_key][yi, xi] = metrics[metric_key] completed_points = (yi + 1) * len(x_values) - row_new_final_runs = _cache_size(cache) - final_cache_before_row + row_new_final_entries = _cache_size(cache) - final_cache_before_row row_new_comparisons = ( _comparison_cache_size(cache) - comparison_cache_before_row ) row_pct = completed_points / total_points * 100.0 + flush_sweep_cache(cache, force=True) print( f"[{progress_label}] row {yi + 1}/{len(y_values)} complete " f"({y_key}={float(y_value):.4f}, {completed_points}/{total_points} " - f"points, {row_pct:.1f}%, {row_new_final_runs} new final-value runs, " + f"points, {row_pct:.1f}%, {row_new_final_entries} new final-value cache entries, " f"{row_new_comparisons} new comparison bundles)" ) @@ -910,6 +1135,7 @@ def build_metric_curve( metric_keys=(metric_key,), ) data[xi] = metrics[metric_key] + flush_sweep_cache(cache, force=True) return data @@ -937,6 +1163,226 @@ def _compute_axis_edges(values, scale="linear"): return edges +def build_fixed_slice_variants(values): + """Pick four representative quarter-range slices from a sweep grid.""" + values = np.asarray(values, dtype=float) + if values.size < len(FIXED_SLICE_FRACTIONS): + raise ValueError("Need at least four grid points to build fixed slices") + + variants = [] + used_indices = set() + for idx, fraction in enumerate(FIXED_SLICE_FRACTIONS): + target_index = int(round(fraction * (values.size - 1))) + while target_index in used_indices and target_index + 1 < values.size: + target_index += 1 + while target_index in used_indices and target_index - 1 >= 0: + target_index -= 1 + if target_index in used_indices: + raise ValueError("Could not build four unique fixed slices from sweep grid") + used_indices.add(target_index) + variants.append( + { + "index": target_index, + "fraction": fraction, + "label": FIXED_SLICE_LABELS[idx], + "slug": f"q{idx + 1}", + "value": float(values[target_index]), + } + ) + return variants + + +def _pair_slice_suffix(pair, slice_variant): + """Build a stable artifact suffix for a pairwise fixed-variable slice.""" + return f"{pair['slug']}_{pair['fixed_slug']}_{slice_variant['slug']}" + + +def _build_heatmap_norm( + data_arrays, + center_zero, + color_norm=None, + symlog_linthresh=None, +): + """Build a color normalizer shared by 2D and 3D heatmaps.""" + finite_parts = [] + for data in data_arrays: + finite = np.asarray(data, dtype=float) + finite = finite[np.isfinite(finite)] + if finite.size: + finite_parts.append(finite) + finite = np.concatenate(finite_parts) if finite_parts else np.array([], dtype=float) + + if center_zero: + if finite.size == 0: + vmax = 1.0 + else: + vmax = max(abs(float(finite.min())), abs(float(finite.max())), 1e-9) + if ( + color_norm == "symlog" + and symlog_linthresh is not None + and vmax > symlog_linthresh + ): + return SymLogNorm( + linthresh=symlog_linthresh, + linscale=1.0, + vmin=-vmax, + vmax=vmax, + base=10.0, + ) + return TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) + + if finite.size == 0: + vmin, vmax = 0.0, 1.0 + else: + vmin = float(finite.min()) + vmax = float(finite.max()) + if np.isclose(vmin, vmax): + pad = max(abs(vmin) * 0.01, 1e-9) + vmin -= pad + vmax += pad + return Normalize(vmin=vmin, vmax=vmax) + + +def get_pair_heatmap_metric_specs(): + """Return the standard thermostat pairwise heatmap metrics.""" + metric_specs = [ + { + "key": "efficiency_pct", + "title": "Efficiency vs heatmap geometric", + "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", + "slug": "efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "launch_geometric_efficiency_pct", + "title": "Efficiency vs launch-style geometric", + "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", + "slug": "launch_geometric_efficiency", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "geometric_vs_launch_geometric_pct", + "title": "Geometric tuning vs launch-style geometric", + "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", + "slug": "geometric_vs_launch_geometric", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "constant_arc_vs_launch_constant_arc_pct", + "title": "Const arc tuning vs launch-style const arc", + "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", + "slug": "constant_arc_vs_launch_constant_arc", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_geometric_final_value_musd", + "title": "Geometric final value with noise model", + "colorbar_label": "Geometric final value with noise model ($M)", + "slug": "noise_geometric_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_constant_arc_final_value_musd", + "title": "Const arc final value with noise model", + "colorbar_label": "Const Arc final value with noise model ($M)", + "slug": "noise_constant_arc_final_value", + "center_zero": False, + "cmap": "viridis", + }, + { + "key": "noise_vs_arb_geometric_improvement_pct", + "title": "Noise-model improvement over arb-only (geometric)", + "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", + "slug": "noise_vs_arb_geometric_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + { + "key": "noise_vs_arb_constant_arc_improvement_pct", + "title": "Noise-model improvement over arb-only (const arc)", + "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", + "slug": "noise_vs_arb_constant_arc_improvement", + "center_zero": True, + "cmap": "RdYlGn", + }, + ] + for spec in metric_specs: + if spec["center_zero"]: + spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM + spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH + spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG + if not RUN_CONSTANT_ARC_LENGTH: + metric_specs = [ + spec + for spec in metric_specs + if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS + ] + return metric_specs + + +def get_pair_heatmap_specs(base_cfg): + """Return the three pairwise thermostat heatmap families plus slice settings.""" + fixed_slice_variants = { + "price_ratio": build_fixed_slice_variants(HEATMAP_PRICE_RATIOS), + "centeredness_margin": build_fixed_slice_variants(HEATMAP_MARGINS), + "daily_price_shift_exponent": build_fixed_slice_variants( + HEATMAP_SHIFT_EXPONENTS + ), + } + return [ + { + "slug": "price_ratio_vs_margin", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_MARGINS, + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "x_label": "Price ratio", + "y_label": "Centeredness margin", + "xticks": PRICE_RATIO_TICKS, + "yticks": MARGIN_TICKS, + "fixed_key": "daily_price_shift_exponent", + "fixed_label": "Shift exponent", + "fixed_slug": "shift_exp", + "fixed_slices": fixed_slice_variants["daily_price_shift_exponent"], + }, + { + "slug": "shift_exp_vs_margin", + "x_values": HEATMAP_SHIFT_EXPONENTS, + "y_values": HEATMAP_MARGINS, + "x_key": "daily_price_shift_exponent", + "y_key": "centeredness_margin", + "x_label": "Shift exponent", + "y_label": "Centeredness margin", + "xticks": SHIFT_EXPONENT_TICKS, + "yticks": MARGIN_TICKS, + "fixed_key": "price_ratio", + "fixed_label": "Price ratio", + "fixed_slug": "price_ratio", + "fixed_slices": fixed_slice_variants["price_ratio"], + }, + { + "slug": "price_ratio_vs_shift_exp", + "x_values": HEATMAP_PRICE_RATIOS, + "y_values": HEATMAP_SHIFT_EXPONENTS, + "x_key": "price_ratio", + "y_key": "daily_price_shift_exponent", + "x_label": "Price ratio", + "y_label": "Shift exponent", + "xticks": PRICE_RATIO_TICKS, + "yticks": SHIFT_EXPONENT_TICKS, + "fixed_key": "centeredness_margin", + "fixed_label": "Centeredness margin", + "fixed_slug": "margin", + "fixed_slices": fixed_slice_variants["centeredness_margin"], + }, + ] + + def plot_heatmap( data, x_values, @@ -955,34 +1401,13 @@ def plot_heatmap( symlog_linthresh=None, ): """Render and save a single heatmap.""" - finite = np.asarray(data, dtype=float) - finite = finite[np.isfinite(finite)] - - if center_zero: - vmax = max(abs(float(np.nanmin(data))), abs(float(np.nanmax(data))), 1e-9) - if color_norm == "symlog" and symlog_linthresh is not None and vmax > symlog_linthresh: - norm = SymLogNorm( - linthresh=symlog_linthresh, - linscale=1.0, - vmin=-vmax, - vmax=vmax, - base=10.0, - ) - else: - norm = TwoSlopeNorm(vcenter=0.0, vmin=-vmax, vmax=vmax) - cmap_name = cmap or "RdYlGn" - else: - if finite.size == 0: - vmin, vmax = 0.0, 1.0 - else: - vmin = float(finite.min()) - vmax = float(finite.max()) - if np.isclose(vmin, vmax): - pad = max(abs(vmin) * 0.01, 1e-9) - vmin -= pad - vmax += pad - norm = Normalize(vmin=vmin, vmax=vmax) - cmap_name = cmap or "viridis" + norm = _build_heatmap_norm( + [data], + center_zero=center_zero, + color_norm=color_norm, + symlog_linthresh=symlog_linthresh, + ) + cmap_name = cmap or ("RdYlGn" if center_zero else "viridis") x_edges = _compute_axis_edges(x_values, scale=xscale) y_edges = _compute_axis_edges(y_values, scale="linear") @@ -1015,6 +1440,112 @@ def plot_heatmap( plt.close(fig) +def plot_three_variable_heatmap_3d( + price_margin_data, + shift_margin_data, + price_shift_data, + fixed_price_ratio, + fixed_margin, + fixed_shift_exponent, + title, + colorbar_label, + filename, + center_zero=True, + cmap=None, + color_norm=None, + symlog_linthresh=None, +): + """Render orthogonal 3D heatmap surfaces across the three thermostat variables.""" + norm = _build_heatmap_norm( + [price_margin_data, shift_margin_data, price_shift_data], + center_zero=center_zero, + color_norm=color_norm, + symlog_linthresh=symlog_linthresh, + ) + cmap_name = cmap or ("RdYlGn" if center_zero else "viridis") + cmap_obj = plt.get_cmap(cmap_name) + + price_margin_x, price_margin_y = np.meshgrid(HEATMAP_PRICE_RATIOS, HEATMAP_MARGINS) + price_margin_z = np.full_like(price_margin_x, fixed_shift_exponent, dtype=float) + + shift_margin_z, shift_margin_y = np.meshgrid( + HEATMAP_SHIFT_EXPONENTS, + HEATMAP_MARGINS, + ) + shift_margin_x = np.full_like(shift_margin_z, fixed_price_ratio, dtype=float) + + price_shift_x, price_shift_z = np.meshgrid( + HEATMAP_PRICE_RATIOS, + HEATMAP_SHIFT_EXPONENTS, + ) + price_shift_y = np.full_like(price_shift_x, fixed_margin, dtype=float) + + fig = plt.figure(figsize=(10.5, 7.2)) + ax = fig.add_subplot(111, projection="3d") + ax.set_facecolor("white") + fig.patch.set_facecolor("white") + + ax.plot_surface( + price_margin_x, + price_margin_y, + price_margin_z, + facecolors=cmap_obj(norm(np.asarray(price_margin_data, dtype=float))), + shade=False, + ) + ax.plot_surface( + shift_margin_x, + shift_margin_y, + shift_margin_z, + facecolors=cmap_obj(norm(np.asarray(shift_margin_data, dtype=float))), + shade=False, + ) + ax.plot_surface( + price_shift_x, + price_shift_y, + price_shift_z, + facecolors=cmap_obj(norm(np.asarray(price_shift_data, dtype=float))), + shade=False, + ) + + ax.set_xlim(float(HEATMAP_PRICE_RATIOS.min()), float(HEATMAP_PRICE_RATIOS.max())) + ax.set_ylim(float(HEATMAP_MARGINS.min()), float(HEATMAP_MARGINS.max())) + ax.set_zlim( + float(HEATMAP_SHIFT_EXPONENTS.min()), + float(HEATMAP_SHIFT_EXPONENTS.max()), + ) + ax.set_xlabel("Price ratio") + ax.set_ylabel("Centeredness margin") + ax.set_zlabel("Shift exponent") + ax.set_xticks(PRICE_RATIO_TICKS) + ax.set_yticks(MARGIN_TICKS[::2]) + ax.set_zticks(SHIFT_EXPONENT_TICKS) + ax.set_title(title) + ax.grid(False) + ax.view_init(elev=THREE_D_VIEW_ELEVATION, azim=THREE_D_VIEW_AZIMUTH) + try: + ax.set_box_aspect( + ( + float(HEATMAP_PRICE_RATIOS.max() - HEATMAP_PRICE_RATIOS.min()), + float(HEATMAP_MARGINS.max() - HEATMAP_MARGINS.min()), + float( + HEATMAP_SHIFT_EXPONENTS.max() - HEATMAP_SHIFT_EXPONENTS.min() + ), + ) + ) + except AttributeError: + pass + + sm = ScalarMappable(norm=norm, cmap=cmap_obj) + sm.set_array([]) + cbar = fig.colorbar(sm, ax=ax, fraction=0.03, pad=0.1, shrink=0.82) + cbar.set_label(colorbar_label) + + plt.tight_layout() + plt.savefig(filename, dpi=150) + print(f"Saved {filename}") + plt.close(fig) + + def plot_arc_speed_line_chart( data, x_values, @@ -1090,128 +1621,11 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): """Generate pairwise heatmaps for thermostat tuning and noise-vs-arb effects.""" owns_cache = cache is None if cache is None: - cache = make_sweep_cache(price_data) - metric_specs = [ - { - "key": "efficiency_pct", - "title": "Efficiency vs heatmap geometric", - "colorbar_label": "Const Arc - heatmap Geo (% of heatmap geometric final value)", - "slug": "efficiency", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "launch_geometric_efficiency_pct", - "title": "Efficiency vs launch-style geometric", - "colorbar_label": "Const Arc - launch Geo (% of launch geometric final value)", - "slug": "launch_geometric_efficiency", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "geometric_vs_launch_geometric_pct", - "title": "Geometric tuning vs launch-style geometric", - "colorbar_label": "Candidate Geo - launch Geo (% of launch geometric final value)", - "slug": "geometric_vs_launch_geometric", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "constant_arc_vs_launch_constant_arc_pct", - "title": "Const arc tuning vs launch-style const arc", - "colorbar_label": "Candidate Const Arc - launch Const Arc (% of launch const arc final value)", - "slug": "constant_arc_vs_launch_constant_arc", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "noise_geometric_final_value_musd", - "title": "Geometric final value with noise model", - "colorbar_label": "Geometric final value with noise model ($M)", - "slug": "noise_geometric_final_value", - "center_zero": False, - "cmap": "viridis", - }, - { - "key": "noise_constant_arc_final_value_musd", - "title": "Const arc final value with noise model", - "colorbar_label": "Const Arc final value with noise model ($M)", - "slug": "noise_constant_arc_final_value", - "center_zero": False, - "cmap": "viridis", - }, - { - "key": "noise_vs_arb_geometric_improvement_pct", - "title": "Noise-model improvement over arb-only (geometric)", - "colorbar_label": "Noise-model Geo - arb-only Geo (% of arb-only final value)", - "slug": "noise_vs_arb_geometric_improvement", - "center_zero": True, - "cmap": "RdYlGn", - }, - { - "key": "noise_vs_arb_constant_arc_improvement_pct", - "title": "Noise-model improvement over arb-only (const arc)", - "colorbar_label": "Noise-model Const Arc - arb-only Const Arc (% of arb-only final value)", - "slug": "noise_vs_arb_constant_arc_improvement", - "center_zero": True, - "cmap": "RdYlGn", - }, - ] - for spec in metric_specs: - if spec["center_zero"]: - spec["color_norm"] = CENTER_ZERO_HEATMAP_COLOR_NORM - spec["symlog_linthresh"] = CENTER_ZERO_HEATMAP_SYMLOG_LINTHRESH - spec["artifact_tag"] = CENTER_ZERO_HEATMAP_COLOR_TAG - if not RUN_CONSTANT_ARC_LENGTH: - metric_specs = [ - spec - for spec in metric_specs - if spec["key"] in GEOMETRIC_ONLY_HEATMAP_METRIC_KEYS - ] - pair_specs = [ - { - "slug": "price_ratio_vs_margin", - "x_values": HEATMAP_PRICE_RATIOS, - "y_values": HEATMAP_MARGINS, - "x_key": "price_ratio", - "y_key": "centeredness_margin", - "x_label": "Price ratio", - "y_label": "Centeredness margin", - "title_suffix": ( - f"shift_exp fixed at {base_cfg['daily_price_shift_exponent']:.2f}" - ), - "xticks": PRICE_RATIO_TICKS, - "yticks": MARGIN_TICKS - }, - { - "slug": "shift_exp_vs_margin", - "x_values": HEATMAP_SHIFT_EXPONENTS, - "y_values": HEATMAP_MARGINS, - "x_key": "daily_price_shift_exponent", - "y_key": "centeredness_margin", - "x_label": "Shift exponent", - "y_label": "Centeredness margin", - "title_suffix": f"price_ratio fixed at {base_cfg['price_ratio']:.2f}", - "xticks": SHIFT_EXPONENT_TICKS, - "yticks": MARGIN_TICKS - }, - { - "slug": "price_ratio_vs_shift_exp", - "x_values": HEATMAP_PRICE_RATIOS, - "y_values": HEATMAP_SHIFT_EXPONENTS, - "x_key": "price_ratio", - "y_key": "daily_price_shift_exponent", - "x_label": "Price ratio", - "y_label": "Shift exponent", - "title_suffix": ( - f"margin fixed at {base_cfg['centeredness_margin']:.2f}" - ), - "xticks": PRICE_RATIO_TICKS, - "yticks": SHIFT_EXPONENT_TICKS, - }, - ] - + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) + metric_specs = get_pair_heatmap_metric_specs() + pair_specs = get_pair_heatmap_specs(base_cfg) metric_spec_map = {spec["key"]: spec for spec in metric_specs} + slice_count = len(pair_specs[0]["fixed_slices"]) if pair_specs else 0 if RUN_CONSTANT_ARC_LENGTH: print( @@ -1222,9 +1636,11 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): ) print( "Running {count} heatmap pair sweeps sequentially " - "(current outputs use cached noise-model runs; improvement heatmaps " - "reuse those values and add cached arb-only runs).".format( - count=len(pair_specs) + "(3 pair grids x {slice_count} fixed-variable quarter slices; " + "cached noise-model runs are reused across the absolute, launch, " + "and arb-only comparison outputs).".format( + count=len(pair_specs) * slice_count, + slice_count=slice_count, ) ) else: @@ -1235,20 +1651,135 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): ) print( "RUN_CONSTANT_ARC_LENGTH=False, so only geometric heatmaps will be generated " - "and only geometric/arb-only geometric runs will be scheduled." + f"across {len(pair_specs) * slice_count} fixed-variable pair sweeps." ) for pair in pair_specs: + for slice_variant in pair["fixed_slices"]: + pair_suffix = _pair_slice_suffix(pair, slice_variant) + slice_cfg = dict(base_cfg) + slice_cfg[pair["fixed_key"]] = float(slice_variant["value"]) + output_files = { + spec["key"]: heatmap_artifact_filename( + spec, + base_cfg, + suffix=pair_suffix, + ) + for spec in metric_specs + } + missing_files = _missing_artifacts( + pair_suffix, + list(output_files.values()), + ) + if not missing_files: + continue + + missing_metric_keys = [ + spec["key"] + for spec in metric_specs + if output_files[spec["key"]] in missing_files + ] + data_by_metric = build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=slice_cfg, + metric_keys=missing_metric_keys, + cache=cache, + progress_label=pair_suffix, + launch_final_values=launch_final_values, + ) + print(f"[{pair_suffix}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: + spec = metric_spec_map[metric_key] + plot_heatmap( + data=data_by_metric[metric_key], + x_values=pair["x_values"], + y_values=pair["y_values"], + x_label=pair["x_label"], + y_label=pair["y_label"], + title=( + f"{spec['title']}: {pair['fixed_label']} {slice_variant['label']} " + f"slice fixed at {format_heatmap_param_value(slice_variant['value'])} | " + f"TVL {format_tvl_millions_label(base_cfg)}" + ), + colorbar_label=spec["colorbar_label"], + filename=output_files[metric_key], + xticks=pair["xticks"], + yticks=pair["yticks"], + center_zero=spec["center_zero"], + cmap=spec["cmap"], + color_norm=spec.get("color_norm"), + symlog_linthresh=spec.get("symlog_linthresh"), + ) + del data_by_metric + gc.collect() + + if owns_cache: + flush_sweep_cache(cache, force=True) + cache.clear() + gc.collect() + print("Released heatmap metric cache.") + + +def generate_three_variable_3d_heatmaps( + base_cfg, + price_data, + launch_final_values, + cache=None, +): + """Render 3D thermostat heatmaps from the three pairwise quarter slices.""" + owns_cache = cache is None + if cache is None: + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) + + metric_specs = get_pair_heatmap_metric_specs() + metric_spec_map = {spec["key"]: spec for spec in metric_specs} + pair_specs = get_pair_heatmap_specs(base_cfg) + pair_by_fixed_key = {pair["fixed_key"]: pair for pair in pair_specs} + price_margin_pair = pair_by_fixed_key["daily_price_shift_exponent"] + shift_margin_pair = pair_by_fixed_key["price_ratio"] + price_shift_pair = pair_by_fixed_key["centeredness_margin"] + slice_count = len(price_margin_pair["fixed_slices"]) + + def build_pair_slice_data(pair, slice_variant, metric_keys): + pair_cfg = dict(base_cfg) + pair_cfg[pair["fixed_key"]] = float(slice_variant["value"]) + return build_heatmap_matrices( + x_values=pair["x_values"], + y_values=pair["y_values"], + x_key=pair["x_key"], + y_key=pair["y_key"], + base_cfg=pair_cfg, + metric_keys=metric_keys, + cache=cache, + progress_label=f"3d_{_pair_slice_suffix(pair, slice_variant)}", + launch_final_values=launch_final_values, + ) + + print( + "\nGenerating 3D thermostat heatmaps " + f"({slice_count} quarter-slice variants, TVL={format_tvl_millions_label(base_cfg)})..." + ) + + for slice_idx in range(slice_count): + shift_slice = price_margin_pair["fixed_slices"][slice_idx] + price_slice = shift_margin_pair["fixed_slices"][slice_idx] + margin_slice = price_shift_pair["fixed_slices"][slice_idx] + slice_slug = shift_slice["slug"] + slice_label = shift_slice["label"] + output_files = { - spec["key"]: heatmap_artifact_filename( + spec["key"]: three_d_heatmap_artifact_filename( spec, base_cfg, - suffix=pair["slug"], + suffix=f"slice_{slice_slug}", ) for spec in metric_specs } missing_files = _missing_artifacts( - pair["slug"], + f"3d_slice_{slice_slug}", list(output_files.values()), ) if not missing_files: @@ -1259,46 +1790,53 @@ def generate_heatmaps(base_cfg, price_data, launch_final_values, cache=None): for spec in metric_specs if output_files[spec["key"]] in missing_files ] - data_by_metric = build_heatmap_matrices( - x_values=pair["x_values"], - y_values=pair["y_values"], - x_key=pair["x_key"], - y_key=pair["y_key"], - base_cfg=base_cfg, - metric_keys=missing_metric_keys, - cache=cache, - progress_label=pair["slug"], - launch_final_values=launch_final_values, + price_margin_data = build_pair_slice_data( + price_margin_pair, + shift_slice, + missing_metric_keys, + ) + shift_margin_data = build_pair_slice_data( + shift_margin_pair, + price_slice, + missing_metric_keys, + ) + price_shift_data = build_pair_slice_data( + price_shift_pair, + margin_slice, + missing_metric_keys, ) - print(f"[{pair['slug']}] plotting missing heatmaps...") + for metric_key in missing_metric_keys: spec = metric_spec_map[metric_key] - plot_heatmap( - data=data_by_metric[metric_key], - x_values=pair["x_values"], - y_values=pair["y_values"], - x_label=pair["x_label"], - y_label=pair["y_label"], + plot_three_variable_heatmap_3d( + price_margin_data=price_margin_data[metric_key], + shift_margin_data=shift_margin_data[metric_key], + price_shift_data=price_shift_data[metric_key], + fixed_price_ratio=float(price_slice["value"]), + fixed_margin=float(margin_slice["value"]), + fixed_shift_exponent=float(shift_slice["value"]), title=( - f"{spec['title']}: {pair['title_suffix']} | " - f"TVL {format_tvl_millions_label(base_cfg)}" + f"{spec['title']} 3D {slice_label} slice | TVL {format_tvl_millions_label(base_cfg)}\n" + f"price_ratio={format_heatmap_param_value(price_slice['value'])}, " + f"margin={format_heatmap_param_value(margin_slice['value'])}, " + f"shift_exp={format_heatmap_param_value(shift_slice['value'])}" ), colorbar_label=spec["colorbar_label"], filename=output_files[metric_key], - xticks=pair["xticks"], - yticks=pair["yticks"], center_zero=spec["center_zero"], cmap=spec["cmap"], color_norm=spec.get("color_norm"), symlog_linthresh=spec.get("symlog_linthresh"), ) - del data_by_metric + + del price_margin_data, shift_margin_data, price_shift_data gc.collect() if owns_cache: + flush_sweep_cache(cache, force=True) cache.clear() gc.collect() - print("Released heatmap metric cache.") + print("Released 3D heatmap cache.") def compute_auto_calibrated_arc_length_speed(cfg, price_data): @@ -1378,7 +1916,7 @@ def generate_arc_speed_efficiency_artifacts( return owns_cache = cache is None if cache is None: - cache = make_sweep_cache(price_data) + cache = make_sweep_cache(price_data, cache_scope_cfg=base_cfg) launch_auto_speed = compute_auto_calibrated_arc_length_speed(launch_cfg, price_data) heatmap_metric_specs = [ { @@ -1559,6 +2097,7 @@ def generate_arc_speed_efficiency_artifacts( gc.collect() if owns_cache: + flush_sweep_cache(cache, force=True) cache.clear() gc.collect() print("Released arc-speed sweep cache.") @@ -1965,7 +2504,10 @@ def plot_comparison(cfg, results, fig_idx): launch_cfg=tvl_configs[0], price_data=shared_price_data, ) - shared_sweep_cache = make_sweep_cache(shared_price_data) + shared_sweep_cache = make_sweep_cache( + shared_price_data, + cache_scope_cfg=tvl_configs[1], + ) print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") generate_heatmaps( @@ -1982,6 +2524,13 @@ def plot_comparison(cfg, results, fig_idx): launch_final_values=launch_final_values, cache=shared_sweep_cache, ) + generate_three_variable_3d_heatmaps( + dict(tvl_configs[1]), + price_data=shared_price_data, + launch_final_values=launch_final_values, + cache=shared_sweep_cache, + ) + flush_sweep_cache(shared_sweep_cache, force=True) shared_sweep_cache.clear() gc.collect() print(f"Released shared sweep cache for TVL {tvl_label}.") diff --git a/scripts/reclamm/find_adjacent_heatmap_pairs.py b/scripts/reclamm/find_adjacent_heatmap_pairs.py new file mode 100644 index 00000000..1bc4ce36 --- /dev/null +++ b/scripts/reclamm/find_adjacent_heatmap_pairs.py @@ -0,0 +1,1264 @@ +"""Scan cached reCLAMM heatmaps for adjacent cells with large value gaps. + +This script reconstructs heatmap cells from the persisted scalar forward-value +cache written by ``compare_reclamm_thermostats.py``. It does not inspect PNG +pixels or rerun the simulator for cache-backed metrics. +""" + +from __future__ import annotations + +import argparse +import hashlib +import importlib.util +import math +import os +from pathlib import Path +from typing import Dict, Iterable, List, Mapping, MutableMapping, Optional, Sequence, Tuple + +import numpy as np +import pandas as pd + + +CACHE_ONLY_METRIC_SPECS = { + "efficiency_pct": { + "sources": ("noise_constant_arc", "noise_geometric"), + "unit": "pct", + "compute": lambda values: ( + values["noise_constant_arc"] / max(abs(values["noise_geometric"]), 1.0e-12) + - 1.0 + ) + * 100.0, + }, + "noise_geometric_final_value_musd": { + "sources": ("noise_geometric",), + "unit": "musd", + "compute": lambda values: values["noise_geometric"] / 1.0e6, + }, + "noise_constant_arc_final_value_musd": { + "sources": ("noise_constant_arc",), + "unit": "musd", + "compute": lambda values: values["noise_constant_arc"] / 1.0e6, + }, + "noise_vs_arb_geometric_improvement_pct": { + "sources": ("noise_geometric", "arb_geometric"), + "unit": "pct", + "compute": lambda values: ( + values["noise_geometric"] / max(abs(values["arb_geometric"]), 1.0e-12) - 1.0 + ) + * 100.0, + }, + "noise_vs_arb_constant_arc_improvement_pct": { + "sources": ("noise_constant_arc", "arb_constant_arc"), + "unit": "pct", + "compute": lambda values: ( + values["noise_constant_arc"] + / max(abs(values["arb_constant_arc"]), 1.0e-12) + - 1.0 + ) + * 100.0, + }, +} + +OUTPUT_COLUMNS = [ + "metric_key", + "metric_unit", + "source_noise_profile", + "pair_slug", + "slice_slug", + "slice_label", + "fixed_key", + "fixed_value", + "adjacency_axis", + "heatmap_value_diff_abs", + "heatmap_value_diff_signed_2_minus_1", + "1_price_ratio", + "1_centeredness_margin", + "1_daily_price_shift_exponent", + "1_tvl_usd", + "1_heatmap_value", + "1_x_index", + "1_y_index", + "2_price_ratio", + "2_centeredness_margin", + "2_daily_price_shift_exponent", + "2_tvl_usd", + "2_heatmap_value", + "2_x_index", + "2_y_index", +] + + +def build_inclusive_sweep(start: float, stop: float, step: float) -> np.ndarray: + """Build a sweep that keeps the requested step and explicitly includes the stop.""" + values = np.arange(start, stop + 1.0e-12, step, dtype=float) + if values.size == 0 or not np.isclose(values[-1], stop): + values = np.append(values, float(stop)) + return values + + +class _LightweightCompareContext: + """Small subset of compare_reclamm_thermostats usable without JAX.""" + + RUN_CONSTANT_ARC_LENGTH = True + DEFAULT_INITIAL_POOL_VALUE = 1_000_000.0 + TVL_SWEEP_VALUES = ( + 1_000_000.0, + 5_000_000.0, + 20_000_000.0, + ) + HEATMAP_PRICE_RATIOS = build_inclusive_sweep(1.01, 3.00, 0.025) + HEATMAP_MARGINS = np.linspace(0.05, 0.90, 39) + HEATMAP_SHIFT_EXPONENTS = build_inclusive_sweep(0.01, 0.50, 0.0125) + FIXED_SLICE_FRACTIONS = (0.125, 0.375, 0.625, 0.875) + FIXED_SLICE_LABELS = ("Q1", "Q2", "Q3", "Q4") + HEATMAP_FORWARD_CACHE_ENABLED = True + HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" + HEATMAP_FORWARD_CACHE_ROOT = os.path.join( + "results", + "reclamm_heatmap_forward_cache", + ) + AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" + DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" + DEFAULT_NOISE_MODEL = "market_linear" + DEFAULT_GAS_COST = 1.0 + DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 + LEGACY_NOISE_COEFFS = [ + -0.453, + 0.025, + -0.060, + 0.310, + -0.149, + 0.359, + 0.061, + 0.060, + ] + LEGACY_LOG_CADENCE = 2.68 + LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) + FIXED_COMPARE_ARB_FREQUENCY = LEGACY_ARB_FREQUENCY + AAVE_ETH_NOISE_SETTINGS = { + "enable_noise_model": True, + "noise_model": DEFAULT_NOISE_MODEL, + "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, + "noise_pool_id": AAVE_WETH_POOL_ID, + "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, + "gas_cost": DEFAULT_GAS_COST, + "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, + } + CONFIGS = [ + { + "name": "AAVE/ETH launch-style range (25bps, reference)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 1.5014, + "centeredness_margin": 0.5, + "daily_price_shift_exponent": 0.1, + "reason": "Original launch-style parameters.", + **AAVE_ETH_NOISE_SETTINGS, + }, + { + "name": "AAVE/ETH aggressive tight range (25bps)", + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, + "price_ratio": 1.10, + "centeredness_margin": 0.60, + "daily_price_shift_exponent": 0.1, + "reason": ( + "Aggressively tightened and moved to an earlier thermostat trigger. " + "At fixed price_ratio=1.10, the shift_exponent sweep still favored " + "0.1, while margin=0.60 widened the non-linear edge materially." + ), + **AAVE_ETH_NOISE_SETTINGS, + }, + ] + + def __init__(self): + self._noise_settings_cache = {} + self.noise_profile = "market_linear" + + @classmethod + def from_compare_module(cls, compare_module): + """Build an analyzer-friendly context from an imported thermostat module.""" + context = cls() + copied_attrs = ( + "RUN_CONSTANT_ARC_LENGTH", + "DEFAULT_INITIAL_POOL_VALUE", + "TVL_SWEEP_VALUES", + "HEATMAP_PRICE_RATIOS", + "HEATMAP_MARGINS", + "HEATMAP_SHIFT_EXPONENTS", + "FIXED_SLICE_FRACTIONS", + "FIXED_SLICE_LABELS", + "HEATMAP_FORWARD_CACHE_ENABLED", + "HEATMAP_FORWARD_CACHE_RUN_NAME", + "HEATMAP_FORWARD_CACHE_ROOT", + "AAVE_WETH_POOL_ID", + "DEFAULT_MARKET_LINEAR_ARTIFACT_DIR", + "DEFAULT_NOISE_MODEL", + "DEFAULT_GAS_COST", + "DEFAULT_PROTOCOL_FEE_SPLIT", + "LEGACY_NOISE_COEFFS", + "LEGACY_LOG_CADENCE", + "LEGACY_ARB_FREQUENCY", + "FIXED_COMPARE_ARB_FREQUENCY", + "AAVE_ETH_NOISE_SETTINGS", + "CONFIGS", + ) + for attr_name in copied_attrs: + if hasattr(compare_module, attr_name): + value = getattr(compare_module, attr_name) + if attr_name == "CONFIGS": + value = [dict(cfg) for cfg in value] + elif isinstance(value, dict): + value = dict(value) + elif isinstance(value, np.ndarray): + value = np.asarray(value, dtype=float).copy() + elif isinstance(value, tuple): + value = tuple(value) + elif isinstance(value, list): + value = list(value) + setattr(context, attr_name, value) + context._noise_settings_cache.clear() + return context + + def set_noise_profile(self, profile): + if profile not in {"market_linear", "legacy_calibrated"}: + raise ValueError(f"Unsupported lightweight noise profile: {profile}") + if profile != self.noise_profile: + self.noise_profile = profile + self._noise_settings_cache.clear() + + def get_initial_pool_value(self, cfg): + return float(cfg.get("initial_pool_value", self.DEFAULT_INITIAL_POOL_VALUE)) + + def get_tvl_millions(self, cfg): + return self.get_initial_pool_value(cfg) / 1_000_000.0 + + def format_tvl_millions_slug(self, cfg): + tvl_millions = self.get_tvl_millions(cfg) + rounded = round(float(tvl_millions), 6) + if np.isclose(rounded, round(rounded)): + return f"{int(round(rounded))}m" + return f"{rounded:.6f}".rstrip("0").rstrip(".").replace(".", "p") + "m" + + def format_tvl_millions_label(self, cfg): + return f"{self.get_tvl_millions(cfg):.1f}M" + + def configs_for_tvl(self, base_configs, initial_pool_value): + configs = [] + for cfg in base_configs: + updated = dict(cfg) + updated["initial_pool_value"] = float(initial_pool_value) + configs.append(updated) + return configs + + def _heatmap_forward_cache_scope_slug(self, cfg): + if cfg is None: + return "unspecified_tvl" + return f"tvl_{self.format_tvl_millions_slug(cfg)}" + + def _heatmap_forward_cache_path(self, cfg): + if not self.HEATMAP_FORWARD_CACHE_ENABLED: + return None + return os.path.join( + self.HEATMAP_FORWARD_CACHE_ROOT, + self.HEATMAP_FORWARD_CACHE_RUN_NAME, + f"forward_values_{self._heatmap_forward_cache_scope_slug(cfg)}.parquet", + ) + + def build_fixed_slice_variants(self, values): + values = np.asarray(values, dtype=float) + if values.size < len(self.FIXED_SLICE_FRACTIONS): + raise ValueError("Need at least four grid points to build fixed slices") + + variants = [] + used_indices = set() + for idx, fraction in enumerate(self.FIXED_SLICE_FRACTIONS): + target_index = int(round(fraction * (values.size - 1))) + while target_index in used_indices and target_index + 1 < values.size: + target_index += 1 + while target_index in used_indices and target_index - 1 >= 0: + target_index -= 1 + if target_index in used_indices: + raise ValueError( + "Could not build four unique fixed slices from sweep grid" + ) + used_indices.add(target_index) + variants.append( + { + "index": target_index, + "fraction": fraction, + "label": self.FIXED_SLICE_LABELS[idx], + "slug": f"q{idx + 1}", + "value": float(values[target_index]), + } + ) + return variants + + def get_pair_heatmap_specs(self, _base_cfg): + fixed_slice_variants = { + "price_ratio": self.build_fixed_slice_variants(self.HEATMAP_PRICE_RATIOS), + "centeredness_margin": self.build_fixed_slice_variants(self.HEATMAP_MARGINS), + "daily_price_shift_exponent": self.build_fixed_slice_variants( + self.HEATMAP_SHIFT_EXPONENTS + ), + } + return [ + { + "slug": "price_ratio_vs_margin", + "x_values": self.HEATMAP_PRICE_RATIOS, + "y_values": self.HEATMAP_MARGINS, + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "fixed_key": "daily_price_shift_exponent", + "fixed_slices": fixed_slice_variants["daily_price_shift_exponent"], + }, + { + "slug": "shift_exp_vs_margin", + "x_values": self.HEATMAP_SHIFT_EXPONENTS, + "y_values": self.HEATMAP_MARGINS, + "x_key": "daily_price_shift_exponent", + "y_key": "centeredness_margin", + "fixed_key": "price_ratio", + "fixed_slices": fixed_slice_variants["price_ratio"], + }, + { + "slug": "price_ratio_vs_shift_exp", + "x_values": self.HEATMAP_PRICE_RATIOS, + "y_values": self.HEATMAP_SHIFT_EXPONENTS, + "x_key": "price_ratio", + "y_key": "daily_price_shift_exponent", + "fixed_key": "centeredness_margin", + "fixed_slices": fixed_slice_variants["centeredness_margin"], + }, + ] + + @staticmethod + def _hashable_noise_params(params): + if params is None: + return None + return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) + + def _normalize_arb_frequency(self, value, default=None): + if value is None: + if default is None: + default = self.FIXED_COMPARE_ARB_FREQUENCY + value = default + return max(int(round(float(value))), 1) + + def get_effective_arb_frequency(self, cfg, noise_cfg=None): + del noise_cfg + return self._normalize_arb_frequency(self.FIXED_COMPARE_ARB_FREQUENCY) + + def normalize_compare_run_cfg(self, cfg, enable_noise_model=None): + updated = dict(cfg) + updated["price_ratio"] = float(cfg["price_ratio"]) + updated["centeredness_margin"] = float(cfg["centeredness_margin"]) + updated["daily_price_shift_exponent"] = float( + cfg["daily_price_shift_exponent"] + ) + updated["initial_pool_value"] = float(self.get_initial_pool_value(cfg)) + updated["gas_cost"] = self.DEFAULT_GAS_COST + updated["protocol_fee_split"] = self.DEFAULT_PROTOCOL_FEE_SPLIT + updated["arb_fees"] = 0.0 + updated["arb_frequency"] = self.get_effective_arb_frequency(cfg) + updated["noise_trader_ratio"] = 0.0 + + arc_length_speed = cfg.get("arc_length_speed") + if arc_length_speed is None: + updated.pop("arc_length_speed", None) + else: + updated["arc_length_speed"] = float(arc_length_speed) + + use_noise = ( + bool(cfg.get("enable_noise_model", False)) + if enable_noise_model is None + else bool(enable_noise_model) + ) + updated["enable_noise_model"] = use_noise + + requested_mode = ( + cfg.get("noise_model", self.DEFAULT_NOISE_MODEL) + or self.DEFAULT_NOISE_MODEL + ) + if use_noise: + canonical_noise_model = ( + requested_mode + if requested_mode != "arb_only" + else self.DEFAULT_NOISE_MODEL + ) + updated["noise_model"] = canonical_noise_model + if canonical_noise_model == "market_linear": + updated["noise_artifact_dir"] = self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + updated["noise_pool_id"] = self.AAVE_WETH_POOL_ID + else: + updated.pop("noise_artifact_dir", None) + updated.pop("noise_pool_id", None) + updated.pop("reclamm_noise_params", None) + updated.pop("noise_arrays_path", None) + else: + updated["noise_model"] = "arb_only" + for key in ( + "reclamm_noise_params", + "noise_arrays_path", + "noise_artifact_dir", + "noise_pool_id", + ): + updated.pop(key, None) + + return updated + + def _legacy_calibrated_noise_settings(self, arb_frequency=None): + arb_frequency = self._normalize_arb_frequency(arb_frequency) + return { + "noise_model": "calibrated", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + f"c_{i}": self.LEGACY_NOISE_COEFFS[i] + for i in range(len(self.LEGACY_NOISE_COEFFS)) + }, + "arb_frequency": arb_frequency, + "noise_summary": ( + "calibrated legacy 8-covariate " + f"(arb_frequency={arb_frequency})" + ), + "noise_cache_key": ( + "calibrated", + tuple(round(float(c), 12) for c in self.LEGACY_NOISE_COEFFS), + arb_frequency, + ), + } + + def resolve_reclamm_noise_settings(self, cfg): + cfg = self.normalize_compare_run_cfg(cfg) + enable_noise_model = cfg.get("enable_noise_model", False) + requested_mode = cfg.get("noise_model", self.DEFAULT_NOISE_MODEL) + requested_arb_frequency = self.get_effective_arb_frequency(cfg) + cache_key = ( + tuple(cfg.get("tokens", [])), + cfg.get("start"), + cfg.get("end"), + enable_noise_model, + requested_mode, + cfg.get("noise_artifact_dir", self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), + cfg.get("noise_pool_id", self.AAVE_WETH_POOL_ID), + requested_arb_frequency, + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + self._hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + ) + if cache_key in self._noise_settings_cache: + return self._noise_settings_cache[cache_key] + + if not enable_noise_model: + result = { + "noise_model": "arb_only", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": None, + "noise_arrays_path": None, + "arb_frequency": requested_arb_frequency, + "noise_summary": "arb_only (noise disabled)", + "noise_cache_key": ("disabled",), + } + elif requested_mode == "market_linear": + if self.noise_profile == "legacy_calibrated": + result = self._legacy_calibrated_noise_settings( + arb_frequency=requested_arb_frequency + ) + self._noise_settings_cache[cache_key] = result + return result + artifact_dir = cfg.get( + "noise_artifact_dir", + self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, + ) + pool_id = cfg.get("noise_pool_id", self.AAVE_WETH_POOL_ID) + start_date = str(cfg["start"]).split(" ")[0] + end_date = str(cfg["end"]).split(" ")[0] + arrays_path = cfg.get("noise_arrays_path") or os.path.join( + artifact_dir, + "_sim_arrays", + f"{pool_id}_{start_date}_{end_date}.npz", + ) + meta_path = os.path.join(artifact_dir, "meta.json") + model_path = os.path.join(artifact_dir, "model.npz") + if not ( + os.path.exists(arrays_path) + and os.path.exists(meta_path) + and os.path.exists(model_path) + ): + result = self._legacy_calibrated_noise_settings( + arb_frequency=requested_arb_frequency + ) + else: + with np.load(arrays_path) as arrays: + tvl_mean = float(arrays["tvl_mean"]) + tvl_std = float(arrays["tvl_std"]) + arb_frequency = requested_arb_frequency + result = { + "noise_model": "market_linear", + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, + }, + "noise_arrays_path": arrays_path, + "arb_frequency": arb_frequency, + "noise_summary": f"market_linear (arb_frequency={arb_frequency})", + "noise_cache_key": ( + "market_linear", + arrays_path, + arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), + ), + } + elif requested_mode == "calibrated": + result = self._legacy_calibrated_noise_settings( + arb_frequency=requested_arb_frequency + ) + else: + arb_frequency = requested_arb_frequency + result = { + "noise_model": requested_mode, + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": cfg.get("reclamm_noise_params"), + "noise_arrays_path": cfg.get("noise_arrays_path"), + "arb_frequency": arb_frequency, + "noise_summary": f"{requested_mode} (arb_frequency={arb_frequency})", + "noise_cache_key": ( + requested_mode, + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + self._hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), + arb_frequency, + ), + } + + self._noise_settings_cache[cache_key] = result + return result + + def make_noise_variant_cfg(self, cfg, enable_noise_model): + return self.normalize_compare_run_cfg( + cfg, + enable_noise_model=enable_noise_model, + ) + + @staticmethod + def _make_method_cache_hash(key): + return hashlib.sha256(repr(key).encode("utf-8")).hexdigest() + + def _make_method_cache_key(self, cfg, method): + cfg = self.normalize_compare_run_cfg(cfg) + noise_cfg = self.resolve_reclamm_noise_settings(cfg) + arb_frequency = self.get_effective_arb_frequency(cfg, noise_cfg) + key = ( + method, + bool(cfg.get("enable_noise_model", False)), + round(float(cfg["price_ratio"]), 6), + round(float(cfg["centeredness_margin"]), 6), + round(float(cfg["daily_price_shift_exponent"]), 6), + round(self.get_initial_pool_value(cfg), 2), + noise_cfg.get("noise_cache_key"), + None if arb_frequency is None else int(arb_frequency), + round( + float( + cfg.get( + "gas_cost", + self.DEFAULT_GAS_COST + if cfg.get("enable_noise_model", False) + else 0.0, + ) + ), + 6, + ), + round( + float( + cfg.get( + "protocol_fee_split", + self.DEFAULT_PROTOCOL_FEE_SPLIT + if cfg.get("enable_noise_model", False) + else 0.0, + ) + ), + 6, + ), + ) + if method == "constant_arc_length": + speed = cfg.get("arc_length_speed") + key += (None if speed is None else round(float(speed), 12),) + return key + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser( + description=( + "Identify horizontally and vertically adjacent reCLAMM heatmap cells " + "whose derived metric values differ by at least the requested threshold." + ) + ) + parser.add_argument( + "--metric-key", + default="noise_vs_arb_geometric_improvement_pct", + choices=sorted(CACHE_ONLY_METRIC_SPECS), + help="Heatmap metric to reconstruct from the persisted forward-value cache.", + ) + parser.add_argument( + "--pair-slug", + default="price_ratio_vs_margin", + help="Pair heatmap family slug, or 'all' to scan every pair family.", + ) + parser.add_argument( + "--slice-slug", + default="all", + help="Quarter-slice slug (q1/q2/q3/q4), or 'all' to scan every slice.", + ) + parser.add_argument( + "--min-diff", + type=float, + default=30.0, + help=( + "Minimum absolute difference between adjacent heatmap values. " + "For the default metric this is in percentage points." + ), + ) + parser.add_argument( + "--adjacency-axis", + default="both", + choices=("both", "horizontal", "vertical"), + help=( + "Which adjacency direction to scan. " + "'both' includes horizontal and vertical neighbors." + ), + ) + parser.add_argument( + "--initial-pool-value", + type=float, + default=1_000_000.0, + help="TVL in USD used for the cached heatmap sweep.", + ) + parser.add_argument( + "--config-index", + type=int, + default=1, + help="Which compare_reclamm_thermostats.py base config to use.", + ) + parser.add_argument( + "--cache-path", + default=None, + help="Optional parquet cache override. Defaults to the compare script's TVL cache.", + ) + parser.add_argument( + "--output-csv", + default=None, + help="Optional CSV output path. Defaults under scripts/results/.", + ) + parser.add_argument( + "--skip-top-row-geometric-comparison", + action="store_true", + help=( + "Skip the follow-up geometric noise comparison for the top CSV row. " + "By default the script attempts that comparison after writing the CSV." + ), + ) + parser.add_argument( + "--top-row-geometric-comparison-output-file", + default=None, + help="Optional PNG output path override for the top-row geometric comparison.", + ) + parser.add_argument( + "--allow-partial-cache", + action="store_true", + help="Write output even if some heatmap cells are missing from the cache.", + ) + return parser.parse_args() + + +def load_compare_module(module_path: Optional[Path] = None): + compare_path = module_path or Path(__file__).with_name("compare_reclamm_thermostats.py") + spec = importlib.util.spec_from_file_location( + "reclamm_compare_reclamm_thermostats", + compare_path, + ) + if spec is not None and spec.loader is not None: + module = importlib.util.module_from_spec(spec) + try: + spec.loader.exec_module(module) + return _LightweightCompareContext.from_compare_module(module) + except ModuleNotFoundError as exc: + if exc.name != "jax": + raise + print( + "compare_reclamm_thermostats.py depends on jax in this environment; " + "using the lightweight cache-key context instead." + ) + return _LightweightCompareContext() + + +def get_metric_spec(metric_key: str) -> Mapping[str, object]: + try: + return CACHE_ONLY_METRIC_SPECS[metric_key] + except KeyError as exc: + raise ValueError( + f"Unsupported metric_key={metric_key!r}. " + f"Supported cache-only metrics: {sorted(CACHE_ONLY_METRIC_SPECS)}" + ) from exc + + +def load_cache_lookup(cache_path: Path) -> Dict[str, float]: + frame = pd.read_parquet(cache_path, columns=["cache_key_hash", "final_value"]) + return { + str(row.cache_key_hash): float(row.final_value) + for row in frame.itertuples(index=False) + } + + +def resolve_existing_cache_path(cache_path: Path) -> Path: + candidates = [Path(cache_path)] + if not cache_path.is_absolute(): + candidates.append(Path("scripts") / cache_path) + for candidate in candidates: + if candidate.exists(): + return candidate + return Path(cache_path) + + +def load_geometric_compare_module(module_path: Optional[Path] = None): + compare_path = module_path or Path(__file__).with_name( + "compare_reclamm_geometric_noise_runs.py" + ) + spec = importlib.util.spec_from_file_location( + "reclamm_compare_reclamm_geometric_noise_runs", + compare_path, + ) + if spec is None or spec.loader is None: + raise RuntimeError(f"Could not load geometric compare module from {compare_path}") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def run_top_row_geometric_comparison( + csv_path: Path, + output_file: Optional[str] = None, + row_index: int = 0, +): + """Run the paired geometric-vs-arb comparison using the top adjacent CSV row.""" + module = load_geometric_compare_module() + if not hasattr(module, "run_adjacent_csv_row_comparison"): + raise RuntimeError( + "compare_reclamm_geometric_noise_runs.py does not expose " + "run_adjacent_csv_row_comparison" + ) + return module.run_adjacent_csv_row_comparison( + csv_path=csv_path, + row_index=row_index, + output_file=output_file, + ) + + +def resolve_pair_specs(compare_module, base_cfg: Mapping[str, object], pair_slug: str): + pair_specs = compare_module.get_pair_heatmap_specs(base_cfg) + if pair_slug == "all": + return pair_specs + + matched = [pair for pair in pair_specs if pair["slug"] == pair_slug] + if not matched: + available = [pair["slug"] for pair in pair_specs] + raise ValueError( + f"Unknown pair slug {pair_slug!r}. Available pair slugs: {available}" + ) + return matched + + +def resolve_slice_variants(pair_spec: Mapping[str, object], slice_slug: str): + slice_variants = pair_spec["fixed_slices"] + if slice_slug == "all": + return list(slice_variants) + + matched = [variant for variant in slice_variants if variant["slug"] == slice_slug] + if not matched: + available = [variant["slug"] for variant in slice_variants] + raise ValueError( + f"Unknown slice slug {slice_slug!r}. Available slice slugs: {available}" + ) + return matched + + +def build_default_output_path( + compare_module, + base_cfg: Mapping[str, object], + metric_key: str, + pair_slug: str, + slice_slug: str, + min_diff: float, +) -> Path: + output_dir = Path("scripts/results/reclamm_heatmap_adjacency") + output_dir.mkdir(parents=True, exist_ok=True) + diff_token = str(float(min_diff)).rstrip("0").rstrip(".").replace(".", "p") + filename = ( + f"reclamm_adjacent_pairs_{metric_key}_{pair_slug}_{slice_slug}" + f"_mindiff_{diff_token}_tvl_{compare_module.format_tvl_millions_slug(base_cfg)}.csv" + ) + return output_dir / filename + + +def autodetect_lightweight_noise_profile( + compare_module, + base_cfg: Mapping[str, object], + pair_specs: Sequence[Mapping[str, object]], + metric_key: str, + slice_slug: str, + cache_lookup: Mapping[str, float], +): + if not hasattr(compare_module, "set_noise_profile"): + return + + metric_spec = get_metric_spec(metric_key) + if not pair_specs: + return + + pair_spec = pair_specs[0] + slice_variants = resolve_slice_variants(pair_spec, slice_slug) + if not slice_variants: + return + + slice_variant = slice_variants[0] + x_values = list(pair_spec["x_values"]) + y_values = list(pair_spec["y_values"]) + sample_x_indices = sorted({0, len(x_values) // 2, len(x_values) - 1}) + sample_y_indices = sorted({0, len(y_values) // 2, len(y_values) - 1}) + + scores = {} + for profile in ("market_linear", "legacy_calibrated"): + compare_module.set_noise_profile(profile) + hit_count = 0 + probe_count = 0 + slice_cfg = dict(base_cfg) + slice_cfg[pair_spec["fixed_key"]] = float(slice_variant["value"]) + for y_index in sample_y_indices: + for x_index in sample_x_indices: + cfg = dict(slice_cfg) + cfg[pair_spec["x_key"]] = float(x_values[x_index]) + cfg[pair_spec["y_key"]] = float(y_values[y_index]) + for source_name in metric_spec["sources"]: + source_cfg, method = _source_variant(compare_module, cfg, source_name) + cache_key = compare_module._make_method_cache_key(source_cfg, method) + cache_key_hash = compare_module._make_method_cache_hash(cache_key) + probe_count += 1 + if cache_key_hash in cache_lookup: + hit_count += 1 + scores[profile] = (hit_count, probe_count) + + best_profile = max( + scores, + key=lambda profile: (scores[profile][0], scores[profile][1], profile == "market_linear"), + ) + compare_module.set_noise_profile(best_profile) + hit_count, probe_count = scores[best_profile] + print( + f"Lightweight noise profile auto-detect chose {best_profile} " + f"({hit_count}/{probe_count} sample cache hits)." + ) + + +def _source_variant(compare_module, cfg: Mapping[str, object], source_name: str): + enable_noise_model = source_name.startswith("noise_") + method = "geometric" if source_name.endswith("geometric") else "constant_arc_length" + source_cfg = compare_module.make_noise_variant_cfg(cfg, enable_noise_model) + source_cfg["noise_model"] = ( + getattr(compare_module, "DEFAULT_NOISE_MODEL", "market_linear") + if enable_noise_model + else "arb_only" + ) + return source_cfg, method + + +def _compute_metric_value(metric_key: str, final_values: Mapping[str, float]) -> float: + metric_spec = get_metric_spec(metric_key) + return float(metric_spec["compute"](final_values)) + + +def build_cell_record( + compare_module, + cfg: Mapping[str, object], + pair_spec: Mapping[str, object], + slice_variant: Mapping[str, object], + metric_key: str, + x_index: int, + y_index: int, + cache_lookup: Mapping[str, float], +): + metric_spec = get_metric_spec(metric_key) + final_values = {} + missing_hashes = [] + for source_name in metric_spec["sources"]: + source_cfg, method = _source_variant(compare_module, cfg, source_name) + cache_key = compare_module._make_method_cache_key(source_cfg, method) + cache_key_hash = compare_module._make_method_cache_hash(cache_key) + cached_value = cache_lookup.get(cache_key_hash) + if cached_value is None: + missing_hashes.append(cache_key_hash) + continue + final_values[source_name] = float(cached_value) + + if missing_hashes: + return None, missing_hashes + + return ( + { + "metric_key": metric_key, + "metric_unit": metric_spec["unit"], + "source_noise_profile": str( + getattr(compare_module, "noise_profile", "unknown") + ), + "pair_slug": pair_spec["slug"], + "slice_slug": slice_variant["slug"], + "slice_label": slice_variant["label"], + "fixed_key": pair_spec["fixed_key"], + "fixed_value": float(slice_variant["value"]), + "price_ratio": float(cfg["price_ratio"]), + "centeredness_margin": float(cfg["centeredness_margin"]), + "daily_price_shift_exponent": float(cfg["daily_price_shift_exponent"]), + "tvl_usd": float(compare_module.get_initial_pool_value(cfg)), + "heatmap_value": _compute_metric_value(metric_key, final_values), + "x_index": int(x_index), + "y_index": int(y_index), + }, + [], + ) + + +def build_slice_cell_grid( + compare_module, + base_cfg: Mapping[str, object], + pair_spec: Mapping[str, object], + slice_variant: Mapping[str, object], + metric_key: str, + cache_lookup: Mapping[str, float], +): + records_by_coord: Dict[Tuple[int, int], MutableMapping[str, object]] = {} + missing_hashes: List[str] = [] + x_values = pair_spec["x_values"] + y_values = pair_spec["y_values"] + slice_cfg = dict(base_cfg) + slice_cfg[pair_spec["fixed_key"]] = float(slice_variant["value"]) + + for y_index, y_value in enumerate(y_values): + for x_index, x_value in enumerate(x_values): + cfg = dict(slice_cfg) + cfg[pair_spec["x_key"]] = float(x_value) + cfg[pair_spec["y_key"]] = float(y_value) + record, missing_for_cell = build_cell_record( + compare_module=compare_module, + cfg=cfg, + pair_spec=pair_spec, + slice_variant=slice_variant, + metric_key=metric_key, + x_index=x_index, + y_index=y_index, + cache_lookup=cache_lookup, + ) + if record is not None: + records_by_coord[(y_index, x_index)] = record + missing_hashes.extend(missing_for_cell) + + expected_cell_count = len(x_values) * len(y_values) + return { + "records_by_coord": records_by_coord, + "expected_cell_count": expected_cell_count, + "resolved_cell_count": len(records_by_coord), + "missing_hash_count": len(missing_hashes), + "missing_hashes": missing_hashes, + } + + +def build_adjacent_row( + metric_key: str, + metric_unit: str, + axis: str, + first_cell: Mapping[str, object], + second_cell: Mapping[str, object], +) -> Dict[str, object]: + signed_diff = float(second_cell["heatmap_value"]) - float(first_cell["heatmap_value"]) + abs_diff = abs(signed_diff) + return { + "metric_key": metric_key, + "metric_unit": metric_unit, + "source_noise_profile": first_cell.get("source_noise_profile", "unknown"), + "pair_slug": first_cell["pair_slug"], + "slice_slug": first_cell["slice_slug"], + "slice_label": first_cell["slice_label"], + "fixed_key": first_cell["fixed_key"], + "fixed_value": float(first_cell["fixed_value"]), + "adjacency_axis": axis, + "heatmap_value_diff_abs": abs_diff, + "heatmap_value_diff_signed_2_minus_1": signed_diff, + "1_price_ratio": float(first_cell["price_ratio"]), + "1_centeredness_margin": float(first_cell["centeredness_margin"]), + "1_daily_price_shift_exponent": float(first_cell["daily_price_shift_exponent"]), + "1_tvl_usd": float(first_cell["tvl_usd"]), + "1_heatmap_value": float(first_cell["heatmap_value"]), + "1_x_index": int(first_cell["x_index"]), + "1_y_index": int(first_cell["y_index"]), + "2_price_ratio": float(second_cell["price_ratio"]), + "2_centeredness_margin": float(second_cell["centeredness_margin"]), + "2_daily_price_shift_exponent": float(second_cell["daily_price_shift_exponent"]), + "2_tvl_usd": float(second_cell["tvl_usd"]), + "2_heatmap_value": float(second_cell["heatmap_value"]), + "2_x_index": int(second_cell["x_index"]), + "2_y_index": int(second_cell["y_index"]), + } + + +def find_adjacent_rows_for_slice( + metric_key: str, + metric_unit: str, + records_by_coord: Mapping[Tuple[int, int], Mapping[str, object]], + x_count: int, + y_count: int, + min_diff: float, + adjacency_axis: str = "both", +) -> List[Dict[str, object]]: + rows = [] + + if adjacency_axis not in {"both", "horizontal", "vertical"}: + raise ValueError( + f"Unsupported adjacency_axis={adjacency_axis!r}; expected both, horizontal, or vertical" + ) + + if adjacency_axis in {"both", "horizontal"}: + for y_index in range(y_count): + for x_index in range(x_count - 1): + first_cell = records_by_coord.get((y_index, x_index)) + second_cell = records_by_coord.get((y_index, x_index + 1)) + if first_cell is None or second_cell is None: + continue + row = build_adjacent_row( + metric_key=metric_key, + metric_unit=metric_unit, + axis="horizontal", + first_cell=first_cell, + second_cell=second_cell, + ) + if row["heatmap_value_diff_abs"] >= min_diff: + rows.append(row) + + if adjacency_axis in {"both", "vertical"}: + for y_index in range(y_count - 1): + for x_index in range(x_count): + first_cell = records_by_coord.get((y_index, x_index)) + second_cell = records_by_coord.get((y_index + 1, x_index)) + if first_cell is None or second_cell is None: + continue + row = build_adjacent_row( + metric_key=metric_key, + metric_unit=metric_unit, + axis="vertical", + first_cell=first_cell, + second_cell=second_cell, + ) + if row["heatmap_value_diff_abs"] >= min_diff: + rows.append(row) + + rows.sort( + key=lambda row: ( + -float(row["heatmap_value_diff_abs"]), + str(row["pair_slug"]), + str(row["slice_slug"]), + str(row["adjacency_axis"]), + int(row["1_y_index"]), + int(row["1_x_index"]), + ) + ) + return rows + + +def scan_heatmap_pairs( + compare_module, + base_cfg: Mapping[str, object], + metric_key: str, + pair_specs: Sequence[Mapping[str, object]], + slice_slug: str, + min_diff: float, + adjacency_axis: str, + cache_lookup: Mapping[str, float], +): + metric_spec = get_metric_spec(metric_key) + all_rows: List[Dict[str, object]] = [] + diagnostics = [] + + for pair_spec in pair_specs: + slice_variants = resolve_slice_variants(pair_spec, slice_slug) + for slice_variant in slice_variants: + slice_scan = build_slice_cell_grid( + compare_module=compare_module, + base_cfg=base_cfg, + pair_spec=pair_spec, + slice_variant=slice_variant, + metric_key=metric_key, + cache_lookup=cache_lookup, + ) + diagnostics.append( + { + "pair_slug": pair_spec["slug"], + "slice_slug": slice_variant["slug"], + "resolved_cell_count": slice_scan["resolved_cell_count"], + "expected_cell_count": slice_scan["expected_cell_count"], + "missing_hash_count": slice_scan["missing_hash_count"], + } + ) + all_rows.extend( + find_adjacent_rows_for_slice( + metric_key=metric_key, + metric_unit=metric_spec["unit"], + records_by_coord=slice_scan["records_by_coord"], + x_count=len(pair_spec["x_values"]), + y_count=len(pair_spec["y_values"]), + min_diff=min_diff, + adjacency_axis=adjacency_axis, + ) + ) + + return all_rows, diagnostics + + +def rows_to_frame(rows: Iterable[Mapping[str, object]]) -> pd.DataFrame: + frame = pd.DataFrame(list(rows)) + if frame.empty: + return pd.DataFrame(columns=OUTPUT_COLUMNS) + frame = frame.loc[:, OUTPUT_COLUMNS] + frame.sort_values( + by=[ + "heatmap_value_diff_abs", + "pair_slug", + "slice_slug", + "adjacency_axis", + "1_y_index", + "1_x_index", + ], + ascending=[False, True, True, True, True, True], + inplace=True, + ignore_index=True, + ) + return frame + + +def main() -> int: + args = parse_args() + compare_module = load_compare_module() + + if not 0 <= args.config_index < len(compare_module.CONFIGS): + raise ValueError( + f"config-index {args.config_index} is out of range for " + f"{len(compare_module.CONFIGS)} available configs" + ) + + base_cfg = compare_module.configs_for_tvl( + compare_module.CONFIGS, + initial_pool_value=args.initial_pool_value, + )[args.config_index] + pair_specs = resolve_pair_specs(compare_module, base_cfg, args.pair_slug) + + cache_path = ( + Path(args.cache_path) + if args.cache_path is not None + else Path(compare_module._heatmap_forward_cache_path(base_cfg)) + ) + cache_path = resolve_existing_cache_path(cache_path) + if not cache_path.exists(): + raise FileNotFoundError(f"Cache parquet not found: {cache_path}") + + cache_lookup = load_cache_lookup(cache_path) + autodetect_lightweight_noise_profile( + compare_module=compare_module, + base_cfg=base_cfg, + pair_specs=pair_specs, + metric_key=args.metric_key, + slice_slug=args.slice_slug, + cache_lookup=cache_lookup, + ) + output_csv = ( + Path(args.output_csv) + if args.output_csv is not None + else build_default_output_path( + compare_module=compare_module, + base_cfg=base_cfg, + metric_key=args.metric_key, + pair_slug=args.pair_slug, + slice_slug=args.slice_slug, + min_diff=args.min_diff, + ) + ) + + print( + f"Loaded {len(cache_lookup):,} cached final values from {cache_path} " + f"for {base_cfg['name']} at TVL {compare_module.format_tvl_millions_label(base_cfg)}." + ) + print( + f"Scanning metric={args.metric_key}, pair_slug={args.pair_slug}, " + f"slice_slug={args.slice_slug}, adjacency_axis={args.adjacency_axis}, " + f"min_diff={args.min_diff} " + f"({get_metric_spec(args.metric_key)['unit']})." + ) + + rows, diagnostics = scan_heatmap_pairs( + compare_module=compare_module, + base_cfg=base_cfg, + metric_key=args.metric_key, + pair_specs=pair_specs, + slice_slug=args.slice_slug, + min_diff=args.min_diff, + adjacency_axis=args.adjacency_axis, + cache_lookup=cache_lookup, + ) + + for diagnostic in diagnostics: + print( + f"[{diagnostic['pair_slug']}:{diagnostic['slice_slug']}] " + f"resolved {diagnostic['resolved_cell_count']}/" + f"{diagnostic['expected_cell_count']} cells " + f"({diagnostic['missing_hash_count']} missing cache hashes)" + ) + + missing_any = any(diagnostic["missing_hash_count"] > 0 for diagnostic in diagnostics) + if missing_any and not args.allow_partial_cache: + raise RuntimeError( + "Cache was incomplete for at least one requested heatmap slice. " + "Re-run with --allow-partial-cache to write the rows that were resolvable." + ) + + frame = rows_to_frame(rows) + output_csv.parent.mkdir(parents=True, exist_ok=True) + frame.to_csv(output_csv, index=False) + print(f"Wrote {len(frame):,} adjacent pairs to {output_csv}") + + if not args.skip_top_row_geometric_comparison: + if frame.empty: + print("Skipping top-row geometric comparison because the CSV is empty.") + else: + print( + "Running geometric noise comparison for the top adjacent-pairs CSV row..." + ) + try: + comparison_output = run_top_row_geometric_comparison( + csv_path=output_csv, + output_file=args.top_row_geometric_comparison_output_file, + row_index=0, + ) + print( + f"Completed top-row geometric comparison using {output_csv} row 0. " + f"Output: {comparison_output}" + ) + except Exception as exc: # pragma: no cover - depends on local runtime deps + print( + "Top-row geometric comparison did not run successfully: " + f"{exc}" + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/scripts/test_compare_reclamm_geometric_noise_runs.py b/tests/scripts/test_compare_reclamm_geometric_noise_runs.py new file mode 100644 index 00000000..8a0fb740 --- /dev/null +++ b/tests/scripts/test_compare_reclamm_geometric_noise_runs.py @@ -0,0 +1,143 @@ +"""Tests for adjacent-row sourcing in compare_reclamm_geometric_noise_runs.py.""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import numpy as np + + +SCRIPT_PATH = ( + Path(__file__).resolve().parents[2] + / "scripts" + / "reclamm" + / "compare_reclamm_geometric_noise_runs.py" +) + + +def load_script_module(): + spec = importlib.util.spec_from_file_location( + "test_compare_reclamm_geometric_noise_runs_module", + SCRIPT_PATH, + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +def test_build_run_specs_from_adjacent_row_maps_csv_cells_to_two_specs(): + module = load_script_module() + row = { + "metric_key": "noise_vs_arb_geometric_improvement_pct", + "metric_unit": "pct", + "source_noise_profile": "legacy_calibrated", + "pair_slug": "price_ratio_vs_margin", + "slice_slug": "q2", + "adjacency_axis": "horizontal", + "heatmap_value_diff_abs": 54.223889391076895, + "1_price_ratio": 1.335, + "1_centeredness_margin": 0.3184210526, + "1_daily_price_shift_exponent": 0.1975, + "1_tvl_usd": 1_000_000.0, + "1_heatmap_value": -53.0543210862, + "2_price_ratio": 1.36, + "2_centeredness_margin": 0.3184210526, + "2_daily_price_shift_exponent": 0.1975, + "2_tvl_usd": 1_000_000.0, + "2_heatmap_value": 1.1695683049, + } + + description, run_specs = module.build_run_specs_from_adjacent_row( + row, + csv_path=Path("adjacent_pairs.csv"), + row_index=0, + ) + + assert "adjacent_pairs.csv row 0" in description + assert "price_ratio_vs_margin q2" in description + assert "horizontal" in description + assert "noise_profile=legacy_calibrated" in description + assert len(run_specs) == 2 + assert run_specs[0]["name"] == "Top diff row cell 1" + assert run_specs[0]["price_ratio"] == 1.335 + assert run_specs[0]["centeredness_margin"] == 0.3184210526 + assert run_specs[0]["daily_price_shift_exponent"] == 0.1975 + assert run_specs[0]["tvl_usd"] == 1_000_000.0 + assert run_specs[0]["color"] == "C0" + assert run_specs[0]["source_noise_profile"] == "legacy_calibrated" + assert "heatmap_value=-53.054321" in run_specs[0]["reason"] + assert run_specs[1]["name"] == "Top diff row cell 2" + assert run_specs[1]["price_ratio"] == 1.36 + assert run_specs[1]["color"] == "C1" + + +def test_default_output_file_for_adjacent_csv_uses_csv_stem_and_row_index(): + module = load_script_module() + output = module.default_output_file_for_adjacent_csv( + Path("scripts/results/reclamm_heatmap_adjacency/example.csv"), + row_index=3, + ) + + assert ( + output.as_posix() + == "scripts/results/reclamm_heatmap_adjacency/example_row_3_geometric_noise_compare.png" + ) + + +def test_build_run_config_honors_legacy_calibrated_noise_profile(): + module = load_script_module() + base_config = { + "name": "base", + "price_ratio": 1.1, + "centeredness_margin": 0.6, + "daily_price_shift_exponent": 0.1, + "initial_pool_value": 1_000_000.0, + "noise_model": "market_linear", + "reclamm_noise_params": {"foo": 1.0}, + "noise_arrays_path": "path.npz", + } + spec = { + "name": "cell", + "price_ratio": 1.335, + "centeredness_margin": 0.3184210526, + "daily_price_shift_exponent": 0.1975, + "tvl_usd": 1_000_000.0, + "source_noise_profile": "legacy_calibrated", + } + + cfg = module.build_run_config(spec, base_config=base_config) + + assert cfg["noise_model"] == "calibrated" + assert "reclamm_noise_params" not in cfg + assert "noise_arrays_path" not in cfg + + +def test_print_run_inputs_to_terminal_includes_fingerprint_and_update_params(capsys): + module = load_script_module() + cfg = { + "name": "cell", + "variant_label": "arb-only", + } + run_fingerprint = { + "tokens": ["AAVE", "ETH"], + "fees": np.float64(0.0025), + "arb_frequency": np.int64(14), + } + update_params = { + "price_ratio": np.array(1.335), + "centeredness_margin": np.array(0.3184210526), + "daily_price_shift_base": np.array(0.99999841596), + } + + module.print_run_inputs_to_terminal(cfg, run_fingerprint, update_params) + + captured = capsys.readouterr().out + assert "Run inputs for cell (arb-only):" in captured + assert '"run_fingerprint"' in captured + assert '"update_params"' in captured + assert '"tokens": [' in captured + assert '"AAVE"' in captured + assert '"arb_frequency": 14' in captured + assert '"price_ratio": 1.335' in captured diff --git a/tests/scripts/test_compare_reclamm_thermostats.py b/tests/scripts/test_compare_reclamm_thermostats.py index 7072f73c..3ea0da79 100644 --- a/tests/scripts/test_compare_reclamm_thermostats.py +++ b/tests/scripts/test_compare_reclamm_thermostats.py @@ -34,6 +34,7 @@ def inject_module(name, module): pandas_module.Timestamp = lambda value: value pandas_module.DatetimeIndex = tuple pandas_module.DataFrame = type("DataFrame", (), {}) + pandas_module.read_parquet = lambda *args, **kwargs: None inject_module("pandas", pandas_module) matplotlib_module = types.ModuleType("matplotlib") @@ -42,6 +43,7 @@ def inject_module(name, module): colors_module = types.ModuleType("matplotlib.colors") colors_module.TwoSlopeNorm = object colors_module.Normalize = object + colors_module.SymLogNorm = object cm_module = types.ModuleType("matplotlib.cm") cm_module.ScalarMappable = object inject_module("matplotlib", matplotlib_module) @@ -123,9 +125,12 @@ def base_cfg(): return { "tokens": ["AAVE", "ETH"], "start": "2024-06-01 00:00:00", + "end": "2025-06-01 00:00:00", + "fees": 0.0025, "price_ratio": 1.10, "centeredness_margin": 0.60, "daily_price_shift_exponent": 0.1, + "initial_pool_value": 5_000_000.0, } @@ -137,6 +142,98 @@ def launch_final_values(): } +def test_make_noise_variant_cfg_disables_noise_fields(script_module, base_cfg): + noisy_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + "noise_artifact_dir": "results/linear_market_noise", + "noise_pool_id": "0x9d1fcf346ea1b0", + "gas_cost": 1.0, + "protocol_fee_split": 0.25, + "reclamm_noise_params": {"tvl_mean": 1.0, "tvl_std": 2.0}, + "noise_arrays_path": "results/linear_market_noise/_sim_arrays/aave_eth.npz", + "arb_frequency": 6, + } + + arb_only_cfg = script_module.make_noise_variant_cfg( + noisy_cfg, + enable_noise_model=False, + ) + resolved = script_module.resolve_reclamm_noise_settings(arb_only_cfg) + + assert arb_only_cfg["enable_noise_model"] is False + assert arb_only_cfg["noise_model"] is None + assert arb_only_cfg["gas_cost"] == script_module.DEFAULT_GAS_COST + assert arb_only_cfg["protocol_fee_split"] == script_module.DEFAULT_PROTOCOL_FEE_SPLIT + assert arb_only_cfg["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY + assert "noise_artifact_dir" not in arb_only_cfg + assert "noise_pool_id" not in arb_only_cfg + assert "reclamm_noise_params" not in arb_only_cfg + assert "noise_arrays_path" not in arb_only_cfg + assert resolved["noise_model"] is None + assert resolved["noise_summary"] == "arb-only (noise disabled)" + + +def test_make_noise_variant_cfg_defaults_to_fixed_compare_arb_cadence( + script_module, + base_cfg, +): + noisy_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + + arb_only_cfg = script_module.make_noise_variant_cfg( + noisy_cfg, + enable_noise_model=False, + ) + resolved_noise = script_module.resolve_reclamm_noise_settings(noisy_cfg) + + assert arb_only_cfg["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY + assert resolved_noise["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY + + +def test_make_fingerprint_ignores_non_axis_override_fields(script_module, base_cfg): + canonical_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + noisy_override_cfg = { + **canonical_cfg, + "arb_frequency": 6, + "gas_cost": 7.0, + "protocol_fee_split": 0.9, + "arb_fees": 3.0, + "noise_artifact_dir": "custom/noise/dir", + "noise_pool_id": "override-pool", + "reclamm_noise_params": {"tvl_mean": 999.0}, + "noise_arrays_path": "custom/path.npz", + } + + canonical_fingerprint = script_module.make_fingerprint(canonical_cfg, "geometric") + overridden_fingerprint = script_module.make_fingerprint( + noisy_override_cfg, + "geometric", + ) + canonical_key = script_module._make_method_cache_key(canonical_cfg, "geometric") + overridden_key = script_module._make_method_cache_key( + noisy_override_cfg, + "geometric", + ) + + assert overridden_fingerprint == canonical_fingerprint + assert overridden_key == canonical_key + assert overridden_fingerprint["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY + assert overridden_fingerprint["gas_cost"] == script_module.DEFAULT_GAS_COST + assert ( + overridden_fingerprint["protocol_fee_split"] + == script_module.DEFAULT_PROTOCOL_FEE_SPLIT + ) + + def test_generate_heatmaps_skips_existing_pairs( monkeypatch, script_module, @@ -169,7 +266,14 @@ def test_generate_heatmaps_only_renders_missing_artifacts( base_cfg, launch_final_values, ): - missing_file = "reclamm_heatmap_efficiency_price_ratio_vs_margin.png" + pair = script_module.get_pair_heatmap_specs(base_cfg)[0] + slice_variant = pair["fixed_slices"][0] + pair_suffix = script_module._pair_slice_suffix(pair, slice_variant) + missing_file = script_module.tvl_artifact_filename( + "reclamm_heatmap_geometric_vs_launch_geometric_symlog20", + base_cfg, + suffix=pair_suffix, + ) def fake_exists(filename): if filename == missing_file: @@ -182,7 +286,7 @@ def fake_exists(filename): def fake_build_heatmap_matrices(**kwargs): build_calls.append(kwargs) return { - "efficiency_pct": np.zeros( + "geometric_vs_launch_geometric_pct": np.zeros( (len(kwargs["y_values"]), len(kwargs["x_values"])), dtype=float, ) @@ -207,18 +311,176 @@ def fake_plot_heatmap(**kwargs): ) assert len(build_calls) == 1 - assert build_calls[0]["progress_label"] == "price_ratio_vs_margin" - assert build_calls[0]["metric_keys"] == ["efficiency_pct"] + assert build_calls[0]["progress_label"] == pair_suffix + assert build_calls[0]["base_cfg"][pair["fixed_key"]] == pytest.approx( + slice_variant["value"] + ) + assert build_calls[0]["metric_keys"] == ["geometric_vs_launch_geometric_pct"] assert plotted_files == [missing_file] +def test_generate_heatmaps_only_renders_missing_improvement_artifacts( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + pair = script_module.get_pair_heatmap_specs(base_cfg)[0] + slice_variant = pair["fixed_slices"][0] + pair_suffix = script_module._pair_slice_suffix(pair, slice_variant) + missing_file = script_module.tvl_artifact_filename( + "reclamm_heatmap_noise_vs_arb_geometric_improvement_symlog20", + base_cfg, + suffix=pair_suffix, + ) + + def fake_exists(filename): + if filename == missing_file: + return False + return filename.startswith("reclamm_heatmap_") + + build_calls = [] + plotted_files = [] + + def fake_build_heatmap_matrices(**kwargs): + build_calls.append(kwargs) + return { + "noise_vs_arb_geometric_improvement_pct": np.zeros( + (len(kwargs["y_values"]), len(kwargs["x_values"])), + dtype=float, + ) + } + + def fake_plot_heatmap(**kwargs): + plotted_files.append(kwargs["filename"]) + + monkeypatch.setattr(script_module.os.path, "exists", fake_exists) + monkeypatch.setattr( + script_module, + "build_heatmap_matrices", + fake_build_heatmap_matrices, + ) + monkeypatch.setattr(script_module, "plot_heatmap", fake_plot_heatmap) + + script_module.generate_heatmaps( + base_cfg, + price_data=None, + launch_final_values=launch_final_values, + cache={}, + ) + + assert len(build_calls) == 1 + assert build_calls[0]["progress_label"] == pair_suffix + assert build_calls[0]["base_cfg"][pair["fixed_key"]] == pytest.approx( + slice_variant["value"] + ) + assert build_calls[0]["metric_keys"] == [ + "noise_vs_arb_geometric_improvement_pct" + ] + assert plotted_files == [missing_file] + + +def test_generate_three_variable_3d_heatmaps_only_renders_missing_slice( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + missing_file = script_module.tvl_artifact_filename( + "reclamm_heatmap_3d_geometric_vs_launch_geometric_symlog20", + base_cfg, + suffix="slice_q1", + ) + + def fake_exists(filename): + if filename == missing_file: + return False + return filename.startswith("reclamm_heatmap_3d_") + + build_calls = [] + plotted_files = [] + + def fake_build_heatmap_matrices(**kwargs): + build_calls.append(kwargs) + return { + "geometric_vs_launch_geometric_pct": np.zeros( + (len(kwargs["y_values"]), len(kwargs["x_values"])), + dtype=float, + ) + } + + def fake_plot_three_variable_heatmap_3d(**kwargs): + plotted_files.append(kwargs["filename"]) + + monkeypatch.setattr(script_module.os.path, "exists", fake_exists) + monkeypatch.setattr( + script_module, + "build_heatmap_matrices", + fake_build_heatmap_matrices, + ) + monkeypatch.setattr( + script_module, + "plot_three_variable_heatmap_3d", + fake_plot_three_variable_heatmap_3d, + ) + + script_module.generate_three_variable_3d_heatmaps( + base_cfg, + price_data=None, + launch_final_values=launch_final_values, + cache={}, + ) + + assert len(build_calls) == 3 + assert {call["progress_label"] for call in build_calls} == { + "3d_price_ratio_vs_margin_shift_exp_q1", + "3d_shift_exp_vs_margin_price_ratio_q1", + "3d_price_ratio_vs_shift_exp_margin_q1", + } + assert plotted_files == [missing_file] + + +def test_run_method_final_value_cached_reuses_persisted_parquet_value( + monkeypatch, + script_module, + base_cfg, +): + cfg = dict(base_cfg) + cache_key = script_module._make_method_cache_key(cfg, "geometric") + cache_key_hash = script_module._make_method_cache_hash(cache_key) + + class FakeFrame: + empty = False + + def itertuples(self, index=False): + return [ + types.SimpleNamespace( + cache_key_hash=cache_key_hash, + final_value=1_234_567.0, + ) + ] + + monkeypatch.setattr(script_module.os.path, "exists", lambda filename: True) + monkeypatch.setattr(script_module.pd, "read_parquet", lambda *args, **kwargs: FakeFrame()) + monkeypatch.setattr( + script_module, + "do_run_on_historic_data", + lambda **kwargs: pytest.fail("persisted forward-value cache should be reused"), + ) + + cache = script_module.make_sweep_cache(price_data=None, cache_scope_cfg=cfg) + value = script_module._run_method_final_value_cached(cfg, "geometric", cache) + + assert value == pytest.approx(1_234_567.0) + + def test_arc_speed_artifacts_only_build_missing_line_output( monkeypatch, script_module, base_cfg, launch_final_values, ): - missing_line = "reclamm_line_efficiency_arc_speed_vs_price_ratio.png" + missing_line = script_module.tvl_artifact_filename("reclamm_line_efficiency", base_cfg, suffix="arc_speed_vs_price_ratio") def fake_exists(filename): if filename == missing_line: @@ -243,6 +505,7 @@ def fake_build_metric_curve(**kwargs): return np.zeros(len(kwargs["x_values"]), dtype=float) monkeypatch.setattr(script_module.os.path, "exists", fake_exists) + monkeypatch.setattr(script_module, "RUN_CONSTANT_ARC_LENGTH", True) monkeypatch.setattr( script_module, "compute_auto_calibrated_arc_length_speed", @@ -282,3 +545,134 @@ def fake_build_metric_curve(**kwargs): assert len(curve_calls) == 1 assert curve_calls[0]["x_key"] == "arc_length_speed" assert plotted_lines == [missing_line] + + +def test_flush_sweep_cache_writes_compact_scalar_parquet(script_module): + captured = {} + + class FakeFrame: + def __init__(self, payload): + captured["payload"] = payload + + def sort_values(self, *args, **kwargs): + captured["sort_values"] = (args, kwargs) + + def to_parquet(self, path, index=False, compression=None): + captured["path"] = path + captured["index"] = index + captured["compression"] = compression + + script_module.pd.DataFrame = FakeFrame + script_module.os.makedirs = lambda *args, **kwargs: captured.setdefault( + "makedirs", args[0] + ) + + cache = { + "_pending_persistent_final_values": { + "abc123": { + "cache_key_hash": "abc123", + "final_value": 123.45, + "method": "geometric", + "enable_noise_model": True, + "noise_model": "market_linear", + "price_ratio": 1.1, + "centeredness_margin": 0.6, + "daily_price_shift_exponent": 0.1, + "initial_pool_value": 5_000_000.0, + "arb_frequency": 15, + } + }, + "_persistent_final_value_cache": {}, + "_persistent_final_value_records": {}, + "_persistent_final_value_cache_loaded": True, + "_persistent_final_value_cache_path": "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_5m.parquet", + } + + script_module.flush_sweep_cache(cache, force=True) + + assert set(captured["payload"].keys()) == set( + script_module.PERSISTED_FORWARD_VALUE_COLUMNS + ) + assert captured["payload"]["cache_key_hash"] == ["abc123"] + assert captured["payload"]["method"] == ["geometric"] + assert captured["payload"]["arb_frequency"] == [15] + assert captured["index"] is False + assert captured["compression"] == "zstd" + assert captured["path"].endswith(".parquet") + assert cache["_pending_persistent_final_values"] == {} + assert cache["_persistent_final_value_cache"] == {"abc123": 123.45} + assert cache["_persistent_final_value_records"]["abc123"]["noise_model"] == "market_linear" + + +def test_load_persistent_final_value_cache_supports_legacy_two_column_parquet( + monkeypatch, + script_module, +): + class FakeFrame: + empty = False + + def itertuples(self, index=False): + return [ + types.SimpleNamespace( + cache_key_hash="legacy123", + final_value=999.0, + ) + ] + + monkeypatch.setattr(script_module.os.path, "exists", lambda filename: True) + monkeypatch.setattr(script_module.pd, "read_parquet", lambda *args, **kwargs: FakeFrame()) + + cache = { + "_persistent_final_value_cache_loaded": False, + "_persistent_final_value_cache_path": "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_1m.parquet", + } + + script_module._load_persistent_final_value_cache(cache) + + assert cache["_persistent_final_value_cache"] == {"legacy123": 999.0} + assert cache["_persistent_final_value_records"]["legacy123"]["final_value"] == pytest.approx(999.0) + assert cache["_persistent_final_value_records"]["legacy123"]["price_ratio"] is None + + +def test_run_comparison_cached_only_uses_geometric_runs_when_constant_arc_disabled( + monkeypatch, + script_module, + base_cfg, + launch_final_values, +): + calls = [] + + monkeypatch.setattr( + script_module, + "_make_comparison_cache_key", + lambda cfg, launch_final_values: ("cache", round(float(cfg["price_ratio"]), 6)), + ) + + def fake_run_method_final_value_cached(cfg, method, cache): + calls.append((cfg.get("enable_noise_model", False), method)) + return { + (True, "geometric"): 1_050_000.0, + (False, "geometric"): 1_000_000.0, + }[(cfg.get("enable_noise_model", False), method)] + + monkeypatch.setattr( + script_module, + "_run_method_final_value_cached", + fake_run_method_final_value_cached, + ) + + metrics = script_module.run_comparison_cached( + base_cfg, + cache={"_comparison_cache": {}, "_final_value_cache": {}, "_shared_price_data": None}, + launch_final_values=launch_final_values, + metric_keys=( + "geometric_vs_launch_geometric_pct", + "noise_vs_arb_geometric_improvement_pct", + ), + ) + + assert calls == [(True, "geometric"), (False, "geometric")] + assert metrics == { + "geometric_vs_launch_geometric_pct": pytest.approx(5.0), + "noise_vs_arb_geometric_improvement_pct": pytest.approx(5.0), + } diff --git a/tests/scripts/test_find_adjacent_heatmap_pairs.py b/tests/scripts/test_find_adjacent_heatmap_pairs.py new file mode 100644 index 00000000..597e65b1 --- /dev/null +++ b/tests/scripts/test_find_adjacent_heatmap_pairs.py @@ -0,0 +1,313 @@ +"""Tests for cache-backed adjacent heatmap pair detection.""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import pytest + + +SCRIPT_PATH = ( + Path(__file__).resolve().parents[2] + / "scripts" + / "reclamm" + / "find_adjacent_heatmap_pairs.py" +) + + +def load_script_module(): + spec = importlib.util.spec_from_file_location( + "test_find_adjacent_heatmap_pairs_module", + SCRIPT_PATH, + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +def make_cell(x_index, y_index, heatmap_value, **overrides): + cell = { + "metric_key": "noise_vs_arb_geometric_improvement_pct", + "metric_unit": "pct", + "pair_slug": "price_ratio_vs_margin", + "slice_slug": "q2", + "slice_label": "Q2", + "fixed_key": "daily_price_shift_exponent", + "fixed_value": 0.1975, + "price_ratio": 1.01 + 0.1 * x_index, + "centeredness_margin": 0.05 + 0.1 * y_index, + "daily_price_shift_exponent": 0.1975, + "tvl_usd": 1_000_000.0, + "heatmap_value": float(heatmap_value), + "x_index": int(x_index), + "y_index": int(y_index), + } + cell.update(overrides) + return cell + + +def test_find_adjacent_rows_for_slice_filters_and_sorts_descending(): + module = load_script_module() + records_by_coord = { + (0, 0): make_cell(0, 0, 0.0), + (0, 1): make_cell(1, 0, 35.0), + (0, 2): make_cell(2, 0, -10.0), + (1, 0): make_cell(0, 1, 5.0), + (1, 1): make_cell(1, 1, -40.0), + (1, 2): make_cell(2, 1, -50.0), + } + + rows = module.find_adjacent_rows_for_slice( + metric_key="noise_vs_arb_geometric_improvement_pct", + metric_unit="pct", + records_by_coord=records_by_coord, + x_count=3, + y_count=2, + min_diff=30.0, + ) + + assert [row["heatmap_value_diff_abs"] for row in rows] == [75.0, 45.0, 45.0, 40.0, 35.0] + assert rows[0]["adjacency_axis"] == "vertical" + assert rows[0]["1_x_index"] == 1 + assert rows[0]["1_y_index"] == 0 + assert rows[0]["2_x_index"] == 1 + assert rows[0]["2_y_index"] == 1 + + horizontal_rows = module.find_adjacent_rows_for_slice( + metric_key="noise_vs_arb_geometric_improvement_pct", + metric_unit="pct", + records_by_coord=records_by_coord, + x_count=3, + y_count=2, + min_diff=30.0, + adjacency_axis="horizontal", + ) + assert {row["adjacency_axis"] for row in horizontal_rows} == {"horizontal"} + + vertical_rows = module.find_adjacent_rows_for_slice( + metric_key="noise_vs_arb_geometric_improvement_pct", + metric_unit="pct", + records_by_coord=records_by_coord, + x_count=3, + y_count=2, + min_diff=30.0, + adjacency_axis="vertical", + ) + assert {row["adjacency_axis"] for row in vertical_rows} == {"vertical"} + + +def test_build_slice_cell_grid_reconstructs_metric_values_from_cache_hashes(): + module = load_script_module() + + class FakeCompareModule: + @staticmethod + def make_noise_variant_cfg(cfg, enable_noise_model): + updated = dict(cfg) + updated["enable_noise_model"] = bool(enable_noise_model) + return updated + + @staticmethod + def _make_method_cache_key(cfg, method): + return ( + method, + bool(cfg["enable_noise_model"]), + round(float(cfg["price_ratio"]), 6), + round(float(cfg["centeredness_margin"]), 6), + round(float(cfg["daily_price_shift_exponent"]), 6), + round(float(cfg["initial_pool_value"]), 2), + ) + + @staticmethod + def _make_method_cache_hash(key): + return repr(key) + + @staticmethod + def get_initial_pool_value(cfg): + return float(cfg["initial_pool_value"]) + + base_cfg = { + "price_ratio": 1.1, + "centeredness_margin": 0.3, + "daily_price_shift_exponent": 0.2, + "initial_pool_value": 1_000_000.0, + } + pair_spec = { + "slug": "price_ratio_vs_margin", + "x_values": [1.1, 1.2], + "y_values": [0.3, 0.4], + "x_key": "price_ratio", + "y_key": "centeredness_margin", + "fixed_key": "daily_price_shift_exponent", + } + slice_variant = { + "slug": "q2", + "label": "Q2", + "value": 0.2, + } + + heatmap_targets = { + (0, 0): (130.0, 100.0), # +30% + (0, 1): (200.0, 100.0), # +100% + (1, 0): (70.0, 100.0), # -30% + (1, 1): (160.0, 100.0), # +60% + } + cache_lookup = {} + for (y_index, x_index), (noise_geo, arb_geo) in heatmap_targets.items(): + cfg = dict(base_cfg) + cfg["price_ratio"] = pair_spec["x_values"][x_index] + cfg["centeredness_margin"] = pair_spec["y_values"][y_index] + noise_cfg, noise_method = FakeCompareModule.make_noise_variant_cfg(cfg, True), "geometric" + arb_cfg, arb_method = FakeCompareModule.make_noise_variant_cfg(cfg, False), "geometric" + + noise_key = FakeCompareModule._make_method_cache_key(noise_cfg, noise_method) + arb_key = FakeCompareModule._make_method_cache_key(arb_cfg, arb_method) + cache_lookup[FakeCompareModule._make_method_cache_hash(noise_key)] = noise_geo + cache_lookup[FakeCompareModule._make_method_cache_hash(arb_key)] = arb_geo + + slice_scan = module.build_slice_cell_grid( + compare_module=FakeCompareModule, + base_cfg=base_cfg, + pair_spec=pair_spec, + slice_variant=slice_variant, + metric_key="noise_vs_arb_geometric_improvement_pct", + cache_lookup=cache_lookup, + ) + + assert slice_scan["resolved_cell_count"] == 4 + assert slice_scan["missing_hash_count"] == 0 + assert slice_scan["records_by_coord"][(0, 0)]["heatmap_value"] == pytest.approx(30.0) + assert slice_scan["records_by_coord"][(0, 1)]["heatmap_value"] == pytest.approx(100.0) + assert slice_scan["records_by_coord"][(1, 0)]["heatmap_value"] == pytest.approx(-30.0) + assert slice_scan["records_by_coord"][(1, 1)]["heatmap_value"] == pytest.approx(60.0) + + rows = module.find_adjacent_rows_for_slice( + metric_key="noise_vs_arb_geometric_improvement_pct", + metric_unit="pct", + records_by_coord=slice_scan["records_by_coord"], + x_count=2, + y_count=2, + min_diff=30.0, + ) + + assert [row["heatmap_value_diff_abs"] for row in rows] == pytest.approx([90.0, 70.0, 60.0, 40.0]) + assert rows[0]["1_heatmap_value"] == pytest.approx(-30.0) + assert rows[0]["2_heatmap_value"] == pytest.approx(60.0) + + +def test_run_top_row_geometric_comparison_dispatches_to_compare_module(monkeypatch): + module = load_script_module() + captured = {} + + class FakeCompareModule: + @staticmethod + def run_adjacent_csv_row_comparison(csv_path, row_index=0, output_file=None): + captured["csv_path"] = csv_path + captured["row_index"] = row_index + captured["output_file"] = output_file + return "fake-output.png" + + monkeypatch.setattr( + module, + "load_geometric_compare_module", + lambda module_path=None: FakeCompareModule, + ) + + output = module.run_top_row_geometric_comparison( + Path("tmp_adjacent.csv"), + output_file="custom.png", + row_index=0, + ) + + assert output == "fake-output.png" + assert captured == { + "csv_path": Path("tmp_adjacent.csv"), + "row_index": 0, + "output_file": "custom.png", + } + + +def test_autodetect_lightweight_noise_profile_switches_to_legacy_calibrated(): + module = load_script_module() + compare_context = module._LightweightCompareContext() + base_cfg = compare_context.configs_for_tvl(compare_context.CONFIGS, 1_000_000.0)[1] + pair_spec = compare_context.get_pair_heatmap_specs(base_cfg)[0] + slice_variant = [variant for variant in pair_spec["fixed_slices"] if variant["slug"] == "q2"][0] + + sample_x_indices = sorted({0, len(pair_spec["x_values"]) // 2, len(pair_spec["x_values"]) - 1}) + sample_y_indices = sorted({0, len(pair_spec["y_values"]) // 2, len(pair_spec["y_values"]) - 1}) + + cache_lookup = {} + compare_context.set_noise_profile("legacy_calibrated") + slice_cfg = dict(base_cfg) + slice_cfg[pair_spec["fixed_key"]] = float(slice_variant["value"]) + for y_index in sample_y_indices: + for x_index in sample_x_indices: + cfg = dict(slice_cfg) + cfg[pair_spec["x_key"]] = float(pair_spec["x_values"][x_index]) + cfg[pair_spec["y_key"]] = float(pair_spec["y_values"][y_index]) + for source_name in ("noise_geometric", "arb_geometric"): + source_cfg, method = module._source_variant(compare_context, cfg, source_name) + cache_key = compare_context._make_method_cache_key(source_cfg, method) + cache_key_hash = compare_context._make_method_cache_hash(cache_key) + cache_lookup[cache_key_hash] = 1_000_000.0 + + compare_context.set_noise_profile("market_linear") + module.autodetect_lightweight_noise_profile( + compare_module=compare_context, + base_cfg=base_cfg, + pair_specs=[pair_spec], + metric_key="noise_vs_arb_geometric_improvement_pct", + slice_slug="q2", + cache_lookup=cache_lookup, + ) + + assert compare_context.noise_profile == "legacy_calibrated" + + +def test_source_variant_sets_explicit_market_and_arb_only_noise_models(): + module = load_script_module() + compare_context = module._LightweightCompareContext() + cfg = compare_context.configs_for_tvl(compare_context.CONFIGS, 1_000_000.0)[1] + + noise_cfg, noise_method = module._source_variant( + compare_context, + cfg, + "noise_geometric", + ) + arb_cfg, arb_method = module._source_variant( + compare_context, + cfg, + "arb_geometric", + ) + + assert noise_method == "geometric" + assert noise_cfg["enable_noise_model"] is True + assert noise_cfg["noise_model"] == "market_linear" + + assert arb_method == "geometric" + assert arb_cfg["enable_noise_model"] is False + assert arb_cfg["noise_model"] == "arb_only" + + resolved_arb = compare_context.resolve_reclamm_noise_settings(arb_cfg) + assert resolved_arb["noise_model"] == "arb_only" + assert resolved_arb["noise_cache_key"] == ("disabled",) + + +def test_lightweight_context_defaults_to_fixed_compare_arb_cadence(): + module = load_script_module() + compare_context = module._LightweightCompareContext() + cfg = compare_context.configs_for_tvl(compare_context.CONFIGS, 1_000_000.0)[1] + cfg["arb_frequency"] = 6 + cfg["gas_cost"] = 99.0 + cfg["protocol_fee_split"] = 0.9 + + arb_cfg = compare_context.make_noise_variant_cfg(cfg, False) + resolved_noise = compare_context.resolve_reclamm_noise_settings(cfg) + + assert arb_cfg["arb_frequency"] == compare_context.FIXED_COMPARE_ARB_FREQUENCY + assert arb_cfg["noise_model"] == "arb_only" + assert resolved_noise["arb_frequency"] == compare_context.FIXED_COMPARE_ARB_FREQUENCY + assert arb_cfg["gas_cost"] == compare_context.DEFAULT_GAS_COST + assert arb_cfg["protocol_fee_split"] == compare_context.DEFAULT_PROTOCOL_FEE_SPLIT From 00144e20dd1b87a1c56bb86d1d334180a9db8756 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Sun, 29 Mar 2026 22:55:46 +0100 Subject: [PATCH 075/115] perforance improvements --- scripts/compare_reclamm_thermostats.py | 487 ++++++++++-------- .../compare_reclamm_geometric_noise_runs.py | 23 +- .../reclamm/compare_reclamm_thermostats.py | 485 +++++++++-------- .../reclamm/find_adjacent_heatmap_pairs.py | 267 +++++----- ...st_compare_reclamm_geometric_noise_runs.py | 62 ++- .../test_compare_reclamm_thermostats.py | 372 ++++++++++++- .../test_find_adjacent_heatmap_pairs.py | 40 +- 7 files changed, 1132 insertions(+), 604 deletions(-) diff --git a/scripts/compare_reclamm_thermostats.py b/scripts/compare_reclamm_thermostats.py index 21352a4b..7fb1770c 100644 --- a/scripts/compare_reclamm_thermostats.py +++ b/scripts/compare_reclamm_thermostats.py @@ -17,8 +17,8 @@ import gc import hashlib -import math import os +from pathlib import Path import jax.numpy as jnp import numpy as np @@ -50,6 +50,15 @@ def build_inclusive_sweep(start, stop, step): return values +def _resolve_repo_root(script_path): + """Locate the repository root from either scripts/ or scripts/reclamm/.""" + script_path = Path(script_path).resolve() + for parent in script_path.parents: + if (parent / "quantammsim").exists() and (parent / "scripts").exists(): + return parent + return script_path.parents[1] + + RUN_CONSTANT_ARC_LENGTH = True INTERPOLATION_METHODS = ( ("geometric", "constant_arc_length") @@ -90,36 +99,39 @@ def build_inclusive_sweep(start, stop, step): THREE_D_VIEW_ELEVATION = 22.0 THREE_D_VIEW_AZIMUTH = 140.0 HEATMAP_FORWARD_CACHE_ENABLED = True -HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" +HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_market_linear_v2" HEATMAP_FORWARD_CACHE_ROOT = os.path.join( "results", "reclamm_heatmap_forward_cache", ) -HEATMAP_FORWARD_CACHE_FLUSH_EVERY = 32 +HEATMAP_FORWARD_CACHE_FLUSH_EVERY = 360 +REPO_ROOT = _resolve_repo_root(__file__) AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" +DEFAULT_MARKET_LINEAR_NOISE_START_DATE = "2024-06-01" +DEFAULT_MARKET_LINEAR_NOISE_END_DATE = "2026-03-01" +DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH = str( + REPO_ROOT + / "results" + / "linear_market_noise" + / "_sim_arrays" + / ( + f"{AAVE_WETH_POOL_ID}_{DEFAULT_MARKET_LINEAR_NOISE_START_DATE}_" + f"{DEFAULT_MARKET_LINEAR_NOISE_END_DATE}.npz" + ) +) DEFAULT_NOISE_MODEL = "market_linear" DEFAULT_GAS_COST = 1.0 DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 -LEGACY_NOISE_COEFFS = [ - -0.453, - 0.025, - -0.060, - 0.310, - -0.149, - 0.359, - 0.061, - 0.060, -] -LEGACY_LOG_CADENCE = 2.68 -LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) -FIXED_COMPARE_ARB_FREQUENCY = LEGACY_ARB_FREQUENCY +FIXED_COMPARE_ARB_FREQUENCY = 15 AAVE_ETH_NOISE_SETTINGS = { "enable_noise_model": True, "noise_model": DEFAULT_NOISE_MODEL, + "noise_reference_model": DEFAULT_NOISE_MODEL, "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, "noise_pool_id": AAVE_WETH_POOL_ID, + "noise_arrays_path": DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, "gas_cost": DEFAULT_GAS_COST, "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, @@ -164,7 +176,7 @@ def build_inclusive_sweep(start, stop, step): } _NOISE_SETTINGS_CACHE = {} -_WARNED_NOISE_FALLBACKS = set() +_MARKET_LINEAR_NOISE_DATA_CACHE = {} def get_initial_pool_value(cfg): @@ -251,6 +263,27 @@ def get_effective_arb_frequency(cfg, noise_cfg=None): return _normalize_arb_frequency(FIXED_COMPARE_ARB_FREQUENCY) +def _canonical_noise_reference_model(cfg): + """Resolve the only supported thermostat noise parametrisation.""" + noise_model = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + reference_model = cfg.get("noise_reference_model") + if reference_model is None: + reference_model = DEFAULT_NOISE_MODEL if noise_model == "arb_only" else noise_model + noise_model = str(noise_model) + reference_model = str(reference_model) + if noise_model not in {DEFAULT_NOISE_MODEL, "arb_only"}: + raise ValueError( + "compare_reclamm_thermostats only supports " + "'market_linear' noise and 'arb_only' baselines." + ) + if reference_model != DEFAULT_NOISE_MODEL: + raise ValueError( + "compare_reclamm_thermostats only supports the " + "'market_linear' noise parametrisation." + ) + return reference_model + + def normalize_compare_run_cfg(cfg, enable_noise_model=None): """Canonicalize the compare-run config so non-axis inputs stay fixed.""" updated = dict(cfg) @@ -277,26 +310,18 @@ def normalize_compare_run_cfg(cfg, enable_noise_model=None): ) updated["enable_noise_model"] = use_noise - requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + reference_mode = _canonical_noise_reference_model(cfg) if use_noise: - updated["noise_model"] = requested_mode - if requested_mode == "market_linear": - updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR - updated["noise_pool_id"] = AAVE_WETH_POOL_ID - else: - updated.pop("noise_artifact_dir", None) - updated.pop("noise_pool_id", None) - updated.pop("reclamm_noise_params", None) - updated.pop("noise_arrays_path", None) + updated["noise_model"] = reference_mode + updated["noise_reference_model"] = reference_mode else: - updated["noise_model"] = None - for key in ( - "reclamm_noise_params", - "noise_arrays_path", - "noise_artifact_dir", - "noise_pool_id", - ): - updated.pop(key, None) + updated["noise_model"] = "arb_only" + updated["noise_reference_model"] = reference_mode + + updated["noise_arrays_path"] = DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + updated.pop("reclamm_noise_params", None) + updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + updated["noise_pool_id"] = AAVE_WETH_POOL_ID return updated @@ -306,13 +331,6 @@ def make_noise_variant_cfg(cfg, enable_noise_model): return normalize_compare_run_cfg(cfg, enable_noise_model=enable_noise_model) -def _warn_noise_fallback(message): - """Print a one-time message when the preferred noise setup is unavailable.""" - if message not in _WARNED_NOISE_FALLBACKS: - print(message) - _WARNED_NOISE_FALLBACKS.add(message) - - def _hashable_noise_params(params): """Convert a noise-params dict into a stable cache key fragment.""" if params is None: @@ -320,38 +338,76 @@ def _hashable_noise_params(params): return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) -def _legacy_calibrated_noise_settings(reason=None, arb_frequency=None): - """Fallback calibrated noise config used when market-linear artifacts are absent.""" - if reason: - _warn_noise_fallback( - "market_linear noise unavailable for thermostat comparison; " - f"falling back to calibrated legacy coefficients ({reason})." - ) +def load_shared_market_linear_noise_data( + arrays_path=DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, +): + """Load the market_linear arrays once so compare runs can reuse them.""" + arrays_path = os.path.abspath(os.fspath(arrays_path)) + cached = _MARKET_LINEAR_NOISE_DATA_CACHE.get(arrays_path) + if cached is not None: + return cached + + if not os.path.exists(arrays_path): + raise FileNotFoundError(f"market_linear arrays file not found: {arrays_path}") + + with np.load(arrays_path) as arrays: + required_keys = {"noise_base", "noise_tvl_coeff", "tvl_mean", "tvl_std"} + missing_keys = sorted(required_keys.difference(arrays.files)) + if missing_keys: + raise KeyError( + f"market_linear arrays file {arrays_path} is missing keys: {missing_keys}" + ) + shared = { + "arrays_path": arrays_path, + "noise_base_array": np.asarray(arrays["noise_base"]), + "noise_tvl_coeff_array": np.asarray(arrays["noise_tvl_coeff"]), + "tvl_mean": float(arrays["tvl_mean"]), + "tvl_std": float(arrays["tvl_std"]), + } + _MARKET_LINEAR_NOISE_DATA_CACHE[arrays_path] = shared + return shared + + +def _load_market_linear_noise_stats(arrays_path=DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH): + """Load the exact arrays file used by the market_linear run fingerprint. + + The simulator consumes ``noise_base`` and ``noise_tvl_coeff`` from + ``run_fingerprint["noise_arrays_path"]`` and uses ``tvl_mean``/``tvl_std`` + from the same file for TVL standardization. + """ + shared = load_shared_market_linear_noise_data(arrays_path=arrays_path) + return shared["arrays_path"], shared["tvl_mean"], shared["tvl_std"] + + +def _market_linear_noise_settings(noise_model="market_linear", arb_frequency=None): + """Build the tuned market_linear fingerprint block from the fixed arrays file.""" + arrays_path, tvl_mean, tvl_std = _load_market_linear_noise_stats() arb_frequency = _normalize_arb_frequency(arb_frequency) return { - "noise_model": "calibrated", + "noise_model": noise_model, "noise_trader_ratio": 0.0, "reclamm_noise_params": { - f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, }, + "noise_arrays_path": arrays_path, "arb_frequency": arb_frequency, - "noise_summary": ( - "calibrated legacy 8-covariate " - f"(arb_frequency={arb_frequency})" - ), + "noise_summary": f"{noise_model} (arb_frequency={arb_frequency})", "noise_cache_key": ( - "calibrated", - tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), + noise_model, + arrays_path, arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), ), } - def resolve_reclamm_noise_settings(cfg): """Resolve the active reCLAMM noise-model fingerprint block for a config.""" cfg = normalize_compare_run_cfg(cfg) enable_noise_model = cfg.get("enable_noise_model", False) requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + reference_mode = cfg.get("noise_reference_model", DEFAULT_NOISE_MODEL) requested_arb_frequency = get_effective_arb_frequency(cfg) cache_key = ( tuple(cfg.get("tokens", [])), @@ -359,6 +415,7 @@ def resolve_reclamm_noise_settings(cfg): cfg.get("end"), enable_noise_model, requested_mode, + reference_mode, cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), requested_arb_frequency, @@ -369,103 +426,21 @@ def resolve_reclamm_noise_settings(cfg): if cache_key in _NOISE_SETTINGS_CACHE: return _NOISE_SETTINGS_CACHE[cache_key] - if not enable_noise_model: - result = { - "noise_model": None, - "noise_trader_ratio": 0.0, - "reclamm_noise_params": None, - "noise_arrays_path": None, - "arb_frequency": requested_arb_frequency, - "noise_summary": "arb-only (noise disabled)", - "noise_cache_key": ("disabled",), - } - elif requested_mode == "market_linear": - artifact_dir = cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR) - pool_id = cfg.get("noise_pool_id", AAVE_WETH_POOL_ID) - start_date = str(cfg["start"]).split(" ")[0] - end_date = str(cfg["end"]).split(" ")[0] - try: - from quantammsim.calibration.noise_model_arrays import build_simulator_arrays - - model_path = os.path.join(artifact_dir, "model.npz") - meta_path = os.path.join(artifact_dir, "meta.json") - if not (os.path.exists(model_path) and os.path.exists(meta_path)): - raise FileNotFoundError( - f"expected {model_path} and {meta_path}" - ) - - cache_dir = os.path.join(artifact_dir, "_sim_arrays") - os.makedirs(cache_dir, exist_ok=True) - arrays_path = os.path.join( - cache_dir, - f"{pool_id}_{start_date}_{end_date}.npz", - ) - if not os.path.exists(arrays_path): - arrays = build_simulator_arrays( - token_a=cfg["tokens"][0], - token_b=cfg["tokens"][1], - pool_id=pool_id, - start_date=start_date, - end_date=end_date, - artifact_dir=artifact_dir, - ) - np.savez( - arrays_path, - noise_base=arrays["noise_base"], - noise_tvl_coeff=arrays["noise_tvl_coeff"], - tvl_mean=arrays["tvl_mean"], - tvl_std=arrays["tvl_std"], - ) - - with np.load(arrays_path) as arrays: - tvl_mean = float(arrays["tvl_mean"]) - tvl_std = float(arrays["tvl_std"]) - - arb_frequency = requested_arb_frequency - result = { - "noise_model": "market_linear", - "noise_trader_ratio": 0.0, - "reclamm_noise_params": { - "tvl_mean": tvl_mean, - "tvl_std": tvl_std, - }, - "noise_arrays_path": arrays_path, - "arb_frequency": arb_frequency, - "noise_summary": f"market_linear (arb_frequency={arb_frequency})", - "noise_cache_key": ( - "market_linear", - arrays_path, - arb_frequency, - round(tvl_mean, 12), - round(tvl_std, 12), - ), - } - except Exception as exc: # pragma: no cover - fallback path depends on local artifacts - result = _legacy_calibrated_noise_settings( - str(exc), - arb_frequency=requested_arb_frequency, - ) - elif requested_mode == "calibrated": - result = _legacy_calibrated_noise_settings( - arb_frequency=requested_arb_frequency + if requested_mode == "arb_only": + result = _market_linear_noise_settings( + noise_model="arb_only", + arb_frequency=requested_arb_frequency, + ) + elif requested_mode == DEFAULT_NOISE_MODEL: + result = _market_linear_noise_settings( + noise_model=DEFAULT_NOISE_MODEL, + arb_frequency=requested_arb_frequency, ) else: - arb_frequency = requested_arb_frequency - result = { - "noise_model": requested_mode, - "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), - "reclamm_noise_params": cfg.get("reclamm_noise_params"), - "noise_arrays_path": cfg.get("noise_arrays_path"), - "arb_frequency": arb_frequency, - "noise_summary": f"{requested_mode} (arb_frequency={arb_frequency})", - "noise_cache_key": ( - requested_mode, - round(float(cfg.get("noise_trader_ratio", 0.0)), 12), - _hashable_noise_params(cfg.get("reclamm_noise_params")), - cfg.get("noise_arrays_path"), - arb_frequency, - ), - } + raise ValueError( + "compare_reclamm_thermostats only supports " + "'market_linear' noise and 'arb_only' baselines." + ) _NOISE_SETTINGS_CACHE[cache_key] = result return result @@ -504,7 +479,29 @@ def resolve_reclamm_noise_settings(cfg): ] -def make_fingerprint(cfg, interpolation_method): +def _attach_market_linear_noise_arrays( + fingerprint, + noise_cfg, + market_linear_noise_data, +): + """Attach preloaded market_linear arrays when the compare flow has them.""" + if market_linear_noise_data is None: + return + expected_path = noise_cfg.get("noise_arrays_path") + if expected_path is None: + return + shared_path = os.path.abspath(os.fspath(market_linear_noise_data["arrays_path"])) + expected_path = os.path.abspath(os.fspath(expected_path)) + if shared_path != expected_path: + raise ValueError( + "Shared market_linear noise arrays path does not match " + f"the resolved compare-run noise path: {shared_path} != {expected_path}" + ) + fingerprint["noise_base_array"] = market_linear_noise_data["noise_base_array"] + fingerprint["noise_tvl_coeff_array"] = market_linear_noise_data["noise_tvl_coeff_array"] + + +def make_fingerprint(cfg, interpolation_method, market_linear_noise_data=None): """Build run fingerprint for a given config and interpolation method.""" cfg = normalize_compare_run_cfg(cfg) speed_override = ( @@ -541,6 +538,11 @@ def make_fingerprint(cfg, interpolation_method): fingerprint["reclamm_noise_params"] = noise_cfg["reclamm_noise_params"] if noise_cfg.get("noise_arrays_path") is not None: fingerprint["noise_arrays_path"] = noise_cfg["noise_arrays_path"] + _attach_market_linear_noise_arrays( + fingerprint, + noise_cfg, + market_linear_noise_data, + ) if arb_frequency is not None: fingerprint["arb_frequency"] = arb_frequency return fingerprint @@ -564,13 +566,22 @@ def load_shared_price_data(configs, root=None): return get_historic_parquet_data(tokens, cols=["close"], root=root) -def run_comparison(cfg, price_data=None, low_data_mode=False): +def run_comparison( + cfg, + price_data=None, + low_data_mode=False, + market_linear_noise_data=None, +): """Run both interpolation variants, return results dict.""" params = make_params(cfg) results = {} for method in INTERPOLATION_METHODS: - fp = make_fingerprint(cfg, method) + fp = make_fingerprint( + cfg, + method, + market_linear_noise_data=market_linear_noise_data, + ) results[method] = do_run_on_historic_data( run_fingerprint=fp, params=params, @@ -662,30 +673,44 @@ def _load_persistent_final_value_cache(cache): return disk_cache = {} - disk_records = {} + next_batch_id = 0 cache_path = cache.get("_persistent_final_value_cache_path") if cache_path and os.path.exists(cache_path): - frame = pd.read_parquet(cache_path) - if not frame.empty: + parquet_files = [] + if os.path.isdir(cache_path): + parquet_files = [ + os.path.join(cache_path, filename) + for filename in sorted(os.listdir(cache_path)) + if filename.endswith(".parquet") + ] + batch_ids = [] + for filename in os.listdir(cache_path): + if not (filename.startswith("batch_") and filename.endswith(".parquet")): + continue + token = filename[len("batch_") : -len(".parquet")] + if token.isdigit(): + batch_ids.append(int(token)) + next_batch_id = (max(batch_ids) + 1) if batch_ids else 0 + else: + parquet_files = [cache_path] + + for parquet_file in parquet_files: + frame = pd.read_parquet( + parquet_file, + columns=["cache_key_hash", "final_value"], + ) + if frame.empty: + continue for row in frame.itertuples(index=False): cache_key_hash = str(row.cache_key_hash) final_value = float(row.final_value) disk_cache[cache_key_hash] = final_value - record = { - "cache_key_hash": cache_key_hash, - "final_value": final_value, - } - for column in PERSISTED_FORWARD_VALUE_COLUMNS: - if column in {"cache_key_hash", "final_value"}: - continue - record[column] = getattr(row, column, None) - disk_records[cache_key_hash] = record print( f"Loaded {len(disk_cache)} persisted heatmap forward values from {cache_path}" ) cache["_persistent_final_value_cache"] = disk_cache - cache["_persistent_final_value_records"] = disk_records + cache["_persistent_final_value_next_batch_id"] = next_batch_id cache["_persistent_final_value_cache_loaded"] = True @@ -702,52 +727,63 @@ def flush_sweep_cache(cache, force=False): _load_persistent_final_value_cache(cache) disk_cache = cache.setdefault("_persistent_final_value_cache", {}) - disk_records = cache.setdefault("_persistent_final_value_records", {}) + batch_records = [] for cache_key_hash, record in pending.items(): - merged = dict(disk_records.get(cache_key_hash, {})) - merged.update(record) - merged["cache_key_hash"] = str(cache_key_hash) - merged["final_value"] = float(merged["final_value"]) - disk_records[cache_key_hash] = merged - disk_cache[cache_key_hash] = merged["final_value"] + normalized = dict(record) + normalized["cache_key_hash"] = str(cache_key_hash) + normalized["final_value"] = float(normalized["final_value"]) + disk_cache[cache_key_hash] = normalized["final_value"] + batch_records.append(normalized) cache_path = cache.get("_persistent_final_value_cache_path") if cache_path is None: pending.clear() return - os.makedirs(os.path.dirname(cache_path), exist_ok=True) - sorted_records = [disk_records[key] for key in sorted(disk_records)] + if os.path.exists(cache_path) and not os.path.isdir(cache_path): + raise RuntimeError( + f"Persistent cache path {cache_path} already exists as a file. " + "Use a fresh cache namespace for append-only parquet shards." + ) + + os.makedirs(cache_path, exist_ok=True) + batch_records.sort(key=lambda record: record["cache_key_hash"]) payload = { - column: [record.get(column) for record in sorted_records] + column: [record.get(column) for record in batch_records] for column in PERSISTED_FORWARD_VALUE_COLUMNS } payload["final_value"] = np.asarray(payload["final_value"], dtype=np.float64) frame = pd.DataFrame(payload) - frame.sort_values("cache_key_hash", inplace=True, ignore_index=True) - frame.to_parquet(cache_path, index=False, compression="zstd") + batch_id = int(cache.setdefault("_persistent_final_value_next_batch_id", 0)) + batch_path = os.path.join(cache_path, f"batch_{batch_id:08d}.parquet") + cache["_persistent_final_value_next_batch_id"] = batch_id + 1 + frame.to_parquet(batch_path, index=False, compression="zstd") print( - f"Persisted {len(pending)} new heatmap forward values to {cache_path} " + f"Persisted {len(pending)} new heatmap forward values to {batch_path} " f"({len(disk_cache)} total cached values)." ) pending.clear() -def make_sweep_cache(price_data, cache_scope_cfg=None): +def make_sweep_cache( + price_data, + cache_scope_cfg=None, + market_linear_noise_data=None, +): """Create a shared cache for heatmap and line sweeps.""" cache = { "_shared_price_data": price_data, + "_shared_market_linear_noise_data": market_linear_noise_data, "_final_value_cache": {}, "_comparison_cache": {}, "_pending_persistent_final_values": {}, "_persistent_final_value_cache": {}, - "_persistent_final_value_records": {}, + "_persistent_final_value_next_batch_id": 0, "_persistent_final_value_cache_loaded": False, "_persistent_final_value_cache_path": _heatmap_forward_cache_path( cache_scope_cfg ), } - _load_persistent_final_value_cache(cache) return cache @@ -781,6 +817,10 @@ def _make_method_cache_key(cfg, method): arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) key = ( method, + tuple(str(token) for token in cfg["tokens"]), + str(cfg["start"]), + str(cfg["end"]), + round(float(cfg["fees"]), 12), bool(cfg.get("enable_noise_model", False)), round(float(cfg["price_ratio"]), 6), round(float(cfg["centeredness_margin"]), 6), @@ -812,6 +852,40 @@ def _make_method_cache_key(cfg, method): return key +def _nearest_price_row(price_data, start_ts): + """Select the closest available price row to the requested start timestamp.""" + if len(price_data.index) == 0: + raise ValueError("price_data is empty") + + if isinstance(price_data.index, pd.DatetimeIndex): + target_ts = start_ts + index_tz = getattr(price_data.index, "tz", None) + if index_tz is not None and target_ts.tzinfo is None: + target_ts = target_ts.tz_localize(index_tz) + elif index_tz is None and target_ts.tzinfo is not None: + target_ts = target_ts.tz_convert(None) + target_value = int(target_ts.value) + index_values = price_data.index.asi8 + else: + target_value = int(start_ts.timestamp() * 1000.0) + index_values = price_data.index.to_numpy(dtype=np.int64) + + row_idx = int(np.searchsorted(index_values, target_value, side="left")) + if row_idx >= len(index_values): + row_idx = len(index_values) - 1 + elif row_idx > 0 and index_values[row_idx] != target_value: + prev_idx = row_idx - 1 + if abs(int(index_values[prev_idx]) - target_value) <= abs( + int(index_values[row_idx]) - target_value + ): + row_idx = prev_idx + + row = price_data.iloc[row_idx] + if isinstance(row, pd.DataFrame): + row = row.iloc[0] + return row + + def _make_comparison_cache_key(cfg, launch_final_values): """Cache key for scalar heatmap metrics at a single parameter point.""" noise_cfg = make_noise_variant_cfg(cfg, True) @@ -847,7 +921,11 @@ def _run_method_final_value_cached(cfg, method, cache): return final_value_cache[key] result = do_run_on_historic_data( - run_fingerprint=make_fingerprint(cfg, method), + run_fingerprint=make_fingerprint( + cfg, + method, + market_linear_noise_data=cache.get("_shared_market_linear_noise_data"), + ), params=make_params(cfg), price_data=cache["_shared_price_data"], low_data_mode=True, @@ -1842,25 +1920,7 @@ def build_pair_slice_data(pair, slice_variant, metric_keys): def compute_auto_calibrated_arc_length_speed(cfg, price_data): """Compute the launch/reference auto-calibrated speed for a config.""" start_ts = pd.Timestamp(cfg["start"]) - - if isinstance(price_data.index, pd.DatetimeIndex): - row = price_data.loc[start_ts] - else: - start_unix_ms = int(start_ts.timestamp() * 1000.0) - index_values = price_data.index.to_numpy(dtype=np.int64) - row_idx = int(np.searchsorted(index_values, start_unix_ms, side="left")) - if row_idx >= len(index_values): - row_idx = len(index_values) - 1 - if row_idx > 0 and index_values[row_idx] != start_unix_ms: - prev_idx = row_idx - 1 - if abs(index_values[prev_idx] - start_unix_ms) <= abs( - index_values[row_idx] - start_unix_ms - ): - row_idx = prev_idx - row = price_data.iloc[row_idx] - - if isinstance(row, pd.DataFrame): - row = row.iloc[0] + row = _nearest_price_row(price_data, start_ts) if isinstance(price_data.columns, pd.MultiIndex): initial_price_values = [ @@ -2103,7 +2163,12 @@ def generate_arc_speed_efficiency_artifacts( print("Released arc-speed sweep cache.") -def get_launch_final_values(all_results, launch_cfg, price_data): +def get_launch_final_values( + all_results, + launch_cfg, + price_data, + market_linear_noise_data=None, +): """Reuse launch-style runs when available; otherwise run them once.""" for cfg, results in all_results: if cfg["name"] == launch_cfg["name"]: @@ -2121,6 +2186,7 @@ def get_launch_final_values(all_results, launch_cfg, price_data): launch_cfg, price_data=price_data, low_data_mode=True, + market_linear_noise_data=market_linear_noise_data, ) launch_final_values = { "geometric": float(launch_results["geometric"]["final_value"]), @@ -2388,6 +2454,7 @@ def plot_comparison(cfg, results, fig_idx): if __name__ == "__main__": shared_price_data = load_shared_price_data(CONFIGS) + shared_market_linear_noise_data = load_shared_market_linear_noise_data() for initial_pool_value in TVL_SWEEP_VALUES: tvl_configs = configs_for_tvl(CONFIGS, initial_pool_value) @@ -2398,7 +2465,11 @@ def plot_comparison(cfg, results, fig_idx): for i, cfg in enumerate(tvl_configs): print(f"\n>>> Running {cfg['name']} at TVL {tvl_label}...") try: - results = run_comparison(cfg, price_data=shared_price_data) + results = run_comparison( + cfg, + price_data=shared_price_data, + market_linear_noise_data=shared_market_linear_noise_data, + ) print_comparison(cfg, results) plot_comparison(cfg, results, i) all_results.append((cfg, results)) @@ -2503,10 +2574,12 @@ def plot_comparison(cfg, results, fig_idx): all_results, launch_cfg=tvl_configs[0], price_data=shared_price_data, + market_linear_noise_data=shared_market_linear_noise_data, ) shared_sweep_cache = make_sweep_cache( shared_price_data, cache_scope_cfg=tvl_configs[1], + market_linear_noise_data=shared_market_linear_noise_data, ) print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") diff --git a/scripts/reclamm/compare_reclamm_geometric_noise_runs.py b/scripts/reclamm/compare_reclamm_geometric_noise_runs.py index 3e385dd0..240482b5 100644 --- a/scripts/reclamm/compare_reclamm_geometric_noise_runs.py +++ b/scripts/reclamm/compare_reclamm_geometric_noise_runs.py @@ -30,7 +30,7 @@ DEFAULT_SOURCE_HEATMAP_DESCRIPTION = ( - "legacy unsuffixed price_ratio_vs_margin heatmap with shift_exp fixed at 0.10" + "market_linear price_ratio_vs_margin heatmap with shift_exp fixed at 0.10" ) DEFAULT_RUN_SPECS = [ { @@ -42,7 +42,7 @@ "color": "C0", "reason": ( "Geometric noise-model run taken from the positive cell in the " - "legacy price_ratio-vs-margin heatmap at price_ratio=1.31 and the " + "market_linear price_ratio-vs-margin heatmap at price_ratio=1.31 and the " "lower adjacent centeredness row." ), }, @@ -55,7 +55,7 @@ "color": "C1", "reason": ( "Geometric noise-model run taken from the negative cell directly " - "above the green cell in the legacy price_ratio-vs-margin heatmap " + "above the green cell in the market_linear price_ratio-vs-margin heatmap " "at price_ratio=1.31." ), }, @@ -205,18 +205,23 @@ def build_run_config(spec, base_config): } ) source_noise_profile = spec.get("source_noise_profile") - if source_noise_profile == "legacy_calibrated": - cfg["noise_model"] = "calibrated" - cfg.pop("reclamm_noise_params", None) - cfg.pop("noise_arrays_path", None) - elif source_noise_profile == "market_linear": + if source_noise_profile == "market_linear": cfg["noise_model"] = "market_linear" + elif source_noise_profile not in (None, "", "unknown"): + raise ValueError( + "compare_reclamm_geometric_noise_runs.py only supports the current " + f"market_linear source profile, got {source_noise_profile!r}. " + "Regenerate the adjacent-pairs CSV with the current heatmap cache." + ) return cfg def build_run_variants(spec, base_config, thermostat_compare): """Build matched noise-model and arb-only variants for one highlighted cell.""" - noise_cfg = build_run_config(spec, base_config=base_config) + noise_cfg = thermostat_compare.make_noise_variant_cfg( + build_run_config(spec, base_config=base_config), + True, + ) noise_cfg["variant_key"] = "noise" noise_cfg["variant_label"] = "noise-model" diff --git a/scripts/reclamm/compare_reclamm_thermostats.py b/scripts/reclamm/compare_reclamm_thermostats.py index 21352a4b..f705c669 100644 --- a/scripts/reclamm/compare_reclamm_thermostats.py +++ b/scripts/reclamm/compare_reclamm_thermostats.py @@ -17,8 +17,8 @@ import gc import hashlib -import math import os +from pathlib import Path import jax.numpy as jnp import numpy as np @@ -50,6 +50,15 @@ def build_inclusive_sweep(start, stop, step): return values +def _resolve_repo_root(script_path): + """Locate the repository root from either scripts/ or scripts/reclamm/.""" + script_path = Path(script_path).resolve() + for parent in script_path.parents: + if (parent / "quantammsim").exists() and (parent / "scripts").exists(): + return parent + return script_path.parents[1] + + RUN_CONSTANT_ARC_LENGTH = True INTERPOLATION_METHODS = ( ("geometric", "constant_arc_length") @@ -90,36 +99,39 @@ def build_inclusive_sweep(start, stop, step): THREE_D_VIEW_ELEVATION = 22.0 THREE_D_VIEW_AZIMUTH = 140.0 HEATMAP_FORWARD_CACHE_ENABLED = True -HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" +HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_market_linear_v2" HEATMAP_FORWARD_CACHE_ROOT = os.path.join( "results", "reclamm_heatmap_forward_cache", ) HEATMAP_FORWARD_CACHE_FLUSH_EVERY = 32 +REPO_ROOT = _resolve_repo_root(__file__) AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" +DEFAULT_MARKET_LINEAR_NOISE_START_DATE = "2024-06-01" +DEFAULT_MARKET_LINEAR_NOISE_END_DATE = "2026-03-01" +DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH = str( + REPO_ROOT + / "results" + / "linear_market_noise" + / "_sim_arrays" + / ( + f"{AAVE_WETH_POOL_ID}_{DEFAULT_MARKET_LINEAR_NOISE_START_DATE}_" + f"{DEFAULT_MARKET_LINEAR_NOISE_END_DATE}.npz" + ) +) DEFAULT_NOISE_MODEL = "market_linear" DEFAULT_GAS_COST = 1.0 DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 -LEGACY_NOISE_COEFFS = [ - -0.453, - 0.025, - -0.060, - 0.310, - -0.149, - 0.359, - 0.061, - 0.060, -] -LEGACY_LOG_CADENCE = 2.68 -LEGACY_ARB_FREQUENCY = max(1, round(math.exp(LEGACY_LOG_CADENCE))) -FIXED_COMPARE_ARB_FREQUENCY = LEGACY_ARB_FREQUENCY +FIXED_COMPARE_ARB_FREQUENCY = 15 AAVE_ETH_NOISE_SETTINGS = { "enable_noise_model": True, "noise_model": DEFAULT_NOISE_MODEL, + "noise_reference_model": DEFAULT_NOISE_MODEL, "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, "noise_pool_id": AAVE_WETH_POOL_ID, + "noise_arrays_path": DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, "gas_cost": DEFAULT_GAS_COST, "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, @@ -164,7 +176,7 @@ def build_inclusive_sweep(start, stop, step): } _NOISE_SETTINGS_CACHE = {} -_WARNED_NOISE_FALLBACKS = set() +_MARKET_LINEAR_NOISE_DATA_CACHE = {} def get_initial_pool_value(cfg): @@ -251,6 +263,27 @@ def get_effective_arb_frequency(cfg, noise_cfg=None): return _normalize_arb_frequency(FIXED_COMPARE_ARB_FREQUENCY) +def _canonical_noise_reference_model(cfg): + """Resolve the only supported thermostat noise parametrisation.""" + noise_model = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + reference_model = cfg.get("noise_reference_model") + if reference_model is None: + reference_model = DEFAULT_NOISE_MODEL if noise_model == "arb_only" else noise_model + noise_model = str(noise_model) + reference_model = str(reference_model) + if noise_model not in {DEFAULT_NOISE_MODEL, "arb_only"}: + raise ValueError( + "compare_reclamm_thermostats only supports " + "'market_linear' noise and 'arb_only' baselines." + ) + if reference_model != DEFAULT_NOISE_MODEL: + raise ValueError( + "compare_reclamm_thermostats only supports the " + "'market_linear' noise parametrisation." + ) + return reference_model + + def normalize_compare_run_cfg(cfg, enable_noise_model=None): """Canonicalize the compare-run config so non-axis inputs stay fixed.""" updated = dict(cfg) @@ -277,26 +310,18 @@ def normalize_compare_run_cfg(cfg, enable_noise_model=None): ) updated["enable_noise_model"] = use_noise - requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) or DEFAULT_NOISE_MODEL + reference_mode = _canonical_noise_reference_model(cfg) if use_noise: - updated["noise_model"] = requested_mode - if requested_mode == "market_linear": - updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR - updated["noise_pool_id"] = AAVE_WETH_POOL_ID - else: - updated.pop("noise_artifact_dir", None) - updated.pop("noise_pool_id", None) - updated.pop("reclamm_noise_params", None) - updated.pop("noise_arrays_path", None) + updated["noise_model"] = reference_mode + updated["noise_reference_model"] = reference_mode else: - updated["noise_model"] = None - for key in ( - "reclamm_noise_params", - "noise_arrays_path", - "noise_artifact_dir", - "noise_pool_id", - ): - updated.pop(key, None) + updated["noise_model"] = "arb_only" + updated["noise_reference_model"] = reference_mode + + updated["noise_arrays_path"] = DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + updated.pop("reclamm_noise_params", None) + updated["noise_artifact_dir"] = DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + updated["noise_pool_id"] = AAVE_WETH_POOL_ID return updated @@ -306,13 +331,6 @@ def make_noise_variant_cfg(cfg, enable_noise_model): return normalize_compare_run_cfg(cfg, enable_noise_model=enable_noise_model) -def _warn_noise_fallback(message): - """Print a one-time message when the preferred noise setup is unavailable.""" - if message not in _WARNED_NOISE_FALLBACKS: - print(message) - _WARNED_NOISE_FALLBACKS.add(message) - - def _hashable_noise_params(params): """Convert a noise-params dict into a stable cache key fragment.""" if params is None: @@ -320,38 +338,76 @@ def _hashable_noise_params(params): return tuple(sorted((str(k), round(float(v), 12)) for k, v in params.items())) -def _legacy_calibrated_noise_settings(reason=None, arb_frequency=None): - """Fallback calibrated noise config used when market-linear artifacts are absent.""" - if reason: - _warn_noise_fallback( - "market_linear noise unavailable for thermostat comparison; " - f"falling back to calibrated legacy coefficients ({reason})." - ) +def load_shared_market_linear_noise_data( + arrays_path=DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, +): + """Load the market_linear arrays once so compare runs can reuse them.""" + arrays_path = os.path.abspath(os.fspath(arrays_path)) + cached = _MARKET_LINEAR_NOISE_DATA_CACHE.get(arrays_path) + if cached is not None: + return cached + + if not os.path.exists(arrays_path): + raise FileNotFoundError(f"market_linear arrays file not found: {arrays_path}") + + with np.load(arrays_path) as arrays: + required_keys = {"noise_base", "noise_tvl_coeff", "tvl_mean", "tvl_std"} + missing_keys = sorted(required_keys.difference(arrays.files)) + if missing_keys: + raise KeyError( + f"market_linear arrays file {arrays_path} is missing keys: {missing_keys}" + ) + shared = { + "arrays_path": arrays_path, + "noise_base_array": np.asarray(arrays["noise_base"]), + "noise_tvl_coeff_array": np.asarray(arrays["noise_tvl_coeff"]), + "tvl_mean": float(arrays["tvl_mean"]), + "tvl_std": float(arrays["tvl_std"]), + } + _MARKET_LINEAR_NOISE_DATA_CACHE[arrays_path] = shared + return shared + + +def _load_market_linear_noise_stats(arrays_path=DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH): + """Load the exact arrays file used by the market_linear run fingerprint. + + The simulator consumes ``noise_base`` and ``noise_tvl_coeff`` from + ``run_fingerprint["noise_arrays_path"]`` and uses ``tvl_mean``/``tvl_std`` + from the same file for TVL standardization. + """ + shared = load_shared_market_linear_noise_data(arrays_path=arrays_path) + return shared["arrays_path"], shared["tvl_mean"], shared["tvl_std"] + + +def _market_linear_noise_settings(noise_model="market_linear", arb_frequency=None): + """Build the tuned market_linear fingerprint block from the fixed arrays file.""" + arrays_path, tvl_mean, tvl_std = _load_market_linear_noise_stats() arb_frequency = _normalize_arb_frequency(arb_frequency) return { - "noise_model": "calibrated", + "noise_model": noise_model, "noise_trader_ratio": 0.0, "reclamm_noise_params": { - f"c_{i}": LEGACY_NOISE_COEFFS[i] for i in range(len(LEGACY_NOISE_COEFFS)) + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, }, + "noise_arrays_path": arrays_path, "arb_frequency": arb_frequency, - "noise_summary": ( - "calibrated legacy 8-covariate " - f"(arb_frequency={arb_frequency})" - ), + "noise_summary": f"{noise_model} (arb_frequency={arb_frequency})", "noise_cache_key": ( - "calibrated", - tuple(round(float(c), 12) for c in LEGACY_NOISE_COEFFS), + noise_model, + arrays_path, arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), ), } - def resolve_reclamm_noise_settings(cfg): """Resolve the active reCLAMM noise-model fingerprint block for a config.""" cfg = normalize_compare_run_cfg(cfg) enable_noise_model = cfg.get("enable_noise_model", False) requested_mode = cfg.get("noise_model", DEFAULT_NOISE_MODEL) + reference_mode = cfg.get("noise_reference_model", DEFAULT_NOISE_MODEL) requested_arb_frequency = get_effective_arb_frequency(cfg) cache_key = ( tuple(cfg.get("tokens", [])), @@ -359,6 +415,7 @@ def resolve_reclamm_noise_settings(cfg): cfg.get("end"), enable_noise_model, requested_mode, + reference_mode, cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), cfg.get("noise_pool_id", AAVE_WETH_POOL_ID), requested_arb_frequency, @@ -369,103 +426,21 @@ def resolve_reclamm_noise_settings(cfg): if cache_key in _NOISE_SETTINGS_CACHE: return _NOISE_SETTINGS_CACHE[cache_key] - if not enable_noise_model: - result = { - "noise_model": None, - "noise_trader_ratio": 0.0, - "reclamm_noise_params": None, - "noise_arrays_path": None, - "arb_frequency": requested_arb_frequency, - "noise_summary": "arb-only (noise disabled)", - "noise_cache_key": ("disabled",), - } - elif requested_mode == "market_linear": - artifact_dir = cfg.get("noise_artifact_dir", DEFAULT_MARKET_LINEAR_ARTIFACT_DIR) - pool_id = cfg.get("noise_pool_id", AAVE_WETH_POOL_ID) - start_date = str(cfg["start"]).split(" ")[0] - end_date = str(cfg["end"]).split(" ")[0] - try: - from quantammsim.calibration.noise_model_arrays import build_simulator_arrays - - model_path = os.path.join(artifact_dir, "model.npz") - meta_path = os.path.join(artifact_dir, "meta.json") - if not (os.path.exists(model_path) and os.path.exists(meta_path)): - raise FileNotFoundError( - f"expected {model_path} and {meta_path}" - ) - - cache_dir = os.path.join(artifact_dir, "_sim_arrays") - os.makedirs(cache_dir, exist_ok=True) - arrays_path = os.path.join( - cache_dir, - f"{pool_id}_{start_date}_{end_date}.npz", - ) - if not os.path.exists(arrays_path): - arrays = build_simulator_arrays( - token_a=cfg["tokens"][0], - token_b=cfg["tokens"][1], - pool_id=pool_id, - start_date=start_date, - end_date=end_date, - artifact_dir=artifact_dir, - ) - np.savez( - arrays_path, - noise_base=arrays["noise_base"], - noise_tvl_coeff=arrays["noise_tvl_coeff"], - tvl_mean=arrays["tvl_mean"], - tvl_std=arrays["tvl_std"], - ) - - with np.load(arrays_path) as arrays: - tvl_mean = float(arrays["tvl_mean"]) - tvl_std = float(arrays["tvl_std"]) - - arb_frequency = requested_arb_frequency - result = { - "noise_model": "market_linear", - "noise_trader_ratio": 0.0, - "reclamm_noise_params": { - "tvl_mean": tvl_mean, - "tvl_std": tvl_std, - }, - "noise_arrays_path": arrays_path, - "arb_frequency": arb_frequency, - "noise_summary": f"market_linear (arb_frequency={arb_frequency})", - "noise_cache_key": ( - "market_linear", - arrays_path, - arb_frequency, - round(tvl_mean, 12), - round(tvl_std, 12), - ), - } - except Exception as exc: # pragma: no cover - fallback path depends on local artifacts - result = _legacy_calibrated_noise_settings( - str(exc), - arb_frequency=requested_arb_frequency, - ) - elif requested_mode == "calibrated": - result = _legacy_calibrated_noise_settings( - arb_frequency=requested_arb_frequency + if requested_mode == "arb_only": + result = _market_linear_noise_settings( + noise_model="arb_only", + arb_frequency=requested_arb_frequency, + ) + elif requested_mode == DEFAULT_NOISE_MODEL: + result = _market_linear_noise_settings( + noise_model=DEFAULT_NOISE_MODEL, + arb_frequency=requested_arb_frequency, ) else: - arb_frequency = requested_arb_frequency - result = { - "noise_model": requested_mode, - "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), - "reclamm_noise_params": cfg.get("reclamm_noise_params"), - "noise_arrays_path": cfg.get("noise_arrays_path"), - "arb_frequency": arb_frequency, - "noise_summary": f"{requested_mode} (arb_frequency={arb_frequency})", - "noise_cache_key": ( - requested_mode, - round(float(cfg.get("noise_trader_ratio", 0.0)), 12), - _hashable_noise_params(cfg.get("reclamm_noise_params")), - cfg.get("noise_arrays_path"), - arb_frequency, - ), - } + raise ValueError( + "compare_reclamm_thermostats only supports " + "'market_linear' noise and 'arb_only' baselines." + ) _NOISE_SETTINGS_CACHE[cache_key] = result return result @@ -504,7 +479,29 @@ def resolve_reclamm_noise_settings(cfg): ] -def make_fingerprint(cfg, interpolation_method): +def _attach_market_linear_noise_arrays( + fingerprint, + noise_cfg, + market_linear_noise_data, +): + """Attach preloaded market_linear arrays when the compare flow has them.""" + if market_linear_noise_data is None: + return + expected_path = noise_cfg.get("noise_arrays_path") + if expected_path is None: + return + shared_path = os.path.abspath(os.fspath(market_linear_noise_data["arrays_path"])) + expected_path = os.path.abspath(os.fspath(expected_path)) + if shared_path != expected_path: + raise ValueError( + "Shared market_linear noise arrays path does not match " + f"the resolved compare-run noise path: {shared_path} != {expected_path}" + ) + fingerprint["noise_base_array"] = market_linear_noise_data["noise_base_array"] + fingerprint["noise_tvl_coeff_array"] = market_linear_noise_data["noise_tvl_coeff_array"] + + +def make_fingerprint(cfg, interpolation_method, market_linear_noise_data=None): """Build run fingerprint for a given config and interpolation method.""" cfg = normalize_compare_run_cfg(cfg) speed_override = ( @@ -541,6 +538,11 @@ def make_fingerprint(cfg, interpolation_method): fingerprint["reclamm_noise_params"] = noise_cfg["reclamm_noise_params"] if noise_cfg.get("noise_arrays_path") is not None: fingerprint["noise_arrays_path"] = noise_cfg["noise_arrays_path"] + _attach_market_linear_noise_arrays( + fingerprint, + noise_cfg, + market_linear_noise_data, + ) if arb_frequency is not None: fingerprint["arb_frequency"] = arb_frequency return fingerprint @@ -564,13 +566,22 @@ def load_shared_price_data(configs, root=None): return get_historic_parquet_data(tokens, cols=["close"], root=root) -def run_comparison(cfg, price_data=None, low_data_mode=False): +def run_comparison( + cfg, + price_data=None, + low_data_mode=False, + market_linear_noise_data=None, +): """Run both interpolation variants, return results dict.""" params = make_params(cfg) results = {} for method in INTERPOLATION_METHODS: - fp = make_fingerprint(cfg, method) + fp = make_fingerprint( + cfg, + method, + market_linear_noise_data=market_linear_noise_data, + ) results[method] = do_run_on_historic_data( run_fingerprint=fp, params=params, @@ -662,30 +673,44 @@ def _load_persistent_final_value_cache(cache): return disk_cache = {} - disk_records = {} + next_batch_id = 0 cache_path = cache.get("_persistent_final_value_cache_path") if cache_path and os.path.exists(cache_path): - frame = pd.read_parquet(cache_path) - if not frame.empty: + parquet_files = [] + if os.path.isdir(cache_path): + parquet_files = [ + os.path.join(cache_path, filename) + for filename in sorted(os.listdir(cache_path)) + if filename.endswith(".parquet") + ] + batch_ids = [] + for filename in os.listdir(cache_path): + if not (filename.startswith("batch_") and filename.endswith(".parquet")): + continue + token = filename[len("batch_") : -len(".parquet")] + if token.isdigit(): + batch_ids.append(int(token)) + next_batch_id = (max(batch_ids) + 1) if batch_ids else 0 + else: + parquet_files = [cache_path] + + for parquet_file in parquet_files: + frame = pd.read_parquet( + parquet_file, + columns=["cache_key_hash", "final_value"], + ) + if frame.empty: + continue for row in frame.itertuples(index=False): cache_key_hash = str(row.cache_key_hash) final_value = float(row.final_value) disk_cache[cache_key_hash] = final_value - record = { - "cache_key_hash": cache_key_hash, - "final_value": final_value, - } - for column in PERSISTED_FORWARD_VALUE_COLUMNS: - if column in {"cache_key_hash", "final_value"}: - continue - record[column] = getattr(row, column, None) - disk_records[cache_key_hash] = record print( f"Loaded {len(disk_cache)} persisted heatmap forward values from {cache_path}" ) cache["_persistent_final_value_cache"] = disk_cache - cache["_persistent_final_value_records"] = disk_records + cache["_persistent_final_value_next_batch_id"] = next_batch_id cache["_persistent_final_value_cache_loaded"] = True @@ -702,52 +727,63 @@ def flush_sweep_cache(cache, force=False): _load_persistent_final_value_cache(cache) disk_cache = cache.setdefault("_persistent_final_value_cache", {}) - disk_records = cache.setdefault("_persistent_final_value_records", {}) + batch_records = [] for cache_key_hash, record in pending.items(): - merged = dict(disk_records.get(cache_key_hash, {})) - merged.update(record) - merged["cache_key_hash"] = str(cache_key_hash) - merged["final_value"] = float(merged["final_value"]) - disk_records[cache_key_hash] = merged - disk_cache[cache_key_hash] = merged["final_value"] + normalized = dict(record) + normalized["cache_key_hash"] = str(cache_key_hash) + normalized["final_value"] = float(normalized["final_value"]) + disk_cache[cache_key_hash] = normalized["final_value"] + batch_records.append(normalized) cache_path = cache.get("_persistent_final_value_cache_path") if cache_path is None: pending.clear() return - os.makedirs(os.path.dirname(cache_path), exist_ok=True) - sorted_records = [disk_records[key] for key in sorted(disk_records)] + if os.path.exists(cache_path) and not os.path.isdir(cache_path): + raise RuntimeError( + f"Persistent cache path {cache_path} already exists as a file. " + "Use a fresh cache namespace for append-only parquet shards." + ) + + os.makedirs(cache_path, exist_ok=True) + batch_records.sort(key=lambda record: record["cache_key_hash"]) payload = { - column: [record.get(column) for record in sorted_records] + column: [record.get(column) for record in batch_records] for column in PERSISTED_FORWARD_VALUE_COLUMNS } payload["final_value"] = np.asarray(payload["final_value"], dtype=np.float64) frame = pd.DataFrame(payload) - frame.sort_values("cache_key_hash", inplace=True, ignore_index=True) - frame.to_parquet(cache_path, index=False, compression="zstd") + batch_id = int(cache.setdefault("_persistent_final_value_next_batch_id", 0)) + batch_path = os.path.join(cache_path, f"batch_{batch_id:08d}.parquet") + cache["_persistent_final_value_next_batch_id"] = batch_id + 1 + frame.to_parquet(batch_path, index=False, compression="zstd") print( - f"Persisted {len(pending)} new heatmap forward values to {cache_path} " + f"Persisted {len(pending)} new heatmap forward values to {batch_path} " f"({len(disk_cache)} total cached values)." ) pending.clear() -def make_sweep_cache(price_data, cache_scope_cfg=None): +def make_sweep_cache( + price_data, + cache_scope_cfg=None, + market_linear_noise_data=None, +): """Create a shared cache for heatmap and line sweeps.""" cache = { "_shared_price_data": price_data, + "_shared_market_linear_noise_data": market_linear_noise_data, "_final_value_cache": {}, "_comparison_cache": {}, "_pending_persistent_final_values": {}, "_persistent_final_value_cache": {}, - "_persistent_final_value_records": {}, + "_persistent_final_value_next_batch_id": 0, "_persistent_final_value_cache_loaded": False, "_persistent_final_value_cache_path": _heatmap_forward_cache_path( cache_scope_cfg ), } - _load_persistent_final_value_cache(cache) return cache @@ -781,6 +817,10 @@ def _make_method_cache_key(cfg, method): arb_frequency = get_effective_arb_frequency(cfg, noise_cfg) key = ( method, + tuple(str(token) for token in cfg["tokens"]), + str(cfg["start"]), + str(cfg["end"]), + round(float(cfg["fees"]), 12), bool(cfg.get("enable_noise_model", False)), round(float(cfg["price_ratio"]), 6), round(float(cfg["centeredness_margin"]), 6), @@ -812,6 +852,40 @@ def _make_method_cache_key(cfg, method): return key +def _nearest_price_row(price_data, start_ts): + """Select the closest available price row to the requested start timestamp.""" + if len(price_data.index) == 0: + raise ValueError("price_data is empty") + + if isinstance(price_data.index, pd.DatetimeIndex): + target_ts = start_ts + index_tz = getattr(price_data.index, "tz", None) + if index_tz is not None and target_ts.tzinfo is None: + target_ts = target_ts.tz_localize(index_tz) + elif index_tz is None and target_ts.tzinfo is not None: + target_ts = target_ts.tz_convert(None) + target_value = int(target_ts.value) + index_values = price_data.index.asi8 + else: + target_value = int(start_ts.timestamp() * 1000.0) + index_values = price_data.index.to_numpy(dtype=np.int64) + + row_idx = int(np.searchsorted(index_values, target_value, side="left")) + if row_idx >= len(index_values): + row_idx = len(index_values) - 1 + elif row_idx > 0 and index_values[row_idx] != target_value: + prev_idx = row_idx - 1 + if abs(int(index_values[prev_idx]) - target_value) <= abs( + int(index_values[row_idx]) - target_value + ): + row_idx = prev_idx + + row = price_data.iloc[row_idx] + if isinstance(row, pd.DataFrame): + row = row.iloc[0] + return row + + def _make_comparison_cache_key(cfg, launch_final_values): """Cache key for scalar heatmap metrics at a single parameter point.""" noise_cfg = make_noise_variant_cfg(cfg, True) @@ -847,7 +921,11 @@ def _run_method_final_value_cached(cfg, method, cache): return final_value_cache[key] result = do_run_on_historic_data( - run_fingerprint=make_fingerprint(cfg, method), + run_fingerprint=make_fingerprint( + cfg, + method, + market_linear_noise_data=cache.get("_shared_market_linear_noise_data"), + ), params=make_params(cfg), price_data=cache["_shared_price_data"], low_data_mode=True, @@ -1842,25 +1920,7 @@ def build_pair_slice_data(pair, slice_variant, metric_keys): def compute_auto_calibrated_arc_length_speed(cfg, price_data): """Compute the launch/reference auto-calibrated speed for a config.""" start_ts = pd.Timestamp(cfg["start"]) - - if isinstance(price_data.index, pd.DatetimeIndex): - row = price_data.loc[start_ts] - else: - start_unix_ms = int(start_ts.timestamp() * 1000.0) - index_values = price_data.index.to_numpy(dtype=np.int64) - row_idx = int(np.searchsorted(index_values, start_unix_ms, side="left")) - if row_idx >= len(index_values): - row_idx = len(index_values) - 1 - if row_idx > 0 and index_values[row_idx] != start_unix_ms: - prev_idx = row_idx - 1 - if abs(index_values[prev_idx] - start_unix_ms) <= abs( - index_values[row_idx] - start_unix_ms - ): - row_idx = prev_idx - row = price_data.iloc[row_idx] - - if isinstance(row, pd.DataFrame): - row = row.iloc[0] + row = _nearest_price_row(price_data, start_ts) if isinstance(price_data.columns, pd.MultiIndex): initial_price_values = [ @@ -2103,7 +2163,12 @@ def generate_arc_speed_efficiency_artifacts( print("Released arc-speed sweep cache.") -def get_launch_final_values(all_results, launch_cfg, price_data): +def get_launch_final_values( + all_results, + launch_cfg, + price_data, + market_linear_noise_data=None, +): """Reuse launch-style runs when available; otherwise run them once.""" for cfg, results in all_results: if cfg["name"] == launch_cfg["name"]: @@ -2121,6 +2186,7 @@ def get_launch_final_values(all_results, launch_cfg, price_data): launch_cfg, price_data=price_data, low_data_mode=True, + market_linear_noise_data=market_linear_noise_data, ) launch_final_values = { "geometric": float(launch_results["geometric"]["final_value"]), @@ -2388,6 +2454,7 @@ def plot_comparison(cfg, results, fig_idx): if __name__ == "__main__": shared_price_data = load_shared_price_data(CONFIGS) + shared_market_linear_noise_data = load_shared_market_linear_noise_data() for initial_pool_value in TVL_SWEEP_VALUES: tvl_configs = configs_for_tvl(CONFIGS, initial_pool_value) @@ -2398,7 +2465,11 @@ def plot_comparison(cfg, results, fig_idx): for i, cfg in enumerate(tvl_configs): print(f"\n>>> Running {cfg['name']} at TVL {tvl_label}...") try: - results = run_comparison(cfg, price_data=shared_price_data) + results = run_comparison( + cfg, + price_data=shared_price_data, + market_linear_noise_data=shared_market_linear_noise_data, + ) print_comparison(cfg, results) plot_comparison(cfg, results, i) all_results.append((cfg, results)) @@ -2503,10 +2574,12 @@ def plot_comparison(cfg, results, fig_idx): all_results, launch_cfg=tvl_configs[0], price_data=shared_price_data, + market_linear_noise_data=shared_market_linear_noise_data, ) shared_sweep_cache = make_sweep_cache( shared_price_data, cache_scope_cfg=tvl_configs[1], + market_linear_noise_data=shared_market_linear_noise_data, ) print(f"\nGenerating thermostat heatmaps for TVL {tvl_label}...") diff --git a/scripts/reclamm/find_adjacent_heatmap_pairs.py b/scripts/reclamm/find_adjacent_heatmap_pairs.py index 1bc4ce36..51a3a8a4 100644 --- a/scripts/reclamm/find_adjacent_heatmap_pairs.py +++ b/scripts/reclamm/find_adjacent_heatmap_pairs.py @@ -96,6 +96,30 @@ def build_inclusive_sweep(start: float, stop: float, step: float) -> np.ndarray: return values +def _resolve_repo_root(script_path): + """Locate the repository root from either scripts/ or scripts/reclamm/.""" + script_path = Path(script_path).resolve() + for parent in script_path.parents: + if (parent / "quantammsim").exists() and (parent / "scripts").exists(): + return parent + return script_path.parents[1] + + +REPO_ROOT = _resolve_repo_root(__file__) +DEFAULT_MARKET_LINEAR_NOISE_START_DATE = "2024-06-01" +DEFAULT_MARKET_LINEAR_NOISE_END_DATE = "2026-03-01" +DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH = str( + REPO_ROOT + / "results" + / "linear_market_noise" + / "_sim_arrays" + / ( + "0x9d1fcf346ea1b0_" + f"{DEFAULT_MARKET_LINEAR_NOISE_START_DATE}_{DEFAULT_MARKET_LINEAR_NOISE_END_DATE}.npz" + ) +) + + class _LightweightCompareContext: """Small subset of compare_reclamm_thermostats usable without JAX.""" @@ -112,13 +136,16 @@ class _LightweightCompareContext: FIXED_SLICE_FRACTIONS = (0.125, 0.375, 0.625, 0.875) FIXED_SLICE_LABELS = ("Q1", "Q2", "Q3", "Q4") HEATMAP_FORWARD_CACHE_ENABLED = True - HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_v1" + HEATMAP_FORWARD_CACHE_RUN_NAME = "aave_eth_thermostat_heatmaps_market_linear_v2" HEATMAP_FORWARD_CACHE_ROOT = os.path.join( "results", "reclamm_heatmap_forward_cache", ) AAVE_WETH_POOL_ID = "0x9d1fcf346ea1b0" DEFAULT_MARKET_LINEAR_ARTIFACT_DIR = "results/linear_market_noise" + DEFAULT_MARKET_LINEAR_NOISE_START_DATE = DEFAULT_MARKET_LINEAR_NOISE_START_DATE + DEFAULT_MARKET_LINEAR_NOISE_END_DATE = DEFAULT_MARKET_LINEAR_NOISE_END_DATE + DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH = DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH DEFAULT_NOISE_MODEL = "market_linear" DEFAULT_GAS_COST = 1.0 DEFAULT_PROTOCOL_FEE_SPLIT = 0.25 @@ -138,8 +165,10 @@ class _LightweightCompareContext: AAVE_ETH_NOISE_SETTINGS = { "enable_noise_model": True, "noise_model": DEFAULT_NOISE_MODEL, + "noise_reference_model": DEFAULT_NOISE_MODEL, "noise_artifact_dir": DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, "noise_pool_id": AAVE_WETH_POOL_ID, + "noise_arrays_path": DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, "arb_frequency": FIXED_COMPARE_ARB_FREQUENCY, "gas_cost": DEFAULT_GAS_COST, "protocol_fee_split": DEFAULT_PROTOCOL_FEE_SPLIT, @@ -197,6 +226,9 @@ def from_compare_module(cls, compare_module): "HEATMAP_FORWARD_CACHE_ROOT", "AAVE_WETH_POOL_ID", "DEFAULT_MARKET_LINEAR_ARTIFACT_DIR", + "DEFAULT_MARKET_LINEAR_NOISE_START_DATE", + "DEFAULT_MARKET_LINEAR_NOISE_END_DATE", + "DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH", "DEFAULT_NOISE_MODEL", "DEFAULT_GAS_COST", "DEFAULT_PROTOCOL_FEE_SPLIT", @@ -225,7 +257,7 @@ def from_compare_module(cls, compare_module): return context def set_noise_profile(self, profile): - if profile not in {"market_linear", "legacy_calibrated"}: + if profile != "market_linear": raise ValueError(f"Unsupported lightweight noise profile: {profile}") if profile != self.noise_profile: self.noise_profile = profile @@ -353,6 +385,13 @@ def get_effective_arb_frequency(self, cfg, noise_cfg=None): del noise_cfg return self._normalize_arb_frequency(self.FIXED_COMPARE_ARB_FREQUENCY) + def _canonical_noise_reference_model(self, cfg): + noise_model = cfg.get("noise_model", self.DEFAULT_NOISE_MODEL) or self.DEFAULT_NOISE_MODEL + reference_model = cfg.get("noise_reference_model") + if reference_model is None: + reference_model = self.DEFAULT_NOISE_MODEL if noise_model == "arb_only" else noise_model + return str(reference_model) + def normalize_compare_run_cfg(self, cfg, enable_noise_model=None): updated = dict(cfg) updated["price_ratio"] = float(cfg["price_ratio"]) @@ -380,53 +419,79 @@ def normalize_compare_run_cfg(self, cfg, enable_noise_model=None): ) updated["enable_noise_model"] = use_noise - requested_mode = ( - cfg.get("noise_model", self.DEFAULT_NOISE_MODEL) - or self.DEFAULT_NOISE_MODEL - ) + reference_mode = self._canonical_noise_reference_model(cfg) if use_noise: - canonical_noise_model = ( - requested_mode - if requested_mode != "arb_only" - else self.DEFAULT_NOISE_MODEL - ) - updated["noise_model"] = canonical_noise_model - if canonical_noise_model == "market_linear": + updated["noise_model"] = reference_mode + updated["noise_reference_model"] = reference_mode + else: + updated["noise_model"] = "arb_only" + updated["noise_reference_model"] = reference_mode + + if reference_mode == "market_linear": + updated["noise_arrays_path"] = self.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + updated.pop("reclamm_noise_params", None) + if use_noise or updated["noise_model"] == "arb_only": updated["noise_artifact_dir"] = self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR updated["noise_pool_id"] = self.AAVE_WETH_POOL_ID - else: - updated.pop("noise_artifact_dir", None) - updated.pop("noise_pool_id", None) + else: updated.pop("reclamm_noise_params", None) updated.pop("noise_arrays_path", None) - else: - updated["noise_model"] = "arb_only" - for key in ( - "reclamm_noise_params", - "noise_arrays_path", - "noise_artifact_dir", - "noise_pool_id", - ): - updated.pop(key, None) + updated.pop("noise_artifact_dir", None) + updated.pop("noise_pool_id", None) return updated - def _legacy_calibrated_noise_settings(self, arb_frequency=None): + def _load_market_linear_noise_stats(self, arrays_path=None): + arrays_path = os.path.abspath( + os.fspath(arrays_path or self.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH) + ) + if not os.path.exists(arrays_path): + raise FileNotFoundError(f"market_linear arrays file not found: {arrays_path}") + + with np.load(arrays_path) as arrays: + required_keys = {"noise_base", "noise_tvl_coeff", "tvl_mean", "tvl_std"} + missing_keys = sorted(required_keys.difference(arrays.files)) + if missing_keys: + raise KeyError( + f"market_linear arrays file {arrays_path} is missing keys: {missing_keys}" + ) + return arrays_path, float(arrays["tvl_mean"]), float(arrays["tvl_std"]) + + def _market_linear_noise_settings(self, noise_model="market_linear", arb_frequency=None): + arrays_path, tvl_mean, tvl_std = self._load_market_linear_noise_stats() + arb_frequency = self._normalize_arb_frequency(arb_frequency) + return { + "noise_model": noise_model, + "noise_trader_ratio": 0.0, + "reclamm_noise_params": { + "tvl_mean": tvl_mean, + "tvl_std": tvl_std, + }, + "noise_arrays_path": arrays_path, + "arb_frequency": arb_frequency, + "noise_summary": f"{noise_model} (arb_frequency={arb_frequency})", + "noise_cache_key": ( + noise_model, + arrays_path, + arb_frequency, + round(tvl_mean, 12), + round(tvl_std, 12), + ), + } + + def _legacy_calibrated_noise_settings(self, arb_frequency=None, noise_model="calibrated"): arb_frequency = self._normalize_arb_frequency(arb_frequency) return { - "noise_model": "calibrated", + "noise_model": noise_model, "noise_trader_ratio": 0.0, "reclamm_noise_params": { f"c_{i}": self.LEGACY_NOISE_COEFFS[i] for i in range(len(self.LEGACY_NOISE_COEFFS)) }, "arb_frequency": arb_frequency, - "noise_summary": ( - "calibrated legacy 8-covariate " - f"(arb_frequency={arb_frequency})" - ), + "noise_summary": f"{noise_model} (arb_frequency={arb_frequency})", "noise_cache_key": ( - "calibrated", + noise_model, tuple(round(float(c), 12) for c in self.LEGACY_NOISE_COEFFS), arb_frequency, ), @@ -436,6 +501,7 @@ def resolve_reclamm_noise_settings(self, cfg): cfg = self.normalize_compare_run_cfg(cfg) enable_noise_model = cfg.get("enable_noise_model", False) requested_mode = cfg.get("noise_model", self.DEFAULT_NOISE_MODEL) + reference_mode = cfg.get("noise_reference_model", self.DEFAULT_NOISE_MODEL) requested_arb_frequency = self.get_effective_arb_frequency(cfg) cache_key = ( tuple(cfg.get("tokens", [])), @@ -443,6 +509,7 @@ def resolve_reclamm_noise_settings(self, cfg): cfg.get("end"), enable_noise_model, requested_mode, + reference_mode, cfg.get("noise_artifact_dir", self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR), cfg.get("noise_pool_id", self.AAVE_WETH_POOL_ID), requested_arb_frequency, @@ -453,68 +520,39 @@ def resolve_reclamm_noise_settings(self, cfg): if cache_key in self._noise_settings_cache: return self._noise_settings_cache[cache_key] - if not enable_noise_model: - result = { - "noise_model": "arb_only", - "noise_trader_ratio": 0.0, - "reclamm_noise_params": None, - "noise_arrays_path": None, - "arb_frequency": requested_arb_frequency, - "noise_summary": "arb_only (noise disabled)", - "noise_cache_key": ("disabled",), - } - elif requested_mode == "market_linear": - if self.noise_profile == "legacy_calibrated": - result = self._legacy_calibrated_noise_settings( + if requested_mode == "arb_only": + if reference_mode == "market_linear": + result = self._market_linear_noise_settings( + noise_model="arb_only", arb_frequency=requested_arb_frequency ) - self._noise_settings_cache[cache_key] = result - return result - artifact_dir = cfg.get( - "noise_artifact_dir", - self.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR, - ) - pool_id = cfg.get("noise_pool_id", self.AAVE_WETH_POOL_ID) - start_date = str(cfg["start"]).split(" ")[0] - end_date = str(cfg["end"]).split(" ")[0] - arrays_path = cfg.get("noise_arrays_path") or os.path.join( - artifact_dir, - "_sim_arrays", - f"{pool_id}_{start_date}_{end_date}.npz", - ) - meta_path = os.path.join(artifact_dir, "meta.json") - model_path = os.path.join(artifact_dir, "model.npz") - if not ( - os.path.exists(arrays_path) - and os.path.exists(meta_path) - and os.path.exists(model_path) - ): + elif reference_mode == "calibrated": result = self._legacy_calibrated_noise_settings( + noise_model="arb_only", arb_frequency=requested_arb_frequency ) else: - with np.load(arrays_path) as arrays: - tvl_mean = float(arrays["tvl_mean"]) - tvl_std = float(arrays["tvl_std"]) arb_frequency = requested_arb_frequency result = { - "noise_model": "market_linear", - "noise_trader_ratio": 0.0, - "reclamm_noise_params": { - "tvl_mean": tvl_mean, - "tvl_std": tvl_std, - }, - "noise_arrays_path": arrays_path, + "noise_model": "arb_only", + "noise_trader_ratio": cfg.get("noise_trader_ratio", 0.0), + "reclamm_noise_params": cfg.get("reclamm_noise_params"), + "noise_arrays_path": cfg.get("noise_arrays_path"), "arb_frequency": arb_frequency, - "noise_summary": f"market_linear (arb_frequency={arb_frequency})", + "noise_summary": f"arb_only (arb_frequency={arb_frequency})", "noise_cache_key": ( - "market_linear", - arrays_path, + "arb_only", + round(float(cfg.get("noise_trader_ratio", 0.0)), 12), + self._hashable_noise_params(cfg.get("reclamm_noise_params")), + cfg.get("noise_arrays_path"), arb_frequency, - round(tvl_mean, 12), - round(tvl_std, 12), ), } + elif requested_mode == "market_linear": + result = self._market_linear_noise_settings( + noise_model="market_linear", + arb_frequency=requested_arb_frequency, + ) elif requested_mode == "calibrated": result = self._legacy_calibrated_noise_settings( arb_frequency=requested_arb_frequency @@ -815,65 +853,15 @@ def autodetect_lightweight_noise_profile( ): if not hasattr(compare_module, "set_noise_profile"): return - - metric_spec = get_metric_spec(metric_key) - if not pair_specs: - return - - pair_spec = pair_specs[0] - slice_variants = resolve_slice_variants(pair_spec, slice_slug) - if not slice_variants: - return - - slice_variant = slice_variants[0] - x_values = list(pair_spec["x_values"]) - y_values = list(pair_spec["y_values"]) - sample_x_indices = sorted({0, len(x_values) // 2, len(x_values) - 1}) - sample_y_indices = sorted({0, len(y_values) // 2, len(y_values) - 1}) - - scores = {} - for profile in ("market_linear", "legacy_calibrated"): - compare_module.set_noise_profile(profile) - hit_count = 0 - probe_count = 0 - slice_cfg = dict(base_cfg) - slice_cfg[pair_spec["fixed_key"]] = float(slice_variant["value"]) - for y_index in sample_y_indices: - for x_index in sample_x_indices: - cfg = dict(slice_cfg) - cfg[pair_spec["x_key"]] = float(x_values[x_index]) - cfg[pair_spec["y_key"]] = float(y_values[y_index]) - for source_name in metric_spec["sources"]: - source_cfg, method = _source_variant(compare_module, cfg, source_name) - cache_key = compare_module._make_method_cache_key(source_cfg, method) - cache_key_hash = compare_module._make_method_cache_hash(cache_key) - probe_count += 1 - if cache_key_hash in cache_lookup: - hit_count += 1 - scores[profile] = (hit_count, probe_count) - - best_profile = max( - scores, - key=lambda profile: (scores[profile][0], scores[profile][1], profile == "market_linear"), - ) - compare_module.set_noise_profile(best_profile) - hit_count, probe_count = scores[best_profile] - print( - f"Lightweight noise profile auto-detect chose {best_profile} " - f"({hit_count}/{probe_count} sample cache hits)." - ) + del base_cfg, pair_specs, metric_key, slice_slug, cache_lookup + compare_module.set_noise_profile("market_linear") + print("Lightweight noise profile fixed to market_linear.") def _source_variant(compare_module, cfg: Mapping[str, object], source_name: str): enable_noise_model = source_name.startswith("noise_") method = "geometric" if source_name.endswith("geometric") else "constant_arc_length" - source_cfg = compare_module.make_noise_variant_cfg(cfg, enable_noise_model) - source_cfg["noise_model"] = ( - getattr(compare_module, "DEFAULT_NOISE_MODEL", "market_linear") - if enable_noise_model - else "arb_only" - ) - return source_cfg, method + return compare_module.make_noise_variant_cfg(cfg, enable_noise_model), method def _compute_metric_value(metric_key: str, final_values: Mapping[str, float]) -> float: @@ -1169,7 +1157,10 @@ def main() -> int: ) cache_path = resolve_existing_cache_path(cache_path) if not cache_path.exists(): - raise FileNotFoundError(f"Cache parquet not found: {cache_path}") + raise FileNotFoundError( + f"Cache parquet not found: {cache_path}. " + "Generate the current market_linear/arb_only heatmap cache first." + ) cache_lookup = load_cache_lookup(cache_path) autodetect_lightweight_noise_profile( @@ -1227,7 +1218,9 @@ def main() -> int: if missing_any and not args.allow_partial_cache: raise RuntimeError( "Cache was incomplete for at least one requested heatmap slice. " - "Re-run with --allow-partial-cache to write the rows that were resolvable." + "This cache does not match the current market_linear/arb_only " + "parameterization. Regenerate the heatmap cache, or re-run with " + "--allow-partial-cache to write only the rows that were resolvable." ) frame = rows_to_frame(rows) diff --git a/tests/scripts/test_compare_reclamm_geometric_noise_runs.py b/tests/scripts/test_compare_reclamm_geometric_noise_runs.py index 8a0fb740..c211e7e1 100644 --- a/tests/scripts/test_compare_reclamm_geometric_noise_runs.py +++ b/tests/scripts/test_compare_reclamm_geometric_noise_runs.py @@ -32,7 +32,7 @@ def test_build_run_specs_from_adjacent_row_maps_csv_cells_to_two_specs(): row = { "metric_key": "noise_vs_arb_geometric_improvement_pct", "metric_unit": "pct", - "source_noise_profile": "legacy_calibrated", + "source_noise_profile": "market_linear", "pair_slug": "price_ratio_vs_margin", "slice_slug": "q2", "adjacency_axis": "horizontal", @@ -58,7 +58,7 @@ def test_build_run_specs_from_adjacent_row_maps_csv_cells_to_two_specs(): assert "adjacent_pairs.csv row 0" in description assert "price_ratio_vs_margin q2" in description assert "horizontal" in description - assert "noise_profile=legacy_calibrated" in description + assert "noise_profile=market_linear" in description assert len(run_specs) == 2 assert run_specs[0]["name"] == "Top diff row cell 1" assert run_specs[0]["price_ratio"] == 1.335 @@ -66,7 +66,7 @@ def test_build_run_specs_from_adjacent_row_maps_csv_cells_to_two_specs(): assert run_specs[0]["daily_price_shift_exponent"] == 0.1975 assert run_specs[0]["tvl_usd"] == 1_000_000.0 assert run_specs[0]["color"] == "C0" - assert run_specs[0]["source_noise_profile"] == "legacy_calibrated" + assert run_specs[0]["source_noise_profile"] == "market_linear" assert "heatmap_value=-53.054321" in run_specs[0]["reason"] assert run_specs[1]["name"] == "Top diff row cell 2" assert run_specs[1]["price_ratio"] == 1.36 @@ -86,7 +86,7 @@ def test_default_output_file_for_adjacent_csv_uses_csv_stem_and_row_index(): ) -def test_build_run_config_honors_legacy_calibrated_noise_profile(): +def test_build_run_config_rejects_legacy_calibrated_noise_profile(): module = load_script_module() base_config = { "name": "base", @@ -107,11 +107,57 @@ def test_build_run_config_honors_legacy_calibrated_noise_profile(): "source_noise_profile": "legacy_calibrated", } - cfg = module.build_run_config(spec, base_config=base_config) + import pytest + + with pytest.raises(ValueError): + module.build_run_config(spec, base_config=base_config) + + +def test_build_run_variants_canonicalizes_noise_and_arb_only_configs(): + module = load_script_module() + fixed_path = str( + Path(__file__).resolve().parents[2] + / "results" + / "linear_market_noise" + / "_sim_arrays" + / "0x9d1fcf346ea1b0_2024-06-01_2026-03-01.npz" + ) + base_config = { + "name": "base", + "price_ratio": 1.1, + "centeredness_margin": 0.6, + "daily_price_shift_exponent": 0.1, + "initial_pool_value": 1_000_000.0, + "noise_model": "market_linear", + "noise_arrays_path": fixed_path, + } + spec = { + "name": "cell", + "price_ratio": 1.335, + "centeredness_margin": 0.3184210526, + "daily_price_shift_exponent": 0.1975, + "tvl_usd": 1_000_000.0, + "source_noise_profile": "market_linear", + } - assert cfg["noise_model"] == "calibrated" - assert "reclamm_noise_params" not in cfg - assert "noise_arrays_path" not in cfg + class FakeThermostatCompare: + @staticmethod + def make_noise_variant_cfg(cfg, enable_noise_model): + updated = dict(cfg) + updated["enable_noise_model"] = bool(enable_noise_model) + updated["noise_model"] = "market_linear" if enable_noise_model else "arb_only" + updated["noise_arrays_path"] = fixed_path + updated["reclamm_noise_params"] = {"tvl_mean": 1.0, "tvl_std": 2.0} + return updated + + variants = module.build_run_variants(spec, base_config, FakeThermostatCompare) + + assert variants["noise"]["noise_model"] == "market_linear" + assert variants["arb"]["noise_model"] == "arb_only" + assert variants["noise"]["noise_arrays_path"] == fixed_path + assert variants["arb"]["noise_arrays_path"] == fixed_path + assert variants["noise"]["reclamm_noise_params"] == {"tvl_mean": 1.0, "tvl_std": 2.0} + assert variants["arb"]["reclamm_noise_params"] == {"tvl_mean": 1.0, "tvl_std": 2.0} def test_print_run_inputs_to_terminal_includes_fingerprint_and_update_params(capsys): diff --git a/tests/scripts/test_compare_reclamm_thermostats.py b/tests/scripts/test_compare_reclamm_thermostats.py index 3ea0da79..caec6ae6 100644 --- a/tests/scripts/test_compare_reclamm_thermostats.py +++ b/tests/scripts/test_compare_reclamm_thermostats.py @@ -163,16 +163,21 @@ def test_make_noise_variant_cfg_disables_noise_fields(script_module, base_cfg): resolved = script_module.resolve_reclamm_noise_settings(arb_only_cfg) assert arb_only_cfg["enable_noise_model"] is False - assert arb_only_cfg["noise_model"] is None + assert arb_only_cfg["noise_model"] == "arb_only" + assert arb_only_cfg["noise_reference_model"] == "market_linear" assert arb_only_cfg["gas_cost"] == script_module.DEFAULT_GAS_COST assert arb_only_cfg["protocol_fee_split"] == script_module.DEFAULT_PROTOCOL_FEE_SPLIT assert arb_only_cfg["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY - assert "noise_artifact_dir" not in arb_only_cfg - assert "noise_pool_id" not in arb_only_cfg + assert arb_only_cfg["noise_artifact_dir"] == script_module.DEFAULT_MARKET_LINEAR_ARTIFACT_DIR + assert arb_only_cfg["noise_pool_id"] == script_module.AAVE_WETH_POOL_ID + assert arb_only_cfg["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH assert "reclamm_noise_params" not in arb_only_cfg - assert "noise_arrays_path" not in arb_only_cfg - assert resolved["noise_model"] is None - assert resolved["noise_summary"] == "arb-only (noise disabled)" + assert resolved["noise_model"] == "arb_only" + assert resolved["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + assert set(resolved["reclamm_noise_params"]) == {"tvl_mean", "tvl_std"} + assert resolved["noise_summary"] == ( + f"arb_only (arb_frequency={script_module.FIXED_COMPARE_ARB_FREQUENCY})" + ) def test_make_noise_variant_cfg_defaults_to_fixed_compare_arb_cadence( @@ -192,6 +197,7 @@ def test_make_noise_variant_cfg_defaults_to_fixed_compare_arb_cadence( resolved_noise = script_module.resolve_reclamm_noise_settings(noisy_cfg) assert arb_only_cfg["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY + assert arb_only_cfg["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH assert resolved_noise["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY @@ -228,12 +234,137 @@ def test_make_fingerprint_ignores_non_axis_override_fields(script_module, base_c assert overridden_key == canonical_key assert overridden_fingerprint["arb_frequency"] == script_module.FIXED_COMPARE_ARB_FREQUENCY assert overridden_fingerprint["gas_cost"] == script_module.DEFAULT_GAS_COST + assert ( + overridden_fingerprint["noise_arrays_path"] + == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + ) assert ( overridden_fingerprint["protocol_fee_split"] == script_module.DEFAULT_PROTOCOL_FEE_SPLIT ) +def test_arb_only_fingerprint_only_changes_noise_model(script_module, base_cfg): + noisy_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + + noise_fingerprint = script_module.make_fingerprint(noisy_cfg, "geometric") + arb_only_cfg = script_module.make_noise_variant_cfg(noisy_cfg, enable_noise_model=False) + arb_fingerprint = script_module.make_fingerprint(arb_only_cfg, "geometric") + + expected_arb = dict(noise_fingerprint) + expected_arb["noise_model"] = "arb_only" + + assert noise_fingerprint["noise_model"] == "market_linear" + assert noise_fingerprint["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + assert arb_fingerprint == expected_arb + + +def test_make_fingerprint_keeps_path_fallback_when_arrays_not_preloaded( + script_module, + base_cfg, +): + noisy_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + + fingerprint = script_module.make_fingerprint(noisy_cfg, "geometric") + + assert fingerprint["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + assert "noise_base_array" not in fingerprint + assert "noise_tvl_coeff_array" not in fingerprint + + +def test_make_fingerprint_includes_preloaded_market_linear_arrays( + script_module, + base_cfg, +): + noisy_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + shared_noise = { + "arrays_path": script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, + "noise_base_array": np.array([1.0, 2.0]), + "noise_tvl_coeff_array": np.array([3.0, 4.0]), + "tvl_mean": 10.0, + "tvl_std": 5.0, + } + + fingerprint = script_module.make_fingerprint( + noisy_cfg, + "geometric", + market_linear_noise_data=shared_noise, + ) + + assert fingerprint["noise_arrays_path"] == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + assert np.array_equal(fingerprint["noise_base_array"], shared_noise["noise_base_array"]) + assert np.array_equal( + fingerprint["noise_tvl_coeff_array"], + shared_noise["noise_tvl_coeff_array"], + ) + + +@pytest.mark.parametrize( + ("field", "updated_value"), + [ + ("fees", 0.01), + ("start", "2024-07-01 00:00:00"), + ("end", "2025-07-01 00:00:00"), + ("tokens", ["WBTC", "ETH"]), + ], +) +def test_method_cache_key_includes_run_identity_fields( + script_module, + base_cfg, + field, + updated_value, +): + canonical_cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + updated_cfg = dict(canonical_cfg) + updated_cfg[field] = updated_value + + canonical_key = script_module._make_method_cache_key(canonical_cfg, "geometric") + updated_key = script_module._make_method_cache_key(updated_cfg, "geometric") + + assert updated_key != canonical_key + + +@pytest.mark.parametrize( + "cfg_overrides", + [ + {"enable_noise_model": True, "noise_model": "calibrated"}, + { + "enable_noise_model": False, + "noise_model": "arb_only", + "noise_reference_model": "calibrated", + }, + ], +) +def test_resolve_reclamm_noise_settings_rejects_legacy_modes( + script_module, + base_cfg, + cfg_overrides, +): + cfg = { + **base_cfg, + **cfg_overrides, + } + + with pytest.raises(ValueError, match="market_linear"): + script_module.resolve_reclamm_noise_settings(cfg) + + def test_generate_heatmaps_skips_existing_pairs( monkeypatch, script_module, @@ -474,6 +605,71 @@ def itertuples(self, index=False): assert value == pytest.approx(1_234_567.0) +def test_run_method_final_value_cached_passes_preloaded_market_linear_arrays( + monkeypatch, + script_module, + base_cfg, +): + captured = {} + shared_noise = { + "arrays_path": script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH, + "noise_base_array": np.array([11.0, 12.0]), + "noise_tvl_coeff_array": np.array([21.0, 22.0]), + "tvl_mean": 100.0, + "tvl_std": 25.0, + } + + monkeypatch.setattr( + script_module, + "_load_persistent_final_value_cache", + lambda cache: cache.update( + { + "_persistent_final_value_cache_loaded": True, + "_persistent_final_value_cache": {}, + "_persistent_final_value_next_batch_id": 0, + } + ), + ) + monkeypatch.setattr(script_module, "flush_sweep_cache", lambda *args, **kwargs: None) + + def fake_do_run_on_historic_data(**kwargs): + captured["run_fingerprint"] = kwargs["run_fingerprint"] + return {"final_value": 1_111_111.0} + + monkeypatch.setattr( + script_module, + "do_run_on_historic_data", + fake_do_run_on_historic_data, + ) + + cfg = { + **base_cfg, + "enable_noise_model": True, + "noise_model": "market_linear", + } + cache = script_module.make_sweep_cache( + price_data=None, + cache_scope_cfg=cfg, + market_linear_noise_data=shared_noise, + ) + + value = script_module._run_method_final_value_cached(cfg, "geometric", cache) + + assert value == pytest.approx(1_111_111.0) + assert np.array_equal( + captured["run_fingerprint"]["noise_base_array"], + shared_noise["noise_base_array"], + ) + assert np.array_equal( + captured["run_fingerprint"]["noise_tvl_coeff_array"], + shared_noise["noise_tvl_coeff_array"], + ) + assert ( + captured["run_fingerprint"]["noise_arrays_path"] + == script_module.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + ) + + def test_arc_speed_artifacts_only_build_missing_line_output( monkeypatch, script_module, @@ -547,6 +743,67 @@ def fake_build_metric_curve(**kwargs): assert plotted_lines == [missing_line] +def test_compute_auto_calibrated_arc_length_speed_uses_nearest_datetime_row( + monkeypatch, + script_module, +): + real_pd = pytest.importorskip("pandas") + monkeypatch.setattr(script_module.pd, "Timestamp", real_pd.Timestamp) + monkeypatch.setattr(script_module.pd, "DatetimeIndex", real_pd.DatetimeIndex) + monkeypatch.setattr(script_module.pd, "DataFrame", real_pd.DataFrame) + monkeypatch.setattr( + script_module.pd, + "MultiIndex", + real_pd.MultiIndex, + raising=False, + ) + + price_data = real_pd.DataFrame( + { + "close_AAVE": [10.0, 30.0], + "close_ETH": [1.0, 1.0], + }, + index=real_pd.DatetimeIndex( + ["2024-06-01 00:01:00", "2024-06-01 00:03:00"] + ), + ) + cfg = { + "tokens": ["AAVE", "ETH"], + "start": "2024-06-01 00:02:10", + "price_ratio": 1.10, + "centeredness_margin": 0.60, + "daily_price_shift_exponent": 0.1, + "initial_pool_value": 1_000_000.0, + } + captured = {} + + monkeypatch.setattr( + script_module, + "initialise_reclamm_reserves", + lambda *args, **kwargs: (np.array([1.0, 1.0]), 1.0, 1.0), + ) + monkeypatch.setattr( + script_module, + "compute_price_ratio", + lambda *args, **kwargs: 1.0, + ) + + def fake_calibrate_arc_length_speed(*args, **kwargs): + captured["market_price_0"] = args[7] + return 123.0 + + monkeypatch.setattr( + script_module, + "calibrate_arc_length_speed", + fake_calibrate_arc_length_speed, + ) + + result = script_module.compute_auto_calibrated_arc_length_speed(cfg, price_data) + + assert result == pytest.approx(123.0) + assert captured["market_price_0"] == pytest.approx(30.0) + + def test_flush_sweep_cache_writes_compact_scalar_parquet(script_module): captured = {} @@ -554,9 +811,6 @@ class FakeFrame: def __init__(self, payload): captured["payload"] = payload - def sort_values(self, *args, **kwargs): - captured["sort_values"] = (args, kwargs) - def to_parquet(self, path, index=False, compression=None): captured["path"] = path captured["index"] = index @@ -583,7 +837,7 @@ def to_parquet(self, path, index=False, compression=None): } }, "_persistent_final_value_cache": {}, - "_persistent_final_value_records": {}, + "_persistent_final_value_next_batch_id": 0, "_persistent_final_value_cache_loaded": True, "_persistent_final_value_cache_path": "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_5m.parquet", } @@ -598,10 +852,81 @@ def to_parquet(self, path, index=False, compression=None): assert captured["payload"]["arb_frequency"] == [15] assert captured["index"] is False assert captured["compression"] == "zstd" - assert captured["path"].endswith(".parquet") + assert captured["makedirs"].endswith("forward_values_tvl_5m.parquet") + assert captured["path"].endswith("forward_values_tvl_5m.parquet/batch_00000000.parquet") assert cache["_pending_persistent_final_values"] == {} assert cache["_persistent_final_value_cache"] == {"abc123": 123.45} - assert cache["_persistent_final_value_records"]["abc123"]["noise_model"] == "market_linear" + assert cache["_persistent_final_value_next_batch_id"] == 1 + + +def test_load_persistent_final_value_cache_reads_sharded_parquet_dir( + monkeypatch, + script_module, +): + frames = { + "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_1m.parquet/batch_00000000.parquet": [ + types.SimpleNamespace( + cache_key_hash="first123", + final_value=111.0, + method="geometric", + enable_noise_model=True, + noise_model="market_linear", + price_ratio=1.1, + centeredness_margin=0.6, + daily_price_shift_exponent=0.1, + initial_pool_value=1_000_000.0, + arb_frequency=15, + ) + ], + "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_1m.parquet/batch_00000001.parquet": [ + types.SimpleNamespace( + cache_key_hash="second456", + final_value=222.0, + method="geometric", + enable_noise_model=False, + noise_model="arb_only", + price_ratio=1.2, + centeredness_margin=0.7, + daily_price_shift_exponent=0.2, + initial_pool_value=1_000_000.0, + arb_frequency=15, + ) + ], + } + + class FakeFrame: + def __init__(self, rows): + self._rows = rows + self.empty = not rows + + def itertuples(self, index=False): + return list(self._rows) + + monkeypatch.setattr(script_module.os.path, "exists", lambda filename: True) + monkeypatch.setattr(script_module.os.path, "isdir", lambda filename: True) + monkeypatch.setattr( + script_module.os, + "listdir", + lambda path: ["batch_00000001.parquet", "batch_00000000.parquet"], + ) + monkeypatch.setattr( + script_module.pd, + "read_parquet", + lambda path, *args, **kwargs: FakeFrame(frames[path]), + ) + + cache = { + "_persistent_final_value_cache_loaded": False, + "_persistent_final_value_cache_path": "results/reclamm_heatmap_forward_cache/test/forward_values_tvl_1m.parquet", + } + + script_module._load_persistent_final_value_cache(cache) + + assert cache["_persistent_final_value_cache"] == { + "first123": 111.0, + "second456": 222.0, + } + assert cache["_persistent_final_value_next_batch_id"] == 2 def test_load_persistent_final_value_cache_supports_legacy_two_column_parquet( @@ -630,8 +955,27 @@ def itertuples(self, index=False): script_module._load_persistent_final_value_cache(cache) assert cache["_persistent_final_value_cache"] == {"legacy123": 999.0} - assert cache["_persistent_final_value_records"]["legacy123"]["final_value"] == pytest.approx(999.0) - assert cache["_persistent_final_value_records"]["legacy123"]["price_ratio"] is None + assert cache["_persistent_final_value_next_batch_id"] == 0 + + +def test_make_sweep_cache_does_not_eagerly_load_persisted_parquet( + monkeypatch, + script_module, +): + calls = [] + + monkeypatch.setattr( + script_module, + "_load_persistent_final_value_cache", + lambda cache: calls.append(dict(cache)), + ) + + cache = script_module.make_sweep_cache(price_data=None, cache_scope_cfg=None) + + assert calls == [] + assert cache["_persistent_final_value_cache"] == {} + assert cache["_persistent_final_value_next_batch_id"] == 0 + assert cache["_persistent_final_value_cache_loaded"] is False def test_run_comparison_cached_only_uses_geometric_runs_when_constant_arc_disabled( diff --git a/tests/scripts/test_find_adjacent_heatmap_pairs.py b/tests/scripts/test_find_adjacent_heatmap_pairs.py index 597e65b1..6ad7a397 100644 --- a/tests/scripts/test_find_adjacent_heatmap_pairs.py +++ b/tests/scripts/test_find_adjacent_heatmap_pairs.py @@ -228,30 +228,11 @@ def run_adjacent_csv_row_comparison(csv_path, row_index=0, output_file=None): } -def test_autodetect_lightweight_noise_profile_switches_to_legacy_calibrated(): +def test_autodetect_lightweight_noise_profile_keeps_market_linear(): module = load_script_module() compare_context = module._LightweightCompareContext() base_cfg = compare_context.configs_for_tvl(compare_context.CONFIGS, 1_000_000.0)[1] pair_spec = compare_context.get_pair_heatmap_specs(base_cfg)[0] - slice_variant = [variant for variant in pair_spec["fixed_slices"] if variant["slug"] == "q2"][0] - - sample_x_indices = sorted({0, len(pair_spec["x_values"]) // 2, len(pair_spec["x_values"]) - 1}) - sample_y_indices = sorted({0, len(pair_spec["y_values"]) // 2, len(pair_spec["y_values"]) - 1}) - - cache_lookup = {} - compare_context.set_noise_profile("legacy_calibrated") - slice_cfg = dict(base_cfg) - slice_cfg[pair_spec["fixed_key"]] = float(slice_variant["value"]) - for y_index in sample_y_indices: - for x_index in sample_x_indices: - cfg = dict(slice_cfg) - cfg[pair_spec["x_key"]] = float(pair_spec["x_values"][x_index]) - cfg[pair_spec["y_key"]] = float(pair_spec["y_values"][y_index]) - for source_name in ("noise_geometric", "arb_geometric"): - source_cfg, method = module._source_variant(compare_context, cfg, source_name) - cache_key = compare_context._make_method_cache_key(source_cfg, method) - cache_key_hash = compare_context._make_method_cache_hash(cache_key) - cache_lookup[cache_key_hash] = 1_000_000.0 compare_context.set_noise_profile("market_linear") module.autodetect_lightweight_noise_profile( @@ -260,10 +241,18 @@ def test_autodetect_lightweight_noise_profile_switches_to_legacy_calibrated(): pair_specs=[pair_spec], metric_key="noise_vs_arb_geometric_improvement_pct", slice_slug="q2", - cache_lookup=cache_lookup, + cache_lookup={}, ) - assert compare_context.noise_profile == "legacy_calibrated" + assert compare_context.noise_profile == "market_linear" + + +def test_lightweight_context_rejects_legacy_noise_profile(): + module = load_script_module() + compare_context = module._LightweightCompareContext() + + with pytest.raises(ValueError): + compare_context.set_noise_profile("legacy_calibrated") def test_source_variant_sets_explicit_market_and_arb_only_noise_models(): @@ -285,14 +274,18 @@ def test_source_variant_sets_explicit_market_and_arb_only_noise_models(): assert noise_method == "geometric" assert noise_cfg["enable_noise_model"] is True assert noise_cfg["noise_model"] == "market_linear" + assert noise_cfg["noise_arrays_path"] == compare_context.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH assert arb_method == "geometric" assert arb_cfg["enable_noise_model"] is False assert arb_cfg["noise_model"] == "arb_only" + assert arb_cfg["noise_arrays_path"] == compare_context.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH resolved_arb = compare_context.resolve_reclamm_noise_settings(arb_cfg) assert resolved_arb["noise_model"] == "arb_only" - assert resolved_arb["noise_cache_key"] == ("disabled",) + assert resolved_arb["noise_arrays_path"] == compare_context.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH + assert set(resolved_arb["reclamm_noise_params"]) == {"tvl_mean", "tvl_std"} + assert resolved_arb["noise_cache_key"][0] == "arb_only" def test_lightweight_context_defaults_to_fixed_compare_arb_cadence(): @@ -308,6 +301,7 @@ def test_lightweight_context_defaults_to_fixed_compare_arb_cadence(): assert arb_cfg["arb_frequency"] == compare_context.FIXED_COMPARE_ARB_FREQUENCY assert arb_cfg["noise_model"] == "arb_only" + assert arb_cfg["noise_arrays_path"] == compare_context.DEFAULT_MARKET_LINEAR_NOISE_ARRAYS_PATH assert resolved_noise["arb_frequency"] == compare_context.FIXED_COMPARE_ARB_FREQUENCY assert arb_cfg["gas_cost"] == compare_context.DEFAULT_GAS_COST assert arb_cfg["protocol_fee_split"] == compare_context.DEFAULT_PROTOCOL_FEE_SPLIT From 4c9eae7e61c18543c57d98da4133cce5becb238a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 31 Mar 2026 11:11:58 +0100 Subject: [PATCH 076/115] feat: feature-appropriate scaling, TVL clamp, protocol fee default MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Feature scaling reform in build_data(): TVL and BTC log_price kept in raw log scale (absolute level matters), returns/trends/volume_zscore unscaled, volatilities lightly centered. Eliminates global z-score that squeezed TVL into [-2,+2] and prevented models from learning TVL response. Add ±3σ clamp on standardized log(TVL) in reclamm_market_linear_noise_volume to prevent extreme concentration from wireheading the noise model. Change default protocol_fee_split from 0.0 to 0.25 to match reClAMM production configuration. --- experiments/run_linear_market_noise.py | 43 +++++++++++++++--- quantammsim/pools/noise_trades.py | 5 +- .../runners/default_run_fingerprint.py | 2 +- results/linear_market_noise/model.npz | Bin 5092 -> 5092 bytes 4 files changed, 42 insertions(+), 8 deletions(-) diff --git a/experiments/run_linear_market_noise.py b/experiments/run_linear_market_noise.py index c71dc6ac..56524ce6 100644 --- a/experiments/run_linear_market_noise.py +++ b/experiments/run_linear_market_noise.py @@ -144,12 +144,43 @@ def build_data(matched_clean, option_c_clean, trend_windows=(7, 14, 30), x_base = np.concatenate([x_obs, x_market], axis=1).astype(np.float32) base_names = [f"xobs_{i}" for i in range(k_obs)] + market_names - # Standardize (except intercept column 0) - x_mean = np.mean(x_base, axis=0) - x_std = np.std(x_base, axis=0) - x_std[x_std < 1e-6] = 1.0 - x_mean[0] = 0.0 # don't center intercept - x_std[0] = 1.0 + # Feature-appropriate scaling: + # - intercept: untouched + # - log_tvl, btc_log_price: raw log scale (absolute level carries info) + # - dow_sin/cos: already [-1,1], no scaling + # - returns, trends, vol_zscore: already comparable, no scaling + # - realized_vol, pair_vol: small positive, light centering + # - cross-pool volumes: z-score (different scales across pool groups) + x_mean = np.zeros(x_base.shape[1], dtype=np.float32) + x_std = np.ones(x_base.shape[1], dtype=np.float32) + + for i, name in enumerate(base_names): + if name == "xobs_0": + # Intercept: leave as-is + pass + elif name in ("xobs_1", "btc_log_price"): + # Log levels: leave in raw log scale (range ~10-20) + pass + elif name in ("xobs_2", "xobs_3"): + # dow_sin, dow_cos: already [-1,1] + pass + elif "volume_zscore" in name: + # Already z-scored by construction + pass + elif "log_return" in name or "trend_" in name: + # Returns and trends: small, centered around 0, comparable + pass + elif "realized_vol" in name or "pair_realized" in name: + # Volatilities: small positive, center but don't squeeze + x_mean[i] = float(np.mean(x_base[:, i])) + # Use std but don't over-compress — floor at 0.01 + x_std[i] = max(float(np.std(x_base[:, i])), 0.01) + elif name.startswith("xobs_") and int(name.split("_")[1]) >= 4: + # Cross-pool volumes (xobs_4,5,6): z-score (different scales) + x_mean[i] = float(np.mean(x_base[:, i])) + x_std[i] = max(float(np.std(x_base[:, i])), 1e-6) + # else: leave untouched + x_base = ((x_base - x_mean) / x_std).astype(np.float32) # Interaction terms (products of standardized features) diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index bfef86c5..db8d1b9c 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -395,7 +395,10 @@ def reclamm_market_linear_noise_volume( Per-minute noise volume (USD), floored at zero. """ log_tvl = jnp.log(jnp.maximum(effective_value_usd, 1.0)) - standardized_log_tvl = (log_tvl - tvl_mean) / tvl_std + # Clamp standardized TVL to training range [-3, +3] std to prevent + # extreme concentration from wireheading the noise model + standardized_log_tvl = jnp.clip( + (log_tvl - tvl_mean) / tvl_std, -3.0, 3.0) log_daily_noise = noise_base + noise_tvl_coeff * standardized_log_tvl daily_noise = jnp.exp(log_daily_noise) return jnp.maximum(0.0, daily_noise / 1440.0) diff --git a/quantammsim/runners/default_run_fingerprint.py b/quantammsim/runners/default_run_fingerprint.py index 45b90940..478dfa5e 100644 --- a/quantammsim/runners/default_run_fingerprint.py +++ b/quantammsim/runners/default_run_fingerprint.py @@ -84,7 +84,7 @@ "return_val": "daily_log_sharpe", "initial_pool_value": 1000000.0, "fees": 0.0, - "protocol_fee_split": 0.0, # fraction of swap fees diverted from LP reserves to protocol treasury + "protocol_fee_split": 0.25, # fraction of swap fees diverted from LP reserves to protocol treasury "arb_fees": 0.0, "gas_cost": 0.0, "use_alt_lamb": False, diff --git a/results/linear_market_noise/model.npz b/results/linear_market_noise/model.npz index c792225202ae753ace18972cf62cf5b677717a4e..f440f04f8bfd94472d98fffc989196f8afb4b3e7 100644 GIT binary patch delta 3610 zcmZ9Pc~p(t|HtJdMVgc}Av9`^a?bPY{XD1NRVbB2L@L7-N~69aPMRcAoJytAKq+p8 zrgP5T&vQ=HCCN-zX3;kyiPF{Zz3%W`>-*bleb#&J*Js%4v;TW{nRJPb*VY7Q6p#{R+g!H2&uMxC17$a0p|av4N;*xNL=sD%v5lclCNWNX<0T@5`|EA z#*~3bk}vJt77RD7{mA^>$ALY|l!?cdM~R$OQ$bu{k%aGfiZ#e+I3^(R_o+Em}OO_=hpXM)*f)Pn=cXw#f{ zDx_Knl%}|{VTIRm{7fqx-q<7!NJQrbos@ndCZFx9Z=$<#Rz;? zFrjt(xuuP{#6p4TUD1Nz^{Bq$Jo)AA0rE}?K^9C%B7&^iV3tjaxVP934BE^V|Gecr z$Tj^$@?ABM-B0u3zZ}}ZaT6W>{mIqL)s05H&Z4!XxidUeyIRjEA`;{%gt;yiUzsgYeBT> z8rgUI^;Gq-Q%LDSIj~J!LS@xYk{Z0@NqOIL`2!uPAfn|oTPs=3e%n$)&k`p{i{m3f zO{Ooct8swT&GVPBO^e|PcXwFm_LM1OETG?-F5uQu$3NS%j0)dljQUq6BZuCZ0_oUR zGln~lpd;@Vv++Cg$;C--;9RgH90&`MRtWF25A$?EjNVnsbU`>0R?0~S6CQ!Q4mqa4 ze*}DQw1HMZ4-k|sr%e^SV3S=NTIMt&sqb(DPQ2elu8Y=jmzJxGE0k5>rK28Xtky)H zE0ct(Cr=e`luU%RTLhQIUVAkWXMeodD}5Lw+eeXJv&wPdT}61mr4LLBy^iz1Z}384 zIxg1^gZfAg*EV0~7BvK*rzd#4DKZ_9pR^GjdKORYaLy)+9Mi>FH}!Cw=PcsndLg;# zFchZ{rm$+K7oObN2NK5!rumpTGaisATPJsxG7;P;C3mZ(qkGqyA;aM=q#Q-Wyxx~y09l?IO zd}LZ}C`+^7MI~N&NalO504+Y6)YoZ6Y{z~(sB!b0wAt+?SWD!B>?3|uB|ioVvTHs_ zZ9kh5PF^P#SMABc<(_97N*3=SGwk}2als52YB7M#JqvKzro*yb4K~!hrb6)KX&=w+xZ8(ujmJ%^W9&A8pjy9z=FEC-D4Qn#h!c>0-+^FBp1ijQi^I z6|mO46~97xU{#AU;sgx~kV@?_C##V%I_U60AiTJkkq9ki%iU=HC#$!#M`ePf#*zXD z>@p>O`qq3?4|Bd+OcFn$=nTAhR9`kb;~cbB^kqfI7~HR+&F%}^$rh|Qi>AmQVyCX1 z*!1k~uax`6Q_yT|5Z?%}r`_pw;9v-VDJj>a+1owgI}?Gv)GTp%qr%6nO#h504ar*& zx*ZpcrtMx$nO3z+kDbGKyLKF^OI$!@Z_kCICKYN4$ABL>c`tLNCW+_YIS4qDH!&O2 z3JE1?l!Q1IKy(}J!ko2QydTV`gD3unRg~3|cYBUw!!R{szepRFR9T9n1-Wj(Tg;N1 z8^4pc-7iR6>bTV4wNTO{eGyKM5JAJ67gz_MFVt2WZ{{ZpMdIH!;nJRycfd}E9QJH! z5X*JThoU%w3e%IvJBsqCJucHhmmk6%wl(10r4`TxDKTpE_eq%oKT6Lo8s`WLB^eP3 z7`f=NjbklX&@mRmh8=~t@!eWDoGzsDx{H|p$T%GTA)3rLuVpn$+QI&2Iq6oX@35Vt!w>8&lXFSDnFi_`$-G?<_<}w^%kLla}#9=|8Ah1p4WiTfjVqBQ}B@4T#!L< z^Gd}_pPCW1oQF7Bs*B2WD+vvG6_VL*K-k`~h1?4a@hsdPssu)Jk1le+{b%$@JM|pG z&dM15UgL~QXH!nXg>5Vv{SHi3Wcby!*X)KTt0ewyro=bfGonW=LQwQ8k9hAJgX}Z{ zBsFD+#inByfS#c1jbzKs96a83P^29_1bX5E0mttbG&(~EmQr7U4y51CVfgNmu-VXGZr^6?39pOy`?@QDYdI;u`Z!9Qd=$=`XvJz=c?<{#XGr??pckXT*mI{6C~j7OpOYlm z!qQh3wmF%a(cMmR-4>Cn`&6jnZhLHc)r4|g2Y6IIxJO5-NlQ_YDA*bjCh2k^7 z`*K%V`ceh5CnO21p52N|SLkJr~&bs>&)UwR|sM3FQRU% z_dF+j3c3225W^Ayp<4A?k`p5)X4O`L+Oi7n@L)I>!1Rk8%fR4 zC(x%Vlm7lT09p#P57XKA7E&vUg_KElvvlm>GPvI?78od%;m1ox8bl|Y$sJ?;XV|fudOT5f5z^8b&y;S+L7lpW&|l!6!c6i_7VDcB5`isOMLAYiNf}r{-a(3lx_3Dm zPdzIRUk~tXcRQjx!~{1v-rxoO?g9fWX>kA8Al`TH7)T9@BF?ROD*d`Onj%^y#4I0k zaM(u^esZakRz?IPM?GD7jr3RYvOb^sc6=`F+r>p+lq0Dt>_Z?Ju)|dJ*i~TMq$=|c z>Ooa?CnfgR7m>BvH2%hkC#tSRNvn+9s7cWuQL>tYq%%x~nIqC^Y*Fweo>wJsYXj=h z?Z6OnbASs^_W7h}`r@hF+e23P?rW+|C z`GzW=I0QN$?V&b4*-F2&P5`m(%Cw$t0v#|^Nc~Z(MhmtNP$C;OS|hNR8Z~vl9HH{F zoP6qoBeR50RO^Hm>QjU^>I(lqCIu|#=TbsL_32tgdQ%IjzirkJp{L{*{>W(k=$SO9 z!?pOoxBr=9udXhiyg&b$aH)E&@Ti8C@UX^odHQZfsxVGNQ5i%$NbhJRG6X?nWKP1NN4ug4SCY4GI#{M`YK*uVCOoWAg{m(EY;{s-3Ln)Ltx delta 3744 zcmY+Hc{G;W`^I0AvCOBE21I0NI5c^m`+lBBk};Z7A~}^N%}UezicTaXQbM7@P=?YV zp1of~C7P6qN*$FF4Ma{!@s0EQ=ePg)tn0q7z1Fq%-fQi3H<&hv?%8-k35Zj*pn2@b-(xL8c?m_QPym<#(e7Ql|R$@tYeom_0JBllhm;yY>u+VE~NlrM>A z9J(T5?~@(k>?ecCQJsCntH==B&nQaRo~PHz&w~E4GJY=3nE6Qz2@XFmn6(4hIbi~( zXREUp9#5t^Sr3@te+)@PO+3Suy%8)IjfGWZVMMh=9;;NUg;)C}(Xy`2I`}*2FNXP{ z&VTr>$mQo6LdWfk!nyIOvIoyh!KTO_SA4g?O)4X(EO0gJw=MRX>LnK(T3`e=pY|Yj zlrh63524Al97Uf;vyD2XK$_n&t^tR*y#ec}(AB`@E#Hgj=fxm>C8FZLQmOtXIcgG} z3RRMr*(BY^lRQ&bM6)rOxZIkfXALi7_Ir@Krbymaqvm#+%8y74M@(vbD_Q zogpsW-eh|?X*-D4D#*~<7bOR>RM_Ane(Z=@LyJ94HnF&El}tBPmEU}K1SkLHn6R5q z0=T6CtD9a5Ip5!+`u>w(|Ehrr9;nLZHU|lL$L(brA^yZ=<`Zx(%Ydmv##5{2sbtNr zV4P#S2Gi|F%50vx3flAgQKx}gd_SY_Y|20#VPNTTL33=pL>AKUk$E>I0oRXrgGGOp zg5CEmtd~zB^L>UZo%*7Iv8@|~#)26<%<+OW>0YvFYz!_qnohS>_h&aAG^hO{_Rxo) zm3Ujn>0nZjP2RiL6}6eG(p>%kvCd6_+{PZ^o(>?SJ;JT0UGZ|M9`!znc>ReYJ5DOe z6XvgeMLym;Em*DCg>5=(!SQ5nk1#YRP9Znla#StLt;M!P=l;?g&0^ zS8zbAhA`8*P4Kv`OW(LWhZ=(*;uI7Xo8W)!M4y}@T=5gmpR#!j{0EX z@DC+g+?Y?|nl_PwYe`Iuo*FN#s$}O~6k+h9Kgfj47CgOf8L{)Lf`7FalC3WCfonB=UN1YP>Ub_MLHZQcduSfqamvGRKSZdum zo5;C_(wdwfr1j9OA~0V@6Gmh)_S4qVm@awVeCi?YfJY=oyQom3&3D)(j&C5u!~kCI z8A`A2u*U(x67AyWKf6iTk1H~@w`nDZulz|ydXFOhu2L>WC6La4_?QeMBeChK9;bI8 z9Y?s|nEH2M8W)y07w%9uq3+{j7~C(L?pTK;`g<${%4FhhDTFlb z0=FtL$;)y;qU$6~D*Z<|J6{_WN_LVXlJ|?q+pufo%aM=F6XSvWM$Zpy!7vH4_PIKl z5EYE}t^LT%xB}d;u35+&fJ}3!x~$mg3Ve92K;MTsK-%+YY<0R{7*exUSa^FV9FOf} z7A&XH|h`h*k z|BAqqsT@=FNuI1^2#x5sAGV4dQLJDE4}7;1vF}{ycGVQ*W+V`GlPB<@b1+0diXbi# zkBI-FY6-C)7fSM*bm;)yRIm~ml2jiRs{8XEr~YIS4$_KeD#j^6TFfh=**Qu`%>&p| zkxVkS7ekzK1yNYOp-A_ZIfgzn5?=4!%uKocuqf$aC`qabhlk8jtTNn7To(A#TCYyX zUl0RV+wXH}GAVZ}{vB*Hp+rX_(UIxIZ3Ej)hSb@6EV}P|O>#v~iM5?MoQaEI8_$@#2CW0)$6mSVy{Q%LbprJh&-_2mL8sko3!A(Qc1 z)jr{cT9h!EWYMShszF}o794t%OO_j{@Ur`sT+feArXo5+yl2b}l<{>%vWelaW4JPL z`u7c}%Jf;*tx`baWfFX_>@0D~D1{WmJaML_JUZ3fB`<9r!Ltk65Sdy)Dsli^!^5ab zi3V@xWW}l9dM}iCAGXc5(3iN?&%Tj-f4=^NZKfTo=nNE$pd${-4<+vmxZ^7RAHZ#}P2{(^f zTr|7A8WaS5qL&^8g%R2aV~xqQg;!Dfv_FmNw1kF_K47n$MC}$2K|ha6=<9M5;u;%- z%(o8El$17OyiMz`fIj`lOt(W6B&R@Y68SvD( zRYaNaeQ5pti7;JzBKdy$3S1d&2kSGUQE7WQ3_jaM#1;e&HkUB0dJZXl6-q2pFG6|D zHZ0C*r4pSn(abI@5gS}Qoo2OZ^Ova_d**F_X7!|Lf@4AsD1IGBj%i$DYVv$Z%WH4A z-n|b!I?Um|15(NP-LS8ghs?H0M&D{0@gAxJLD$bf)w?hvZQ6lilUcOK+L2zV*Wv@d zj^upA+dyHfJ!A1L2eaoUGR47LCAKbdlNrC$8c3IhvIP=bvSf`61Gu5Ir{o44-g=Gc zv2GP&w+^7T#ippFx`b%aG-@gAr|&0K(d#Sw@h3I=bDv!kgj7+k*rBSGyxBW~NM5UB zqPK`D>8OHA5Mi4Pc5lgohFt&Li|&=k=Z-@CpV#M4z4shVe2?EgxM{r zZKC>ZJIEcoyYPS;2j{~SG3QbyY`!x9O6Clu@!Fa+qB#?SZSvuzceilkMkq!-yobsC z)tS%Bo#0kUt@wEKA2fif^Oc<@oSmtYFnq-@$kP>LS4lqc>}Z3eRmX{FZWJ?N$32Xx zRzmZ%RGJax3>URlp~NLlr?6~(02*C*Ko(@GkQ#YC+To>(@7s@1dHG@V? zRHbPYK3^~3gjzMstG)(3Qii#}FQNZ8ji~j_BZeZsH8+!DZ3gB_t~F|LB0dDybbP}} z-#PB+E}oNAE3%7?-MD`gLb&Kg2QK#Jc+M^1GW&7+T&^W(DHo<>%+1g5#`^W+I8BF% z+`+F4Ir(;1?oDMf_U_rdVhX36<%ETH&FlcDR`l&S!rs+i&M9mf#ZajRH+y&pm*o4B z{q5D{UWe&$$MiHgadsQ)*yEnRNqvQ!!ud}&;@CQk%Mt^{#TMM}%*9^wwzI!!FleAh zFH^qNnDo}A{nuLe)t&i-UKY3HnD;T>2xPs!iE1Z$wWvwr*sJR`lKQm4y+Eg4T|94P zTp#a!bLi_=x}>tb{?kWi{ML#hX@yUsEH*~dAa;_Ly28*xeQB(t)+LESoTno9+a{fo z@~+n}_56<_C4<+!dMO4Z^|Jo)p##0-Iu!QxGUeUw);>P=i|eH{KjKs`m!^;D%P-v> zbg-{~uD;L5sfM2J)tt-3qkW96iTq7PQOe14t!(TUEdyN-7dd&|e*b^cjJ>RFtI&7H T?b^HkJI&2HHvh#xr!)0`Y!akZ From b44c228e34cb0730a52000cd7745edd9b9d5788a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 31 Mar 2026 11:12:11 +0100 Subject: [PATCH 077/115] feat: CMA-ES optimiser + Optuna min_train_returns_over_hodl rejection Add CMA-ES as third optimisation method in tune_reclamm_calibrated_noise.py alongside Optuna and BFGS, with population_size, sigma0, n_generations controls. Add min_train_returns_over_hodl rejection in Optuna objective: trials with catastrophic in-sample returns_over_hodl are rejected early (return -inf) to avoid wasting evaluation budget. --- experiments/tune_reclamm_calibrated_noise.py | 30 ++++++++++++++++++-- quantammsim/runners/jax_runners.py | 11 +++++++ 2 files changed, 39 insertions(+), 2 deletions(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index 9e999bd6..c2d712c4 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -115,7 +115,7 @@ def _build_market_linear_arrays(args): def _build_opt_settings(args): - """Build optimisation_settings for either optuna or bfgs.""" + """Build optimisation_settings for optuna, bfgs, or cma_es.""" if args.method == "bfgs": return { "method": "bfgs", @@ -128,6 +128,20 @@ def _build_opt_settings(args): "compute_dtype": "float64", }, } + elif args.method == "cma_es": + return { + "method": "cma_es", + "n_parameter_sets": args.n_parameter_sets, + **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + "cma_es_settings": { + "population_size": args.cma_pop_size, + "n_generations": args.cma_generations, + "sigma0": args.cma_sigma0, + "tol": 1e-8, + "n_evaluation_points": args.cma_eval_points, + "compute_dtype": "float32", + }, + } else: return { "method": "optuna", @@ -141,6 +155,8 @@ def _build_opt_settings(args): "parameter_config": PARAMETER_CONFIG, **({"overfitting_penalty": args.overfitting_penalty} if args.overfitting_penalty is not None else {}), + **({"min_train_returns_over_hodl": args.min_train_ret} + if args.min_train_ret is not None else {}), }, } @@ -222,7 +238,8 @@ def main(): parser = argparse.ArgumentParser( description="Tune reClAMM params with calibrated 8-covariate noise model" ) - parser.add_argument("--method", default="optuna", choices=["optuna", "bfgs"], + parser.add_argument("--method", default="optuna", + choices=["optuna", "bfgs", "cma_es"], help="Optimisation method") parser.add_argument("--n-trials", type=int, default=50, help="Optuna trials (ignored for bfgs)") @@ -232,6 +249,15 @@ def main(): parser.add_argument("--bfgs-tol", type=float, default=1e-6) parser.add_argument("--bfgs-eval-points", type=int, default=20, help="Number of evaluation points for bfgs") + # CMA-ES + parser.add_argument("--cma-generations", type=int, default=300) + parser.add_argument("--cma-sigma0", type=float, default=0.5, + help="Initial step size for CMA-ES") + parser.add_argument("--cma-pop-size", type=int, default=None, + help="Population size (None = auto)") + parser.add_argument("--cma-eval-points", type=int, default=20) + parser.add_argument("--min-train-ret", type=float, default=-0.5, + help="Reject trials with IS returns_over_hodl below this") parser.add_argument("--noise-model", default="market_linear", choices=["calibrated", "market_linear"], help="Noise model variant") diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index 09378845..27669a29 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -1456,6 +1456,17 @@ def objective(trial): initial_reserves=train_outputs["reserves"][0], ) + # Reject catastrophic in-sample configurations + min_train_ret_over_hodl = run_fingerprint["optimisation_settings"][ + "optuna_settings"].get("min_train_returns_over_hodl", None) + if min_train_ret_over_hodl is not None: + if float(train_returns_over_hodl) < min_train_ret_over_hodl: + optuna_manager.logger.info( + f"Training {trial.number}, REJECTED:" + f" ret_over_hodl={train_returns_over_hodl:.4f}" + f" < {min_train_ret_over_hodl}") + return float("-inf") + # Test period evaluation using continuous forward pass # This ensures test metrics reflect continuous simulation from training continuous_outputs = partial_forward_pass_continuous_optuna( From b89fa64ec058e2f6c4e6cb8c6dc9978d9f506beb Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 31 Mar 2026 11:15:33 +0100 Subject: [PATCH 078/115] feat: Michaelis-Menten noise model + MLP sweep + comparison tooling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Michaelis-Menten noise model (run_mm_noise.py): structural TVL saturation via V_noise = alpha_i * TVL/(K_i + TVL) * exp(x_market @ gamma_i), with learned EWMA smoothing on TVL (discovery lag). Per-pool alpha, K, gamma; shared lambda. Achieves R²=0.64 matching per-pool linear while adding saturation (K_med ~$19M). MLP noise model (run_mlp_noise.py): Optuna sweep with per-trial model saving, TVL response check at sweep end. Investigation showed shared MLP cannot learn TVL relationship due to cross-pool confounding. Model comparison (run_model_comparison.py): linear vs MLP noise model across TVL levels with time series and summary plots. MM fit plotting (plot_mm_noise_fit.py): 6-panel per-pool time series, cross-pool TVL response/elasticity curves, K distribution analysis. --- experiments/run_mlp_noise.py | 117 ++++- experiments/run_mm_noise.py | 635 ++++++++++++++++++++++++++++ experiments/run_model_comparison.py | 338 +++++++++++++++ scripts/plot_mm_noise_fit.py | 472 +++++++++++++++++++++ 4 files changed, 1555 insertions(+), 7 deletions(-) create mode 100644 experiments/run_mm_noise.py create mode 100644 experiments/run_model_comparison.py create mode 100644 scripts/plot_mm_noise_fit.py diff --git a/experiments/run_mlp_noise.py b/experiments/run_mlp_noise.py index bde323e1..35d54181 100644 --- a/experiments/run_mlp_noise.py +++ b/experiments/run_mlp_noise.py @@ -257,20 +257,21 @@ def run_optuna(data, n_trials): def objective(trial): # Architecture - n_layers = trial.suggest_int("n_layers", 1, 5) - first_hidden = trial.suggest_categorical("first_hidden", [8, 16, 32, 64]) + n_layers = trial.suggest_int("n_layers", 1, 7) + first_hidden = trial.suggest_categorical("first_hidden", [8, 16, 32, 64, 128, 256]) # Bottleneck: each layer is half the previous (min 2) + bottleneck_ratio = trial.suggest_categorical("bottleneck_ratio", [0.5, 0.75, 1.0]) hidden = [] h = first_hidden for _ in range(n_layers): hidden.append(h) - h = max(h // 2, 2) + h = max(int(h * bottleneck_ratio), 2) # Training - lr = trial.suggest_float("lr", 1e-4, 1e-2, log=True) - l2_alpha = trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True) + lr = trial.suggest_float("lr", 1e-4, 5e-2, log=True) + l2_alpha = trial.suggest_float("l2_alpha", 1e-5, 5e-1, log=True) huber_delta = trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5, 2.0]) - n_epochs = trial.suggest_categorical("n_epochs", [2000, 5000, 10000]) + n_epochs = trial.suggest_categorical("n_epochs", [2000, 5000, 10000, 20000]) use_cosine = trial.suggest_categorical("use_cosine", [True, False]) per_pool = trial.suggest_categorical("per_pool", [True, False]) @@ -341,6 +342,33 @@ def objective(trial): f" {'per_pool' if per_pool else 'shared'}" f" lr={lr:.1e} l2={l2_alpha:.1e}" f" hub={huber_delta} ep={n_epochs}") + + # Save every trial's model + trial_dir = os.path.join("results", "mlp_noise", "trials", f"trial_{trial.number:04d}") + os.makedirs(trial_dir, exist_ok=True) + save_dict = {k: np.array(v) for k, v in params.items()} + save_dict["x_mean"] = data.get("x_mean", np.zeros(n_feat)) + save_dict["x_std"] = data.get("x_std", np.ones(n_feat)) + np.savez(os.path.join(trial_dir, "model.npz"), **save_dict) + import json as _json + _meta = { + "pool_ids": data["pool_ids"], + "feat_names": data["feat_names"], + "n_feat": n_feat, + "hidden": hidden, + "per_pool": per_pool, + "eval_r2": med_r2, + "hparams": { + "hidden": hidden, "lr": lr, "l2_alpha": l2_alpha, + "huber_delta": huber_delta, "n_epochs": n_epochs, + "use_cosine": use_cosine, "per_pool": per_pool, + "bottleneck_ratio": bottleneck_ratio, + "first_hidden": first_hidden, "n_layers": n_layers, + }, + } + with open(os.path.join(trial_dir, "meta.json"), "w") as _f: + _json.dump(_meta, _f, indent=2) + return med_r2 study = optuna.create_study(direction="maximize") @@ -365,13 +393,55 @@ def objective(trial): arch = [] for _ in range(n_l): arch.append(h) - h = max(h // 2, 2) + h = max(int(h * t.params.get("bottleneck_ratio", 0.5)), 2) print(f" #{t.number}: eval={t.value:.4f}" f" arch={arch}" f" ep={t.params['n_epochs']}" f" {'cos' if t.params['use_cosine'] else 'cst'}" f" {'pp' if t.params['per_pool'] else 'sh'}") + # Copy best trial to top-level artifact + best_trial = study.best_trial + best_trial_dir = os.path.join("results", "mlp_noise", "trials", + f"trial_{best_trial.number:04d}") + save_dir = "results/mlp_noise" + if os.path.exists(os.path.join(best_trial_dir, "model.npz")): + import shutil + shutil.copy2(os.path.join(best_trial_dir, "model.npz"), + os.path.join(save_dir, "model.npz")) + shutil.copy2(os.path.join(best_trial_dir, "meta.json"), + os.path.join(save_dir, "meta.json")) + print(f"\n Copied best trial ({best_trial.number}) to: {save_dir}") + + # TVL response check on best model + import json as _json + art = dict(np.load(os.path.join(save_dir, "model.npz"), allow_pickle=True)) + with open(os.path.join(save_dir, "meta.json")) as _f: + meta = _json.load(_f) + best_params = {k: jnp.array(art[k]) for k in art + if k.startswith("W") or k.startswith("b") + or k == "log_cadence" or k == "pool_bias"} + per_pool = meta.get("per_pool", False) + pool_idx_probe = jnp.array([0]) if per_pool else None + + print(f"\n TVL Response Check (best model, trial {best_trial.number}):") + x_probe = np.zeros((1, n_feat), dtype=np.float32) + x_probe[0, 0] = 1.0 + x_probe[0, 4] = 10.5 # typical btc_log_price + prev_noise = None + for tvl in [1e4, 1e5, 5e5, 1e6, 5e6, 1e7, 5e7, 1e8, 5e8]: + x_probe[0, 1] = np.log(tvl) + out = np.array(forward_mlp(best_params, jnp.array(x_probe), + pool_idx_probe)) + noise = np.exp(out[0]) + if prev_noise is not None and prev_noise > 0: + ratio = noise / prev_noise + print(f" TVL=${tvl:>11,.0f} noise=${noise:>12,.0f}/day" + f" ({ratio:.2f}x prev)") + else: + print(f" TVL=${tvl:>11,.0f} noise=${noise:>12,.0f}/day") + prev_noise = noise + return study @@ -395,6 +465,8 @@ def main(): help="Append static pool attributes to input") parser.add_argument("--no-split", action="store_true", help="Train on all data") + parser.add_argument("--save-artifact", default=None, + help="Save model to this directory") args = parser.parse_args() os.environ.setdefault("JAX_PLATFORMS", "cpu") @@ -515,6 +587,37 @@ def main(): print(f" Linear (with cross-pool): median R² ≈ 0.53") print(f" Per-pool linear: median R² ≈ 0.61") + # Save artifact + if args.save_artifact: + import json + os.makedirs(args.save_artifact, exist_ok=True) + # Save params as npz + save_dict = {k: np.array(v) for k, v in params.items()} + save_dict["x_mean"] = data["x_mean"] if "x_mean" in data else np.zeros(n_feat) + save_dict["x_std"] = data["x_std"] if "x_std" in data else np.ones(n_feat) + np.savez(os.path.join(args.save_artifact, "model.npz"), **save_dict) + # Save meta + meta = { + "pool_ids": data["pool_ids"], + "feat_names": data["feat_names"], + "n_feat": n_feat, + "hidden": args.hidden, + "per_pool": args.per_pool, + "hparams": { + "hidden": args.hidden, + "lr": args.lr, + "l2_alpha": args.l2_alpha, + "huber_delta": args.huber_delta, + "epochs": args.epochs, + "trend_windows": args.trend_windows, + "use_cosine": args.cosine, + "per_pool": args.per_pool, + }, + } + with open(os.path.join(args.save_artifact, "meta.json"), "w") as f: + json.dump(meta, f, indent=2) + print(f"\n Saved artifact to: {args.save_artifact}") + if __name__ == "__main__": main() diff --git a/experiments/run_mm_noise.py b/experiments/run_mm_noise.py new file mode 100644 index 00000000..abdc2b18 --- /dev/null +++ b/experiments/run_mm_noise.py @@ -0,0 +1,635 @@ +"""Michaelis-Menten noise model with market features. + +Replaces the linear TVL term with a Michaelis-Menten saturation curve +while keeping all market features for temporal fit: + + log(V_noise) = log_alpha_i + x_market @ gamma + + log(TVL) - log(K_i + TVL) + + V_total = V_arb(cadence_i) + exp(log_V_noise) + Loss = Huber(log(V_total) - log(V_obs)) + +The TVL feature (xobs_1) is removed from x_market and handled +structurally via the MM saturation term. All other features (dow, +BTC, token, pair vol, interactions) remain as shared linear covariates. + +Parameters: + log_alpha_i : per-pool intercept + log_K_i : per-pool half-saturation TVL + gamma : shared coefficients on non-TVL features + log_cadence_i: per-pool arb frequency (via PCHIP) + +Usage: + python experiments/run_mm_noise.py + python experiments/run_mm_noise.py --epochs 5000 --lr 3e-4 + python experiments/run_mm_noise.py --per-pool-gamma # per-pool market coeffs +""" + +import argparse +import json +import os +import pickle +import sys +import time + +import jax +import jax.numpy as jnp +import numpy as np +import pandas as pd + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_stage1(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + with open(path, "rb") as f: + data = pickle.load(f) + return data["matched_clean"], data["option_c_clean"] + + +def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), + include_cross_pool=False): + """Build data with MM structure: separate TVL from market features.""" + from experiments.run_linear_market_noise import build_data + + # Get full feature matrix from linear model's pipeline + data = build_data( + matched_clean, option_c_clean, + trend_windows=trend_windows, + include_market=True, + include_cross_pool=include_cross_pool, + ) + + # Separate TVL from other features + feat_names = data["feat_names"] + x_full = data["x"] + + # Find TVL column (xobs_1) and TVL interaction columns + tvl_col = feat_names.index("xobs_1") + tvl_interaction_cols = [i for i, name in enumerate(feat_names) + if name.startswith("xobs_1\u00d7")] + + # Remove TVL and its interactions from market features + remove_cols = {tvl_col} | set(tvl_interaction_cols) + keep_cols = [i for i in range(len(feat_names)) if i not in remove_cols] + x_market = x_full[:, keep_cols].astype(np.float32) + market_names = [feat_names[i] for i in keep_cols] + + # TVL comes from the raw panel data (unstandardized log_tvl) + # x_full[:, tvl_col] might be standardized, so get raw from panel + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + + # Rebuild raw log_tvl from panel + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + date_to_idx = {d: i for i, d in enumerate(date_list)} + n_dates = len(date_list) + + tvl_grid = np.full((n_dates, n_pools), np.nan) + for j, pid in enumerate(pool_ids): + panel = matched_clean[pid]["panel"] + dates = panel["date"].values + log_tvls = panel["log_tvl_lag1"].values.astype(float) + for k, date in enumerate(dates): + tvl_grid[date_to_idx[date], j] = log_tvls[k] + + pool_idx = data["pool_idx"] + day_idx = data["day_idx"] + log_tvl = np.array([tvl_grid[day_idx[s], pool_idx[s]] + for s in range(len(pool_idx))], dtype=np.float32) + + # Token info for display + from quantammsim.calibration.pool_data import _parse_tokens + pool_tokens = [] + for pid in pool_ids: + toks = _parse_tokens(matched_clean[pid]["tokens"]) + tok_a = toks[0] + tok_b = toks[1] if len(toks) > 1 else toks[0] + pool_tokens.append((tok_a, tok_b)) + + removed_names = [feat_names[i] for i in sorted(remove_cols)] + print(f" Removed TVL features: {removed_names}") + print(f" Market features ({len(market_names)}): {market_names}") + + # Build per-pool temporal ordering for EWMA + # For each pool, store the sample indices sorted by day_idx + # Pad to max length so we can use lax.scan uniformly + pool_time_indices = [] # (n_pools, max_T) — sample indices in time order + pool_time_lengths = [] # (n_pools,) — actual length per pool + for i in range(n_pools): + mask = pool_idx == i + idxs = np.where(mask)[0] + # Sort by day_idx + order = np.argsort(day_idx[idxs]) + pool_time_indices.append(idxs[order]) + pool_time_lengths.append(len(idxs)) + + max_T = max(pool_time_lengths) if pool_time_lengths else 0 + # Pad to uniform length (pad with 0, masked later) + pool_time_padded = np.zeros((n_pools, max_T), dtype=np.int32) + pool_time_mask = np.zeros((n_pools, max_T), dtype=np.float32) + for i in range(n_pools): + L = pool_time_lengths[i] + pool_time_padded[i, :L] = pool_time_indices[i] + pool_time_mask[i, :L] = 1.0 + + print(f" EWMA: max_T={max_T}, pools with data: " + f"{sum(1 for l in pool_time_lengths if l > 0)}") + + return { + "x_market": x_market, + "log_tvl": log_tvl, + "y_total": data["y_total"], + "pool_idx": pool_idx, + "day_idx": day_idx, + "sample_grid_days": data["sample_grid_days"], + "pool_coeffs": data["pool_coeffs"], + "pool_gas": data["pool_gas"], + "init_log_cadences": data["init_log_cadences"], + "n_pools": n_pools, + "n_market_feat": x_market.shape[1], + "pool_ids": pool_ids, + "pool_tokens": pool_tokens, + "market_names": market_names, + "x_mean": data["x_mean"], + "x_std": data["x_std"], + "pool_time_padded": pool_time_padded, + "pool_time_mask": pool_time_mask, + } + + +# ---- Model ---- + +def ewma_smooth(log_tvl, raw_lambda, pool_time_padded, pool_time_mask): + """Apply learned EWMA smoothing to log_tvl, per pool. + + smooth_t = λ * log_tvl_t + (1-λ) * smooth_{t-1} + + Returns smoothed log_tvl in the same sample order as input. + """ + lam = jax.nn.sigmoid(raw_lambda) # constrain to (0, 1) + n_pools = pool_time_padded.shape[0] + smoothed = jnp.array(log_tvl) # copy + + for i in range(n_pools): + idxs = pool_time_padded[i] # (max_T,) sample indices + mask = pool_time_mask[i] # (max_T,) 1.0 or 0.0 + raw_vals = log_tvl[idxs] # (max_T,) raw log_tvl in time order + + # lax.scan for EWMA + def step(carry, x): + prev_smooth, = carry + raw_val, m = x + new_smooth = jnp.where( + m > 0, + lam * raw_val + (1.0 - lam) * prev_smooth, + prev_smooth) + return (new_smooth,), new_smooth + + init = (raw_vals[0],) + _, smooth_vals = jax.lax.scan(step, init, (raw_vals, mask)) + + # Scatter smoothed values back to sample positions + smoothed = smoothed.at[idxs].set( + jnp.where(mask > 0, smooth_vals, smoothed[idxs])) + + return smoothed + + +def forward_mm(params, x_market, log_tvl_smooth, pool_idx): + """MM forward pass → log(V_noise) per sample. + + log(V_noise) = log_alpha_i + x_market @ gamma[_i] + + log(TVL_smooth) - log(K_i + TVL_smooth) + """ + log_alpha = params["log_alpha"] + log_K = params["log_K"] + gamma = params["gamma"] + + # Per-sample pool params + alpha_i = log_alpha[pool_idx] + K_i = jnp.exp(log_K[pool_idx]) + tvl = jnp.exp(log_tvl_smooth) + + # Market features: shared or per-pool gamma + if gamma.ndim == 2: + per_sample_gamma = gamma[pool_idx] + market_term = jnp.sum(x_market * per_sample_gamma, axis=1) + else: + market_term = x_market @ gamma + + # MM saturation on smoothed TVL + log_saturation = log_tvl_smooth - jnp.log(K_i + tvl) + + return alpha_i + market_term + log_saturation + + +def make_loss_fn(pool_coeffs, pool_gas, n_pools): + """Loss with PCHIP arb + MM noise.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + def loss_fn(params, x_market, log_tvl, y_total, + sample_grid_days, pool_idx, pool_time_padded, + pool_time_mask, l2_alpha, huber_delta): + log_cadence = params["log_cadence"] + + # EWMA smooth TVL + log_tvl_smooth = ewma_smooth( + log_tvl, params["raw_lambda"], + pool_time_padded, pool_time_mask) + + # V_arb from PCHIP + n_samples = x_market.shape[0] + log_v_arb = jnp.zeros(n_samples) + for i in range(n_pools): + v_arb_all = interpolate_pool_daily( + pool_coeffs[i], jnp.float64(log_cadence[i]), pool_gas[i]) + safe_days = jnp.clip(sample_grid_days, 0, v_arb_all.shape[0] - 1) + log_v_arb = jnp.where( + pool_idx == i, + jnp.log(jnp.maximum(v_arb_all[safe_days], 1e-10)), + log_v_arb) + + # V_noise from MM with smoothed TVL + log_v_noise = forward_mm(params, x_market, log_tvl_smooth, pool_idx) + + # V_total + log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) + + # Huber + residual = log_v_total - y_total + abs_r = jnp.abs(residual) + huber = jnp.where( + abs_r <= huber_delta, + 0.5 * residual ** 2, + huber_delta * (abs_r - 0.5 * huber_delta)) + + # Per-pool equal weighting + pool_counts = jnp.zeros(n_pools).at[pool_idx].add( + jnp.ones_like(pool_idx, dtype=jnp.float32)) + active = (pool_counts > 0).astype(jnp.float32) + n_active = jnp.maximum(jnp.sum(active), 1.0) + pool_counts = jnp.maximum(pool_counts, 1.0) + pool_sums = jnp.zeros(n_pools).at[pool_idx].add(huber) + mean_loss = jnp.sum((pool_sums / pool_counts) * active) / n_active + + # L2 on gamma and log_alpha + reg = l2_alpha * ( + jnp.mean(params["gamma"] ** 2) + + jnp.mean(params["log_alpha"] ** 2) + ) + + return mean_loss + reg + + return jax.jit(jax.value_and_grad(loss_fn)) + + +# ---- Training ---- + +def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, + verbose=True): + """Adam training loop.""" + m = {k: jnp.zeros_like(v) for k, v in params.items()} + v = {k: jnp.zeros_like(v) for k, v in params.items()} + b1, b2, eps = 0.9, 0.999, 1e-8 + + x_market = jnp.array(data["x_market"]) + log_tvl = jnp.array(data["log_tvl"]) + y_total = jnp.array(data["y_total"]) + sgd = jnp.array(data["sample_grid_days"]) + pidx = jnp.array(data["pool_idx"]) + pt_padded = jnp.array(data["pool_time_padded"]) + pt_mask = jnp.array(data["pool_time_mask"]) + + for epoch in range(n_epochs): + loss, grads = grad_fn( + params, x_market, log_tvl, y_total, sgd, pidx, + pt_padded, pt_mask, l2_alpha, huber_delta) + + for k in params: + g = grads[k] + m[k] = b1 * m[k] + (1 - b1) * g + v[k] = b2 * v[k] + (1 - b2) * g ** 2 + m_hat = m[k] / (1 - b1 ** (epoch + 1)) + v_hat = v[k] / (1 - b2 ** (epoch + 1)) + params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + eps) + + if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): + log_K_med = float(jnp.median(params["log_K"])) + K_med = float(jnp.exp(log_K_med)) + cad = np.exp(np.array(params["log_cadence"])) + gamma_norm = float(jnp.sqrt(jnp.mean(params["gamma"] ** 2))) + lam = float(jax.nn.sigmoid(params["raw_lambda"])) + print(f" epoch {epoch:5d} loss={float(loss):.4f}" + f" K_med=${K_med:,.0f}" + f" λ={lam:.3f}" + f" |γ|={gamma_norm:.3f}" + f" cad=[{cad.min():.0f},{np.median(cad):.0f},{cad.max():.0f}]") + + return params + + +# ---- Evaluation ---- + +def evaluate(params, data): + """Per-pool R² and diagnostics.""" + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + n_pools = data["n_pools"] + pool_idx = np.array(data["pool_idx"]) + sgd = np.array(data["sample_grid_days"]) + y = np.array(data["y_total"]) + log_cadence = np.array(params["log_cadence"]) + + # V_arb + v_arb = np.zeros(len(y)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + safe = np.clip(sgd[mask], 0, len(v_arb_all) - 1) + v_arb[mask] = v_arb_all[safe] + log_v_arb = np.log(np.maximum(v_arb, 1e-10)) + + # Smooth TVL with learned lambda + log_tvl_smooth = np.array(ewma_smooth( + jnp.array(data["log_tvl"]), params["raw_lambda"], + jnp.array(data["pool_time_padded"]), + jnp.array(data["pool_time_mask"]))) + + log_v_noise = np.array(forward_mm( + params, jnp.array(data["x_market"]), + jnp.array(log_tvl_smooth), + jnp.array(data["pool_idx"]))) + + log_v_total = np.logaddexp(log_v_arb, log_v_noise) + v_noise = np.exp(log_v_noise) + v_total = np.exp(log_v_total) + + r2s = {} + noise_shares = {} + for i in range(n_pools): + mask = pool_idx == i + if mask.sum() < 2: + continue + yt = y[mask] + pt = log_v_total[mask] + ss_res = np.sum((yt - pt) ** 2) + ss_tot = np.sum((yt - yt.mean()) ** 2) + r2s[data["pool_ids"][i]] = 1 - ss_res / max(ss_tot, 1e-10) + noise_shares[data["pool_ids"][i]] = float(np.median( + v_noise[mask] / v_total[mask])) + + K_values = {data["pool_ids"][i]: float(np.exp(params["log_K"][i])) + for i in range(n_pools)} + + return { + "r2s": r2s, + "noise_shares": noise_shares, + "K_values": K_values, + "median_r2": float(np.median(list(r2s.values()))), + } + + +def tvl_response_check(params, data): + """Print predicted noise at various TVL levels.""" + n_pools = data["n_pools"] + pool_idx = np.array(data["pool_idx"]) + + # Median market features per pool + print(f"\n TVL Response Check (per-pool median market features):") + print(f" {'Pool':>20s} {'K ($M)':>10s} {'TVL=100K':>10s}" + f" {'TVL=1M':>10s} {'TVL=10M':>10s} {'TVL=100M':>10s}" + f" {'TVL=1B':>10s} {'ε@1M':>6s} {'ε@100M':>6s}") + + tvl_test = [1e5, 1e6, 1e7, 1e8, 1e9] + + for i in range(min(n_pools, 15)): + pid = data["pool_ids"][i] + toks = data["pool_tokens"][i] + label = f"{toks[0]}/{toks[1]}" + mask = pool_idx == i + if mask.sum() == 0: + continue + + K_i = float(np.exp(params["log_K"][i])) + x_med = np.median(data["x_market"][mask], axis=0) + + gamma = np.array(params["gamma"]) + if gamma.ndim == 2: + market_term = float(x_med @ gamma[i]) + else: + market_term = float(x_med @ gamma) + log_alpha_i = float(params["log_alpha"][i]) + + vols = [] + for tvl in tvl_test: + log_sat = np.log(tvl) - np.log(K_i + tvl) + log_v = log_alpha_i + market_term + log_sat + vols.append(np.exp(log_v)) + + # Elasticity at 1M and 100M + eps_1m = K_i / (K_i + 1e6) + eps_100m = K_i / (K_i + 1e8) + + print(f" {label:>20s} ${K_i/1e6:>9.1f}" + + "".join(f" ${v:>9,.0f}" for v in vols) + + f" {eps_1m:>6.3f} {eps_100m:>6.3f}") + + +# ---- Main ---- + +def main(): + parser = argparse.ArgumentParser( + description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--epochs", type=int, default=3000) + parser.add_argument("--lr", type=float, default=3e-4) + parser.add_argument("--l2-alpha", type=float, default=1e-3) + parser.add_argument("--huber-delta", type=float, default=1.0) + parser.add_argument("--init-log-K", type=float, default=17.0, + help="Initial log(K) ~ log($24M)") + parser.add_argument("--per-pool-gamma", action="store_true", + help="Per-pool market feature coefficients") + parser.add_argument("--no-split", action="store_true") + parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) + parser.add_argument("--include-cross-pool", action="store_true") + parser.add_argument("--save-artifact", default="results/mm_noise") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + + print("=" * 70) + print("Michaelis-Menten Noise Model + Market Features") + print(f" epochs={args.epochs}, lr={args.lr}, l2={args.l2_alpha}") + print(f" init log(K)={args.init_log_K} (K=${np.exp(args.init_log_K):,.0f})") + print(f" per_pool_gamma={args.per_pool_gamma}") + print("=" * 70) + + matched_clean, option_c_clean = load_stage1() + + print("\nBuilding data...") + t0 = time.time() + data = build_mm_data(matched_clean, option_c_clean, + trend_windows=tuple(args.trend_windows), + include_cross_pool=args.include_cross_pool) + n_pools = data["n_pools"] + n_market = data["n_market_feat"] + n_samples = len(data["pool_idx"]) + print(f" {n_samples} samples, {n_pools} pools," + f" {n_market} market features, {time.time() - t0:.1f}s") + + # Pool summary + pool_idx = data["pool_idx"] + for i, (pid, toks) in enumerate( + zip(data["pool_ids"], data["pool_tokens"])): + mask = pool_idx == i + n = mask.sum() + if n > 0: + med_tvl = np.exp(np.median(data["log_tvl"][mask])) + print(f" {pid[:16]} {toks[0]:>8s}/{toks[1]:<8s}" + f" {n:>4d} days TVL=${med_tvl:>12,.0f}") + + # Split + if args.no_split: + train_data = data + eval_data = None + else: + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = {k: v[train_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + print(f"\n Split: {train_mask.sum()} train, {eval_mask.sum()} eval") + + # Init + if args.per_pool_gamma: + gamma_init = jnp.zeros((n_pools, n_market)) + else: + gamma_init = jnp.zeros(n_market) + + params = { + "log_alpha": jnp.zeros(n_pools), + "log_K": jnp.full(n_pools, args.init_log_K), + "gamma": gamma_init, + "log_cadence": jnp.array(data["init_log_cadences"]), + "raw_lambda": jnp.array(2.0), # sigmoid(2) ≈ 0.88 — mostly raw + } + n_params = sum(v.size for v in params.values()) + print(f"\n Parameters: {n_params}" + f" (α: {n_pools}, K: {n_pools}," + f" γ: {gamma_init.size}, cadence: {n_pools})") + + # Warm-start gamma via Ridge (numpy, no sklearn) + print(" Warm-starting γ via Ridge on residuals...") + + def _ridge(X, y, alpha=1.0): + """Ridge regression: (X'X + αI)^-1 X'y.""" + XtX = X.T @ X + alpha * np.eye(X.shape[1]) + Xty = X.T @ y + return np.linalg.solve(XtX, Xty) + + x_trn = data["x_market"] if args.no_split else train_data["x_market"] + y_trn = data["y_total"] if args.no_split else train_data["y_total"] + if args.per_pool_gamma: + pidx = data["pool_idx"] if args.no_split else train_data["pool_idx"] + for i in range(n_pools): + mask = pidx == i + if mask.sum() < 5: + continue + # Add intercept column for warm-start + X_i = np.concatenate([x_trn[mask], np.ones((mask.sum(), 1))], 1) + w = _ridge(X_i, y_trn[mask]) + params["gamma"] = params["gamma"].at[i].set( + jnp.array(w[:-1].astype(np.float32))) + params["log_alpha"] = params["log_alpha"].at[i].set(float(w[-1])) + else: + X_all = np.concatenate([x_trn, np.ones((len(y_trn), 1))], 1) + w = _ridge(X_all, y_trn) + params["gamma"] = jnp.array(w[:-1].astype(np.float32)) + + # Loss + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + + print(f"\nTraining ({args.epochs} epochs)...") + t0 = time.time() + params = train(params, train_data, grad_fn, args.epochs, args.lr, + args.l2_alpha, args.huber_delta) + print(f" Training time: {time.time() - t0:.1f}s") + + # Evaluate + print("\n" + "=" * 70) + print("Results (train)") + print("=" * 70) + train_eval = evaluate(params, train_data) + print(f" Median R²: {train_eval['median_r2']:.4f}") + + print(f"\n {'Pool':>16s} {'Tokens':>16s} {'R²':>6s}" + f" {'Noise%':>7s} {'K ($M)':>10s}") + for pid in data["pool_ids"]: + i = data["pool_ids"].index(pid) + toks = data["pool_tokens"][i] + r2 = train_eval["r2s"].get(pid, float("nan")) + ns = train_eval["noise_shares"].get(pid, float("nan")) + K = train_eval["K_values"][pid] + print(f" {pid[:16]} {toks[0]:>8s}/{toks[1]:<6s}" + f" {r2:>6.3f} {ns*100:>6.1f}% ${K/1e6:>9.1f}") + + if eval_data is not None: + print("\n" + "=" * 70) + print("Results (eval)") + print("=" * 70) + eval_result = evaluate(params, eval_data) + print(f" Median R²: {eval_result['median_r2']:.4f}") + + # TVL response + tvl_response_check(params, data) + + # Gamma coefficients + gamma = np.array(params["gamma"]) + if gamma.ndim == 1: + print(f"\n Shared γ coefficients:") + for j, name in enumerate(data["market_names"]): + print(f" {name:>30s}: {gamma[j]:>8.4f}") + + # Save + if args.save_artifact: + os.makedirs(args.save_artifact, exist_ok=True) + save_dict = {k: np.array(v) for k, v in params.items()} + np.savez(os.path.join(args.save_artifact, "model.npz"), **save_dict) + meta = { + "model": "michaelis_menten", + "pool_ids": data["pool_ids"], + "pool_tokens": data["pool_tokens"], + "market_names": data["market_names"], + "n_pools": n_pools, + "n_market_feat": n_market, + "per_pool_gamma": args.per_pool_gamma, + "hparams": { + "epochs": args.epochs, "lr": args.lr, + "l2_alpha": args.l2_alpha, "huber_delta": args.huber_delta, + "init_log_K": args.init_log_K, + }, + } + with open(os.path.join(args.save_artifact, "meta.json"), "w") as f: + json.dump(meta, f, indent=2) + print(f"\n Saved: {args.save_artifact}/") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_model_comparison.py b/experiments/run_model_comparison.py new file mode 100644 index 00000000..398ec48e --- /dev/null +++ b/experiments/run_model_comparison.py @@ -0,0 +1,338 @@ +"""Compare linear vs MLP noise models across TVL levels. + +Evaluates both noise models for a given pool over the same date range, +sweeping initial TVL. Uses real price data, the PCHIP arb grid, and +both noise models to predict daily volume decomposition. + +Produces a plot: predicted daily noise volume vs TVL for each model, +with the real observed volume overlaid where available. + +Usage: + python experiments/run_model_comparison.py + python experiments/run_model_comparison.py --tvl-range 1e5 1e6 5e6 7e6 20e6 50e6 +""" + +import argparse +import json +import os +import pickle +import time + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +import jax.numpy as jnp + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) +LINEAR_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "linear_market_noise", +) +MLP_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "mlp_noise", +) + + +def load_linear_model(artifact_dir, pool_id): + """Load linear noise model for a pool.""" + art = np.load(os.path.join(artifact_dir, "model.npz")) + with open(os.path.join(artifact_dir, "meta.json")) as f: + meta = json.load(f) + pool_ids = meta["pool_ids"] + idx = next((i for i, p in enumerate(pool_ids) + if p.startswith(pool_id) or pool_id.startswith(p)), -1) + nc = art["noise_coeffs"] + coeffs = nc[idx] if nc.ndim == 2 and idx >= 0 else (nc if nc.ndim == 1 else np.median(nc, axis=0)) + return { + "coeffs": coeffs, + "log_cadence": art["log_cadence"][idx] if idx >= 0 else np.log(10.0), + "x_mean": art["x_mean"], + "x_std": art["x_std"], + "feat_names": meta["feat_names"], + "type": "linear", + } + + +def load_mlp_model(artifact_dir, pool_id): + """Load MLP noise model.""" + art = dict(np.load(os.path.join(artifact_dir, "model.npz"), allow_pickle=True)) + with open(os.path.join(artifact_dir, "meta.json")) as f: + meta = json.load(f) + pool_ids = meta["pool_ids"] + idx = next((i for i, p in enumerate(pool_ids) + if p.startswith(pool_id) or pool_id.startswith(p)), -1) + + # Extract MLP params + params = {} + for k in art: + if k.startswith("W") or k.startswith("b") or k == "log_cadence" or k == "pool_bias": + params[k] = art[k] + + return { + "params": params, + "log_cadence": art["log_cadence"][idx] if idx >= 0 else np.log(10.0), + "pool_idx": idx, + "x_mean": art["x_mean"], + "x_std": art["x_std"], + "feat_names": meta["feat_names"], + "hidden": meta["hidden"], + "per_pool": meta.get("per_pool", False), + "type": "mlp", + } + + +def predict_noise_linear(model, x_daily, tvl_values): + """Predict noise volume at multiple TVL levels using linear model.""" + tvl_col = 1 # xobs_1 + results = {} + for tvl in tvl_values: + x = x_daily.copy() + x[:, tvl_col] = (np.log(tvl) - model["x_mean"][tvl_col]) / model["x_std"][tvl_col] + # Update TVL interaction terms + for i, name in enumerate(model["feat_names"]): + if name.startswith("xobs_1\u00d7"): + paired = name.split("\u00d7")[1] + if paired in model["feat_names"]: + j = model["feat_names"].index(paired) + x[:, i] = x[:, tvl_col] * x_daily[:, j] + log_noise = x @ model["coeffs"] + results[tvl] = np.exp(log_noise) + return results + + +def predict_noise_mlp(model, x_daily, tvl_values): + """Predict noise volume at multiple TVL levels using MLP model.""" + from experiments.run_mlp_noise import forward_mlp + tvl_col = 1 + params = model["params"] + pool_idx_arr = (jnp.full(x_daily.shape[0], model["pool_idx"]) + if model["per_pool"] and model["pool_idx"] >= 0 else None) + results = {} + for tvl in tvl_values: + x = x_daily.copy() + x[:, tvl_col] = (np.log(tvl) - model["x_mean"][tvl_col]) / model["x_std"][tvl_col] + for i, name in enumerate(model["feat_names"]): + if name.startswith("xobs_1\u00d7"): + paired = name.split("\u00d7")[1] + if paired in model["feat_names"]: + j = model["feat_names"].index(paired) + x[:, i] = x[:, tvl_col] * x_daily[:, j] + log_noise = np.array(forward_mlp(params, jnp.array(x), pool_idx_arr)) + results[tvl] = np.exp(log_noise) + return results + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--pool", default="0x9d1fcf346ea1b0") + parser.add_argument("--tvl-range", type=float, nargs="+", + default=[100_000, 500_000, 1_000_000, 5_000_000, + 7_000_000, 20_000_000, 50_000_000]) + parser.add_argument("--linear-dir", default=LINEAR_DIR) + parser.add_argument("--mlp-dir", default=MLP_DIR) + parser.add_argument("--output-dir", default="results/model_comparison") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + os.makedirs(args.output_dir, exist_ok=True) + + # Load data + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + data = pickle.load(f) + mc = data["matched_clean"] + oc = data["option_c_clean"] + + pid = args.pool + entry = mc[pid] + panel = entry["panel"] + dates = pd.to_datetime(panel["date"]) + vol_obs = np.exp(panel["log_volume"].values.astype(float)) + tvl_obs = np.exp(panel["log_tvl_lag1"].values.astype(float)) + + print(f"Pool: {pid} ({entry['tokens']}, {entry['chain']})") + print(f"{len(dates)} days: {dates.min().date()} → {dates.max().date()}") + + # Build feature matrix (same for both models) + from experiments.run_linear_market_noise import build_data + data_full = build_data(mc, oc, trend_windows=(7,), + include_market=True, include_cross_pool=False) + pool_ids = data_full["pool_ids"] + pool_i = pool_ids.index(pid) + pool_mask = data_full["pool_idx"] == pool_i + x_pool = data_full["x"][pool_mask] + day_idx = data_full["day_idx"][pool_mask] + sgd = data_full["sample_grid_days"][pool_mask] + + all_dates = set() + for p in pool_ids: + all_dates.update(mc[p]["panel"]["date"].values) + date_list = sorted(all_dates) + sample_dates = np.array([pd.Timestamp(date_list[d]) for d in day_idx]) + + n_days = len(sample_dates) + print(f"Feature samples: {n_days}") + + # Load models + print("\nLoading models...") + linear_model = load_linear_model(args.linear_dir, pid) + print(f" Linear: {len(linear_model['coeffs'])} coefficients," + f" cadence={np.exp(linear_model['log_cadence']):.1f}min") + + has_mlp = os.path.exists(os.path.join(args.mlp_dir, "model.npz")) + if has_mlp: + mlp_model = load_mlp_model(args.mlp_dir, pid) + print(f" MLP: hidden={mlp_model['hidden']}," + f" cadence={np.exp(mlp_model['log_cadence']):.1f}min") + else: + print(f" MLP: no artifact at {args.mlp_dir}") + mlp_model = None + + # V_arb from PCHIP (same for both — uses linear model's cadence) + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + cadence = float(np.exp(linear_model["log_cadence"])) + gas = float(np.exp(oc[pid]["log_gas"])) + v_arb_all = np.array(interpolate_pool_daily( + entry["coeffs"], jnp.float64(np.log(cadence)), jnp.float64(gas))) + v_arb = v_arb_all[sgd] + + # Predict at each TVL + print(f"\nPredicting noise at {len(args.tvl_range)} TVL levels...") + linear_noise = predict_noise_linear(linear_model, x_pool, args.tvl_range) + mlp_noise = predict_noise_mlp(mlp_model, x_pool, args.tvl_range) if mlp_model else {} + + # Real observed volume for comparison + tvl_for_samples = np.zeros(n_days) + vol_for_samples = np.zeros(n_days) + for i, sd in enumerate(sample_dates): + matches = np.where(dates == sd)[0] + if len(matches) > 0: + tvl_for_samples[i] = tvl_obs[matches[0]] + vol_for_samples[i] = vol_obs[matches[0]] + + # ---- Plot 1: Median noise volume vs TVL ---- + fig, axes = plt.subplots(1, 2, figsize=(14, 6)) + + tvls = np.array(args.tvl_range) + lin_medians = np.array([np.median(linear_noise[t]) for t in tvls]) + lin_q25 = np.array([np.percentile(linear_noise[t], 25) for t in tvls]) + lin_q75 = np.array([np.percentile(linear_noise[t], 75) for t in tvls]) + + ax = axes[0] + ax.fill_between(tvls / 1e6, lin_q25 / 1e6, lin_q75 / 1e6, + alpha=0.2, color="steelblue") + ax.plot(tvls / 1e6, lin_medians / 1e6, "o-", color="steelblue", + linewidth=2, label="Linear noise (median)") + + if mlp_noise: + mlp_medians = np.array([np.median(mlp_noise[t]) for t in tvls]) + mlp_q25 = np.array([np.percentile(mlp_noise[t], 25) for t in tvls]) + mlp_q75 = np.array([np.percentile(mlp_noise[t], 75) for t in tvls]) + ax.fill_between(tvls / 1e6, mlp_q25 / 1e6, mlp_q75 / 1e6, + alpha=0.2, color="coral") + ax.plot(tvls / 1e6, mlp_medians / 1e6, "s-", color="coral", + linewidth=2, label="MLP noise (median)") + + # Add real observed volume at real TVL + valid = tvl_for_samples > 100 + ax.scatter(tvl_for_samples[valid] / 1e6, vol_for_samples[valid] / 1e6, + c="black", s=3, alpha=0.2, label="Observed total vol", zorder=1) + + ax.set_xlabel("Effective TVL ($M)") + ax.set_ylabel("Daily volume ($M)") + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_title(f"{entry['tokens']} — Noise Volume vs TVL") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + # ---- Plot 2: Noise/TVL ratio vs TVL ---- + ax = axes[1] + ax.plot(tvls / 1e6, lin_medians / tvls * 100, "o-", color="steelblue", + linewidth=2, label="Linear noise/TVL") + if mlp_noise: + ax.plot(tvls / 1e6, mlp_medians / tvls * 100, "s-", color="coral", + linewidth=2, label="MLP noise/TVL") + + # Real vol/TVL + ax.scatter(tvl_for_samples[valid] / 1e6, + vol_for_samples[valid] / tvl_for_samples[valid] * 100, + c="black", s=3, alpha=0.2, label="Observed vol/TVL") + + ax.set_xlabel("Effective TVL ($M)") + ax.set_ylabel("Noise / TVL (%)") + ax.set_xscale("log") + ax.set_title("Noise as Fraction of TVL") + ax.legend(fontsize=8) + ax.grid(True, alpha=0.3) + + fig.suptitle(f"Linear vs MLP Noise Model — {entry['tokens']}", fontsize=12) + fig.tight_layout() + out = os.path.join(args.output_dir, f"{pid[:16]}_model_comparison.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"\nSaved: {out}") + + # ---- Plot 3: Time series at selected TVLs ---- + fig, axes = plt.subplots(len(args.tvl_range), 1, + figsize=(14, 3 * len(args.tvl_range)), + sharex=True) + if len(args.tvl_range) == 1: + axes = [axes] + + for k, tvl in enumerate(args.tvl_range): + ax = axes[k] + v_total_lin = v_arb + linear_noise[tvl] + ax.plot(sample_dates, v_total_lin / 1e6, "b-", linewidth=0.6, + alpha=0.7, label="Linear (arb+noise)") + if mlp_noise: + v_total_mlp = v_arb + mlp_noise[tvl] + ax.plot(sample_dates, v_total_mlp / 1e6, "r-", linewidth=0.6, + alpha=0.7, label="MLP (arb+noise)") + ax.plot(sample_dates, vol_for_samples / 1e6, "k-", linewidth=0.5, + alpha=0.3, label="Observed (at real TVL)") + ax.set_ylabel(f"$M/day\nTVL=${tvl/1e6:.1f}M") + ax.set_yscale("log") + if k == 0: + ax.legend(fontsize=7, loc="upper right") + ax.grid(True, alpha=0.3) + + axes[-1].set_xlabel("Date") + fig.suptitle(f"Volume Time Series at Different TVLs — {entry['tokens']}", fontsize=11) + fig.tight_layout() + out2 = os.path.join(args.output_dir, f"{pid[:16]}_tvl_sweep_timeseries.png") + fig.savefig(out2, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"Saved: {out2}") + + # Summary table + print(f"\n{'='*70}") + print(f"Summary: Median daily noise volume by TVL") + print(f"{'='*70}") + print(f"{'TVL':>14s} {'Linear':>12s} {'Lin/TVL':>8s}", end="") + if mlp_noise: + print(f" {'MLP':>12s} {'MLP/TVL':>8s} {'MLP/Lin':>8s}") + else: + print() + + for tvl in args.tvl_range: + lin = np.median(linear_noise[tvl]) + print(f"${tvl:>13,.0f} ${lin:>11,.0f} {lin/tvl*100:>7.1f}%", end="") + if mlp_noise: + mlp = np.median(mlp_noise[tvl]) + ratio = mlp / lin if lin > 0 else 0 + print(f" ${mlp:>11,.0f} {mlp/tvl*100:>7.1f}% {ratio:>7.2f}x") + else: + print() + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_mm_noise_fit.py b/scripts/plot_mm_noise_fit.py new file mode 100644 index 00000000..a5184ad1 --- /dev/null +++ b/scripts/plot_mm_noise_fit.py @@ -0,0 +1,472 @@ +"""Plot MM noise model fit: per-pool time series + TVL response curves. + +Produces two types of plots: +1. Per-pool 6-panel time series (like plot_model_vs_real_reclamm.py): + TVL, volume decomposition, V_noise, fee revenue, vol/TVL, pred/obs +2. Cross-pool TVL response curves showing MM saturation + +Usage: + python scripts/plot_mm_noise_fit.py + python scripts/plot_mm_noise_fit.py --pool 0x9d1fcf346ea1b0 + python scripts/plot_mm_noise_fit.py --all-pools +""" + +import argparse +import json +import os +import pickle +import sys + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +import jax.numpy as jnp + + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def load_model(artifact_dir): + """Load MM model artifact.""" + art = dict(np.load(os.path.join(artifact_dir, "model.npz"), + allow_pickle=True)) + with open(os.path.join(artifact_dir, "meta.json")) as f: + meta = json.load(f) + params = {} + for k in art: + params[k] = jnp.array(art[k]) + return params, meta + + +def compute_decomposition(params, meta, matched_clean, option_c_clean): + """Compute V_arb, V_noise, V_total for all pools.""" + from experiments.run_mm_noise import build_mm_data, forward_mm + from quantammsim.calibration.grid_interpolation import interpolate_pool_daily + + data = build_mm_data(matched_clean, option_c_clean, + trend_windows=(7,), + include_cross_pool=False) + + pool_ids = data["pool_ids"] + n_pools = data["n_pools"] + pool_idx = np.array(data["pool_idx"]) + sgd = np.array(data["sample_grid_days"]) + day_idx = np.array(data["day_idx"]) + y = np.array(data["y_total"]) + log_tvl = np.array(data["log_tvl"]) + + log_cadence = np.array(params["log_cadence"]) + + # V_arb + v_arb = np.zeros(len(y)) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + continue + v_arb_all = np.array(interpolate_pool_daily( + data["pool_coeffs"][i], jnp.float64(log_cadence[i]), + data["pool_gas"][i])) + safe = np.clip(sgd[mask], 0, len(v_arb_all) - 1) + v_arb[mask] = v_arb_all[safe] + + # V_noise + log_v_noise = np.array(forward_mm( + params, jnp.array(data["x_market"]), + jnp.array(data["log_tvl"]), + jnp.array(data["pool_idx"]))) + v_noise = np.exp(log_v_noise) + v_total = v_arb + v_noise + v_obs = np.exp(y) + + # Reconstruct dates + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + + dates = np.array([pd.Timestamp(date_list[d]) for d in day_idx]) + + return { + "pool_ids": pool_ids, + "pool_tokens": data["pool_tokens"], + "pool_idx": pool_idx, + "dates": dates, + "v_arb": v_arb, + "v_noise": v_noise, + "v_total": v_total, + "v_obs": v_obs, + "log_tvl": log_tvl, + "tvl": np.exp(log_tvl), + } + + +def plot_pool_timeseries(decomp, params, pool_i, output_dir): + """6-panel time series for a single pool.""" + pid = decomp["pool_ids"][pool_i] + toks = decomp["pool_tokens"][pool_i] + label = f"{toks[0]}/{toks[1]}" + + mask = decomp["pool_idx"] == pool_i + if mask.sum() < 10: + print(f" Skipping {pid[:16]} ({label}): {mask.sum()} samples") + return + + dates = decomp["dates"][mask] + v_arb = decomp["v_arb"][mask] + v_noise = decomp["v_noise"][mask] + v_total = decomp["v_total"][mask] + v_obs = decomp["v_obs"][mask] + tvl = decomp["tvl"][mask] + + K_i = float(np.exp(params["log_K"][pool_i])) + + # R² + log_pred = np.log(np.maximum(v_total, 1e-10)) + log_obs = np.log(np.maximum(v_obs, 1e-10)) + ss_res = np.sum((log_pred - log_obs) ** 2) + ss_tot = np.sum((log_obs - log_obs.mean()) ** 2) + r2 = 1 - ss_res / max(ss_tot, 1e-10) + + fig, axes = plt.subplots(6, 1, figsize=(14, 18), sharex=True) + + # 1. TVL + ax = axes[0] + ax.plot(dates, tvl / 1e6, "k-", linewidth=0.7) + ax.axhline(K_i / 1e6, color="red", linestyle="--", alpha=0.5, + label=f"K = ${K_i/1e6:.1f}M") + ax.set_ylabel("TVL ($M)") + ax.set_yscale("log") + ax.legend(fontsize=8) + ax.set_title(f"{label} — TVL (K={K_i/1e6:.1f}M)") + ax.grid(True, alpha=0.3) + + # 2. Volume decomposition + ax = axes[1] + ax.fill_between(dates, 0, v_arb / 1e6, alpha=0.4, color="steelblue", + label="V_arb") + ax.fill_between(dates, v_arb / 1e6, v_total / 1e6, alpha=0.4, + color="coral", label="V_noise (MM)") + ax.plot(dates, v_obs / 1e6, "k-", linewidth=0.5, alpha=0.7, + label="V_obs") + ax.plot(dates, v_total / 1e6, "r--", linewidth=0.5, alpha=0.7, + label="V_pred") + ax.set_ylabel("Volume ($M/day)") + ax.set_yscale("log") + ax.legend(fontsize=7) + ax.set_title(f"Volume Decomposition (R²={r2:.3f})") + ax.grid(True, alpha=0.3) + + # 3. V_noise only + ax = axes[2] + ax.fill_between(dates, 0, v_noise / 1e6, alpha=0.4, color="coral") + ax.plot(dates, v_noise / 1e6, "r-", linewidth=0.5) + noise_med = np.median(v_noise) + ax.axhline(noise_med / 1e6, color="red", linestyle=":", alpha=0.5, + label=f"median=${noise_med:,.0f}") + ax.set_ylabel("V_noise ($M/day)") + ax.set_yscale("log") + ax.legend(fontsize=8) + ax.set_title("Noise Volume (MM model)") + ax.grid(True, alpha=0.3) + + # 4. Fee revenue (assuming 0.25% fee, 25% protocol take) + fee_rate = 0.0025 + protocol_take = 0.25 + fee_arb = v_arb * fee_rate * (1 - protocol_take) + fee_noise = v_noise * fee_rate * (1 - protocol_take) + fee_obs = v_obs * fee_rate * (1 - protocol_take) + ax = axes[3] + ax.fill_between(dates, 0, fee_arb, alpha=0.4, color="steelblue", + label="Arb fees") + ax.fill_between(dates, fee_arb, fee_arb + fee_noise, alpha=0.4, + color="coral", label="Noise fees") + ax.plot(dates, fee_obs, "k-", linewidth=0.5, alpha=0.7, + label="Obs fees (approx)") + ax.set_ylabel("Fee revenue ($/day)") + ax.legend(fontsize=7) + ax.set_title("Fee Revenue (0.25% fee, 75% LP)") + ax.grid(True, alpha=0.3) + + # 5. Vol/TVL + ax = axes[4] + vol_tvl_obs = v_obs / tvl + vol_tvl_pred = v_total / tvl + ax.plot(dates, vol_tvl_obs * 100, "k-", linewidth=0.5, alpha=0.5, + label="Observed") + ax.plot(dates, vol_tvl_pred * 100, "r-", linewidth=0.5, alpha=0.5, + label="Predicted") + ax.axhline(np.median(vol_tvl_obs) * 100, color="black", linestyle=":", + alpha=0.3) + ax.axhline(np.median(vol_tvl_pred) * 100, color="red", linestyle=":", + alpha=0.3) + ax.set_ylabel("Vol/TVL (%)") + ax.legend(fontsize=8) + ax.set_title("Volume as % of TVL") + ax.grid(True, alpha=0.3) + + # 6. Pred/Obs ratio + ax = axes[5] + ratio = v_total / np.maximum(v_obs, 1) + ax.plot(dates, ratio, "b-", linewidth=0.5, alpha=0.5) + ax.axhline(1.0, color="black", linestyle="-", alpha=0.3) + med_ratio = np.median(ratio) + ax.axhline(med_ratio, color="blue", linestyle=":", alpha=0.5, + label=f"median={med_ratio:.2f}") + ax.set_ylabel("Pred / Obs") + ax.set_yscale("log") + ax.set_ylim(0.05, 20) + ax.legend(fontsize=8) + ax.set_title("Prediction Ratio") + ax.grid(True, alpha=0.3) + ax.set_xlabel("Date") + + fig.suptitle(f"MM Noise Model — {label} ({pid[:16]})", fontsize=13) + fig.tight_layout() + out = os.path.join(output_dir, f"{pid[:16]}_mm_fit.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_tvl_response(params, meta, decomp, output_dir): + """Cross-pool TVL response curves showing MM saturation.""" + pool_ids = meta["pool_ids"] + pool_tokens = meta["pool_tokens"] + n_pools = len(pool_ids) + + tvl_range = np.logspace(4, 10, 200) # $10K to $10B + + fig, axes = plt.subplots(1, 3, figsize=(18, 6)) + + # Select pools with enough data for interesting plots + pool_idx_arr = np.array(decomp["pool_idx"]) + interesting = [] + for i in range(n_pools): + n = (pool_idx_arr == i).sum() + if n >= 50: + interesting.append(i) + + colors = plt.cm.tab20(np.linspace(0, 1, len(interesting))) + + # Panel 1: Noise volume vs TVL (absolute) + ax = axes[0] + for ci, i in enumerate(interesting): + K_i = float(np.exp(params["log_K"][i])) + + mask = pool_idx_arr == i + actual_noise = np.median(decomp["v_noise"][mask]) + actual_tvl = np.median(decomp["tvl"][mask]) + # Scale: noise(TVL) = actual_noise * [TVL/(K+TVL)] / [actual_TVL/(K+actual_TVL)] + mm_actual = actual_tvl / (K_i + actual_tvl) + mm_curve = tvl_range / (K_i + tvl_range) + noise_curve = actual_noise * mm_curve / mm_actual + + label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" + ax.plot(tvl_range / 1e6, noise_curve / 1e6, color=colors[ci], + linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + # Mark actual TVL + ax.scatter([actual_tvl / 1e6], [actual_noise / 1e6], + color=colors[ci], s=20, zorder=5) + + ax.set_xlabel("TVL ($M)") + ax.set_ylabel("Daily Noise Volume ($M)") + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_title("Noise Volume vs TVL (MM saturation)") + ax.legend(fontsize=6, ncol=2) + ax.grid(True, alpha=0.3) + + # Panel 2: Noise/TVL ratio vs TVL + ax = axes[1] + for ci, i in enumerate(interesting): + K_i = float(np.exp(params["log_K"][i])) + mask = pool_idx_arr == i + actual_noise = np.median(decomp["v_noise"][mask]) + actual_tvl = np.median(decomp["tvl"][mask]) + mm_actual = actual_tvl / (K_i + actual_tvl) + mm_curve = tvl_range / (K_i + tvl_range) + noise_curve = actual_noise * mm_curve / mm_actual + ratio_curve = noise_curve / tvl_range * 100 + + label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" + ax.plot(tvl_range / 1e6, ratio_curve, color=colors[ci], + linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + + ax.set_xlabel("TVL ($M)") + ax.set_ylabel("Noise / TVL (%)") + ax.set_xscale("log") + ax.set_title("Noise as Fraction of TVL") + ax.legend(fontsize=6, ncol=2) + ax.grid(True, alpha=0.3) + + # Panel 3: Elasticity vs TVL + ax = axes[2] + for ci, i in enumerate(interesting): + K_i = float(np.exp(params["log_K"][i])) + eps_curve = K_i / (K_i + tvl_range) + + label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" + ax.plot(tvl_range / 1e6, eps_curve, color=colors[ci], + linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + # Mark actual TVL + actual_tvl = np.median(decomp["tvl"][pool_idx_arr == i]) + eps_actual = K_i / (K_i + actual_tvl) + ax.scatter([actual_tvl / 1e6], [eps_actual], + color=colors[ci], s=20, zorder=5) + + ax.axhline(0.5, color="gray", linestyle="--", alpha=0.3, label="ε=0.5") + ax.set_xlabel("TVL ($M)") + ax.set_ylabel("Elasticity ε(TVL)") + ax.set_xscale("log") + ax.set_ylim(0, 1.05) + ax.set_title("TVL Elasticity (K/(K+TVL))") + ax.legend(fontsize=6, ncol=2) + ax.grid(True, alpha=0.3) + + fig.suptitle("Michaelis-Menten Noise Model — TVL Response", fontsize=13) + fig.tight_layout() + out = os.path.join(output_dir, "mm_tvl_response.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def plot_K_distribution(params, meta, decomp, output_dir): + """Plot K values across pools with data quality indicator.""" + pool_ids = meta["pool_ids"] + pool_tokens = meta["pool_tokens"] + n_pools = len(pool_ids) + pool_idx_arr = np.array(decomp["pool_idx"]) + + K_vals = [] + labels = [] + n_days = [] + tvl_ranges = [] + for i in range(n_pools): + mask = pool_idx_arr == i + n = mask.sum() + K_i = float(np.exp(params["log_K"][i])) + K_vals.append(K_i) + tok = pool_tokens[i] + labels.append(f"{tok[0]}/{tok[1]}") + n_days.append(n) + if n > 0: + tvl = decomp["tvl"][mask] + tvl_ranges.append(np.log10(tvl.max() / max(tvl.min(), 1))) + else: + tvl_ranges.append(0) + + K_vals = np.array(K_vals) + n_days = np.array(n_days) + tvl_ranges = np.array(tvl_ranges) + + # Sort by K + order = np.argsort(K_vals) + + fig, axes = plt.subplots(1, 2, figsize=(16, 8)) + + # Panel 1: K values as horizontal bar + ax = axes[0] + y_pos = np.arange(n_pools) + colors = plt.cm.viridis(tvl_ranges[order] / max(tvl_ranges.max(), 1)) + ax.barh(y_pos, K_vals[order] / 1e6, color=colors, height=0.7) + ax.set_yticks(y_pos) + ax.set_yticklabels([labels[i] for i in order], fontsize=7) + ax.set_xlabel("K ($M)") + ax.set_title("Half-Saturation TVL by Pool\n(color = log10 TVL range)") + ax.axvline(np.median(K_vals) / 1e6, color="red", linestyle="--", + alpha=0.5, label=f"median=${np.median(K_vals)/1e6:.1f}M") + ax.legend() + ax.grid(True, alpha=0.3, axis="x") + + # Panel 2: K vs data quality (n_days and TVL range) + ax = axes[1] + valid = n_days > 0 + sc = ax.scatter(n_days[valid], K_vals[valid] / 1e6, + c=tvl_ranges[valid], cmap="viridis", + s=50, alpha=0.7) + for i in range(n_pools): + if n_days[i] > 0: + ax.annotate(labels[i], (n_days[i], K_vals[i] / 1e6), + fontsize=5, alpha=0.7) + ax.set_xlabel("Number of training days") + ax.set_ylabel("K ($M)") + ax.set_title("K vs Data Quantity\n(color = log10 TVL range)") + plt.colorbar(sc, ax=ax, label="log10(TVL_max/TVL_min)") + ax.grid(True, alpha=0.3) + + fig.suptitle("Michaelis-Menten K Distribution", fontsize=13) + fig.tight_layout() + out = os.path.join(output_dir, "mm_K_distribution.png") + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f" Saved: {out}") + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--pool", default=None, + help="Plot single pool (prefix match)") + parser.add_argument("--all-pools", action="store_true") + parser.add_argument("--artifact-dir", default="results/mm_noise") + parser.add_argument("--output-dir", default="results/mm_noise/plots") + parser.add_argument("--top-n", type=int, default=10, + help="Plot top N pools by sample count") + args = parser.parse_args() + + os.environ.setdefault("JAX_PLATFORMS", "cpu") + os.makedirs(args.output_dir, exist_ok=True) + + # Load + print("Loading model...") + params, meta = load_model(args.artifact_dir) + + print("Loading data...") + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + stage1 = pickle.load(f) + matched_clean = stage1["matched_clean"] + option_c_clean = stage1["option_c_clean"] + + print("Computing decomposition...") + decomp = compute_decomposition(params, meta, matched_clean, option_c_clean) + + pool_ids = decomp["pool_ids"] + pool_idx = decomp["pool_idx"] + + # Which pools to plot + if args.pool: + targets = [i for i, pid in enumerate(pool_ids) + if pid.startswith(args.pool)] + elif args.all_pools: + targets = list(range(len(pool_ids))) + else: + # Top N by sample count + counts = [(pool_idx == i).sum() for i in range(len(pool_ids))] + targets = sorted(range(len(pool_ids)), key=lambda i: -counts[i]) + targets = targets[:args.top_n] + + # Per-pool time series + print(f"\nPlotting {len(targets)} pools...") + for i in targets: + plot_pool_timeseries(decomp, params, i, args.output_dir) + + # TVL response curves + print("\nPlotting TVL response...") + plot_tvl_response(params, meta, decomp, args.output_dir) + + # K distribution + print("\nPlotting K distribution...") + plot_K_distribution(params, meta, decomp, args.output_dir) + + print(f"\nDone. Plots in {args.output_dir}/") + + +if __name__ == "__main__": + main() From 1fae546f7c6b8e15ec89d4c7b9507e0c1106d4f1 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 7 Apr 2026 11:07:51 +0100 Subject: [PATCH 079/115] =?UTF-8?q?feat:=20MM=20noise=20model=20=E2=80=94?= =?UTF-8?q?=20per-pool=20K,=20Optuna=20sweep,=20cross-pool=20TVL=20analysi?= =?UTF-8?q?s?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MM model now supports per-pool log_K (default) and shared Binance-volume K (--shared-K). Per-pool K with per-pool gamma achieves R²=0.66 at 20K epochs with structural TVL saturation (median K≈$2M). Optuna sweep searches lr, l2, huber_delta, init_log_K, per_pool_gamma. Best shared-gamma eval R²=0.42 (huber=0.5, lr=1e-4 consistently). Removed learned EWMA (lambda stayed near 1, no benefit). Removed TVL interaction features (hurt eval R², subsumed by K). Plot script handles both K modes, shows all pools by default. New: verify_vol_volume_slope.py — cross-pool volume/TVL analysis confirming sublinear scaling (TVL^0.80, R²=0.79) and TVL elasticity ~0.9 across fee tiers. --- experiments/run_mm_noise.py | 411 ++++++++++++++++++------- experiments/verify_vol_volume_slope.py | 372 ++++++++++++++++++++++ scripts/plot_mm_noise_fit.py | 58 ++-- 3 files changed, 703 insertions(+), 138 deletions(-) create mode 100644 experiments/verify_vol_volume_slope.py diff --git a/experiments/run_mm_noise.py b/experiments/run_mm_noise.py index abdc2b18..219c5f28 100644 --- a/experiments/run_mm_noise.py +++ b/experiments/run_mm_noise.py @@ -53,8 +53,15 @@ def load_stage1(): def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), include_cross_pool=False): - """Build data with MM structure: separate TVL from market features.""" + """Build data with MM structure: separate TVL from market features. + + Also builds per-sample Binance log-volumes for predicting K. + """ from experiments.run_linear_market_noise import build_data + from quantammsim.calibration.market_features import ( + _load_binance_daily, TOKEN_MAP, + ) + from quantammsim.calibration.pool_data import _parse_tokens # Get full feature matrix from linear model's pipeline data = build_data( @@ -64,23 +71,18 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), include_cross_pool=include_cross_pool, ) - # Separate TVL from other features + # Remove TVL column and TVL interaction terms — TVL handled by MM, + # interactions subsumed by time-varying K from Binance volumes feat_names = data["feat_names"] x_full = data["x"] - - # Find TVL column (xobs_1) and TVL interaction columns tvl_col = feat_names.index("xobs_1") tvl_interaction_cols = [i for i, name in enumerate(feat_names) if name.startswith("xobs_1\u00d7")] - - # Remove TVL and its interactions from market features remove_cols = {tvl_col} | set(tvl_interaction_cols) keep_cols = [i for i in range(len(feat_names)) if i not in remove_cols] x_market = x_full[:, keep_cols].astype(np.float32) market_names = [feat_names[i] for i in keep_cols] - # TVL comes from the raw panel data (unstandardized log_tvl) - # x_full[:, tvl_col] might be standardized, so get raw from panel pool_ids = data["pool_ids"] n_pools = data["n_pools"] @@ -105,47 +107,77 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), log_tvl = np.array([tvl_grid[day_idx[s], pool_idx[s]] for s in range(len(pool_idx))], dtype=np.float32) - # Token info for display - from quantammsim.calibration.pool_data import _parse_tokens + # Binance daily log-volumes per token, for predicting K + binance_cache = {} + + def _get_binance_daily(symbol): + mapped = TOKEN_MAP.get(symbol, symbol) + if mapped not in binance_cache: + daily = _load_binance_daily(mapped) + if daily is not None: + binance_cache[mapped] = { + d: float(np.log(max(v, 1.0))) + for d, v in daily["volume_usd"].items() + } + else: + binance_cache[mapped] = {} + return binance_cache[mapped] + pool_tokens = [] - for pid in pool_ids: + log_vol_a_grid = np.full((n_dates, n_pools), np.nan) + log_vol_b_grid = np.full((n_dates, n_pools), np.nan) + + for j, pid in enumerate(pool_ids): toks = _parse_tokens(matched_clean[pid]["tokens"]) tok_a = toks[0] tok_b = toks[1] if len(toks) > 1 else toks[0] pool_tokens.append((tok_a, tok_b)) - removed_names = [feat_names[i] for i in sorted(remove_cols)] - print(f" Removed TVL features: {removed_names}") - print(f" Market features ({len(market_names)}): {market_names}") + bvol_a = _get_binance_daily(tok_a) + bvol_b = _get_binance_daily(tok_b) - # Build per-pool temporal ordering for EWMA - # For each pool, store the sample indices sorted by day_idx - # Pad to max length so we can use lax.scan uniformly - pool_time_indices = [] # (n_pools, max_T) — sample indices in time order - pool_time_lengths = [] # (n_pools,) — actual length per pool + panel = matched_clean[pid]["panel"] + for k, date in enumerate(panel["date"].values): + t = date_to_idx[date] + day = pd.Timestamp(date).normalize() + if day in bvol_a: + log_vol_a_grid[t, j] = bvol_a[day] + if day in bvol_b: + log_vol_b_grid[t, j] = bvol_b[day] + + # Per-sample Binance volumes (impute missing with per-pool median) + n_samples = len(pool_idx) + log_vol_a = np.array([log_vol_a_grid[day_idx[s], pool_idx[s]] + for s in range(n_samples)], dtype=np.float32) + log_vol_b = np.array([log_vol_b_grid[day_idx[s], pool_idx[s]] + for s in range(n_samples)], dtype=np.float32) + + # Impute NaN with per-pool median for i in range(n_pools): mask = pool_idx == i - idxs = np.where(mask)[0] - # Sort by day_idx - order = np.argsort(day_idx[idxs]) - pool_time_indices.append(idxs[order]) - pool_time_lengths.append(len(idxs)) - - max_T = max(pool_time_lengths) if pool_time_lengths else 0 - # Pad to uniform length (pad with 0, masked later) - pool_time_padded = np.zeros((n_pools, max_T), dtype=np.int32) - pool_time_mask = np.zeros((n_pools, max_T), dtype=np.float32) - for i in range(n_pools): - L = pool_time_lengths[i] - pool_time_padded[i, :L] = pool_time_indices[i] - pool_time_mask[i, :L] = 1.0 + for arr in (log_vol_a, log_vol_b): + pool_vals = arr[mask] + if np.isnan(pool_vals).all(): + arr[mask] = 20.0 # fallback ~$500M daily vol + elif np.isnan(pool_vals).any(): + arr[mask] = np.where(np.isnan(pool_vals), + np.nanmedian(pool_vals), pool_vals) + + n_missing = np.isnan(log_vol_a).sum() + np.isnan(log_vol_b).sum() + if n_missing > 0: + print(f" WARNING: {n_missing} NaN in Binance volumes after imputation") + log_vol_a = np.nan_to_num(log_vol_a, nan=20.0) + log_vol_b = np.nan_to_num(log_vol_b, nan=20.0) - print(f" EWMA: max_T={max_T}, pools with data: " - f"{sum(1 for l in pool_time_lengths if l > 0)}") + removed_names = [feat_names[i] for i in sorted(remove_cols)] + print(f" Removed: {removed_names}") + print(f" Market features ({len(market_names)}): {market_names}") return { "x_market": x_market, "log_tvl": log_tvl, + "log_vol_a": log_vol_a, + "log_vol_b": log_vol_b, "y_total": data["y_total"], "pool_idx": pool_idx, "day_idx": day_idx, @@ -160,63 +192,34 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), "market_names": market_names, "x_mean": data["x_mean"], "x_std": data["x_std"], - "pool_time_padded": pool_time_padded, - "pool_time_mask": pool_time_mask, } # ---- Model ---- -def ewma_smooth(log_tvl, raw_lambda, pool_time_padded, pool_time_mask): - """Apply learned EWMA smoothing to log_tvl, per pool. - - smooth_t = λ * log_tvl_t + (1-λ) * smooth_{t-1} - - Returns smoothed log_tvl in the same sample order as input. - """ - lam = jax.nn.sigmoid(raw_lambda) # constrain to (0, 1) - n_pools = pool_time_padded.shape[0] - smoothed = jnp.array(log_tvl) # copy - - for i in range(n_pools): - idxs = pool_time_padded[i] # (max_T,) sample indices - mask = pool_time_mask[i] # (max_T,) 1.0 or 0.0 - raw_vals = log_tvl[idxs] # (max_T,) raw log_tvl in time order - - # lax.scan for EWMA - def step(carry, x): - prev_smooth, = carry - raw_val, m = x - new_smooth = jnp.where( - m > 0, - lam * raw_val + (1.0 - lam) * prev_smooth, - prev_smooth) - return (new_smooth,), new_smooth - - init = (raw_vals[0],) - _, smooth_vals = jax.lax.scan(step, init, (raw_vals, mask)) - - # Scatter smoothed values back to sample positions - smoothed = smoothed.at[idxs].set( - jnp.where(mask > 0, smooth_vals, smoothed[idxs])) - - return smoothed - - -def forward_mm(params, x_market, log_tvl_smooth, pool_idx): +def forward_mm(params, x_market, log_tvl, pool_idx, + log_vol_a=None, log_vol_b=None): """MM forward pass → log(V_noise) per sample. - log(V_noise) = log_alpha_i + x_market @ gamma[_i] - + log(TVL_smooth) - log(K_i + TVL_smooth) + Supports two K modes: + - Per-pool: params contains "log_K" (n_pools,) + - Binance-volume: params contains "k_params" (3,) + log_vol_a/b """ log_alpha = params["log_alpha"] - log_K = params["log_K"] gamma = params["gamma"] - # Per-sample pool params alpha_i = log_alpha[pool_idx] - K_i = jnp.exp(log_K[pool_idx]) - tvl = jnp.exp(log_tvl_smooth) + tvl = jnp.exp(log_tvl) + + # K: per-pool or from Binance volumes + if "k_params" in params: + k_params = params["k_params"] + vol_min = jnp.minimum(log_vol_a, log_vol_b) + vol_max = jnp.maximum(log_vol_a, log_vol_b) + log_K = k_params[0] + k_params[1] * vol_min + k_params[2] * vol_max + K = jnp.exp(log_K) + else: + K = jnp.exp(params["log_K"][pool_idx]) # Market features: shared or per-pool gamma if gamma.ndim == 2: @@ -225,9 +228,7 @@ def forward_mm(params, x_market, log_tvl_smooth, pool_idx): else: market_term = x_market @ gamma - # MM saturation on smoothed TVL - log_saturation = log_tvl_smooth - jnp.log(K_i + tvl) - + log_saturation = log_tvl - jnp.log(K + tvl) return alpha_i + market_term + log_saturation @@ -235,16 +236,10 @@ def make_loss_fn(pool_coeffs, pool_gas, n_pools): """Loss with PCHIP arb + MM noise.""" from quantammsim.calibration.grid_interpolation import interpolate_pool_daily - def loss_fn(params, x_market, log_tvl, y_total, - sample_grid_days, pool_idx, pool_time_padded, - pool_time_mask, l2_alpha, huber_delta): + def loss_fn(params, x_market, log_tvl, log_vol_a, log_vol_b, y_total, + sample_grid_days, pool_idx, l2_alpha, huber_delta): log_cadence = params["log_cadence"] - # EWMA smooth TVL - log_tvl_smooth = ewma_smooth( - log_tvl, params["raw_lambda"], - pool_time_padded, pool_time_mask) - # V_arb from PCHIP n_samples = x_market.shape[0] log_v_arb = jnp.zeros(n_samples) @@ -257,8 +252,10 @@ def loss_fn(params, x_market, log_tvl, y_total, jnp.log(jnp.maximum(v_arb_all[safe_days], 1e-10)), log_v_arb) - # V_noise from MM with smoothed TVL - log_v_noise = forward_mm(params, x_market, log_tvl_smooth, pool_idx) + # V_noise from MM + log_v_noise = forward_mm( + params, x_market, log_tvl, pool_idx, + log_vol_a=log_vol_a, log_vol_b=log_vol_b) # V_total log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) @@ -302,16 +299,16 @@ def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, x_market = jnp.array(data["x_market"]) log_tvl = jnp.array(data["log_tvl"]) + log_vol_a = jnp.array(data["log_vol_a"]) + log_vol_b = jnp.array(data["log_vol_b"]) y_total = jnp.array(data["y_total"]) sgd = jnp.array(data["sample_grid_days"]) pidx = jnp.array(data["pool_idx"]) - pt_padded = jnp.array(data["pool_time_padded"]) - pt_mask = jnp.array(data["pool_time_mask"]) for epoch in range(n_epochs): loss, grads = grad_fn( - params, x_market, log_tvl, y_total, sgd, pidx, - pt_padded, pt_mask, l2_alpha, huber_delta) + params, x_market, log_tvl, log_vol_a, log_vol_b, + y_total, sgd, pidx, l2_alpha, huber_delta) for k in params: g = grads[k] @@ -322,15 +319,17 @@ def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, params[k] = params[k] - lr * m_hat / (jnp.sqrt(v_hat) + eps) if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): - log_K_med = float(jnp.median(params["log_K"])) - K_med = float(jnp.exp(log_K_med)) + ev = evaluate(params, data) + if "k_params" in params: + k_p = np.array(params["k_params"]) + k_str = f" k=[{k_p[0]:.2f},{k_p[1]:.3f},{k_p[2]:.3f}]" + else: + k_str = "" + K_med = float(np.median(list(ev["K_values"].values()))) cad = np.exp(np.array(params["log_cadence"])) - gamma_norm = float(jnp.sqrt(jnp.mean(params["gamma"] ** 2))) - lam = float(jax.nn.sigmoid(params["raw_lambda"])) print(f" epoch {epoch:5d} loss={float(loss):.4f}" - f" K_med=${K_med:,.0f}" - f" λ={lam:.3f}" - f" |γ|={gamma_norm:.3f}" + f" R²={ev['median_r2']:.3f}" + f" K_med=${K_med/1e6:.1f}M{k_str}" f" cad=[{cad.min():.0f},{np.median(cad):.0f},{cad.max():.0f}]") return params @@ -361,16 +360,12 @@ def evaluate(params, data): v_arb[mask] = v_arb_all[safe] log_v_arb = np.log(np.maximum(v_arb, 1e-10)) - # Smooth TVL with learned lambda - log_tvl_smooth = np.array(ewma_smooth( - jnp.array(data["log_tvl"]), params["raw_lambda"], - jnp.array(data["pool_time_padded"]), - jnp.array(data["pool_time_mask"]))) - log_v_noise = np.array(forward_mm( params, jnp.array(data["x_market"]), - jnp.array(log_tvl_smooth), - jnp.array(data["pool_idx"]))) + jnp.array(data["log_tvl"]), + jnp.array(data["pool_idx"]), + log_vol_a=jnp.array(data["log_vol_a"]), + log_vol_b=jnp.array(data["log_vol_b"]))) log_v_total = np.logaddexp(log_v_arb, log_v_noise) v_noise = np.exp(log_v_noise) @@ -390,8 +385,22 @@ def evaluate(params, data): noise_shares[data["pool_ids"][i]] = float(np.median( v_noise[mask] / v_total[mask])) - K_values = {data["pool_ids"][i]: float(np.exp(params["log_K"][i])) - for i in range(n_pools)} + # Per-pool K values + K_values = {} + if "k_params" in params: + k_p = np.array(params["k_params"]) + for i in range(n_pools): + mask = pool_idx == i + if not mask.any(): + K_values[data["pool_ids"][i]] = float(np.exp(k_p[0])) + continue + va = data["log_vol_a"][mask] + vb = data["log_vol_b"][mask] + log_K_i = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) + K_values[data["pool_ids"][i]] = float(np.exp(np.median(log_K_i))) + else: + for i in range(n_pools): + K_values[data["pool_ids"][i]] = float(np.exp(params["log_K"][i])) return { "r2s": r2s, @@ -414,7 +423,7 @@ def tvl_response_check(params, data): tvl_test = [1e5, 1e6, 1e7, 1e8, 1e9] - for i in range(min(n_pools, 15)): + for i in range(n_pools): pid = data["pool_ids"][i] toks = data["pool_tokens"][i] label = f"{toks[0]}/{toks[1]}" @@ -422,7 +431,15 @@ def tvl_response_check(params, data): if mask.sum() == 0: continue - K_i = float(np.exp(params["log_K"][i])) + # Per-pool K + if "k_params" in params: + k_p = np.array(params["k_params"]) + va = data["log_vol_a"][mask] + vb = data["log_vol_b"][mask] + log_K_i = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) + K_i = float(np.exp(np.median(log_K_i))) + else: + K_i = float(np.exp(params["log_K"][i])) x_med = np.median(data["x_market"][mask], axis=0) gamma = np.array(params["gamma"]) @@ -459,11 +476,16 @@ def main(): parser.add_argument("--huber-delta", type=float, default=1.0) parser.add_argument("--init-log-K", type=float, default=17.0, help="Initial log(K) ~ log($24M)") + parser.add_argument("--shared-K", action="store_true", + help="Predict K from Binance volumes (3 shared params)" + " instead of per-pool K") parser.add_argument("--per-pool-gamma", action="store_true", help="Per-pool market feature coefficients") parser.add_argument("--no-split", action="store_true") parser.add_argument("--trend-windows", type=int, nargs="+", default=[7]) parser.add_argument("--include-cross-pool", action="store_true") + parser.add_argument("--tune", type=int, default=0, + help="Optuna sweep (0 = single run)") parser.add_argument("--save-artifact", default="results/mm_noise") args = parser.parse_args() @@ -489,6 +511,10 @@ def main(): print(f" {n_samples} samples, {n_pools} pools," f" {n_market} market features, {time.time() - t0:.1f}s") + if args.tune > 0: + run_optuna(data, args.tune) + return + # Pool summary pool_idx = data["pool_idx"] for i, (pid, toks) in enumerate( @@ -525,11 +551,13 @@ def main(): params = { "log_alpha": jnp.zeros(n_pools), - "log_K": jnp.full(n_pools, args.init_log_K), "gamma": gamma_init, "log_cadence": jnp.array(data["init_log_cadences"]), - "raw_lambda": jnp.array(2.0), # sigmoid(2) ≈ 0.88 — mostly raw } + if args.shared_K: + params["k_params"] = jnp.array([args.init_log_K, 0.0, 0.0]) + else: + params["log_K"] = jnp.full(n_pools, args.init_log_K) n_params = sum(v.size for v in params.values()) print(f"\n Parameters: {n_params}" f" (α: {n_pools}, K: {n_pools}," @@ -579,6 +607,13 @@ def _ridge(X, y, alpha=1.0): train_eval = evaluate(params, train_data) print(f" Median R²: {train_eval['median_r2']:.4f}") + if "k_params" in params: + k_p = np.array(params["k_params"]) + print(f" k_params: k_0={k_p[0]:.2f}, k_min={k_p[1]:.4f}, k_max={k_p[2]:.4f}") + else: + K_med = float(np.exp(np.median(np.array(params["log_K"])))) + print(f" Per-pool K: median=${K_med/1e6:.1f}M") + print(f"\n {'Pool':>16s} {'Tokens':>16s} {'R²':>6s}" f" {'Noise%':>7s} {'K ($M)':>10s}") for pid in data["pool_ids"]: @@ -631,5 +666,145 @@ def _ridge(X, y, alpha=1.0): print(f"\n Saved: {args.save_artifact}/") +def run_optuna(data, n_trials): + """Optuna hyperparameter sweep for MM noise model.""" + import optuna + optuna.logging.set_verbosity(optuna.logging.WARNING) + + n_pools = data["n_pools"] + n_market = data["n_market_feat"] + n_samples = len(data["pool_idx"]) + + # 70/30 temporal split + day_idx = data["day_idx"] + split_day = int(day_idx.max() * 0.7) + train_mask = day_idx <= split_day + eval_mask = day_idx > split_day + train_data = {k: v[train_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + eval_data = {k: v[eval_mask] if isinstance(v, np.ndarray) + and v.shape[0] == n_samples else v + for k, v in data.items()} + print(f" Optuna split: {train_mask.sum()} train, {eval_mask.sum()} eval") + + def _ridge(X, y, alpha=1.0): + XtX = X.T @ X + alpha * np.eye(X.shape[1]) + return np.linalg.solve(XtX, X.T @ y) + + def objective(trial): + lr = trial.suggest_float("lr", 1e-4, 3e-2, log=True) + l2_alpha = trial.suggest_float("l2_alpha", 1e-5, 1e-1, log=True) + huber_delta = trial.suggest_categorical("huber_delta", [0.5, 1.0, 1.5]) + init_log_K = trial.suggest_float("init_log_K", 14.0, 20.0) + n_epochs = trial.suggest_categorical("n_epochs", [2000, 3000, 5000]) + per_pool_gamma = trial.suggest_categorical("per_pool_gamma", [True, False]) + if per_pool_gamma: + gamma_init = jnp.zeros((n_pools, n_market)) + else: + gamma_init = jnp.zeros(n_market) + + params = { + "log_alpha": jnp.zeros(n_pools), + "k_params": jnp.array([init_log_K, 0.0, 0.0]), + "gamma": gamma_init, + "log_cadence": jnp.array(data["init_log_cadences"]), + } + + # Warm-start gamma + x_trn = train_data["x_market"] + y_trn = train_data["y_total"] + if per_pool_gamma: + pidx = train_data["pool_idx"] + for i in range(n_pools): + mask_i = pidx == i + if mask_i.sum() < 5: + continue + X_i = np.concatenate([x_trn[mask_i], + np.ones((mask_i.sum(), 1))], 1) + w = _ridge(X_i, y_trn[mask_i]) + params["gamma"] = params["gamma"].at[i].set( + jnp.array(w[:-1].astype(np.float32))) + params["log_alpha"] = params["log_alpha"].at[i].set( + float(w[-1])) + else: + X_all = np.concatenate([x_trn, np.ones((len(y_trn), 1))], 1) + w = _ridge(X_all, y_trn) + params["gamma"] = jnp.array(w[:-1].astype(np.float32)) + + grad_fn = make_loss_fn(data["pool_coeffs"], data["pool_gas"], n_pools) + params = train(params, train_data, grad_fn, n_epochs, lr, + l2_alpha, huber_delta, verbose=False) + + # Eval + eval_result = evaluate(params, eval_data) + med_r2 = eval_result["median_r2"] + + K_med = float(np.median([v for v in eval_result["K_values"].values()])) + k_p = np.array(params["k_params"]) + pp_str = "pp" if per_pool_gamma else "sh" + print(f" Trial {trial.number}: eval={med_r2:.4f}" + f" K_med=${K_med/1e6:.1f}M" + f" k=[{k_p[0]:.1f},{k_p[1]:.3f},{k_p[2]:.3f}]" + f" {pp_str} ep={n_epochs} lr={lr:.1e} l2={l2_alpha:.1e}" + f" hub={huber_delta}") + + # Save every trial + trial_dir = os.path.join("results", "mm_noise", "trials", + f"trial_{trial.number:04d}") + os.makedirs(trial_dir, exist_ok=True) + save_dict = {k: np.array(v) for k, v in params.items()} + np.savez(os.path.join(trial_dir, "model.npz"), **save_dict) + meta = { + "pool_ids": data["pool_ids"], + "pool_tokens": data["pool_tokens"], + "market_names": data["market_names"], + "n_pools": n_pools, + "n_market_feat": n_market, + "per_pool_gamma": per_pool_gamma, + "eval_r2": med_r2, + "hparams": { + "lr": lr, "l2_alpha": l2_alpha, "huber_delta": huber_delta, + "init_log_K": init_log_K, "n_epochs": n_epochs, + "per_pool_gamma": per_pool_gamma, + }, + } + with open(os.path.join(trial_dir, "meta.json"), "w") as f: + json.dump(meta, f, indent=2) + + return med_r2 + + study = optuna.create_study(direction="maximize") + study.optimize(objective, n_trials=n_trials) + + print(f"\n{'='*70}") + print(f"Optuna Results (MM noise)") + print(f"{'='*70}") + print(f" Best eval R²: {study.best_value:.4f}") + print(f" Best params:") + for k, v in sorted(study.best_params.items()): + print(f" {k}: {v}") + + trials = sorted(study.trials, key=lambda t: t.value if t.value else -999, + reverse=True) + print(f"\n Top 10:") + for t in trials[:10]: + if t.value is not None: + print(f" #{t.number}: eval={t.value:.4f} {t.params}") + + # Copy best to top-level + best_dir = os.path.join("results", "mm_noise", "trials", + f"trial_{study.best_trial.number:04d}") + if os.path.exists(os.path.join(best_dir, "model.npz")): + import shutil + for fn in ("model.npz", "meta.json"): + shutil.copy2(os.path.join(best_dir, fn), + os.path.join("results", "mm_noise", fn)) + print(f"\n Copied best trial ({study.best_trial.number})" + f" to results/mm_noise/") + + return study + + if __name__ == "__main__": main() diff --git a/experiments/verify_vol_volume_slope.py b/experiments/verify_vol_volume_slope.py new file mode 100644 index 00000000..1758bd73 --- /dev/null +++ b/experiments/verify_vol_volume_slope.py @@ -0,0 +1,372 @@ +"""Verify: is the volatility-volume slope identical across fee tiers? + +The claim (from noise_calibration_review.md): "the relationship between +price volatility and swap volume is identical across fee tiers (slope 0.91 +for both low-fee and high-fee pools)." + +This script tests the claim by: +1. Loading all pool panel data +2. Splitting pools by fee tier (low vs high) +3. Regressing log(volume) on log(volatility) within each group +4. Comparing slopes + +If the slopes are similar, it means volatility drives organic volume +identically regardless of fee — supporting the arb/noise decomposition +(since arb intensity differs across fee tiers but noise doesn't). + +Usage: + python experiments/verify_vol_volume_slope.py +""" + +import os +import pickle +import sys + +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + + +def main(): + path = os.path.join(CACHE_DIR, "stage1.pkl") + if not os.path.exists(path): + print("ERROR: no stage1 cache.") + sys.exit(1) + with open(path, "rb") as f: + data = pickle.load(f) + matched_clean = data["matched_clean"] + + # Collect per-observation data + rows = [] + for pid, entry in matched_clean.items(): + panel = entry["panel"] + fee = entry.get("fee", np.exp(panel["log_fee"].values[0])) + chain = entry.get("chain", "unknown") + tokens = entry.get("tokens", "?") + + log_vol = panel["log_volume"].values.astype(float) + vol_raw = panel["volatility"].values.astype(float) + log_tvl = panel["log_tvl_lag1"].values.astype(float) + + for i in range(len(log_vol)): + if vol_raw[i] > 1e-10 and np.isfinite(log_vol[i]): + rows.append({ + "pool_id": pid, + "tokens": tokens, + "chain": chain, + "fee": fee, + "log_fee": np.log(fee), + "log_volume": log_vol[i], + "log_sigma": np.log(max(vol_raw[i], 1e-10)), + "log_tvl": log_tvl[i] if np.isfinite(log_tvl[i]) else np.nan, + }) + + df = pd.DataFrame(rows).dropna() + print(f"Loaded {len(df)} observations from {df['pool_id'].nunique()} pools") + + # Fee tier split + fees = df.groupby("pool_id")["fee"].first() + median_fee = fees.median() + print(f"\nFee distribution:") + print(f" min={fees.min():.5f} median={median_fee:.5f} max={fees.max():.5f}") + print(f" Unique fees: {sorted(fees.unique())}") + + low_fee_pools = set(fees[fees <= median_fee].index) + high_fee_pools = set(fees[fees > median_fee].index) + print(f" Low-fee pools (≤{median_fee:.4f}): {len(low_fee_pools)}") + print(f" High-fee pools (>{median_fee:.4f}): {len(high_fee_pools)}") + + df_low = df[df["pool_id"].isin(low_fee_pools)] + df_high = df[df["pool_id"].isin(high_fee_pools)] + + # OLS: log_volume ~ intercept + log_sigma + def ols_slope(x, y): + X = np.column_stack([np.ones(len(x)), x]) + beta = np.linalg.lstsq(X, y, rcond=None)[0] + y_hat = X @ beta + ss_res = np.sum((y - y_hat) ** 2) + ss_tot = np.sum((y - y.mean()) ** 2) + r2 = 1 - ss_res / ss_tot + # Standard error of slope + n = len(x) + se = np.sqrt(ss_res / (n - 2) / np.sum((x - x.mean()) ** 2)) + return beta[1], beta[0], r2, se + + print(f"\n{'='*60}") + print("OLS: log(volume) ~ intercept + log(sigma)") + print(f"{'='*60}") + + # All pools + slope, intercept, r2, se = ols_slope(df["log_sigma"].values, df["log_volume"].values) + print(f"\n All pools ({len(df)} obs):") + print(f" slope = {slope:.4f} ± {1.96*se:.4f} (95% CI)") + print(f" R² = {r2:.4f}") + + # Low fee + slope_l, int_l, r2_l, se_l = ols_slope(df_low["log_sigma"].values, df_low["log_volume"].values) + print(f"\n Low-fee pools ({len(df_low)} obs, {len(low_fee_pools)} pools):") + print(f" slope = {slope_l:.4f} ± {1.96*se_l:.4f}") + print(f" R² = {r2_l:.4f}") + + # High fee + slope_h, int_h, r2_h, se_h = ols_slope(df_high["log_sigma"].values, df_high["log_volume"].values) + print(f"\n High-fee pools ({len(df_high)} obs, {len(high_fee_pools)} pools):") + print(f" slope = {slope_h:.4f} ± {1.96*se_h:.4f}") + print(f" R² = {r2_h:.4f}") + + print(f"\n Difference: {abs(slope_l - slope_h):.4f}") + print(f" Ratio: {slope_l/slope_h:.3f}") + + # Per-pool slopes + print(f"\n{'='*60}") + print("Per-pool OLS slopes") + print(f"{'='*60}") + pool_slopes = [] + print(f"\n {'Pool':>16s} {'Tokens':>20s} {'Fee':>8s} {'Slope':>8s}" + f" {'R²':>6s} {'N':>5s}") + for pid in sorted(matched_clean.keys()): + pool_df = df[df["pool_id"] == pid] + if len(pool_df) < 20: + continue + s, _, r, se = ols_slope(pool_df["log_sigma"].values, pool_df["log_volume"].values) + fee = pool_df["fee"].iloc[0] + tokens = pool_df["tokens"].iloc[0] + pool_slopes.append({"pool_id": pid, "tokens": tokens, "fee": fee, + "slope": s, "r2": r, "n": len(pool_df)}) + print(f" {pid[:16]} {tokens:>20s} {fee:>8.5f} {s:>8.4f}" + f" {r:>6.3f} {len(pool_df):>5d}") + + ps = pd.DataFrame(pool_slopes) + if len(ps) > 0: + low_slopes = ps[ps["fee"] <= median_fee]["slope"] + high_slopes = ps[ps["fee"] > median_fee]["slope"] + print(f"\n Per-pool slope summary:") + print(f" Low-fee: median={low_slopes.median():.4f}," + f" mean={low_slopes.mean():.4f} (n={len(low_slopes)})") + print(f" High-fee: median={high_slopes.median():.4f}," + f" mean={high_slopes.mean():.4f} (n={len(high_slopes)})") + + # Also try with TVL control: log_volume ~ log_sigma + log_tvl + print(f"\n{'='*60}") + print("OLS: log(volume) ~ intercept + log(sigma) + log(tvl)") + print(f"{'='*60}") + + def ols_multi(df_sub): + x = np.column_stack([ + np.ones(len(df_sub)), + df_sub["log_sigma"].values, + df_sub["log_tvl"].values, + ]) + y = df_sub["log_volume"].values + beta = np.linalg.lstsq(x, y, rcond=None)[0] + y_hat = x @ beta + ss_res = np.sum((y - y_hat) ** 2) + ss_tot = np.sum((y - y.mean()) ** 2) + return beta, 1 - ss_res / ss_tot + + beta_all, r2_all = ols_multi(df) + print(f"\n All: σ_slope={beta_all[1]:.4f}, tvl_slope={beta_all[2]:.4f}, R²={r2_all:.4f}") + + beta_l, r2_l = ols_multi(df_low) + print(f" Low: σ_slope={beta_l[1]:.4f}, tvl_slope={beta_l[2]:.4f}, R²={r2_l:.4f}") + + beta_h, r2_h = ols_multi(df_high) + print(f" High: σ_slope={beta_h[1]:.4f}, tvl_slope={beta_h[2]:.4f}, R²={r2_h:.4f}") + + print(f"\n σ slope difference (low-high): {beta_l[1] - beta_h[1]:.4f}") + + + # ---- Volume/TVL vs TVL (cross-pool) ---- + print(f"\n{'='*60}") + print("Volume/TVL vs TVL (cross-pool)") + print(f"{'='*60}") + + # Per-pool median volume and TVL + pool_stats = [] + for pid in sorted(matched_clean.keys()): + pool_df = df[df["pool_id"] == pid] + if len(pool_df) < 20: + continue + med_vol = np.exp(np.median(pool_df["log_volume"].values)) + med_tvl = np.exp(np.median(pool_df["log_tvl"].values)) + tokens = pool_df["tokens"].iloc[0] + fee = pool_df["fee"].iloc[0] + vol_tvl = med_vol / med_tvl + pool_stats.append({ + "pool_id": pid, "tokens": tokens, "fee": fee, + "med_vol": med_vol, "med_tvl": med_tvl, + "vol_tvl_pct": vol_tvl * 100, + "log_med_tvl": np.log(med_tvl), + }) + + ps = pd.DataFrame(pool_stats) + print(f"\n {'Pool':>16s} {'Tokens':>20s} {'TVL':>14s} {'Vol/day':>14s} {'Vol/TVL':>8s}") + for _, row in ps.sort_values("med_tvl").iterrows(): + print(f" {row['pool_id'][:16]} {row['tokens']:>20s}" + f" ${row['med_tvl']:>13,.0f} ${row['med_vol']:>13,.0f}" + f" {row['vol_tvl_pct']:>7.1f}%") + + # OLS: log(vol/tvl) ~ log(tvl) + log_vol_tvl = np.log(ps["med_vol"].values / ps["med_tvl"].values) + log_tvl_vals = ps["log_med_tvl"].values + slope_vt, int_vt, r2_vt, se_vt = ols_slope(log_tvl_vals, log_vol_tvl) + print(f"\n OLS: log(Vol/TVL) ~ log(TVL)") + print(f" slope = {slope_vt:.4f} ± {1.96*se_vt:.4f}") + print(f" R² = {r2_vt:.4f}") + print(f" (slope < 0 means Vol/TVL declines with TVL)") + + # Equivalent: log(Vol) ~ α + β*log(TVL), β < 1 means sublinear + slope_v, int_v, r2_v, se_v = ols_slope(log_tvl_vals, np.log(ps["med_vol"].values)) + print(f"\n OLS: log(Vol) ~ log(TVL)") + print(f" slope = {slope_v:.4f} ± {1.96*se_v:.4f}") + print(f" R² = {r2_v:.4f}") + print(f" (slope < 1 means sublinear = Vol/TVL declines)") + + # ---- TVL elasticity by TVL quartile (MM signature) ---- + print(f"\n{'='*60}") + print("TVL Elasticity by TVL Quartile") + print(f"{'='*60}") + + # Use observation-level data, not pool medians — more power + # Within-quartile regression: log(vol) ~ log(tvl) for pools in each bin + ps_sorted = ps.sort_values("med_tvl") + n_q = len(ps_sorted) // 4 + quartiles = [] + for q in range(4): + start = q * n_q + end = (q + 1) * n_q if q < 3 else len(ps_sorted) + q_pools = set(ps_sorted.iloc[start:end]["pool_id"]) + q_df = df[df["pool_id"].isin(q_pools)] + if len(q_df) < 20: + continue + s, intercept, r2, se = ols_slope(q_df["log_tvl"].values, + q_df["log_volume"].values) + tvl_lo = np.exp(q_df["log_tvl"].min()) + tvl_hi = np.exp(q_df["log_tvl"].max()) + tvl_med = np.exp(q_df["log_tvl"].median()) + quartiles.append({ + "q": q + 1, "n_pools": len(q_pools), "n_obs": len(q_df), + "tvl_lo": tvl_lo, "tvl_hi": tvl_hi, "tvl_med": tvl_med, + "slope": s, "se": se, "r2": r2, + }) + print(f"\n Q{q+1}: TVL ${tvl_lo:,.0f} – ${tvl_hi:,.0f}" + f" (median ${tvl_med:,.0f})") + print(f" {len(q_pools)} pools, {len(q_df)} obs") + print(f" slope = {s:.4f} ± {1.96*se:.4f} R² = {r2:.4f}") + + if len(quartiles) >= 2: + print(f"\n Summary:") + print(f" {'Quartile':>10s} {'Med TVL':>14s} {'Slope':>8s} {'95% CI':>16s}") + for q in quartiles: + ci = f"[{q['slope']-1.96*q['se']:.3f}, {q['slope']+1.96*q['se']:.3f}]" + print(f" Q{q['q']:>9d} ${q['tvl_med']:>13,.0f} {q['slope']:>8.4f} {ci:>16s}") + + slope_q1 = quartiles[0]["slope"] + slope_q4 = quartiles[-1]["slope"] + print(f"\n Q1→Q4 slope change: {slope_q4 - slope_q1:+.4f}") + print(f" (Negative = elasticity declines with TVL = MM signature)") + + # Also try: rolling window across pools sorted by TVL + print(f"\n Rolling 10-pool window:") + print(f" {'Window':>8s} {'Med TVL':>14s} {'Slope':>8s} {'R²':>6s}") + window = 10 + for start_i in range(0, len(ps_sorted) - window + 1, 3): + w_pools = set(ps_sorted.iloc[start_i:start_i + window]["pool_id"]) + w_df = df[df["pool_id"].isin(w_pools)] + if len(w_df) < 30: + continue + s, _, r2, se = ols_slope(w_df["log_tvl"].values, + w_df["log_volume"].values) + tvl_med = np.exp(w_df["log_tvl"].median()) + print(f" {start_i:>3d}-{start_i+window:>3d} ${tvl_med:>13,.0f} {s:>8.4f} {r2:>6.3f}") + + # Plot + try: + import matplotlib + matplotlib.use("Agg") + import matplotlib.pyplot as plt + + fig, axes = plt.subplots(1, 3, figsize=(16, 5)) + + # Panel 1: Vol/TVL vs TVL + ax = axes[0] + ax.scatter(ps["med_tvl"] / 1e6, ps["vol_tvl_pct"], + s=30, alpha=0.7, c="steelblue") + for _, row in ps.iterrows(): + ax.annotate(row["tokens"].split(",")[0], + (row["med_tvl"] / 1e6, row["vol_tvl_pct"]), + fontsize=5, alpha=0.6) + ax.set_xscale("log") + ax.set_xlabel("Median TVL ($M)") + ax.set_ylabel("Median Vol/TVL (%)") + ax.set_title(f"Volume/TVL Declines with TVL\n" + f"log(Vol/TVL) ~ {slope_vt:.2f}·log(TVL), R²={r2_vt:.2f}") + ax.grid(True, alpha=0.3) + + # Fit line + tvl_fit = np.logspace(np.log10(ps["med_tvl"].min()), + np.log10(ps["med_tvl"].max()), 100) + vol_tvl_fit = np.exp(int_vt + slope_vt * np.log(tvl_fit)) * 100 + ax.plot(tvl_fit / 1e6, vol_tvl_fit, "r--", linewidth=1, alpha=0.7) + + # Panel 2: Vol vs TVL (log-log) + ax = axes[1] + ax.scatter(ps["med_tvl"] / 1e6, ps["med_vol"] / 1e6, + s=30, alpha=0.7, c="coral") + for _, row in ps.iterrows(): + ax.annotate(row["tokens"].split(",")[0], + (row["med_tvl"] / 1e6, row["med_vol"] / 1e6), + fontsize=5, alpha=0.6) + ax.set_xscale("log") + ax.set_yscale("log") + ax.set_xlabel("Median TVL ($M)") + ax.set_ylabel("Median Daily Volume ($M)") + ax.set_title(f"Volume vs TVL (cross-pool)\n" + f"log(Vol) ~ {slope_v:.2f}·log(TVL), R²={r2_v:.2f}") + ax.grid(True, alpha=0.3) + + # Fit line + linear reference + vol_fit = np.exp(int_v + slope_v * np.log(tvl_fit)) + ax.plot(tvl_fit / 1e6, vol_fit / 1e6, "r--", linewidth=1, + alpha=0.7, label=f"slope={slope_v:.2f}") + # Linear reference (slope=1) + vol_linear = np.exp(int_v + 1.0 * np.log(tvl_fit)) + ax.plot(tvl_fit / 1e6, vol_linear / 1e6, "k:", linewidth=0.5, + alpha=0.3, label="slope=1 (linear)") + ax.legend(fontsize=8) + + # Panel 3: by fee tier + ax = axes[2] + for _, row in ps.iterrows(): + color = "steelblue" if row["fee"] <= median_fee else "coral" + ax.scatter(row["med_tvl"] / 1e6, row["vol_tvl_pct"], + s=30, alpha=0.7, c=color) + ax.annotate(row["tokens"].split(",")[0], + (row["med_tvl"] / 1e6, row["vol_tvl_pct"]), + fontsize=5, alpha=0.6) + ax.set_xscale("log") + ax.set_xlabel("Median TVL ($M)") + ax.set_ylabel("Median Vol/TVL (%)") + ax.set_title("Vol/TVL by Fee Tier\n" + f"blue=low fee (≤{median_fee:.4f}), red=high fee") + ax.grid(True, alpha=0.3) + + fig.suptitle("Cross-Pool Evidence for Volume Saturation", fontsize=13) + fig.tight_layout() + out = os.path.join(os.path.dirname(os.path.dirname(__file__)), + "results", "mm_noise", "plots", + "cross_pool_vol_tvl.png") + os.makedirs(os.path.dirname(out), exist_ok=True) + fig.savefig(out, dpi=150, bbox_inches="tight") + plt.close(fig) + print(f"\n Saved: {out}") + except Exception as e: + print(f" Plot failed: {e}") + + +if __name__ == "__main__": + main() diff --git a/scripts/plot_mm_noise_fit.py b/scripts/plot_mm_noise_fit.py index a5184ad1..34818e0c 100644 --- a/scripts/plot_mm_noise_fit.py +++ b/scripts/plot_mm_noise_fit.py @@ -44,6 +44,21 @@ def load_model(artifact_dir): return params, meta +def get_pool_K(params, decomp, pool_i): + """Get median K for a pool, handling both per-pool and k_params modes.""" + if "k_params" in params: + k_p = np.array(params["k_params"]) + mask = decomp["pool_idx"] == pool_i + if not mask.any(): + return float(np.exp(k_p[0])) + va = decomp.get("log_vol_a", np.zeros(mask.sum()))[mask] + vb = decomp.get("log_vol_b", np.zeros(mask.sum()))[mask] + log_K = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) + return float(np.exp(np.median(log_K))) + else: + return float(np.exp(params["log_K"][pool_i])) + + def compute_decomposition(params, meta, matched_clean, option_c_clean): """Compute V_arb, V_noise, V_total for all pools.""" from experiments.run_mm_noise import build_mm_data, forward_mm @@ -79,7 +94,9 @@ def compute_decomposition(params, meta, matched_clean, option_c_clean): log_v_noise = np.array(forward_mm( params, jnp.array(data["x_market"]), jnp.array(data["log_tvl"]), - jnp.array(data["pool_idx"]))) + jnp.array(data["pool_idx"]), + log_vol_a=jnp.array(data["log_vol_a"]), + log_vol_b=jnp.array(data["log_vol_b"]))) v_noise = np.exp(log_v_noise) v_total = v_arb + v_noise v_obs = np.exp(y) @@ -102,6 +119,8 @@ def compute_decomposition(params, meta, matched_clean, option_c_clean): "v_total": v_total, "v_obs": v_obs, "log_tvl": log_tvl, + "log_vol_a": data["log_vol_a"], + "log_vol_b": data["log_vol_b"], "tvl": np.exp(log_tvl), } @@ -124,7 +143,7 @@ def plot_pool_timeseries(decomp, params, pool_i, output_dir): v_obs = decomp["v_obs"][mask] tvl = decomp["tvl"][mask] - K_i = float(np.exp(params["log_K"][pool_i])) + K_i = get_pool_K(params, decomp, pool_i) # R² log_pred = np.log(np.maximum(v_total, 1e-10)) @@ -244,12 +263,11 @@ def plot_tvl_response(params, meta, decomp, output_dir): fig, axes = plt.subplots(1, 3, figsize=(18, 6)) - # Select pools with enough data for interesting plots pool_idx_arr = np.array(decomp["pool_idx"]) interesting = [] for i in range(n_pools): n = (pool_idx_arr == i).sum() - if n >= 50: + if n > 0: interesting.append(i) colors = plt.cm.tab20(np.linspace(0, 1, len(interesting))) @@ -257,7 +275,7 @@ def plot_tvl_response(params, meta, decomp, output_dir): # Panel 1: Noise volume vs TVL (absolute) ax = axes[0] for ci, i in enumerate(interesting): - K_i = float(np.exp(params["log_K"][i])) + K_i = get_pool_K(params, decomp, i) mask = pool_idx_arr == i actual_noise = np.median(decomp["v_noise"][mask]) @@ -269,7 +287,7 @@ def plot_tvl_response(params, meta, decomp, output_dir): label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" ax.plot(tvl_range / 1e6, noise_curve / 1e6, color=colors[ci], - linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + linewidth=1.0, alpha=0.7, label=label) # Mark actual TVL ax.scatter([actual_tvl / 1e6], [actual_noise / 1e6], color=colors[ci], s=20, zorder=5) @@ -279,13 +297,13 @@ def plot_tvl_response(params, meta, decomp, output_dir): ax.set_xscale("log") ax.set_yscale("log") ax.set_title("Noise Volume vs TVL (MM saturation)") - ax.legend(fontsize=6, ncol=2) + ax.legend(fontsize=5, ncol=3, loc="best") ax.grid(True, alpha=0.3) # Panel 2: Noise/TVL ratio vs TVL ax = axes[1] for ci, i in enumerate(interesting): - K_i = float(np.exp(params["log_K"][i])) + K_i = get_pool_K(params, decomp, i) mask = pool_idx_arr == i actual_noise = np.median(decomp["v_noise"][mask]) actual_tvl = np.median(decomp["tvl"][mask]) @@ -296,24 +314,24 @@ def plot_tvl_response(params, meta, decomp, output_dir): label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" ax.plot(tvl_range / 1e6, ratio_curve, color=colors[ci], - linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + linewidth=1.0, alpha=0.7, label=label) ax.set_xlabel("TVL ($M)") ax.set_ylabel("Noise / TVL (%)") ax.set_xscale("log") ax.set_title("Noise as Fraction of TVL") - ax.legend(fontsize=6, ncol=2) + ax.legend(fontsize=5, ncol=3, loc="best") ax.grid(True, alpha=0.3) # Panel 3: Elasticity vs TVL ax = axes[2] for ci, i in enumerate(interesting): - K_i = float(np.exp(params["log_K"][i])) + K_i = get_pool_K(params, decomp, i) eps_curve = K_i / (K_i + tvl_range) label = f"{pool_tokens[i][0]}/{pool_tokens[i][1]}" ax.plot(tvl_range / 1e6, eps_curve, color=colors[ci], - linewidth=1.0, alpha=0.7, label=label if ci < 10 else None) + linewidth=1.0, alpha=0.7, label=label) # Mark actual TVL actual_tvl = np.median(decomp["tvl"][pool_idx_arr == i]) eps_actual = K_i / (K_i + actual_tvl) @@ -326,7 +344,7 @@ def plot_tvl_response(params, meta, decomp, output_dir): ax.set_xscale("log") ax.set_ylim(0, 1.05) ax.set_title("TVL Elasticity (K/(K+TVL))") - ax.legend(fontsize=6, ncol=2) + ax.legend(fontsize=5, ncol=3, loc="best") ax.grid(True, alpha=0.3) fig.suptitle("Michaelis-Menten Noise Model — TVL Response", fontsize=13) @@ -351,7 +369,7 @@ def plot_K_distribution(params, meta, decomp, output_dir): for i in range(n_pools): mask = pool_idx_arr == i n = mask.sum() - K_i = float(np.exp(params["log_K"][i])) + K_i = get_pool_K(params, decomp, i) K_vals.append(K_i) tok = pool_tokens[i] labels.append(f"{tok[0]}/{tok[1]}") @@ -417,8 +435,8 @@ def main(): parser.add_argument("--all-pools", action="store_true") parser.add_argument("--artifact-dir", default="results/mm_noise") parser.add_argument("--output-dir", default="results/mm_noise/plots") - parser.add_argument("--top-n", type=int, default=10, - help="Plot top N pools by sample count") + parser.add_argument("--top-n", type=int, default=None, + help="Plot top N pools by sample count (default: all)") args = parser.parse_args() os.environ.setdefault("JAX_PLATFORMS", "cpu") @@ -444,13 +462,13 @@ def main(): if args.pool: targets = [i for i, pid in enumerate(pool_ids) if pid.startswith(args.pool)] - elif args.all_pools: - targets = list(range(len(pool_ids))) - else: - # Top N by sample count + elif args.top_n is not None: counts = [(pool_idx == i).sum() for i in range(len(pool_ids))] targets = sorted(range(len(pool_ids)), key=lambda i: -counts[i]) targets = targets[:args.top_n] + else: + # Default: all pools + targets = list(range(len(pool_ids))) # Per-pool time series print(f"\nPlotting {len(targets)} pools...") From 629ff8301270ce368846f01b00a9ed07f74e1e1e Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 7 Apr 2026 15:29:08 +0100 Subject: [PATCH 080/115] feat: observed competitor TVL as K via DeFi Llama network conductance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New: fetch_competitor_tvl.py fetches historical TVL from DeFi Llama for all competing pools per token pair (same-chain), computes effective K via network conductance model (direct + multi-hop through hub tokens WETH, WSTETH, USDC, USDT, DAI, WBTC with harmonic mean for series combination). Self-exclusion via own TVL subtraction. MM model (run_mm_noise.py) now supports --observed-K flag: K is fixed from DeFi Llama data (0 learned K params), with per-pool alpha + gamma learning the noise level and temporal variation. R²=0.625 with economically meaningful K values (AAVE/WETH: $89M, USDC/WETH: $432M). New noise function: reclamm_mm_observed_noise_volume() in noise_trades.py evaluates V_noise = exp(base) * TVL/(K+TVL) per minute. New array builder: build_mm_simulator_arrays() in noise_model_arrays.py precomputes noise_base + competitor_tvl minute arrays for the simulator. --- experiments/fetch_competitor_tvl.py | 513 ++++++++++++++++++ experiments/run_mm_noise.py | 269 +++++---- quantammsim/calibration/noise_model_arrays.py | 167 ++++++ quantammsim/pools/noise_trades.py | 42 ++ scripts/plot_mm_noise_fit.py | 30 +- 5 files changed, 901 insertions(+), 120 deletions(-) create mode 100644 experiments/fetch_competitor_tvl.py diff --git a/experiments/fetch_competitor_tvl.py b/experiments/fetch_competitor_tvl.py new file mode 100644 index 00000000..7ad3cb3b --- /dev/null +++ b/experiments/fetch_competitor_tvl.py @@ -0,0 +1,513 @@ +"""Fetch competitor TVL for each token pair from DeFi Llama. + +For each of our 36 calibration pools, finds all other DEX pools trading +the same token pair, sums their daily TVL, and saves as a time series. + +K_i(t) = sum_{j != i} TVL_j(t) for all pools trading pool i's pair + +Output: results/competitor_tvl/competitor_tvl.npz + - pool_ids: list of our pool IDs + - dates: array of dates (days since epoch or ISO strings) + - competitor_tvl: (n_dates, n_pools) array of daily competitor TVL in USD + +Usage: + python experiments/fetch_competitor_tvl.py + python experiments/fetch_competitor_tvl.py --cache-dir results/competitor_tvl +""" + +import argparse +import json +import os +import pickle +import sys +import time +from collections import defaultdict + +import numpy as np +import pandas as pd + +CACHE_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "token_factored_calibration", "_cache", +) + +# Map Balancer token names to DeFi Llama symbol conventions +# DeFi Llama symbols are typically uppercase, no dots, no "W" prefix inconsistencies +SYMBOL_MAP = { + "WETH": "WETH", + "WBTC": "WBTC", + "wstETH": "WSTETH", + "waEthLidowstETH": "WSTETH", + "waEthLidoWETH": "WETH", + "waGnowstETH": "WSTETH", + "waGnoGNO": "GNO", + "waBasUSDC": "USDC", + "waBasWETH": "WETH", + "sDAI": "DAI", + "scUSD": "USDC", + "stS": "S", + "JitoSOL": "JITOSOL", + # Common DeFi Llama variants + "USDC.e": "USDC", + "USDT.e": "USDT", + "WETH.e": "WETH", + "WBTC.e": "WBTC", +} + + +def _normalize_symbol(token): + """Normalize Balancer token name to DeFi Llama symbol.""" + return SYMBOL_MAP.get(token, token.upper()) + + +def _fetch_json(url, retries=5, delay=3.0): + """Fetch JSON from URL with exponential backoff.""" + import urllib.request + for attempt in range(retries): + try: + req = urllib.request.Request(url) + req.add_header("User-Agent", "quantammsim/1.0") + with urllib.request.urlopen(req, timeout=30) as resp: + return json.loads(resp.read().decode()) + except Exception as e: + wait = delay * (2 ** attempt) # 3, 6, 12, 24, 48s + if attempt < retries - 1: + print(f" Retry {attempt+1} (wait {wait:.0f}s): {e}") + time.sleep(wait) + else: + raise + + +def fetch_all_pools(local_path=None): + """Load DeFi Llama yield pools from local file or API.""" + if local_path and os.path.exists(local_path): + print(f"Loading DeFi Llama pools from {local_path}...") + with open(local_path) as f: + data = json.load(f) + else: + print("Fetching DeFi Llama pool list from API...") + data = _fetch_json("https://yields.llama.fi/pools") + pools = data.get("data", []) if isinstance(data, dict) else data + print(f" {len(pools)} pools") + return pools + + +# Map Balancer chain names to DeFi Llama chain names +CHAIN_MAP = { + "mainnet": "Ethereum", + "ethereum": "Ethereum", + "arbitrum": "Arbitrum", + "polygon": "Polygon", + "gnosis": "Gnosis", + "base": "Base", + "optimism": "Optimism", + "avalanche": "Avalanche", + "sonic": "Sonic", +} + + +def match_pools(our_pools, llama_pools): + """Match our token pairs to DeFi Llama pools. + + Returns dict: pool_id -> {pair_key, chain, llama_pools_same_chain, + llama_pools_all_chains, tokens} + """ + # Normalize DeFi Llama token symbols to match our convention + LLAMA_NORMALIZE = { + "ETH": "WETH", + "BTC": "WBTC", + "STETH": "WSTETH", + } + + # Index llama pools by (pair, chain) + pair_chain_to_llama = defaultdict(list) + pair_to_llama = defaultdict(list) + for p in llama_pools: + symbol = p.get("symbol", "") + if not symbol or "-" not in symbol: + continue + tokens = symbol.split("-") + if len(tokens) != 2: + continue + normed = [LLAMA_NORMALIZE.get(t.upper(), t.upper()) for t in tokens] + pair_key = tuple(sorted(normed)) + chain = p.get("chain", "") + pair_to_llama[pair_key].append(p) + pair_chain_to_llama[(pair_key, chain)].append(p) + + from quantammsim.calibration.pool_data import _parse_tokens + + matches = {} + for pid, entry in our_pools.items(): + toks = _parse_tokens(entry["tokens"]) + tok_a = _normalize_symbol(toks[0]) + tok_b = _normalize_symbol(toks[1]) if len(toks) > 1 else tok_a + pair_key = tuple(sorted([tok_a, tok_b])) + + our_chain = entry.get("chain", "mainnet") + llama_chain = CHAIN_MAP.get(our_chain.lower(), our_chain) + + matches[pid] = { + "pair_key": pair_key, + "chain": llama_chain, + "llama_pools_same_chain": pair_chain_to_llama.get( + (pair_key, llama_chain), []), + "llama_pools_all_chains": pair_to_llama.get(pair_key, []), + "tokens": (tok_a, tok_b), + } + + return matches, pair_chain_to_llama + + +def fetch_pool_history(pool_id): + """Fetch daily TVL history for a DeFi Llama pool.""" + url = f"https://yields.llama.fi/chart/{pool_id}" + data = _fetch_json(url) + points = data.get("data", []) + if not points: + return None + + dates = [] + tvls = [] + for p in points: + ts = p.get("timestamp", "")[:10] + tvl = p.get("tvlUsd", 0) + if ts and tvl is not None: + dates.append(pd.Timestamp(ts)) + tvls.append(float(tvl)) + + return pd.Series(tvls, index=pd.DatetimeIndex(dates), name="tvl") + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--cache-dir", default="results/competitor_tvl") + parser.add_argument("--max-pools-per-pair", type=int, default=30, + help="Max DeFi Llama pools to fetch per pair") + parser.add_argument("--min-tvl", type=float, default=10000, + help="Skip pools with current TVL below this") + parser.add_argument("--pools-json", default=None, + help="Local DeFi Llama pools.json (skip API fetch)") + args = parser.parse_args() + + os.makedirs(args.cache_dir, exist_ok=True) + + # Load our pools + with open(os.path.join(CACHE_DIR, "stage1.pkl"), "rb") as f: + stage1 = pickle.load(f) + matched_clean = stage1["matched_clean"] + pool_ids = sorted(matched_clean.keys()) + print(f"Our pools: {len(pool_ids)}") + + # Fetch DeFi Llama pools + llama_pools = fetch_all_pools(args.pools_json) + + # Match + matches, pair_chain_to_llama = match_pools(matched_clean, llama_pools) + + # Summary + print(f"\nPair matching (same-chain / all-chains):") + seen = set() + for pid in pool_ids: + m = matches[pid] + pair = m["pair_key"] + chain = m["chain"] + key = (pair, chain) + if key in seen: + continue + seen.add(key) + toks = matched_clean[pid].get("tokens", "?") + n_same = len(m["llama_pools_same_chain"]) + n_all = len(m["llama_pools_all_chains"]) + tvl_same = sum(p.get("tvlUsd", 0) for p in m["llama_pools_same_chain"]) + tvl_all = sum(p.get("tvlUsd", 0) for p in m["llama_pools_all_chains"]) + print(f" {'/'.join(pair):>20s} {chain:>10s}" + f" same={n_same:>3d} (${tvl_same/1e6:>7.1f}M)" + f" all={n_all:>3d} (${tvl_all/1e6:>7.1f}M)" + f" [{toks}]") + + # Flag zero-match pairs — likely symbol mapping issues + zero_matches = [] + for pid in pool_ids: + m = matches[pid] + if len(m["llama_pools_same_chain"]) == 0 and len(m["llama_pools_all_chains"]) == 0: + toks = matched_clean[pid].get("tokens", "?") + zero_matches.append((pid[:16], toks, m["pair_key"], m["chain"])) + if zero_matches: + print(f"\n WARNING: {len(zero_matches)} pools with zero DeFi Llama matches" + f" (check SYMBOL_MAP):") + for pid, toks, pair, chain in zero_matches: + print(f" {pid} {toks:>20s} → {'/'.join(pair)} ({chain})") + + # Fetch historical TVL for each (pair, chain) combination + print(f"\nFetching historical TVL (same-chain)...") + # Key: (pair, chain) -> pd.Series of daily total TVL + pair_chain_histories = {} + + fetched = set() + for pid in pool_ids: + m = matches[pid] + pair = m["pair_key"] + chain = m["chain"] + key = (pair, chain) + if key in fetched: + continue + fetched.add(key) + + # Same-chain pools, sorted by TVL + llama = sorted(m["llama_pools_same_chain"], + key=lambda p: p.get("tvlUsd", 0), reverse=True) + llama = [p for p in llama if p.get("tvlUsd", 0) >= args.min_tvl] + llama = llama[:args.max_pools_per_pair] + + cache_name = f"{'_'.join(pair)}_{chain}_history.pkl" + pair_cache = os.path.join(args.cache_dir, cache_name) + + if not llama: + print(f" {'/'.join(pair)} ({chain}): no qualifying pools") + continue + + if os.path.exists(pair_cache): + with open(pair_cache, "rb") as f: + pair_chain_histories[key] = pickle.load(f) + print(f" {'/'.join(pair)} ({chain}): loaded from cache" + f" ({len(pair_chain_histories[key])} days)") + continue + + print(f" {'/'.join(pair)} ({chain}): fetching {len(llama)} pools...", + end="", flush=True) + pool_series = [] + for lp in llama: + lid = lp["pool"] + try: + hist = fetch_pool_history(lid) + if hist is not None and len(hist) > 10: + pool_series.append(hist) + except Exception as e: + print(f"\n Skip {lid}: {e}", end="") + time.sleep(3.0) # rate limit — DeFi Llama allows ~1 req/3s + print(f" got {len(pool_series)} histories") + + if pool_series: + df = pd.concat(pool_series, axis=1).sort_index() + pair_chain_histories[key] = df.sum(axis=1) # skipna=True: pre-launch = 0 + + with open(pair_cache, "wb") as f: + pickle.dump(pair_chain_histories[key], f) + + # --- Network conductance: fetch hub-pair TVL for multi-hop K --- + HUB_TOKENS = ["WETH", "WSTETH", "USDC", "USDT", "DAI", "WBTC"] + + # Identify all (token, hub) pairs we need across all pools + hub_pairs_needed = set() + for pid in pool_ids: + m = matches[pid] + tok_a, tok_b = m["tokens"] + chain = m["chain"] + for hub in HUB_TOKENS: + if hub in (tok_a, tok_b): + continue + # Need L(tok_a, hub) and L(hub, tok_b) on same chain + pair_ah = tuple(sorted([tok_a, hub])) + pair_hb = tuple(sorted([hub, tok_b])) + hub_pairs_needed.add((pair_ah, chain)) + hub_pairs_needed.add((pair_hb, chain)) + + # Remove pairs we already have + hub_pairs_to_fetch = hub_pairs_needed - fetched + print(f"\nFetching hub-pair TVL for network conductance...") + print(f" {len(hub_pairs_needed)} hub pairs needed," + f" {len(hub_pairs_to_fetch)} to fetch") + + for pair, chain in sorted(hub_pairs_to_fetch): + key = (pair, chain) + cache_name = f"{'_'.join(pair)}_{chain}_history.pkl" + pair_cache = os.path.join(args.cache_dir, cache_name) + + if os.path.exists(pair_cache): + with open(pair_cache, "rb") as f: + pair_chain_histories[key] = pickle.load(f) + continue + + # Find matching DeFi Llama pools + llama = pair_chain_to_llama.get((pair, chain), []) + llama = sorted(llama, key=lambda p: p.get("tvlUsd", 0), reverse=True) + llama = [p for p in llama if p.get("tvlUsd", 0) >= args.min_tvl] + llama = llama[:args.max_pools_per_pair] + + if not llama: + continue + + print(f" {'/'.join(pair)} ({chain}): fetching {len(llama)} pools...", + end="", flush=True) + pool_series = [] + for lp in llama: + lid = lp["pool"] + try: + hist = fetch_pool_history(lid) + if hist is not None and len(hist) > 10: + pool_series.append(hist) + except Exception as e: + print(f"\n Skip {lid}: {e}", end="") + time.sleep(3.0) + print(f" got {len(pool_series)} histories") + + if pool_series: + df = pd.concat(pool_series, axis=1).sort_index() + pair_chain_histories[key] = df.sum(axis=1) # skipna=True: pre-launch = 0 + with open(pair_cache, "wb") as f: + pickle.dump(pair_chain_histories[key], f) + + # Build per-pool competitor TVL arrays aligned to our panel dates + print(f"\nBuilding competitor TVL arrays...") + + # Common date grid + all_dates = set() + for pid in pool_ids: + all_dates.update(matched_clean[pid]["panel"]["date"].values) + date_list = sorted(all_dates) + n_dates = len(date_list) + date_to_idx = {d: i for i, d in enumerate(date_list)} + n_pools = len(pool_ids) + + competitor_tvl = np.full((n_dates, n_pools), np.nan) + + for j, pid in enumerate(pool_ids): + m = matches[pid] + pair = m["pair_key"] + chain = m["chain"] + key = (pair, chain) + if key not in pair_chain_histories: + continue + + hist = pair_chain_histories[key] + panel = matched_clean[pid]["panel"] + panel_dates = panel["date"].values + # Own TVL for self-exclusion: K_i = total_pair_tvl - own_tvl + own_tvl = np.exp(panel["log_tvl_lag1"].values.astype(float)) + + for k, date in enumerate(panel_dates): + t = date_to_idx[date] + day = pd.Timestamp(date).normalize() + if day in hist.index: + total = hist.loc[day] + own = own_tvl[k] if k < len(own_tvl) else 0 + # Competitor TVL = total pair TVL - own TVL (floor at 0) + competitor_tvl[t, j] = max(total - own, 0) + + # Forward-fill then back-fill gaps + for j in range(n_pools): + col = competitor_tvl[:, j] + mask = np.isfinite(col) + if mask.any() and not mask.all(): + s = pd.Series(col, index=date_list).ffill().bfill() + competitor_tvl[:, j] = s.values + + valid = np.isfinite(competitor_tvl) + n_valid = valid.sum() + n_total = n_dates * n_pools + print(f" Coverage: {n_valid}/{n_total} ({100*n_valid/n_total:.0f}%)") + + # Warn about pools with surprisingly low coverage despite having pair data + for j, pid in enumerate(pool_ids): + m = matches[pid] + key = (m["pair_key"], m["chain"]) + if key in pair_chain_histories and len(pair_chain_histories[key]) > 100: + col = competitor_tvl[:, j] + cov = np.isfinite(col).sum() / n_dates + if cov < 0.5: + print(f" WARNING: {pid[:16]} has pair data but only" + f" {cov*100:.0f}% coverage — possible date mismatch") + + # --- Compute K_eff = K_direct + multi-hop contributions --- + print(f"\nComputing network K_eff (direct + multi-hop)...") + # Compute multi-hop contribution over ALL dates in the grid + # (not just panel dates — so forward-fill works correctly) + k_eff = np.full((n_dates, n_pools), np.nan) + + def _get_pair_tvl_on_date(pair_key, chain, day): + """Get total TVL for a pair on a given date.""" + key = (pair_key, chain) + if key not in pair_chain_histories: + return 0.0 + hist = pair_chain_histories[key] + if day in hist.index: + return float(hist.loc[day]) + return 0.0 + + for j, pid in enumerate(pool_ids): + m = matches[pid] + tok_a, tok_b = m["tokens"] + chain = m["chain"] + + for t, date in enumerate(date_list): + day = pd.Timestamp(date).normalize() + + # Direct (from already-computed competitor_tvl) + direct = competitor_tvl[t, j] if np.isfinite(competitor_tvl[t, j]) else 0.0 + + # Multi-hop through hub tokens + multihop = 0.0 + for hub in HUB_TOKENS: + if hub in (tok_a, tok_b): + continue + pair_ah = tuple(sorted([tok_a, hub])) + pair_hb = tuple(sorted([hub, tok_b])) + L_ah = _get_pair_tvl_on_date(pair_ah, chain, day) + L_hb = _get_pair_tvl_on_date(pair_hb, chain, day) + if L_ah > 0 and L_hb > 0: + multihop += L_ah * L_hb / (L_ah + L_hb) + + total = direct + multihop + if total > 0: + k_eff[t, j] = total + + # Forward-fill / back-fill K_eff gaps + for j in range(n_pools): + col = k_eff[:, j] + mask = np.isfinite(col) & (col > 0) + if mask.any() and not mask.all(): + s = pd.Series(col, index=date_list).ffill().bfill() + k_eff[:, j] = s.values + + # Per-pool stats + print(f"\n {'Pool':>16s} {'Tokens':>20s} {'Pair':>20s}" + f" {'K_direct med':>14s} {'K_eff med':>14s} {'Multi/Dir':>10s}") + for j, pid in enumerate(pool_ids): + toks = matched_clean[pid].get("tokens", "?") + pair = matches[pid]["pair_key"] + # Use only panel dates for display (not forward-filled grid dates) + panel_dates = matched_clean[pid]["panel"]["date"].values + panel_t = [date_to_idx[d] for d in panel_dates if d in date_to_idx] + if panel_t: + d_vals = competitor_tvl[panel_t, j] + e_vals = k_eff[panel_t, j] + valid_d = d_vals[np.isfinite(d_vals)] + valid_e = e_vals[np.isfinite(e_vals)] + else: + valid_d = valid_e = np.array([]) + med_d = np.median(valid_d) if len(valid_d) > 0 else 0 + med_e = np.median(valid_e) if len(valid_e) > 0 else 0 + ratio = med_e / med_d if med_d > 0 else float("inf") + print(f" {pid[:16]} {toks:>20s} {'/'.join(pair):>20s}" + f" ${med_d:>13,.0f} ${med_e:>13,.0f} {ratio:>9.1f}x") + + # Save + out_path = os.path.join(args.cache_dir, "competitor_tvl.npz") + np.savez(out_path, + pool_ids=pool_ids, + date_list=np.array([str(d) for d in date_list]), + competitor_tvl=competitor_tvl, + k_eff=k_eff) + print(f"\nSaved: {out_path}") + + # Also save raw pair-chain histories for inspection + pair_path = os.path.join(args.cache_dir, "pair_chain_histories.pkl") + with open(pair_path, "wb") as f: + pickle.dump(pair_chain_histories, f) + print(f"Saved: {pair_path}") + + +if __name__ == "__main__": + main() diff --git a/experiments/run_mm_noise.py b/experiments/run_mm_noise.py index 219c5f28..45323a67 100644 --- a/experiments/run_mm_noise.py +++ b/experiments/run_mm_noise.py @@ -51,16 +51,19 @@ def load_stage1(): return data["matched_clean"], data["option_c_clean"] +COMPETITOR_TVL_PATH = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "competitor_tvl", "competitor_tvl.npz", +) + + def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), - include_cross_pool=False): + include_cross_pool=False, competitor_tvl_path=None): """Build data with MM structure: separate TVL from market features. - Also builds per-sample Binance log-volumes for predicting K. + Loads observed competitor TVL from DeFi Llama for K. """ from experiments.run_linear_market_noise import build_data - from quantammsim.calibration.market_features import ( - _load_binance_daily, TOKEN_MAP, - ) from quantammsim.calibration.pool_data import _parse_tokens # Get full feature matrix from linear model's pipeline @@ -71,8 +74,7 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), include_cross_pool=include_cross_pool, ) - # Remove TVL column and TVL interaction terms — TVL handled by MM, - # interactions subsumed by time-varying K from Binance volumes + # Remove TVL column and TVL interaction terms — TVL handled by MM feat_names = data["feat_names"] x_full = data["x"] tvl_col = feat_names.index("xobs_1") @@ -86,7 +88,7 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), pool_ids = data["pool_ids"] n_pools = data["n_pools"] - # Rebuild raw log_tvl from panel + # Common date grid all_dates = set() for pid in pool_ids: all_dates.update(matched_clean[pid]["panel"]["date"].values) @@ -94,6 +96,7 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), date_to_idx = {d: i for i, d in enumerate(date_list)} n_dates = len(date_list) + # Rebuild raw log_tvl from panel tvl_grid = np.full((n_dates, n_pools), np.nan) for j, pid in enumerate(pool_ids): panel = matched_clean[pid]["panel"] @@ -104,71 +107,88 @@ def build_mm_data(matched_clean, option_c_clean, trend_windows=(7,), pool_idx = data["pool_idx"] day_idx = data["day_idx"] + n_samples = len(pool_idx) log_tvl = np.array([tvl_grid[day_idx[s], pool_idx[s]] - for s in range(len(pool_idx))], dtype=np.float32) - - # Binance daily log-volumes per token, for predicting K - binance_cache = {} - - def _get_binance_daily(symbol): - mapped = TOKEN_MAP.get(symbol, symbol) - if mapped not in binance_cache: - daily = _load_binance_daily(mapped) - if daily is not None: - binance_cache[mapped] = { - d: float(np.log(max(v, 1.0))) - for d, v in daily["volume_usd"].items() - } - else: - binance_cache[mapped] = {} - return binance_cache[mapped] + for s in range(n_samples)], dtype=np.float32) + + # Load observed competitor TVL (K) + comp_path = competitor_tvl_path or COMPETITOR_TVL_PATH + if os.path.exists(comp_path): + comp_data = np.load(comp_path, allow_pickle=True) + comp_pool_ids = list(comp_data["pool_ids"]) + comp_dates = list(comp_data["date_list"]) + # Use K_eff (network conductance) if available, else direct competitor TVL + if "k_eff" in comp_data: + comp_tvl_matrix = comp_data["k_eff"] + print(f" Using network K_eff (direct + multi-hop)") + else: + comp_tvl_matrix = comp_data["competitor_tvl"] + print(f" Using direct competitor TVL only") + + # Build date index for competitor data (normalize to YYYY-MM-DD) + comp_date_to_idx = {} + for ci, d in enumerate(comp_dates): + comp_date_to_idx[str(d)[:10]] = ci + + # Map competitor TVL to our (n_dates, n_pools) grid + comp_tvl_grid = np.full((n_dates, n_pools), np.nan) + for j, pid in enumerate(pool_ids): + if pid not in comp_pool_ids: + continue + cj = comp_pool_ids.index(pid) + for t, date in enumerate(date_list): + date_str = str(pd.Timestamp(date))[:10] + if date_str in comp_date_to_idx: + ci = comp_date_to_idx[date_str] + val = comp_tvl_matrix[ci, cj] + if np.isfinite(val) and val > 0: + comp_tvl_grid[t, j] = val + + # Forward-fill / back-fill gaps per pool + for j in range(n_pools): + col = comp_tvl_grid[:, j] + mask = np.isfinite(col) + if mask.any() and not mask.all(): + s = pd.Series(col, index=date_list).ffill().bfill() + comp_tvl_grid[:, j] = s.values + + # Flag pools with no competitor data + has_comp = np.zeros(n_pools, dtype=bool) + for j in range(n_pools): + has_comp[j] = np.isfinite(comp_tvl_grid[:, j]).any() + + n_with = has_comp.sum() + print(f" Competitor TVL: {n_with}/{n_pools} pools with data") + + # Per-sample log(competitor_tvl), floor at $1 + raw_comp = np.array([ + comp_tvl_grid[day_idx[s], pool_idx[s]] + for s in range(n_samples)], dtype=np.float64) + + # For pools without data, impute with median of pools that have data + valid_comp = raw_comp[np.isfinite(raw_comp) & (raw_comp > 0)] + fallback_val = float(np.median(valid_comp)) if len(valid_comp) > 0 else 1e6 + raw_comp = np.where(np.isfinite(raw_comp) & (raw_comp > 0), + raw_comp, fallback_val) + log_comp_tvl = np.log(np.maximum(raw_comp, 1.0)).astype(np.float32) + print(f" Fallback comp TVL for missing pools: ${fallback_val:,.0f}") + for j in range(n_pools): + if not has_comp[j]: + print(f" No competitor data: {pool_ids[j][:16]}" + f" ({matched_clean[pool_ids[j]].get('tokens', '?')})") + else: + print(f" WARNING: no competitor TVL file at {comp_path}") + log_comp_tvl = np.full(n_samples, np.log(1e6), dtype=np.float32) + has_comp = np.zeros(n_pools, dtype=bool) + # Token info pool_tokens = [] - log_vol_a_grid = np.full((n_dates, n_pools), np.nan) - log_vol_b_grid = np.full((n_dates, n_pools), np.nan) - - for j, pid in enumerate(pool_ids): + for pid in pool_ids: toks = _parse_tokens(matched_clean[pid]["tokens"]) tok_a = toks[0] tok_b = toks[1] if len(toks) > 1 else toks[0] pool_tokens.append((tok_a, tok_b)) - bvol_a = _get_binance_daily(tok_a) - bvol_b = _get_binance_daily(tok_b) - - panel = matched_clean[pid]["panel"] - for k, date in enumerate(panel["date"].values): - t = date_to_idx[date] - day = pd.Timestamp(date).normalize() - if day in bvol_a: - log_vol_a_grid[t, j] = bvol_a[day] - if day in bvol_b: - log_vol_b_grid[t, j] = bvol_b[day] - - # Per-sample Binance volumes (impute missing with per-pool median) - n_samples = len(pool_idx) - log_vol_a = np.array([log_vol_a_grid[day_idx[s], pool_idx[s]] - for s in range(n_samples)], dtype=np.float32) - log_vol_b = np.array([log_vol_b_grid[day_idx[s], pool_idx[s]] - for s in range(n_samples)], dtype=np.float32) - - # Impute NaN with per-pool median - for i in range(n_pools): - mask = pool_idx == i - for arr in (log_vol_a, log_vol_b): - pool_vals = arr[mask] - if np.isnan(pool_vals).all(): - arr[mask] = 20.0 # fallback ~$500M daily vol - elif np.isnan(pool_vals).any(): - arr[mask] = np.where(np.isnan(pool_vals), - np.nanmedian(pool_vals), pool_vals) - - n_missing = np.isnan(log_vol_a).sum() + np.isnan(log_vol_b).sum() - if n_missing > 0: - print(f" WARNING: {n_missing} NaN in Binance volumes after imputation") - log_vol_a = np.nan_to_num(log_vol_a, nan=20.0) - log_vol_b = np.nan_to_num(log_vol_b, nan=20.0) - removed_names = [feat_names[i] for i in sorted(remove_cols)] print(f" Removed: {removed_names}") print(f" Market features ({len(market_names)}): {market_names}") @@ -176,8 +196,8 @@ def _get_binance_daily(symbol): return { "x_market": x_market, "log_tvl": log_tvl, - "log_vol_a": log_vol_a, - "log_vol_b": log_vol_b, + "log_comp_tvl": log_comp_tvl, + "has_comp": has_comp, "y_total": data["y_total"], "pool_idx": pool_idx, "day_idx": day_idx, @@ -197,13 +217,14 @@ def _get_binance_daily(symbol): # ---- Model ---- -def forward_mm(params, x_market, log_tvl, pool_idx, - log_vol_a=None, log_vol_b=None): +def forward_mm(params, x_market, log_tvl, pool_idx, log_comp_tvl=None): """MM forward pass → log(V_noise) per sample. - Supports two K modes: + K modes (checked in order): + - Observed: log_comp_tvl provided + params has "k_scale" (2,) + K = exp(k_scale[0] + k_scale[1] * log_comp_tvl) - Per-pool: params contains "log_K" (n_pools,) - - Binance-volume: params contains "k_params" (3,) + log_vol_a/b + - Shared k_params: params contains "k_params" (3,) [legacy] """ log_alpha = params["log_alpha"] gamma = params["gamma"] @@ -211,15 +232,22 @@ def forward_mm(params, x_market, log_tvl, pool_idx, alpha_i = log_alpha[pool_idx] tvl = jnp.exp(log_tvl) - # K: per-pool or from Binance volumes - if "k_params" in params: - k_params = params["k_params"] - vol_min = jnp.minimum(log_vol_a, log_vol_b) - vol_max = jnp.maximum(log_vol_a, log_vol_b) - log_K = k_params[0] + k_params[1] * vol_min + k_params[2] * vol_max + # K + if log_comp_tvl is not None and "k_scale" in params: + # Observed competitor TVL with learned scale/offset + k_s = params["k_scale"] + log_K = k_s[0] + k_s[1] * log_comp_tvl K = jnp.exp(log_K) - else: + elif log_comp_tvl is not None and "k_scale" not in params and "log_K" not in params: + # Observed competitor TVL, used directly as K + K = jnp.exp(log_comp_tvl) + elif "log_K" in params: K = jnp.exp(params["log_K"][pool_idx]) + elif "k_params" in params: + # Legacy Binance-volume mode (kept for loading old models) + K = jnp.exp(params["k_params"][0]) + else: + K = jnp.exp(jnp.array(14.5)) # fallback # Market features: shared or per-pool gamma if gamma.ndim == 2: @@ -236,7 +264,7 @@ def make_loss_fn(pool_coeffs, pool_gas, n_pools): """Loss with PCHIP arb + MM noise.""" from quantammsim.calibration.grid_interpolation import interpolate_pool_daily - def loss_fn(params, x_market, log_tvl, log_vol_a, log_vol_b, y_total, + def loss_fn(params, x_market, log_tvl, log_comp_tvl, y_total, sample_grid_days, pool_idx, l2_alpha, huber_delta): log_cadence = params["log_cadence"] @@ -255,7 +283,7 @@ def loss_fn(params, x_market, log_tvl, log_vol_a, log_vol_b, y_total, # V_noise from MM log_v_noise = forward_mm( params, x_market, log_tvl, pool_idx, - log_vol_a=log_vol_a, log_vol_b=log_vol_b) + log_comp_tvl=log_comp_tvl) # V_total log_v_total = jnp.logaddexp(log_v_arb, log_v_noise) @@ -299,15 +327,14 @@ def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, x_market = jnp.array(data["x_market"]) log_tvl = jnp.array(data["log_tvl"]) - log_vol_a = jnp.array(data["log_vol_a"]) - log_vol_b = jnp.array(data["log_vol_b"]) + log_comp_tvl = jnp.array(data["log_comp_tvl"]) y_total = jnp.array(data["y_total"]) sgd = jnp.array(data["sample_grid_days"]) pidx = jnp.array(data["pool_idx"]) for epoch in range(n_epochs): loss, grads = grad_fn( - params, x_market, log_tvl, log_vol_a, log_vol_b, + params, x_market, log_tvl, log_comp_tvl, y_total, sgd, pidx, l2_alpha, huber_delta) for k in params: @@ -320,7 +347,10 @@ def train(params, data, grad_fn, n_epochs, lr, l2_alpha, huber_delta, if verbose and (epoch % 200 == 0 or epoch == n_epochs - 1): ev = evaluate(params, data) - if "k_params" in params: + if "k_scale" in params: + ks = np.array(params["k_scale"]) + k_str = f" k_s=[{ks[0]:.2f},{ks[1]:.3f}]" + elif "k_params" in params: k_p = np.array(params["k_params"]) k_str = f" k=[{k_p[0]:.2f},{k_p[1]:.3f},{k_p[2]:.3f}]" else: @@ -364,8 +394,7 @@ def evaluate(params, data): params, jnp.array(data["x_market"]), jnp.array(data["log_tvl"]), jnp.array(data["pool_idx"]), - log_vol_a=jnp.array(data["log_vol_a"]), - log_vol_b=jnp.array(data["log_vol_b"]))) + log_comp_tvl=jnp.array(data["log_comp_tvl"]))) log_v_total = np.logaddexp(log_v_arb, log_v_noise) v_noise = np.exp(log_v_noise) @@ -385,22 +414,34 @@ def evaluate(params, data): noise_shares[data["pool_ids"][i]] = float(np.median( v_noise[mask] / v_total[mask])) - # Per-pool K values + # Per-pool median K K_values = {} - if "k_params" in params: - k_p = np.array(params["k_params"]) + if "k_scale" in params: + ks = np.array(params["k_scale"]) for i in range(n_pools): mask = pool_idx == i if not mask.any(): - K_values[data["pool_ids"][i]] = float(np.exp(k_p[0])) + K_values[data["pool_ids"][i]] = 0 continue - va = data["log_vol_a"][mask] - vb = data["log_vol_b"][mask] - log_K_i = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) + lc = data["log_comp_tvl"][mask] + log_K_i = ks[0] + ks[1] * lc K_values[data["pool_ids"][i]] = float(np.exp(np.median(log_K_i))) - else: + elif "log_K" in params: for i in range(n_pools): K_values[data["pool_ids"][i]] = float(np.exp(params["log_K"][i])) + elif "k_params" in params: + k_p = np.array(params["k_params"]) + for i in range(n_pools): + K_values[data["pool_ids"][i]] = float(np.exp(k_p[0])) + else: + # Observed K: compute from log_comp_tvl directly + for i in range(n_pools): + mask = pool_idx == i + if mask.any(): + K_values[data["pool_ids"][i]] = float( + np.exp(np.median(data["log_comp_tvl"][mask]))) + else: + K_values[data["pool_ids"][i]] = 1e6 return { "r2s": r2s, @@ -431,15 +472,18 @@ def tvl_response_check(params, data): if mask.sum() == 0: continue - # Per-pool K - if "k_params" in params: - k_p = np.array(params["k_params"]) - va = data["log_vol_a"][mask] - vb = data["log_vol_b"][mask] - log_K_i = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) - K_i = float(np.exp(np.median(log_K_i))) - else: + # Per-pool K (median) + if "k_scale" in params: + ks = np.array(params["k_scale"]) + lc = data["log_comp_tvl"][mask] + K_i = float(np.exp(np.median(ks[0] + ks[1] * lc))) + elif "log_K" in params: K_i = float(np.exp(params["log_K"][i])) + elif "k_params" in params: + K_i = float(np.exp(np.array(params["k_params"])[0])) + else: + # Observed K directly from competitor TVL + K_i = float(np.exp(np.median(data["log_comp_tvl"][mask]))) x_med = np.median(data["x_market"][mask], axis=0) gamma = np.array(params["gamma"]) @@ -477,8 +521,9 @@ def main(): parser.add_argument("--init-log-K", type=float, default=17.0, help="Initial log(K) ~ log($24M)") parser.add_argument("--shared-K", action="store_true", - help="Predict K from Binance volumes (3 shared params)" - " instead of per-pool K") + help="Predict K from Binance volumes (3 shared params)") + parser.add_argument("--observed-K", action="store_true", + help="Use observed competitor TVL from DeFi Llama as K") parser.add_argument("--per-pool-gamma", action="store_true", help="Per-pool market feature coefficients") parser.add_argument("--no-split", action="store_true") @@ -554,7 +599,11 @@ def main(): "gamma": gamma_init, "log_cadence": jnp.array(data["init_log_cadences"]), } - if args.shared_K: + if args.observed_K: + # K = competitor_tvl directly. No learned params for K. + # log_comp_tvl is passed as data, not as a parameter. + pass + elif args.shared_K: params["k_params"] = jnp.array([args.init_log_K, 0.0, 0.0]) else: params["log_K"] = jnp.full(n_pools, args.init_log_K) @@ -607,12 +656,18 @@ def _ridge(X, y, alpha=1.0): train_eval = evaluate(params, train_data) print(f" Median R²: {train_eval['median_r2']:.4f}") - if "k_params" in params: + if "k_scale" in params: + ks = np.array(params["k_scale"]) + print(f" Observed K: offset={ks[0]:.3f}, slope={ks[1]:.3f}") + elif "k_params" in params: k_p = np.array(params["k_params"]) print(f" k_params: k_0={k_p[0]:.2f}, k_min={k_p[1]:.4f}, k_max={k_p[2]:.4f}") - else: + elif "log_K" in params: K_med = float(np.exp(np.median(np.array(params["log_K"])))) print(f" Per-pool K: median=${K_med/1e6:.1f}M") + else: + K_med = float(np.median(list(train_eval["K_values"].values()))) + print(f" Observed K (fixed): median=${K_med/1e6:.1f}M") print(f"\n {'Pool':>16s} {'Tokens':>16s} {'R²':>6s}" f" {'Noise%':>7s} {'K ($M)':>10s}") diff --git a/quantammsim/calibration/noise_model_arrays.py b/quantammsim/calibration/noise_model_arrays.py index 0c7c904e..7b73dc8e 100644 --- a/quantammsim/calibration/noise_model_arrays.py +++ b/quantammsim/calibration/noise_model_arrays.py @@ -291,3 +291,170 @@ def build_simulator_arrays( "coeffs": coeffs, "tvl_col": tvl_col, } + + +def build_mm_simulator_arrays( + token_a: str, + token_b: str, + start_date: str, + end_date: str, + mm_artifact_dir: str = "results/mm_noise", + competitor_tvl_path: str = "results/competitor_tvl/competitor_tvl.npz", + pool_id: Optional[str] = None, +) -> Dict: + """Build noise_base and competitor_tvl arrays for the MM simulator. + + The MM noise model evaluates:: + + V_noise = exp(noise_base_t) * TVL / (K_t + TVL) + + where noise_base_t = alpha_i + gamma_i @ x_market_t absorbs all + non-TVL terms, and K_t = competitor_tvl_t is observed from DeFi + Llama (network conductance model: direct + multi-hop). + + Parameters + ---------- + token_a, token_b : str + Token symbols. + start_date, end_date : str + Date range. + mm_artifact_dir : str + Directory with MM model.npz and meta.json. + competitor_tvl_path : str + Path to competitor_tvl.npz from fetch_competitor_tvl.py. + pool_id : str, optional + Pool ID for per-pool alpha/gamma. + + Returns + ------- + dict with noise_base, competitor_tvl (minute arrays), dates, etc. + """ + # Load MM model + art, meta = load_artifact(mm_artifact_dir) + pool_ids = meta["pool_ids"] + market_names = meta["market_names"] + n_market = meta["n_market_feat"] + per_pool_gamma = meta.get("per_pool_gamma", False) + + pool_idx = -1 + if pool_id is not None: + pool_idx = _find_pool_index(pool_id, pool_ids) + + log_alpha = art["log_alpha"] + gamma = art["gamma"] + + if pool_idx >= 0: + alpha_i = float(log_alpha[pool_idx]) + gamma_i = gamma[pool_idx] if per_pool_gamma else gamma + print(f" MM model: pool idx {pool_idx}, alpha={alpha_i:.3f}") + else: + alpha_i = float(np.median(log_alpha)) + gamma_i = np.median(gamma, axis=0) if per_pool_gamma else gamma + print(f" MM model: pool not found, using median alpha={alpha_i:.3f}") + + # Build daily market features from Binance + # The MM model uses the same features as the linear model minus TVL + # We need x_mean/x_std from the linear model artifact for standardization + linear_art_dir = os.path.join( + os.path.dirname(os.path.dirname(mm_artifact_dir)), + "results", "linear_market_noise") + if os.path.exists(os.path.join(linear_art_dir, "model.npz")): + lin_art, lin_meta = load_artifact(linear_art_dir) + x_mean = lin_art["x_mean"] + x_std = lin_art["x_std"] + feat_names = lin_meta["feat_names"] + else: + # Fallback: try to get from MM artifact + x_mean = art.get("x_mean", np.zeros(n_market)) + x_std = art.get("x_std", np.ones(n_market)) + feat_names = market_names + + trend_windows = (7,) + + print(f" Building features from Binance: {token_a}/{token_b}," + f" {start_date} → {end_date}") + x_daily, dates = build_daily_features_from_binance( + token_a, token_b, start_date, end_date, + feat_names, x_mean, x_std, trend_windows, + ) + n_days = len(dates) + + # Extract market features (exclude TVL and TVL interactions) + tvl_col = None + tvl_interaction_cols = set() + for i, name in enumerate(feat_names): + if name == "xobs_1": + tvl_col = i + elif name.startswith("xobs_1\u00d7"): + tvl_interaction_cols.add(i) + + keep_cols = [i for i in range(len(feat_names)) + if i != tvl_col and i not in tvl_interaction_cols] + + # Map market_names to x_daily columns + x_market_daily = np.zeros((n_days, n_market), dtype=np.float32) + for mi, mname in enumerate(market_names): + # Find mname in feat_names + for fi, fname in enumerate(feat_names): + if fname == mname and fi in keep_cols: + col_in_daily = fi + x_market_daily[:, mi] = x_daily[:, col_in_daily] + break + + # Compute noise_base = alpha_i + gamma_i @ x_market + noise_base_daily = alpha_i + x_market_daily @ gamma_i + noise_base_daily = noise_base_daily.astype(np.float64) + + # Load competitor TVL (K) + print(f" Loading competitor TVL from {competitor_tvl_path}") + comp_data = np.load(competitor_tvl_path, allow_pickle=True) + comp_pool_ids = list(comp_data["pool_ids"]) + comp_dates = list(comp_data["date_list"]) + k_eff = comp_data["k_eff"] # (n_comp_dates, n_comp_pools) + + # Find pool in competitor data + comp_pool_idx = -1 + if pool_id is not None: + comp_pool_idx = _find_pool_index(pool_id, comp_pool_ids) + + if comp_pool_idx < 0: + print(f" WARNING: pool not in competitor TVL data, using K=$10M") + K_daily = np.full(n_days, 10e6, dtype=np.float64) + else: + # Build date index for competitor data + comp_date_to_idx = {} + for ci, d in enumerate(comp_dates): + comp_date_to_idx[str(d)[:10]] = ci + + K_daily = np.full(n_days, np.nan, dtype=np.float64) + for k, day in enumerate(dates): + ds = str(pd.Timestamp(day))[:10] + if ds in comp_date_to_idx: + ci = comp_date_to_idx[ds] + val = k_eff[ci, comp_pool_idx] + if np.isfinite(val) and val > 0: + K_daily[k] = val + + # Forward-fill / back-fill + s = pd.Series(K_daily).ffill().bfill() + K_daily = s.values.astype(np.float64) + + # Floor + K_daily = np.maximum(K_daily, 1.0) + med_K = np.median(K_daily[np.isfinite(K_daily)]) + print(f" K (competitor TVL): median=${med_K:,.0f}," + f" range=[${K_daily.min():,.0f}, ${K_daily.max():,.0f}]") + + # Expand to minute resolution + n_minutes = n_days * 1440 + noise_base = np.repeat(noise_base_daily, 1440) + competitor_tvl_array = np.repeat(K_daily, 1440) + + return { + "noise_base": noise_base, + "competitor_tvl": competitor_tvl_array, + "dates": dates, + "pool_index": pool_idx, + "n_days": n_days, + "n_minutes": n_minutes, + } diff --git a/quantammsim/pools/noise_trades.py b/quantammsim/pools/noise_trades.py index db8d1b9c..9c9f00f9 100644 --- a/quantammsim/pools/noise_trades.py +++ b/quantammsim/pools/noise_trades.py @@ -404,3 +404,45 @@ def reclamm_market_linear_noise_volume( return jnp.maximum(0.0, daily_noise / 1440.0) +@jit +def reclamm_mm_observed_noise_volume( + effective_value_usd, + noise_base, + competitor_tvl, +): + """Michaelis-Menten noise model with observed competitor TVL as K. + + Derived from optimal routing (Diamandis et al. 2023):: + + V_noise = exp(base_t) * TVL / (K_t + TVL) + + where K_t is observed total competitor liquidity (direct + multi-hop + network conductance) from DeFi Llama, and base_t absorbs per-pool + intercept + market feature effects. + + The MM form guarantees: + - Elasticity ≈ 1 at low TVL (TVL << K) + - Structural saturation at high TVL (V_noise → exp(base_t)) + - No wireheading: V_noise is bounded regardless of concentration + + Parameters + ---------- + effective_value_usd : float + Effective TVL in USD: (Ra+Va)*pA + (Rb+Vb)*pB. + noise_base : float + Precomputed log(V_max_daily) = alpha_i + gamma_i @ x_market_t. + competitor_tvl : float + Observed competitor TVL (K) for this step, from DeFi Llama + network conductance model. + + Returns + ------- + float + Per-minute noise volume (USD), floored at zero. + """ + tvl = jnp.maximum(effective_value_usd, 1.0) + K = jnp.maximum(competitor_tvl, 1.0) + daily_noise = jnp.exp(noise_base) * tvl / (K + tvl) + return jnp.maximum(0.0, daily_noise / 1440.0) + + diff --git a/scripts/plot_mm_noise_fit.py b/scripts/plot_mm_noise_fit.py index 34818e0c..6d1b3319 100644 --- a/scripts/plot_mm_noise_fit.py +++ b/scripts/plot_mm_noise_fit.py @@ -45,18 +45,24 @@ def load_model(artifact_dir): def get_pool_K(params, decomp, pool_i): - """Get median K for a pool, handling both per-pool and k_params modes.""" - if "k_params" in params: - k_p = np.array(params["k_params"]) - mask = decomp["pool_idx"] == pool_i + """Get median K for a pool, handling all K modes.""" + mask = decomp["pool_idx"] == pool_i + if "k_scale" in params: + ks = np.array(params["k_scale"]) if not mask.any(): - return float(np.exp(k_p[0])) - va = decomp.get("log_vol_a", np.zeros(mask.sum()))[mask] - vb = decomp.get("log_vol_b", np.zeros(mask.sum()))[mask] - log_K = k_p[0] + k_p[1] * np.minimum(va, vb) + k_p[2] * np.maximum(va, vb) + return float(np.exp(ks[0])) + lc = decomp.get("log_comp_tvl", np.zeros(mask.sum()))[mask] + log_K = ks[0] + ks[1] * lc return float(np.exp(np.median(log_K))) - else: + elif "log_K" in params: return float(np.exp(params["log_K"][pool_i])) + elif "k_params" in params: + k_p = np.array(params["k_params"]) + return float(np.exp(k_p[0])) + elif "log_comp_tvl" in decomp and mask.any(): + # Observed K directly from competitor TVL + return float(np.exp(np.median(decomp["log_comp_tvl"][mask]))) + return np.exp(14.5) def compute_decomposition(params, meta, matched_clean, option_c_clean): @@ -95,8 +101,7 @@ def compute_decomposition(params, meta, matched_clean, option_c_clean): params, jnp.array(data["x_market"]), jnp.array(data["log_tvl"]), jnp.array(data["pool_idx"]), - log_vol_a=jnp.array(data["log_vol_a"]), - log_vol_b=jnp.array(data["log_vol_b"]))) + log_comp_tvl=jnp.array(data["log_comp_tvl"]))) v_noise = np.exp(log_v_noise) v_total = v_arb + v_noise v_obs = np.exp(y) @@ -119,8 +124,7 @@ def compute_decomposition(params, meta, matched_clean, option_c_clean): "v_total": v_total, "v_obs": v_obs, "log_tvl": log_tvl, - "log_vol_a": data["log_vol_a"], - "log_vol_b": data["log_vol_b"], + "log_comp_tvl": data["log_comp_tvl"], "tvl": np.exp(log_tvl), } From d3632eca915fcf88fad07086b4fbba8eecf7b79c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 7 Apr 2026 15:45:41 +0100 Subject: [PATCH 081/115] feat: integrate mm_observed noise model into simulator pipeline Wire reclamm_mm_observed_noise_volume through the full simulator: - reclamm.py: load noise_base + competitor_tvl arrays from npz - reclamm_reserves.py: dispatch mm_observed in scan step, append competitor_tvl to scan_inputs alongside noise_base - noise_model_arrays.py: build_mm_simulator_arrays() precomputes both arrays from MM model artifact + DeFi Llama competitor TVL - jax_runner_utils.py: add noise array keys to _TRAINING_ONLY_FIELDS Fingerprint usage: noise_model="mm_observed", noise_arrays_path="path/to/arrays.npz" --- quantammsim/pools/reCLAMM/reclamm.py | 26 +++++++++++++++- quantammsim/pools/reCLAMM/reclamm_reserves.py | 31 +++++++++++++++++++ quantammsim/runners/jax_runner_utils.py | 4 +++ 3 files changed, 60 insertions(+), 1 deletion(-) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index 32e0c3e8..c9186045 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -223,10 +223,32 @@ def _prepare_noise_arrays(self, prices, run_fingerprint, start_index, - "tsoukalas_*"/"loglinear": {"volatility": array} - "calibrated": {"volatility": array, "dow_sin": array, "dow_cos": array} - "market_linear": {"noise_base": array, "noise_tvl_coeff": array} + - "mm_observed": {"noise_base": array, "competitor_tvl": array} """ noise_model = run_fingerprint.get("noise_model", "ratio") result = {"volatility": None, "dow_sin": None, "dow_cos": None, - "noise_base": None, "noise_tvl_coeff": None} + "noise_base": None, "noise_tvl_coeff": None, + "competitor_tvl": None} + + if noise_model == "mm_observed": + # MM model with observed competitor TVL as K + nb = run_fingerprint.get("noise_base_array") + ct = run_fingerprint.get("competitor_tvl_array") + if nb is None and "noise_arrays_path" in run_fingerprint: + path = run_fingerprint["noise_arrays_path"] + if not hasattr(self, "_mm_observed_cache") or self._mm_observed_cache[0] != path: + arrays = np.load(path) + self._mm_observed_cache = ( + path, arrays["noise_base"], arrays["competitor_tvl"]) + nb = self._mm_observed_cache[1] + ct = self._mm_observed_cache[2] + if nb is not None: + result["noise_base"] = _prepare_dynamic_array( + jnp.array(nb), start_index, bout_length, arb_freq, max_len) + if ct is not None: + result["competitor_tvl"] = _prepare_dynamic_array( + jnp.array(ct), start_index, bout_length, arb_freq, max_len) + return result if noise_model == "market_linear": # Load precomputed arrays from path (cached on instance) or direct @@ -340,6 +362,7 @@ def calculate_reserves_with_fees( dow_cos_array=dow_cos, noise_base_array=_na["noise_base"], noise_tvl_coeff_array=_na["noise_tvl_coeff"], + competitor_tvl_array=_na.get("competitor_tvl"), ) return jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape) @@ -411,6 +434,7 @@ def calculate_reserves_and_fee_revenue_with_fees( dow_cos_array=dow_cos, noise_base_array=_na["noise_base"], noise_tvl_coeff_array=_na["noise_tvl_coeff"], + competitor_tvl_array=_na.get("competitor_tvl"), ) return ( jnp.broadcast_to(s.initial_reserves, s.arb_prices.shape), diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 93632551..ac264e80 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -37,6 +37,7 @@ reclamm_loglinear_noise_volume, reclamm_calibrated_noise_volume, reclamm_market_linear_noise_volume, + reclamm_mm_observed_noise_volume, ) # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) @@ -1044,6 +1045,20 @@ def _skip_schedule_state(_): tvl_std=_np.get("tvl_std", 1.0), ) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + Ra_new = (Ra_new + Va) * scale - Va + Rb_new = (Rb_new + Vb) * scale - Vb + elif noise_model == "mm_observed": + noise_base = input_list[9] + competitor_tvl = input_list[10] + effective_value = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + + noise_vol = reclamm_mm_observed_noise_volume( + effective_value, noise_base, competitor_tvl, + ) + minutes_per_step = seconds_per_step / 60.0 noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) @@ -1306,6 +1321,7 @@ def _jax_calc_reclamm_reserves_with_fees( dow_cos_array=None, noise_base_array=None, noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """Calculate reClAMM reserves over time with fees. @@ -1371,6 +1387,9 @@ def _jax_calc_reclamm_reserves_with_fees( elif noise_model == "market_linear": scan_inputs.append(noise_base_array) scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) carry_init = [ initial_reserves, @@ -1416,6 +1435,7 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( dow_cos_array=None, noise_base_array=None, noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """Calculate reClAMM reserves with time-varying fees/arb arrays.""" if lp_supply_array is None: @@ -1490,6 +1510,9 @@ def _jax_calc_reclamm_reserves_with_dynamic_inputs( elif noise_model == "market_linear": scan_inputs.append(noise_base_array) scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) carry_init = [ initial_reserves, @@ -1625,6 +1648,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( dow_cos_array=None, noise_base_array=None, noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """Calculate reClAMM reserves and LP fee revenue over time with fees. @@ -1692,6 +1716,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( elif noise_model == "market_linear": scan_inputs.append(noise_base_array) scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) carry_init = [ initial_reserves, @@ -1737,6 +1764,7 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( dow_cos_array=None, noise_base_array=None, noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """Calculate reClAMM reserves and LP fee revenue with time-varying fees/arb arrays. @@ -1817,6 +1845,9 @@ def _jax_calc_reclamm_reserves_and_fee_revenue_with_dynamic_inputs( elif noise_model == "market_linear": scan_inputs.append(noise_base_array) scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) carry_init = [ initial_reserves, diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index 1c93887f..2f931903 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -675,6 +675,10 @@ def __eq__(self, other): "initial_raw_width", "initial_raw_exponents", "initial_pre_exp_scaling", + # Noise model arrays — loaded from path at runtime, not hashable + "noise_base_array", + "noise_tvl_coeff_array", + "competitor_tvl_array", }) From 76bcc96d2647fc7dab4add4cbd9fd0f86b246a74 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 7 Apr 2026 15:48:29 +0100 Subject: [PATCH 082/115] feat: add mm_observed noise model to reClAMM tuning pipeline tune_reclamm_calibrated_noise.py now supports --noise-model mm_observed which uses the Michaelis-Menten model with observed competitor TVL from DeFi Llama as K. Precomputes noise_base + competitor_tvl arrays via build_mm_simulator_arrays, saves to npz, passes path in fingerprint. Usage: python experiments/tune_reclamm_calibrated_noise.py \ --noise-model mm_observed \ --artifact-dir results/mm_noise \ --competitor-tvl-path results/competitor_tvl/competitor_tvl.npz --- experiments/tune_reclamm_calibrated_noise.py | 69 ++++++++++++++++++-- 1 file changed, 64 insertions(+), 5 deletions(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index c2d712c4..55ca99b8 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -114,6 +114,54 @@ def _build_market_linear_arrays(args): return arrays_path, max(1, round(learned_cadence)) +def _build_mm_observed_arrays(args): + """Precompute noise arrays from the MM model + DeFi Llama competitor TVL.""" + from quantammsim.calibration.noise_model_arrays import ( + build_mm_simulator_arrays, load_artifact, _find_pool_index, + ) + + start = args.start_date.split(" ")[0] + end = args.end_test_date.split(" ")[0] + + print(f" Building mm_observed noise arrays for {POOL_ID}...") + print(f" Date range: {start} → {end}") + arrays = build_mm_simulator_arrays( + token_a="AAVE", + token_b="ETH", + start_date=start, + end_date=end, + mm_artifact_dir=args.artifact_dir, + competitor_tvl_path=args.competitor_tvl_path, + pool_id=POOL_ID, + ) + print(f" {arrays['n_days']} days, {arrays['n_minutes']} minutes") + print(f" noise_base range: [{arrays['noise_base'].min():.2f}," + f" {arrays['noise_base'].max():.2f}]") + print(f" competitor_tvl range: [${np.exp(np.log(arrays['competitor_tvl'].max())):.0f}]") + + # Save arrays to disk + import os + cache_dir = os.path.join(args.artifact_dir, "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join(cache_dir, f"{POOL_ID}_{start}_{end}_mm.npz") + np.savez(arrays_path, + noise_base=arrays["noise_base"], + competitor_tvl=arrays["competitor_tvl"]) + print(f" Saved arrays: {arrays_path}") + + # Get cadence from MM model artifact + art, meta = load_artifact(args.artifact_dir) + pool_idx = _find_pool_index(POOL_ID, meta["pool_ids"]) + if pool_idx >= 0 and "log_cadence" in art: + learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) + print(f" Learned cadence: {learned_cadence:.1f} min") + else: + learned_cadence = 5.0 + print(f" Using default cadence: {learned_cadence}") + + return arrays_path, max(1, round(learned_cadence)) + + def _build_opt_settings(args): """Build optimisation_settings for optuna, bfgs, or cma_es.""" if args.method == "bfgs": @@ -163,8 +211,14 @@ def _build_opt_settings(args): def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): """Build run fingerprint with calibrated noise model.""" - if args.noise_model == "market_linear" and noise_arrays_path is not None: - # Load tvl standardization stats from the saved arrays + if args.noise_model == "mm_observed" and noise_arrays_path is not None: + noise_block = { + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": noise_arrays_path, + } + freq = arb_freq or 5 + elif args.noise_model == "market_linear" and noise_arrays_path is not None: _arr = np.load(noise_arrays_path) noise_block = { "noise_trader_ratio": 0.0, @@ -259,11 +313,14 @@ def main(): parser.add_argument("--min-train-ret", type=float, default=-0.5, help="Reject trials with IS returns_over_hodl below this") parser.add_argument("--noise-model", default="market_linear", - choices=["calibrated", "market_linear"], + choices=["calibrated", "market_linear", "mm_observed"], help="Noise model variant") parser.add_argument("--artifact-dir", default="results/linear_market_noise", - help="Artifact dir for market_linear model") + help="Artifact dir for market_linear or mm_observed model") + parser.add_argument("--competitor-tvl-path", + default="results/competitor_tvl/competitor_tvl.npz", + help="Path to competitor TVL data (mm_observed only)") parser.add_argument("--initial-pool-value", type=float, default=20_000_000.0, help="Initial pool TVL in USD (default: 20M)") parser.add_argument("--fees", type=float, default=0.0025, @@ -293,11 +350,13 @@ def main(): else: objectives = [args.objective] - # Precompute noise arrays once (if using market_linear) + # Precompute noise arrays once noise_arrays_path = None arb_freq = None if args.noise_model == "market_linear": noise_arrays_path, arb_freq = _build_market_linear_arrays(args) + elif args.noise_model == "mm_observed": + noise_arrays_path, arb_freq = _build_mm_observed_arrays(args) all_results = {} for obj in objectives: From b06162fabfd43256f96c2a84cc9c02597faa2c3b Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 10 Apr 2026 10:46:03 +0100 Subject: [PATCH 083/115] =?UTF-8?q?feat:=20daily=5Flog=5Fsharpe=5Fexcess?= =?UTF-8?q?=20=E2=80=94=20Sharpe=20on=20excess=20returns=20over=20HODL?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New metric isolates LP alpha by computing Sharpe on the difference between pool and HODL daily log returns. Removes crypto beta so the optimizer can't get credit for market direction — only the pool strategy's actual contribution to returns. e_t = log(V_pool_t / V_pool_{t-1}) - log(V_hodl_t / V_hodl_{t-1}) S_excess = sqrt(365) * mean(e_t) / (std(e_t) + eps) Also adds calmar, sterling, weekly_rovar to tune script objectives. --- experiments/tune_reclamm_calibrated_noise.py | 6 ++- quantammsim/core_simulator/forward_pass.py | 45 ++++++++++++++++++++ 2 files changed, 50 insertions(+), 1 deletion(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index 55ca99b8..c0d04771 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -61,7 +61,11 @@ "shift_exponent": {"low": 1e-5, "high": 125.0, "log_scale": True, "scalar": True}, } -OBJECTIVES = ["daily_log_sharpe", "returns_over_hodl", "fee_revenue_over_value"] +OBJECTIVES = [ + "daily_log_sharpe", "daily_log_sharpe_excess", + "returns_over_hodl", "fee_revenue_over_value", + "calmar", "sterling", "weekly_rovar", +] def _build_market_linear_arrays(args): diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index e99f0517..f207a657 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -112,6 +112,7 @@ def _apply_price_noise(prices, sigma, seed_int): DAILY_COMPATIBLE_METRICS = frozenset({ # Sharpe / VaR / ROVAR metrics naturally operate on day-boundary values. "daily_log_sharpe", + # "daily_log_sharpe_excess" excluded — needs HODL value series from prices "daily_sharpe", "daily_var_95%_trad", "daily_var_99%_trad", @@ -289,6 +290,46 @@ def _daily_log_sharpe(values: jnp.ndarray) -> jnp.ndarray: # Annualize daily stats (calendar days) return jnp.sqrt(365.0) * (mean / (std + 1e-8)) +def _daily_log_sharpe_excess( + pool_values: jnp.ndarray, + hodl_values: jnp.ndarray, +) -> jnp.ndarray: + r"""Annualized Sharpe ratio on daily log *excess* returns over HODL. + + Isolates LP alpha by subtracting the HODL benchmark return at each + daily interval. This removes crypto beta — a strategy that's just + long ETH no longer scores well in a bull-market window. + + .. math:: + + e_t = \log(V^{\mathrm{pool}}_t / V^{\mathrm{pool}}_{t-1}) + - \log(V^{\mathrm{hodl}}_t / V^{\mathrm{hodl}}_{t-1}) + + S_{\mathrm{excess}} = \sqrt{365} \cdot + \frac{\mu(e_t)}{\sigma(e_t) + \epsilon} + + Parameters + ---------- + pool_values : jnp.ndarray + Pool value time series at minute resolution, shape ``(T,)``. + hodl_values : jnp.ndarray + HODL value time series at minute resolution, shape ``(T,)``. + + Returns + ------- + jnp.ndarray + Scalar annualized excess-over-HODL log Sharpe. + """ + daily_pool = pool_values[::1440] + daily_hodl = hodl_values[::1440] + + log_ret_pool = jnp.diff(jnp.log(daily_pool + 1e-12)) + log_ret_hodl = jnp.diff(jnp.log(daily_hodl + 1e-12)) + + excess = log_ret_pool - log_ret_hodl + return jnp.sqrt(365.0) * (excess.mean() / (excess.std() + 1e-8)) + + def _calculate_max_drawdown(value_over_time, duration=7 * 24 * 60): """Calculate worst maximum drawdown across non-overlapping chunks. @@ -743,6 +784,10 @@ def _calculate_return_value( "daily_sharpe": lambda: jnp.sqrt(365) * (daily_returns.mean() / daily_returns.std()), "daily_log_sharpe": lambda: _daily_log_sharpe(value_over_time), + "daily_log_sharpe_excess": lambda: _daily_log_sharpe_excess( + value_over_time, + jnp.sum(stop_gradient(initial_reserves) * local_prices, axis=-1), + ), "returns": lambda: value_over_time[-1] / value_over_time[0] - 1.0, "annualised_returns": lambda: ( (value_over_time[-1] / value_over_time[0]) From 29429b1eeb7747d57945ee54be163a8c8205dae1 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 10 Apr 2026 12:51:18 +0100 Subject: [PATCH 084/115] feat: daily_log_sharpe_excess objective + sweep infrastructure MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New metric: daily_log_sharpe_excess computes Sharpe on excess returns over HODL, isolating LP alpha from crypto beta. Fixes shape mismatch between pool values and HODL values in Optuna training path. Sweep infrastructure: - run_period_sweep.sh: 7 objectives × 3 periods × 400 trials with MAX_PARALLEL=8 job limiter - plot_sweep_results.py: loads sweep results, reruns forward passes with period-specific noise arrays and on-chain baselines, produces per-period comparison plots (value, test-only normalised + fee rev) Also adds calmar, sterling, weekly_rovar to tune script objectives. --- quantammsim/core_simulator/forward_pass.py | 3 +- scripts/plot_sweep_results.py | 385 +++++++++++++++++++++ scripts/run_period_sweep.sh | 72 ++++ 3 files changed, 459 insertions(+), 1 deletion(-) create mode 100644 scripts/plot_sweep_results.py create mode 100644 scripts/run_period_sweep.sh diff --git a/quantammsim/core_simulator/forward_pass.py b/quantammsim/core_simulator/forward_pass.py index f207a657..9fd8e57a 100644 --- a/quantammsim/core_simulator/forward_pass.py +++ b/quantammsim/core_simulator/forward_pass.py @@ -44,6 +44,7 @@ config.update("jax_platform_name", "cpu") +import jax import jax.numpy as jnp import jax.random from jax import jit, vmap, devices @@ -786,7 +787,7 @@ def _calculate_return_value( "daily_log_sharpe": lambda: _daily_log_sharpe(value_over_time), "daily_log_sharpe_excess": lambda: _daily_log_sharpe_excess( value_over_time, - jnp.sum(stop_gradient(initial_reserves) * local_prices, axis=-1), + jnp.sum(stop_gradient(initial_reserves) * local_prices[:value_over_time.shape[0]], axis=-1), ), "returns": lambda: value_over_time[-1] / value_over_time[0] - 1.0, "annualised_returns": lambda: ( diff --git a/scripts/plot_sweep_results.py b/scripts/plot_sweep_results.py new file mode 100644 index 00000000..b5a8778c --- /dev/null +++ b/scripts/plot_sweep_results.py @@ -0,0 +1,385 @@ +"""Plot all sweep results: rerun each (objective, period) combo and compare. + +Reads results/sweep/*.json, groups by period, reruns forward passes over +the full date range (train + test to end_test_date), and produces per-period +comparison plots. + +Usage: + python scripts/plot_sweep_results.py + python scripts/plot_sweep_results.py --end-test-date "2026-03-01 00:00:00" +""" + +import argparse +import glob +import json +import os +import re +import sys + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from datetime import datetime + +import jax.numpy as jnp +from quantammsim.runners.jax_runners import do_run_on_historic_data + + +SWEEP_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "results", "sweep", +) +OUTPUT_DIR = os.path.join( + os.path.dirname(os.path.dirname(__file__)), "results", "sweep", "plots", +) + +# Period definitions: (start, train_end, default_test_end) +# --end-test-date overrides default_test_end for all periods +PERIODS = { + "bull_2023": ("2023-06-01", "2024-06-01", "2026-03-01"), + "default_2024": ("2024-06-01", "2025-06-01", "2026-03-01"), + "recent_2025": ("2025-01-01", "2025-09-01", "2026-03-01"), +} + +# Base fingerprint for reClAMM AAVE/ETH with MM noise +BASE_FP = { + "rule": "reclamm", + "tokens": ["AAVE", "ETH"], + "do_arb": True, + "arb_frequency": 4, + "fees": 0.0025, + "gas_cost": 1.0, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": "results/mm_noise/_sim_arrays/" + "0x9d1fcf346ea1b0_2024-06-01_2026-03-01_mm.npz", + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "reclamm_learn_arc_length_speed": False, + "reclamm_use_shift_exponent": True, + "initial_pool_value": 20_000_000.0, +} + +OBJ_SHORT = { + "daily_log_sharpe": "sharpe", + "daily_log_sharpe_excess": "excess_sharpe", + "fee_revenue_over_value": "fee_rev", + "returns_over_hodl": "ret_hodl", + "calmar": "calmar", + "sterling": "sterling", + "weekly_rovar": "rovar", +} + +BG = "#162536" +TEXT_COLOR = "#E6CE97" +COLORS = [ + "#3498db", "#2ecc71", "#e74c3c", "#f39c12", "#9b59b6", + "#1abc9c", "#e67e22", "#2980b9", "#c0392b", "#8e44ad", +] + + +def load_sweep_results(): + """Load all sweep result JSONs, grouped by period.""" + results = {} # period -> [(obj_name, params)] + for path in sorted(glob.glob(os.path.join(SWEEP_DIR, "*.json"))): + fname = os.path.basename(path).replace(".json", "") + # Parse: {objective}_{period_name} + for period_name in PERIODS: + if fname.endswith(f"_{period_name}"): + obj_name = fname[: -(len(period_name) + 1)] + with open(path) as f: + data = json.load(f) + # Extract params from the first (only) objective key + if isinstance(data, dict): + params_raw = list(data.values())[0] + if isinstance(params_raw, dict): + params = { + k: v for k, v in params_raw.items() + if k in ("price_ratio", "centeredness_margin", + "shift_exponent", "arc_length_speed") + } + if period_name not in results: + results[period_name] = [] + results[period_name].append((obj_name, params)) + break + return results + + +_arrays_cache = {} + + +def _get_noise_arrays_path(start_date, end_date): + """Build or retrieve MM noise arrays for this date range.""" + key = (start_date, end_date) + if key in _arrays_cache: + return _arrays_cache[key] + + from quantammsim.calibration.noise_model_arrays import build_mm_simulator_arrays + + cache_dir = os.path.join( + os.path.dirname(os.path.dirname(__file__)), + "results", "mm_noise", "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join( + cache_dir, f"0x9d1fcf346ea1b0_{start_date}_{end_date}_mm.npz") + + if not os.path.exists(arrays_path): + print(f"\n Building noise arrays for {start_date}→{end_date}...", + end=" ", flush=True) + arrays = build_mm_simulator_arrays( + token_a="AAVE", token_b="ETH", + start_date=start_date, end_date=end_date, + mm_artifact_dir="results/mm_noise", + competitor_tvl_path="results/competitor_tvl/competitor_tvl.npz", + pool_id="0x9d1fcf346ea1b0", + ) + np.savez(arrays_path, + noise_base=arrays["noise_base"], + competitor_tvl=arrays["competitor_tvl"]) + print("done") + + _arrays_cache[key] = arrays_path + return arrays_path + + +def run_config(params, start_date, end_date, end_test_date=None): + """Run a forward pass over start→end_test (or end if no test).""" + actual_end = end_test_date or end_date + arrays_path = _get_noise_arrays_path(start_date, actual_end) + + fp = dict(BASE_FP) + fp["startDateString"] = f"{start_date} 00:00:00" + fp["endDateString"] = f"{actual_end} 00:00:00" + fp["noise_arrays_path"] = arrays_path + jax_params = {k: jnp.array(v) for k, v in params.items()} + return do_run_on_historic_data(run_fingerprint=fp, params=jax_params) + + +def _style_axis(ax): + ax.set_facecolor(BG) + ax.tick_params(colors=TEXT_COLOR) + for spine in ax.spines.values(): + spine.set_color(TEXT_COLOR) + spine.set_alpha(0.3) + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.grid(True, alpha=0.15, color=TEXT_COLOR) + + +def plot_period(period_name, obj_results, end_test_date, output_dir): + """Plot all objectives for one period.""" + start, train_end, default_test_end = PERIODS[period_name] + test_end = end_test_date or default_test_end + + print(f"\n{'='*80}") + print(f"Period: {period_name} ({start} → {train_end} → {test_end})") + print(f"{'='*80}") + + # On-chain baselines + ONCHAIN_CONFIGS = { + "OnChain-launch": { + "price_ratio": 1.5, "centeredness_margin": 0.5, + "shift_exponent": 0.1, + }, + "OnChain-current": { + "price_ratio": 4.0, "centeredness_margin": 0.1, + "shift_exponent": 0.001, + }, + } + + # Run all configs + baselines + runs = {} # label -> forward pass output + run_params = {} # label -> pool params dict + all_configs = ( + [(name, params) for name, params in ONCHAIN_CONFIGS.items()] + + [(f"{OBJ_SHORT.get(obj, obj)} (pr={p.get('price_ratio', 0):.2f})", p) + for obj, p in obj_results] + ) + for label, params in all_configs: + print(f" Running {label}...", end=" ", flush=True) + try: + out = run_config(params, start, train_end, test_end) + runs[label] = out + run_params[label] = params + fv = float(out["final_value"]) + hodl = float((out["reserves"][0] * out["prices"][-1]).sum()) + print(f"final=${fv:,.0f} RoH={fv/hodl - 1:+.2%}") + except Exception as e: + print(f"FAILED: {e}") + + if not runs: + print(" No successful runs!") + return + + # HODL baseline + first_out = next(iter(runs.values())) + hodl_reserves = first_out["reserves"][0] + hodl_values = np.sum( + np.array(hodl_reserves) * np.array(first_out["prices"]), axis=1) + + n_minutes = len(first_out["value"]) + start_dt = datetime.strptime(f"{start} 00:00:00", "%Y-%m-%d %H:%M:%S") + train_end_dt = datetime.strptime(f"{train_end} 00:00:00", "%Y-%m-%d %H:%M:%S") + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") + step = 1440 + dates_daily = dates[::step] + + # ── Plot 1: Full period value ── + fig, axes = plt.subplots(2, 1, figsize=(16, 10), sharex=True, + gridspec_kw={"height_ratios": [3, 1]}) + + ax = axes[0] + for ci, (label, out) in enumerate(runs.items()): + vals = np.array(out["value"][::step]) / 1e6 + is_baseline = label.startswith("OnChain") + ax.plot(dates_daily[:len(vals)], vals, + linewidth=1.5 if is_baseline else 1.8, + linestyle="--" if is_baseline else "-", + alpha=0.7 if is_baseline else 1.0, + color=COLORS[ci % len(COLORS)], label=label) + + hodl_daily = hodl_values[::step] / 1e6 + ax.plot(dates_daily[:len(hodl_daily)], hodl_daily, linewidth=2, + color="white", alpha=0.7, linestyle="--", label="HODL") + + ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5) + _style_axis(ax) + ax.set_ylabel("Pool Value ($M)", color=TEXT_COLOR) + ax.set_title(f"reClAMM AAVE/ETH — {period_name}", + color=TEXT_COLOR, fontsize=14, pad=10) + ax.legend(loc="upper left", fontsize=7, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR, ncol=2) + + # Fee revenue + ax = axes[1] + for ci, (label, out) in enumerate(runs.items()): + fr = out.get("fee_revenue") + if fr is None: + continue + cumfee = np.cumsum(np.array(fr))[::step] / 1e3 + is_baseline = label.startswith("OnChain") + ax.plot(dates_daily[:len(cumfee)], cumfee, + linewidth=1.2 if is_baseline else 1.5, + linestyle="--" if is_baseline else "-", + alpha=0.7 if is_baseline else 1.0, + color=COLORS[ci % len(COLORS)], label=label) + + ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5) + _style_axis(ax) + ax.set_ylabel("Cum. Fee Revenue ($K)", color=TEXT_COLOR) + ax.set_xlabel("Date", color=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + out_path = os.path.join(output_dir, f"sweep_{period_name}_value.png") + fig.savefig(out_path, dpi=200, bbox_inches="tight", facecolor=BG) + plt.close(fig) + print(f" Saved: {out_path}") + + # ── Plot 2: Test-only normalised value + cumulative fee revenue ── + train_minutes = int((train_end_dt - start_dt).total_seconds() / 60) + test_start = min(train_minutes, n_minutes - 1) + + fig, axes = plt.subplots(2, 1, figsize=(16, 10), sharex=True, + gridspec_kw={"height_ratios": [3, 1]}) + ax = axes[0] + test_dates = dates[test_start::step] + + for ci, (label, out) in enumerate(runs.items()): + vals = np.array(out["value"]) + test_vals = vals[test_start::step] + if len(test_vals) < 2: + continue + normed = test_vals / test_vals[0] + is_baseline = label.startswith("OnChain") + ax.plot(test_dates[:len(normed)], normed, + linewidth=1.5 if is_baseline else 1.8, + linestyle="--" if is_baseline else "-", + alpha=0.7 if is_baseline else 1.0, + color=COLORS[ci % len(COLORS)], label=label) + + hodl_test = hodl_values[test_start::step] + if len(hodl_test) > 1: + ax.plot(test_dates[:len(hodl_test)], hodl_test / hodl_test[0], + linewidth=2, color="white", alpha=0.7, linestyle="--", + label="HODL") + + ax.axhline(1.0, color="white", linestyle=":", alpha=0.3) + _style_axis(ax) + ax.set_title(f"Test Period (normalised) — {period_name}", + color=TEXT_COLOR, fontsize=14, pad=10) + ax.set_ylabel("Normalised Value", color=TEXT_COLOR) + ax.legend(loc="best", fontsize=7, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR, ncol=2) + + # Cumulative fee revenue (test period only) + ax = axes[1] + for ci, (label, out) in enumerate(runs.items()): + fr = out.get("fee_revenue") + if fr is None: + continue + fr = np.array(fr) + test_fr = fr[test_start:] + cumfee = np.cumsum(test_fr)[::step] / 1e3 + is_baseline = label.startswith("OnChain") + ax.plot(test_dates[:len(cumfee)], cumfee, + linewidth=1.2 if is_baseline else 1.5, + linestyle="--" if is_baseline else "-", + alpha=0.7 if is_baseline else 1.0, + color=COLORS[ci % len(COLORS)], label=label) + + _style_axis(ax) + ax.set_ylabel("Cum. Fee Revenue ($K)", color=TEXT_COLOR) + ax.set_xlabel("Date", color=TEXT_COLOR) + + fig.patch.set_facecolor(BG) + plt.tight_layout() + out_path = os.path.join(output_dir, f"sweep_{period_name}_test.png") + fig.savefig(out_path, dpi=200, bbox_inches="tight", facecolor=BG) + plt.close(fig) + print(f" Saved: {out_path}") + + # ── Summary table ── + print(f"\n {'Objective':<25s} {'PR':>8s} {'Margin':>7s} {'ShiftExp':>9s}" + f" {'Final $M':>10s} {'HODL $M':>10s} {'RoH':>8s} {'Fee $K':>8s}") + for label, out in runs.items(): + fv = float(out["final_value"]) / 1e6 + hodl_end = float(hodl_values[-1]) / 1e6 + roh = float(out["final_value"]) / float(hodl_values[-1]) - 1 + fr = float(np.array(out.get("fee_revenue", [0])).sum()) / 1e3 + p = run_params.get(label, {}) + pr = p.get("price_ratio", float("nan")) + margin = p.get("centeredness_margin", float("nan")) + se = p.get("shift_exponent", float("nan")) + print(f" {label:<25s} {pr:>8.2f} {margin:>7.3f} {se:>9.4g}" + f" ${fv:>9.2f} ${hodl_end:>9.2f} {roh:>+7.1%} ${fr:>7.0f}") + + +def main(): + parser = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--end-test-date", default=None, + help="Override test end date for all periods") + parser.add_argument("--output-dir", default=OUTPUT_DIR) + parser.add_argument("--periods", nargs="+", default=None, + help="Only plot these periods (default: all)") + args = parser.parse_args() + + os.makedirs(args.output_dir, exist_ok=True) + + results = load_sweep_results() + print(f"Loaded {sum(len(v) for v in results.values())} results" + f" across {len(results)} periods") + + for period_name in sorted(results.keys()): + if args.periods and period_name not in args.periods: + continue + plot_period(period_name, results[period_name], + args.end_test_date, args.output_dir) + + +if __name__ == "__main__": + main() diff --git a/scripts/run_period_sweep.sh b/scripts/run_period_sweep.sh new file mode 100644 index 00000000..ea6931b4 --- /dev/null +++ b/scripts/run_period_sweep.sh @@ -0,0 +1,72 @@ +#!/bin/bash +# Full sweep: all objectives × all periods × 400 trials +# Usage: bash scripts/run_period_sweep.sh +# Monitor: tail -5 /tmp/tune_*.log +# Results: results/sweep/ + +set -e +source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public + +TRIALS=400 +MAX_PARALLEL=8 +COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS" + +OBJECTIVES=( + daily_log_sharpe + daily_log_sharpe_excess + fee_revenue_over_value + returns_over_hodl + calmar + sterling + weekly_rovar +) + +# period_name start_date end_date(train) end_test_date +PERIODS=( + "bull_2023 2023-06-01 2024-06-01 2025-06-01" + "default_2024 2024-06-01 2025-06-01 2026-03-01" + "recent_2025 2025-01-01 2025-09-01 2026-03-01" +) + +OUTDIR="results/sweep" +mkdir -p "$OUTDIR" + +wait_for_slot() { + while [ "$(jobs -rp | wc -l)" -ge "$MAX_PARALLEL" ]; do + sleep 10 + done +} + +N=0 +for period_line in "${PERIODS[@]}"; do + read -r period_name start_date end_date end_test_date <<< "$period_line" + for obj in "${OBJECTIVES[@]}"; do + tag="${obj}_${period_name}" + logfile="/tmp/tune_${tag}.log" + outfile="${OUTDIR}/${tag}.json" + + wait_for_slot + + echo "[$N] Launching: ${tag}" + $COMMON --objective "$obj" \ + --start-date "${start_date} 00:00:00" \ + --end-date "${end_date} 00:00:00" \ + --end-test-date "${end_test_date} 00:00:00" \ + --output "$outfile" \ + > "$logfile" 2>&1 & + + N=$((N + 1)) + done +done + +echo "" +echo "$N jobs queued (${#OBJECTIVES[@]} objectives × ${#PERIODS[@]} periods × $TRIALS trials)" +echo "Max parallel: $MAX_PARALLEL" +echo "" +echo "Monitor: tail -5 /tmp/tune_*.log" +echo "Results: ls $OUTDIR/" +echo "Summary: grep -A3 'Best trial' /tmp/tune_*.log" +echo "" +echo "Waiting for all jobs to finish..." +wait +echo "Done." From 920eb1f5aaf067d05fb927c1cc30d16d8ddac094 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 10 Apr 2026 12:56:20 +0100 Subject: [PATCH 085/115] feat: distributionally robust objective aggregation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New batched_robust_objective_factory uses softmin-weighted averaging over training windows instead of mean: weights = softmax(-outputs / temperature) objective = sum(weights * outputs) At temperature→∞: recovers mean (standard). At temperature→0: pure worst-case. Intermediate values up-weight bad windows, encouraging the optimizer to survive crashes rather than squeeze calm-period gains. Wired through update_from_partial_training_step_factory (vanilla SGD) and update_from_partial_training_step_factory_with_optax (production). Activated via optimisation_settings.robust_temperature in fingerprint, or --robust-temperature flag in tune_reclamm_calibrated_noise.py. --- experiments/tune_reclamm_calibrated_noise.py | 9 +++ quantammsim/runners/jax_runners.py | 9 ++- quantammsim/runners/multi_period_sgd.py | 3 + quantammsim/training/backpropagation.py | 58 ++++++++++++++++++++ 4 files changed, 78 insertions(+), 1 deletion(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index c0d04771..cdaa5145 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -168,11 +168,15 @@ def _build_mm_observed_arrays(args): def _build_opt_settings(args): """Build optimisation_settings for optuna, bfgs, or cma_es.""" + robust = ({"robust_temperature": args.robust_temperature} + if args.robust_temperature is not None else {}) + if args.method == "bfgs": return { "method": "bfgs", "n_parameter_sets": args.n_parameter_sets, **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + **robust, "bfgs_settings": { "maxiter": args.bfgs_maxiter, "tol": args.bfgs_tol, @@ -185,6 +189,7 @@ def _build_opt_settings(args): "method": "cma_es", "n_parameter_sets": args.n_parameter_sets, **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + **robust, "cma_es_settings": { "population_size": args.cma_pop_size, "n_generations": args.cma_generations, @@ -199,6 +204,7 @@ def _build_opt_settings(args): "method": "optuna", "n_parameter_sets": 1, **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), + **robust, "optuna_settings": { "make_scalar": True, "expand_around": False, @@ -345,6 +351,9 @@ def main(): parser.add_argument("--bout-offset", type=int, default=None) parser.add_argument("--val-fraction", type=float, default=None) parser.add_argument("--overfitting-penalty", type=float, default=None) + parser.add_argument("--robust-temperature", type=float, default=None, + help="Robust aggregation temperature (lower=more robust)." + " None=standard mean. Try 0.5-2.0.") parser.add_argument("--output", type=str, default=None, help="Save results to JSON file") args = parser.parse_args() diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index 27669a29..da8892b2 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -1419,7 +1419,14 @@ def objective(trial): ) train_objectives.append(train_value) - mean_train_value = jnp.sum(jnp.array(train_objectives)) / len(train_objectives) + _train_arr = jnp.array(train_objectives) + _robust_temp = run_fingerprint.get("optimisation_settings", {}).get( + "robust_temperature", None) + if _robust_temp is not None: + _weights = jax.nn.softmax(-_train_arr / _robust_temp) + mean_train_value = jnp.sum(_weights * _train_arr) + else: + mean_train_value = jnp.mean(_train_arr) train_value = _calculate_return_value( run_fingerprint["return_val"], train_outputs["reserves"], diff --git a/quantammsim/runners/multi_period_sgd.py b/quantammsim/runners/multi_period_sgd.py index c81e00a1..90d5f4a6 100644 --- a/quantammsim/runners/multi_period_sgd.py +++ b/quantammsim/runners/multi_period_sgd.py @@ -466,11 +466,14 @@ def multi_period_sgd_training( opt_state = optimizer.init(params) # Use existing factory - it handles batching, gradients, optimizer application + robust_temp = run_fingerprint["optimisation_settings"].get( + "robust_temperature", None) update_fn = update_from_partial_training_step_factory_with_optax( partial_training_step, optimizer, run_fingerprint["optimisation_settings"]["train_on_hessian_trace"], Partial(partial_training_step, start_index=(data_dict["start_idx"], 0)), + robust_temperature=robust_temp, ) # Training loop diff --git a/quantammsim/training/backpropagation.py b/quantammsim/training/backpropagation.py index 07d63110..bc34f664 100644 --- a/quantammsim/training/backpropagation.py +++ b/quantammsim/training/backpropagation.py @@ -49,6 +49,7 @@ # jax.set_cpu_device_count(n) # print(devices("cpu")) +import jax import jax.numpy as jnp from jax import grad, value_and_grad, jit, vmap from jax.tree_util import tree_map @@ -170,6 +171,45 @@ def batched_objective(params, start_indexes): return batched_objective +def batched_robust_objective_factory(batched_partial_training_step, temperature=1.0): + """Creates an objective with distributionally robust aggregation. + + Instead of ``mean(outputs)``, uses a softmin-weighted average that + up-weights bad windows and down-weights good ones:: + + weights = softmax(-outputs / temperature) + objective = sum(weights * outputs) + + At ``temperature → ∞``: recovers the mean (standard behavior). + At ``temperature → 0``: recovers the min (pure worst-case). + + This encourages the optimizer to spend gradient budget on surviving + crashes rather than squeezing marginal gains in calm periods. + + Parameters + ---------- + batched_partial_training_step : callable + A vectorized function that processes batches of inputs. + temperature : float + Controls robustness. Lower = more robust / pessimistic. + Recommended range: 0.1 (very robust) to 10.0 (near mean). + Default 1.0 is a moderate robustness level. + + Returns + ------- + callable + JIT-compiled robust objective function. + """ + + @jit + def batched_robust_objective(params, start_indexes): + output = batched_partial_training_step(params, start_indexes) + weights = jax.nn.softmax(-output / temperature) + return jnp.sum(weights * output) + + return batched_robust_objective + + def batched_objective_with_hessian_factory( batched_partial_training_step, partial_fixed_training_step ): @@ -326,6 +366,7 @@ def update_from_partial_training_step_factory( partial_training_step, train_on_hessian_trace=False, partial_fixed_training_step=None, + robust_temperature=None, ): """Creates a complete update function from a partial training step. @@ -342,6 +383,11 @@ def update_from_partial_training_step_factory( partial_fixed_training_step : callable, optional The function used to compute Hessian trace when train_on_hessian_trace is True. Required if train_on_hessian_trace is True. + robust_temperature : float, optional + If set, uses distributionally robust aggregation (softmin-weighted + average) instead of mean over training windows. Lower values are + more robust / pessimistic. Recommended range: 0.1–10.0. + None (default) uses standard mean aggregation. Returns ------- @@ -358,6 +404,10 @@ def update_from_partial_training_step_factory( batched_partial_training_step, partial_fixed_training_step ) update = update_with_hessian_factory(batched_objective_with_hessian) + elif robust_temperature is not None: + batched_objective = batched_robust_objective_factory( + batched_partial_training_step, temperature=robust_temperature) + update = update_factory(batched_objective) else: batched_objective = batched_objective_factory(batched_partial_training_step) update = update_factory(batched_objective) @@ -522,6 +572,7 @@ def update_from_partial_training_step_factory_with_optax( optimizer, train_on_hessian_trace=False, partial_fixed_training_step=None, + robust_temperature=None, ): """Creates a complete update function from a partial training step using optax optimizer. @@ -539,6 +590,9 @@ def update_from_partial_training_step_factory_with_optax( partial_fixed_training_step : callable, optional The function used to compute Hessian trace when train_on_hessian_trace is True. Required if train_on_hessian_trace is True. + robust_temperature : float, optional + If set, uses distributionally robust aggregation (softmin-weighted + average) instead of mean. See ``batched_robust_objective_factory``. Returns ------- @@ -555,6 +609,10 @@ def update_from_partial_training_step_factory_with_optax( batched_partial_training_step, partial_fixed_training_step ) update = update_with_hessian_factory_with_optax(batched_objective_with_hessian, optimizer) + elif robust_temperature is not None: + batched_objective = batched_robust_objective_factory( + batched_partial_training_step, temperature=robust_temperature) + update = update_factory_with_optax(batched_objective, optimizer) else: batched_objective = batched_objective_factory(batched_partial_training_step) update = update_factory_with_optax(batched_objective, optimizer) From 3fad415d5c2946b1ca4c152e88e3bb953794949d Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 21 Apr 2026 16:14:23 +0100 Subject: [PATCH 086/115] feat: noise fee fold, in-range gate, blessed-arb experiment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three changes to the reClAMM simulation step: 1. Noise fee revenue folded into fee_revenue output — all four noise branches now split fees via protocol_fee_split consistently with arb fees, and add LP's share to lp_fee_revenue_usd. 2. In-range gate on mm_observed noise model — noise volume is zeroed when the pool is out of range (centeredness < margin), preventing the optimizer from stacking virtual reserves without recentering. 3. Blessed-arb monkey patch — module-level flag enabling zero-fee arb trades with LVR returned to pool via proportional reserve scaling. Controlled by set_blessed_arb(); default off. --- quantammsim/pools/reCLAMM/reclamm_reserves.py | 65 ++++++++++++++++--- 1 file changed, 56 insertions(+), 9 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index ac264e80..1a2cafda 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -43,6 +43,19 @@ # Reference balance for initialisation (matches Solidity _INITIALIZATION_MAX_BALANCE_A) _INITIALIZATION_MAX_BALANCE_A = 1e6 +# MONKEY PATCH — blessed-arb experiment: +# If True, arb trades execute zero-fee (internal gamma = 1) and the +# arbitrageur returns their LVR back to the pool (minus gas + external cost). +# Set via set_blessed_arb() before constructing scan closures. Cache must be +# cleared between toggled values (jax.clear_caches()) because the global is +# captured at Partial-construction time. +_BLESSED_ARB = False + + +def set_blessed_arb(enabled: bool) -> None: + global _BLESSED_ARB + _BLESSED_ARB = bool(enabled) + # Virtual balance decay is capped at 30 days to prevent overflow _MAX_DECAY_DURATION_SECONDS = 30 * 86400 @@ -961,7 +974,11 @@ def _skip_schedule_state(_): 0, ) - optimal_arb_trade = jnp.where(fees_are_being_charged, fee_trade, zero_fee_trade) + # Blessed-arb monkey patch: zero-fee trade regardless of pool fee setting. + if _BLESSED_ARB: + optimal_arb_trade = zero_fee_trade + else: + optimal_arb_trade = jnp.where(fees_are_being_charged, fee_trade, zero_fee_trade) # Check profitability for arb profit_to_arb = -(optimal_arb_trade * prices).sum() - arb_thresh @@ -976,6 +993,7 @@ def _skip_schedule_state(_): # --- Noise model dispatch --- # noise_model is a concrete Python string (passed via Partial as static # aux_data), so if/elif branches resolve at trace time. + noise_fee_income = jnp.asarray(0.0, dtype=prices.dtype) if noise_model == "ratio": noisy_reserves = calculate_reserves_after_noise_trade( applied_trade, jnp.array([Ra_new, Rb_new]), prices, @@ -1011,7 +1029,8 @@ def _skip_schedule_state(_): # effective reserves (Ra+Va, Rb+Vb) by the same factor, then # subtract back the fixed virtual reserves. minutes_per_step = seconds_per_step / 60.0 - noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) Ra_new = (Ra_new + Va) * scale - Va Rb_new = (Rb_new + Vb) * scale - Vb @@ -1029,7 +1048,8 @@ def _skip_schedule_state(_): ) minutes_per_step = seconds_per_step / 60.0 - noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) Ra_new = (Ra_new + Va) * scale - Va Rb_new = (Rb_new + Vb) * scale - Vb @@ -1046,7 +1066,8 @@ def _skip_schedule_state(_): ) minutes_per_step = seconds_per_step / 60.0 - noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) Ra_new = (Ra_new + Va) * scale - Va Rb_new = (Rb_new + Vb) * scale - Vb @@ -1059,8 +1080,18 @@ def _skip_schedule_state(_): effective_value, noise_base, competitor_tvl, ) + # In-range gate: noise traders only route here if the pool is + # in range (post-arb, post-recentering state). Hard-indicator + # limit of the routing-with-misquote convex program. + centeredness_post, _ = compute_centeredness(Ra_new, Rb_new, Va, Vb) + in_range_gate = (centeredness_post >= centeredness_margin).astype( + noise_vol.dtype + ) + noise_vol = noise_vol * in_range_gate + minutes_per_step = seconds_per_step / 60.0 - noise_fee_income = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) Ra_new = (Ra_new + Va) * scale - Va Rb_new = (Rb_new + Vb) * scale - Vb @@ -1091,17 +1122,33 @@ def _skip_schedule_state(_): Rb_new = jnp.where(clamp_a, Rb + edge_a[1], jnp.where(clamp_b, Rb + edge_b[1], Rb_new)) # Protocol fee: divert protocol_fee_split of inbound swap fees from LP reserves. - # Computed on the final trade (normal arb or edge trade). + # Computed on the final trade (normal arb or edge trade). Blessed arb pays + # no swap fee, so fee_rate collapses to 0 in that mode. final_trade = jnp.array([Ra_new - Ra, Rb_new - Rb]) - fee_rate = 1.0 - gamma + fee_rate = 0.0 if _BLESSED_ARB else (1.0 - gamma) inbound = jnp.maximum(final_trade, 0.0) protocol_fee = inbound * fee_rate * protocol_fee_split Ra_new = Ra_new - protocol_fee[0] Rb_new = Rb_new - protocol_fee[1] - # LP fee revenue: total fee income minus protocol's share, in USD. + # LP fee revenue: arb swap fees (zero under blessed) + noise-trader fees + # (unchanged; noise traders still pay the pool's fee rate). lp_fee_income = inbound * fee_rate * (1.0 - protocol_fee_split) - lp_fee_revenue_usd = (lp_fee_income * prices).sum() + lp_fee_revenue_usd = (lp_fee_income * prices).sum() + noise_fee_income + + # Blessed-arb LVR return: arb returns gross profit minus gas + external cost + # to the pool, scaled across effective reserves to preserve quoted price. + if _BLESSED_ARB: + arb_profit_usd = -(applied_trade * prices).sum() + external_cost_applied = 0.5 * arb_fees * (jnp.abs(applied_trade) * prices).sum() + returned_profit = jnp.maximum( + arb_profit_usd - arb_thresh - external_cost_applied, 0.0 + ) + eff_val = (Ra_new + Va) * prices[0] + (Rb_new + Vb) * prices[1] + scale_lvr = 1.0 + returned_profit / jnp.maximum(eff_val, 1e-8) + Ra_new = (Ra_new + Va) * scale_lvr - Va + Rb_new = (Rb_new + Vb) * scale_lvr - Vb + lp_fee_revenue_usd = lp_fee_revenue_usd + returned_profit new_reserves = jnp.array([Ra_new, Rb_new]) return [ From 236bce34d87b823b9621501d5ef42c6d1255d9a2 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 21 Apr 2026 16:16:32 +0100 Subject: [PATCH 087/115] fix: declare binance_historical_data and gdown in pyproject.toml MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit These are required by the data-download path (historic_data_utils imports binance_historical_data; other scripts use gdown) and were declared in the legacy setup.py but not in pyproject.toml — so fresh `pip install -e .` installs missing them surfaced only at data-fetch time. --- pyproject.toml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/pyproject.toml b/pyproject.toml index 8859b551..2cec9077 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -24,6 +24,8 @@ dependencies = [ "plotly", "dask", "Historic-Crypto", + "binance_historical_data", + "gdown", "bidask", "optax", "jsonpickle", From ad05e6def6c32d486b5c209285e7b0aa6c0877d7 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 21 Apr 2026 16:17:06 +0100 Subject: [PATCH 088/115] chore: drop unused gdown dependency gdown is declared in pyproject.toml and setup.py but not imported anywhere in the codebase (no gdown.* calls, no drive.google URLs). Removing from both. --- pyproject.toml | 1 - setup.py | 1 - 2 files changed, 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 2cec9077..b2ad7aaa 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -25,7 +25,6 @@ dependencies = [ "dask", "Historic-Crypto", "binance_historical_data", - "gdown", "bidask", "optax", "jsonpickle", diff --git a/setup.py b/setup.py index 20eb403f..d80b355b 100644 --- a/setup.py +++ b/setup.py @@ -21,7 +21,6 @@ "plotly", "bidask", "Historic_Crypto", - "gdown", "binance_historical_data", "dask", "jsonpickle", From 181bd9c6962c055d51c1d3bc596adb2a39e65735 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 15:44:37 +0100 Subject: [PATCH 089/115] fix: align noise feature standardization with training-time pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two related fixes in noise_model_arrays.py addressing the same root issue: inference-time feature processing was inconsistent with how features were standardized during MM noise model training. 1. build_daily_features_from_binance: switch from blanket standardization to selective (realized_vol + cross-pool volume columns only), matching build_data() in run_linear_market_noise.py. The intercept, log_price, dow encoding, returns, trends, and volume_zscore are kept raw. 2. build_mm_simulator_arrays: rewrite to build the 18 MM features directly from Binance helpers (build_btc_daily_features, build_token_daily_features, _compute_pair_volatility), then standardize using the MM model's own training stats — mapping from the 22-feature linear space (saved as x_mean/x_std in model.npz) to the 18 MM features by removing the TVL column and TVL-interaction columns. Falls back to local stats with a warning if training stats are unavailable. --- quantammsim/calibration/noise_model_arrays.py | 167 +++++++++++------- 1 file changed, 102 insertions(+), 65 deletions(-) diff --git a/quantammsim/calibration/noise_model_arrays.py b/quantammsim/calibration/noise_model_arrays.py index 7b73dc8e..d04431f8 100644 --- a/quantammsim/calibration/noise_model_arrays.py +++ b/quantammsim/calibration/noise_model_arrays.py @@ -171,8 +171,19 @@ def build_daily_features_from_binance( x_base[k, col] = val col += 1 - # Standardize base features - x_base = ((x_base - x_mean[:x_base_cols]) / x_std[:x_base_cols]).astype(np.float32) + # Selective standardization — must match build_data() in run_linear_market_noise.py. + # Only realized vols and cross-pool volumes are standardized; everything else is raw. + for i, name in enumerate(feat_names[:x_base_cols]): + if i >= len(x_mean): + break + if "realized_vol" in name or "pair_realized" in name: + x_base[:, i] = (x_base[:, i] - x_mean[i]) / max(float(x_std[i]), 0.01) + elif name.startswith("xobs_") and name != "xobs_0": + parts = name.split("_") + if len(parts) >= 2 and parts[1].isdigit() and int(parts[1]) >= 4: + x_base[:, i] = (x_base[:, i] - x_mean[i]) / max(float(x_std[i]), 1e-6) + # else: leave untouched (intercept, log_price, dow, returns, trends, vol_zscore) + x_base = x_base.astype(np.float32) # Interaction terms base_feat_names = feat_names[:x_base_cols] @@ -312,24 +323,18 @@ def build_mm_simulator_arrays( non-TVL terms, and K_t = competitor_tvl_t is observed from DeFi Llama (network conductance model: direct + multi-hop). - Parameters - ---------- - token_a, token_b : str - Token symbols. - start_date, end_date : str - Date range. - mm_artifact_dir : str - Directory with MM model.npz and meta.json. - competitor_tvl_path : str - Path to competitor_tvl.npz from fetch_competitor_tvl.py. - pool_id : str, optional - Pool ID for per-pool alpha/gamma. - - Returns - ------- - dict with noise_base, competitor_tvl (minute arrays), dates, etc. + Builds the 18 market features directly from Binance data using the + same helper functions as the calibration pipeline. Standardization + follows the same selective rules as build_data(): only realized_vol + features are centered/scaled, everything else is left raw. """ - # Load MM model + from quantammsim.calibration.market_features import ( + build_btc_daily_features, + build_token_daily_features, + _compute_pair_volatility, + TOKEN_MAP, + ) + art, meta = load_artifact(mm_artifact_dir) pool_ids = meta["pool_ids"] market_names = meta["market_names"] @@ -352,65 +357,97 @@ def build_mm_simulator_arrays( gamma_i = np.median(gamma, axis=0) if per_pool_gamma else gamma print(f" MM model: pool not found, using median alpha={alpha_i:.3f}") - # Build daily market features from Binance - # The MM model uses the same features as the linear model minus TVL - # We need x_mean/x_std from the linear model artifact for standardization - linear_art_dir = os.path.join( - os.path.dirname(os.path.dirname(mm_artifact_dir)), - "results", "linear_market_noise") - if os.path.exists(os.path.join(linear_art_dir, "model.npz")): - lin_art, lin_meta = load_artifact(linear_art_dir) - x_mean = lin_art["x_mean"] - x_std = lin_art["x_std"] - feat_names = lin_meta["feat_names"] - else: - # Fallback: try to get from MM artifact - x_mean = art.get("x_mean", np.zeros(n_market)) - x_std = art.get("x_std", np.ones(n_market)) - feat_names = market_names + # Build daily market features from Binance (same helpers as calibration) + start_ts = pd.Timestamp(start_date) + end_ts = pd.Timestamp(end_date) + date_range = pd.date_range(start_ts, end_ts, freq="D") + n_days = len(date_range) - trend_windows = (7,) + mapped_a = TOKEN_MAP.get(token_a, token_a) + mapped_b = TOKEN_MAP.get(token_b, token_b) + btc_feat = build_btc_daily_features([7]) + feat_a = build_token_daily_features(mapped_a, [7]) + feat_b = build_token_daily_features(mapped_b, [7]) + pair_vol = _compute_pair_volatility(mapped_a, mapped_b) print(f" Building features from Binance: {token_a}/{token_b}," f" {start_date} → {end_date}") - x_daily, dates = build_daily_features_from_binance( - token_a, token_b, start_date, end_date, - feat_names, x_mean, x_std, trend_windows, - ) - n_days = len(dates) - - # Extract market features (exclude TVL and TVL interactions) - tvl_col = None - tvl_interaction_cols = set() - for i, name in enumerate(feat_names): - if name == "xobs_1": - tvl_col = i - elif name.startswith("xobs_1\u00d7"): - tvl_interaction_cols.add(i) - keep_cols = [i for i in range(len(feat_names)) - if i != tvl_col and i not in tvl_interaction_cols] + x_market = np.zeros((n_days, n_market), dtype=np.float64) + for k, day in enumerate(date_range): + day_norm = day.normalize() + for mi, mname in enumerate(market_names): + val = 0.0 + if mname == "xobs_0": + val = 1.0 + elif mname == "xobs_2": + val = np.sin(2 * np.pi * day.weekday() / 7) + elif mname == "xobs_3": + val = np.cos(2 * np.pi * day.weekday() / 7) + elif mname.startswith("btc_") and btc_feat is not None: + if day_norm in btc_feat.index and mname in btc_feat.columns: + v = btc_feat.loc[day_norm, mname] + if np.isfinite(v): + val = v + elif mname.startswith("tok_a_") and feat_a is not None: + acol = mname[6:] + if day_norm in feat_a.index and acol in feat_a.columns: + v = feat_a.loc[day_norm, acol] + if np.isfinite(v): + val = v + elif mname.startswith("tok_b_") and feat_b is not None: + bcol = mname[6:] + if day_norm in feat_b.index and bcol in feat_b.columns: + v = feat_b.loc[day_norm, bcol] + if np.isfinite(v): + val = v + elif mname == "pair_realized_vol_7d" and pair_vol is not None: + if day_norm in pair_vol.index: + v = pair_vol.loc[day_norm, "pair_realized_vol_7d"] + if np.isfinite(v): + val = v + x_market[k, mi] = val + + # Selective standardization: only realized_vol features get centered/scaled. + # Must use the TRAINING panel's stats (saved in model.npz as x_mean/x_std + # in the 22-feature linear space). Map to the 18 MM features by removing + # the TVL column (index 1) and 3 TVL-interaction columns (indices 19-21). + x_mean_22 = art.get("x_mean") + x_std_22 = art.get("x_std") + if x_mean_22 is not None and x_std_22 is not None and len(x_mean_22) == 22: + keep_22_to_18 = [i for i in range(22) if i not in {1, 19, 20, 21}] + x_mean_18 = x_mean_22[keep_22_to_18] + x_std_18 = x_std_22[keep_22_to_18] + for mi, mname in enumerate(market_names): + if x_mean_18[mi] != 0 or x_std_18[mi] != 1: + x_market[:, mi] = (x_market[:, mi] - x_mean_18[mi]) / max(x_std_18[mi], 0.01) + else: + # Fallback: compute local stats (less accurate but won't crash) + print(" WARNING: training stats not found, using local standardization") + for mi, mname in enumerate(market_names): + if ("realized_vol" in mname or "pair_realized" in mname) and "\u00d7" not in mname: + col_mean = float(np.mean(x_market[:, mi])) + col_std = max(float(np.std(x_market[:, mi])), 0.01) + x_market[:, mi] = (x_market[:, mi] - col_mean) / col_std - # Map market_names to x_daily columns - x_market_daily = np.zeros((n_days, n_market), dtype=np.float32) + # Interaction terms + name_to_col = {n: i for i, n in enumerate(market_names)} for mi, mname in enumerate(market_names): - # Find mname in feat_names - for fi, fname in enumerate(feat_names): - if fname == mname and fi in keep_cols: - col_in_daily = fi - x_market_daily[:, mi] = x_daily[:, col_in_daily] - break + if "\u00d7" in mname: + parts = mname.split("\u00d7") + if parts[0] in name_to_col and parts[1] in name_to_col: + x_market[:, mi] = (x_market[:, name_to_col[parts[0]]] * + x_market[:, name_to_col[parts[1]]]) # Compute noise_base = alpha_i + gamma_i @ x_market - noise_base_daily = alpha_i + x_market_daily @ gamma_i - noise_base_daily = noise_base_daily.astype(np.float64) + noise_base_daily = alpha_i + x_market @ gamma_i # Load competitor TVL (K) print(f" Loading competitor TVL from {competitor_tvl_path}") comp_data = np.load(competitor_tvl_path, allow_pickle=True) comp_pool_ids = list(comp_data["pool_ids"]) comp_dates = list(comp_data["date_list"]) - k_eff = comp_data["k_eff"] # (n_comp_dates, n_comp_pools) + k_eff = comp_data["k_eff"] # Find pool in competitor data comp_pool_idx = -1 @@ -427,7 +464,7 @@ def build_mm_simulator_arrays( comp_date_to_idx[str(d)[:10]] = ci K_daily = np.full(n_days, np.nan, dtype=np.float64) - for k, day in enumerate(dates): + for k, day in enumerate(date_range): ds = str(pd.Timestamp(day))[:10] if ds in comp_date_to_idx: ci = comp_date_to_idx[ds] @@ -453,7 +490,7 @@ def build_mm_simulator_arrays( return { "noise_base": noise_base, "competitor_tvl": competitor_tvl_array, - "dates": dates, + "dates": date_range, "pool_index": pool_idx, "n_days": n_days, "n_minutes": n_minutes, From 656ecc410999bf3fdb64b9b15830f5b7c48e8d77 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 15:46:38 +0100 Subject: [PATCH 090/115] feat: add optional box constraints to CMA-ES MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit run_cmaes now accepts lower_bounds and upper_bounds arrays. When provided, each generation's population is clipped to those bounds after sampling but before evaluation. The mean update is a convex combination of selected population members, so clipping the population keeps both mean and best_x in bounds without further intervention. The runner builds the bounds vector generically from optimisation_settings.optuna_settings.parameter_config (matching the ravel_pytree layout of params_single), so any tuning script that already populates parameter_config gets CMA-ES bounds for free. Unbounded dimensions fall back to ±1e30. --- quantammsim/runners/jax_runners.py | 35 +++++++++++++++++++++++++++++- quantammsim/training/cma_es.py | 9 ++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index da8892b2..d8cddd32 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -2290,10 +2290,43 @@ def eval_single(flat_x): # Standalone jitted version kept for any verbose/diagnostic use eval_population = jit(eval_fn_raw) + # Build box constraints from parameter_config (if available) + param_config = run_fingerprint.get("optimisation_settings", {}).get( + "optuna_settings", {} + ).get("parameter_config", {}) + if param_config: + # Construct lower/upper bound pytrees matching params_single structure + lb_dict = {} + ub_dict = {} + for k, v in params_single.items(): + if k == "subsidary_params": + continue + cfg = param_config.get(k) + if cfg is not None: + lo = jnp.full_like(jnp.asarray(v, dtype=flat_x0_template.dtype), cfg["low"]) + hi = jnp.full_like(jnp.asarray(v, dtype=flat_x0_template.dtype), cfg["high"]) + else: + lo = jnp.full_like(jnp.asarray(v, dtype=flat_x0_template.dtype), -1e30) + hi = jnp.full_like(jnp.asarray(v, dtype=flat_x0_template.dtype), 1e30) + lb_dict[k] = lo + ub_dict[k] = hi + lb_dict["subsidary_params"] = params_single.get("subsidary_params", []) + ub_dict["subsidary_params"] = params_single.get("subsidary_params", []) + flat_lb, _ = ravel_pytree(lb_dict) + flat_ub, _ = ravel_pytree(ub_dict) + if verbose: + print(f"[CMA-ES] Box constraints: {n_flat} dims bounded") + else: + flat_lb = None + flat_ub = None + @jit def _run_one_restart(flat_x0, rng_key): state = init_cmaes(flat_x0, sigma0) - return run_cmaes(state, rng_key, eval_fn_raw, cma_params, n_generations, tol) + return run_cmaes( + state, rng_key, eval_fn_raw, cma_params, n_generations, tol, + lower_bounds=flat_lb, upper_bounds=flat_ub, + ) # Keep initial params for saving initial_params = deepcopy(params) diff --git a/quantammsim/training/cma_es.py b/quantammsim/training/cma_es.py index 53e92847..f31c4321 100644 --- a/quantammsim/training/cma_es.py +++ b/quantammsim/training/cma_es.py @@ -288,6 +288,8 @@ def run_cmaes( params: dict, n_generations: int, tol: float = 1e-8, + lower_bounds: jnp.ndarray = None, + upper_bounds: jnp.ndarray = None, ) -> CMAESState: """Run CMA-ES via ``lax.while_loop``. JIT-compatible. @@ -308,6 +310,10 @@ def run_cmaes( Maximum number of generations. tol : float Convergence tolerance passed to :func:`_should_stop_jax`. + lower_bounds : jax.Array, optional + Per-dimension lower bounds, shape ``(n,)``. None = unbounded. + upper_bounds : jax.Array, optional + Per-dimension upper bounds, shape ``(n,)``. None = unbounded. Returns ------- @@ -315,6 +321,7 @@ def run_cmaes( Final state after convergence or ``n_generations``. """ lam = params["lam"] + use_bounds = lower_bounds is not None and upper_bounds is not None def cond_fn(carry): state, _key = carry @@ -324,6 +331,8 @@ def body_fn(carry): state, key = carry key, subkey = random.split(key) pop = ask(state, subkey, lam) + if use_bounds: + pop = jnp.clip(pop, lower_bounds, upper_bounds) fitness = eval_fn(pop) state = tell(state, pop, fitness, params) return (state, key) From f5bd5d275167ac72549ed2684bded3d3be5d7611 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 15:55:52 +0100 Subject: [PATCH 091/115] feat: extend reClAMM tune script for CMA-ES bounds and arbitrary pairs Several related changes to experiments/tune_reclamm_calibrated_noise.py: 1. Bounds wiring for CMA-ES: include PARAMETER_CONFIG in the CMA-ES optimisation_settings as optuna_settings.parameter_config, so the generic bounds machinery in jax_runners.py picks up reClAMM's price_ratio / centeredness_margin / shift_exponent bounds. 2. New --pr-max CLI flag to override the upper bound on price_ratio at run time (useful for capping the search away from degenerate wide-band solutions). 3. Generalize the script from hardcoded AAVE/ETH (POOL_ID + ["AAVE", "ETH"]) to --tokens / --pool-id CLI args. The helper functions _build_market_linear_arrays, _build_mm_observed_arrays, build_fingerprint, and run_single now take tokens and pool_id as parameters; main() reads them from argparse and threads them through. 4. Docstring update to mention the new "none" and "mm_observed" noise model modes. --- experiments/tune_reclamm_calibrated_noise.py | 80 ++++++++++++-------- 1 file changed, 49 insertions(+), 31 deletions(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index cdaa5145..fa9fd198 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -1,8 +1,10 @@ """Optuna tuning of reClAMM pool parameters with calibrated noise models. -Supports two noise model modes: - --noise-model calibrated (legacy 8-covariate model) - --noise-model market_linear (new per-pool model with market features) +Supports noise model modes: + --noise-model none (pure arb, no noise traders) + --noise-model calibrated (legacy 8-covariate model, AAVE/ETH only) + --noise-model market_linear (per-pool model with market features) + --noise-model mm_observed (MM model + DeFi Llama competitor TVL) The market_linear model uses precomputed daily arrays from the per-pool calibrated noise model artifact (results/linear_market_noise/). It evaluates: @@ -13,19 +15,19 @@ pair volatility, day-of-week, cross-pool volumes) and tvl_coeff_t is the effective TVL coefficient including interaction terms. -Pool: 0x9d1fcf346ea1b0 = AAVE/WETH Mainnet +Default pool: AAVE/ETH (0x9d1fcf346ea1b0). Use --tokens to override. Usage: cd source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public - # New market_linear model (default) + # AAVE/ETH with market_linear noise (default) python experiments/tune_reclamm_calibrated_noise.py - # Legacy 8-covariate model - python experiments/tune_reclamm_calibrated_noise.py --noise-model calibrated + # COW/ETH with no noise model + python experiments/tune_reclamm_calibrated_noise.py --tokens COW ETH --noise-model none - # All three objectives + # All objectives python experiments/tune_reclamm_calibrated_noise.py --all-objectives # More trials @@ -39,7 +41,8 @@ from pathlib import Path from quantammsim.runners.jax_runners import train_on_historic_data -POOL_ID = "0x9d1fcf346ea1b0" # AAVE/WETH Mainnet +DEFAULT_POOL_ID = "0x9d1fcf346ea1b0" # AAVE/WETH Mainnet +DEFAULT_TOKENS = ["AAVE", "ETH"] # --- Legacy 8-covariate noise coefficients --- NOISE_COEFFS_LEGACY = [ @@ -68,7 +71,7 @@ ] -def _build_market_linear_arrays(args): +def _build_market_linear_arrays(args, pool_id, tokens): """Precompute noise arrays from the per-pool market noise model artifact.""" from quantammsim.calibration.noise_model_arrays import build_simulator_arrays @@ -76,15 +79,15 @@ def _build_market_linear_arrays(args): start = args.start_date.split(" ")[0] end = args.end_test_date.split(" ")[0] - print(f" Building market_linear noise arrays for {POOL_ID}...") + print(f" Building market_linear noise arrays for {pool_id}...") print(f" Date range: {start} → {end}") arrays = build_simulator_arrays( - token_a="AAVE", - token_b="ETH", + token_a=tokens[0], + token_b=tokens[1], start_date=start, end_date=end, artifact_dir=args.artifact_dir, - pool_id=POOL_ID, + pool_id=pool_id, ) print(f" {arrays['n_days']} days, {arrays['n_minutes']} minutes") print(f" noise_base range: [{arrays['noise_base'].min():.2f}," @@ -96,7 +99,7 @@ def _build_market_linear_arrays(args): import os cache_dir = os.path.join(args.artifact_dir, "_sim_arrays") os.makedirs(cache_dir, exist_ok=True) - arrays_path = os.path.join(cache_dir, f"{POOL_ID}_{start}_{end}.npz") + arrays_path = os.path.join(cache_dir, f"{pool_id}_{start}_{end}.npz") np.savez(arrays_path, noise_base=arrays["noise_base"], noise_tvl_coeff=arrays["noise_tvl_coeff"], @@ -107,7 +110,7 @@ def _build_market_linear_arrays(args): # Get learned cadence from artifact from quantammsim.calibration.noise_model_arrays import load_artifact, _find_pool_index art, meta = load_artifact(args.artifact_dir) - pool_idx = _find_pool_index(POOL_ID, meta["pool_ids"]) + pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) if pool_idx >= 0: learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) print(f" Learned cadence: {learned_cadence:.1f} min") @@ -118,7 +121,7 @@ def _build_market_linear_arrays(args): return arrays_path, max(1, round(learned_cadence)) -def _build_mm_observed_arrays(args): +def _build_mm_observed_arrays(args, pool_id, tokens): """Precompute noise arrays from the MM model + DeFi Llama competitor TVL.""" from quantammsim.calibration.noise_model_arrays import ( build_mm_simulator_arrays, load_artifact, _find_pool_index, @@ -127,16 +130,16 @@ def _build_mm_observed_arrays(args): start = args.start_date.split(" ")[0] end = args.end_test_date.split(" ")[0] - print(f" Building mm_observed noise arrays for {POOL_ID}...") + print(f" Building mm_observed noise arrays for {pool_id}...") print(f" Date range: {start} → {end}") arrays = build_mm_simulator_arrays( - token_a="AAVE", - token_b="ETH", + token_a=tokens[0], + token_b=tokens[1], start_date=start, end_date=end, mm_artifact_dir=args.artifact_dir, competitor_tvl_path=args.competitor_tvl_path, - pool_id=POOL_ID, + pool_id=pool_id, ) print(f" {arrays['n_days']} days, {arrays['n_minutes']} minutes") print(f" noise_base range: [{arrays['noise_base'].min():.2f}," @@ -147,7 +150,7 @@ def _build_mm_observed_arrays(args): import os cache_dir = os.path.join(args.artifact_dir, "_sim_arrays") os.makedirs(cache_dir, exist_ok=True) - arrays_path = os.path.join(cache_dir, f"{POOL_ID}_{start}_{end}_mm.npz") + arrays_path = os.path.join(cache_dir, f"{pool_id}_{start}_{end}_mm.npz") np.savez(arrays_path, noise_base=arrays["noise_base"], competitor_tvl=arrays["competitor_tvl"]) @@ -155,7 +158,7 @@ def _build_mm_observed_arrays(args): # Get cadence from MM model artifact art, meta = load_artifact(args.artifact_dir) - pool_idx = _find_pool_index(POOL_ID, meta["pool_ids"]) + pool_idx = _find_pool_index(pool_id, meta["pool_ids"]) if pool_idx >= 0 and "log_cadence" in art: learned_cadence = float(np.exp(art["log_cadence"][pool_idx])) print(f" Learned cadence: {learned_cadence:.1f} min") @@ -190,6 +193,9 @@ def _build_opt_settings(args): "n_parameter_sets": args.n_parameter_sets, **({"val_fraction": args.val_fraction} if args.val_fraction is not None else {}), **robust, + "optuna_settings": { + "parameter_config": PARAMETER_CONFIG, + }, "cma_es_settings": { "population_size": args.cma_pop_size, "n_generations": args.cma_generations, @@ -219,7 +225,7 @@ def _build_opt_settings(args): } -def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): +def build_fingerprint(objective, args, tokens, noise_arrays_path=None, arb_freq=None): """Build run fingerprint with calibrated noise model.""" if args.noise_model == "mm_observed" and noise_arrays_path is not None: noise_block = { @@ -252,7 +258,7 @@ def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): return { "rule": "reclamm", - "tokens": ["AAVE", "ETH"], + "tokens": tokens, "startDateString": args.start_date, "endDateString": args.end_date, "endTestDateString": args.end_test_date, @@ -274,20 +280,20 @@ def build_fingerprint(objective, args, noise_arrays_path=None, arb_freq=None): } -def run_single(objective, args, noise_arrays_path=None, arb_freq=None): +def run_single(objective, args, tokens, pool_id, noise_arrays_path=None, arb_freq=None): """Run Optuna tuning for a single objective.""" print(f"\n{'='*60}") print(f" Objective: {objective}") print(f" Noise model: {args.noise_model}") print(f" Method: {args.method}") - print(f" Pool: AAVE/WETH Mainnet ({POOL_ID})") + print(f" Tokens: {'/'.join(tokens)} ({pool_id})") print(f" Train: {args.start_date} → {args.end_date}") print(f" Test: {args.end_date} → {args.end_test_date}") if arb_freq: print(f" Arb frequency: {arb_freq} min (learned)") print(f"{'='*60}\n") - fp = build_fingerprint(objective, args, noise_arrays_path, arb_freq) + fp = build_fingerprint(objective, args, tokens, noise_arrays_path, arb_freq) result = train_on_historic_data(fp, verbose=True) if result is not None: @@ -322,6 +328,10 @@ def main(): parser.add_argument("--cma-eval-points", type=int, default=20) parser.add_argument("--min-train-ret", type=float, default=-0.5, help="Reject trials with IS returns_over_hodl below this") + parser.add_argument("--tokens", nargs=2, default=DEFAULT_TOKENS, + help="Token pair (default: AAVE ETH)") + parser.add_argument("--pool-id", default=DEFAULT_POOL_ID, + help="Pool ID prefix for noise model lookup") parser.add_argument("--noise-model", default="market_linear", choices=["calibrated", "market_linear", "mm_observed"], help="Noise model variant") @@ -356,24 +366,32 @@ def main(): " None=standard mean. Try 0.5-2.0.") parser.add_argument("--output", type=str, default=None, help="Save results to JSON file") + parser.add_argument("--pr-max", type=float, default=None, + help="Override max price_ratio (default: 200)") args = parser.parse_args() + if args.pr_max is not None: + PARAMETER_CONFIG["price_ratio"]["high"] = args.pr_max + if args.all_objectives: objectives = OBJECTIVES else: objectives = [args.objective] + tokens = args.tokens + pool_id = args.pool_id + # Precompute noise arrays once noise_arrays_path = None arb_freq = None if args.noise_model == "market_linear": - noise_arrays_path, arb_freq = _build_market_linear_arrays(args) + noise_arrays_path, arb_freq = _build_market_linear_arrays(args, pool_id, tokens) elif args.noise_model == "mm_observed": - noise_arrays_path, arb_freq = _build_mm_observed_arrays(args) + noise_arrays_path, arb_freq = _build_mm_observed_arrays(args, pool_id, tokens) all_results = {} for obj in objectives: - result = run_single(obj, args, noise_arrays_path, arb_freq) + result = run_single(obj, args, tokens, pool_id, noise_arrays_path, arb_freq) all_results[obj] = result if args.output: From 51a8ab7e3115ed68d8f02ac01b6952205313ca5c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:02:14 +0100 Subject: [PATCH 092/115] feat: report only noise-trader fees as lp_fee_revenue_usd MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Switch lp_fee_revenue_usd to noise_fee_income only (excluding the swap fees collected on arb trades). Arb fees are still accumulated into pool reserves via the existing inbound/protocol-fee logic; they are simply not counted as "revenue" for the purposes of optimisation metrics. Rationale: optimising for total fees (arb + noise) leads to strategies that monetise pool decay — wide bands let arbs slosh trades through the pool, paying fees that look like revenue but are dwarfed by the LVR they extract. The pool value correctly reflects the net of fees vs IL, so any fee-based objective that uses lp_fee_revenue_usd should see only the non-toxic (noise-trader) flow. Otherwise the optimiser is rewarded for sloshing volume through a decaying pool. --- quantammsim/pools/reCLAMM/reclamm_reserves.py | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm_reserves.py b/quantammsim/pools/reCLAMM/reclamm_reserves.py index 1a2cafda..be1b1ad0 100644 --- a/quantammsim/pools/reCLAMM/reclamm_reserves.py +++ b/quantammsim/pools/reCLAMM/reclamm_reserves.py @@ -1131,10 +1131,11 @@ def _skip_schedule_state(_): Ra_new = Ra_new - protocol_fee[0] Rb_new = Rb_new - protocol_fee[1] - # LP fee revenue: arb swap fees (zero under blessed) + noise-trader fees - # (unchanged; noise traders still pay the pool's fee rate). - lp_fee_income = inbound * fee_rate * (1.0 - protocol_fee_split) - lp_fee_revenue_usd = (lp_fee_income * prices).sum() + noise_fee_income + # LP fee revenue: noise-trader fees only. + # Arb fee income is excluded — arb trades are net-negative for LPs + # (IL exceeds the fee collected), so reporting arb fees as "revenue" + # is misleading. + lp_fee_revenue_usd = noise_fee_income # Blessed-arb LVR return: arb returns gross profit minus gas + external cost # to the pool, scaled across effective reserves to preserve quoted price. From f3054b9d94f2e0d12692fd71f22fb9821325bb2f Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:04:49 +0100 Subject: [PATCH 093/115] feat: plot_reclamm_optuna_result improvements for final-sim panels MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Several enhancements to the value / weight plots used by run_final_sims.py: 1. Normalised mode: detect when input values start near 1.0 (vs millions) and switch y-axis label and scaling accordingly. Fee revenue and cumulative volume are then expressed as fraction of initial TVL. 2. Cumulative volume panel: new third panel under value + fee revenue, computed as 0.5 * sum(|Δreserves| * prices) per step. Useful for diagnosing strategies that monetise sloshing (high volume) vs strategies that capture noise flow (lower volume, better RoH). 3. Date range in title: titles now show "MMM YYYY — MMM YYYY" so the plotted period is clear without consulting the filename. 4. reCLAMM capitalisation (not reClAMM) in titles to match the spec. 5. Train/test divider only drawn when train_end_dt actually falls inside the plotted range — prevents stray vertical lines on test-only or train-only plots. 6. Plot grid spacing fix: switch to hspace=0.3 with bigger per-panel ratios so y-axis labels of stacked panels don't clash. --- scripts/plot_reclamm_optuna_result.py | 117 +++++++++++++++++++------- 1 file changed, 88 insertions(+), 29 deletions(-) diff --git a/scripts/plot_reclamm_optuna_result.py b/scripts/plot_reclamm_optuna_result.py index b90ba408..c9bf96ca 100644 --- a/scripts/plot_reclamm_optuna_result.py +++ b/scripts/plot_reclamm_optuna_result.py @@ -163,24 +163,39 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): """Two-panel plot: value-over-time + cumulative fee revenue.""" train_end_str = ref_config["endDateString"] train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") + start_dt = datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S") first_out = next(iter(time_series.values())) n_minutes = len(first_out["value"]) - dates = pd.date_range( - start=datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S"), - periods=n_minutes, freq="1min", - ) + dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") step = 1440 dates_daily = dates[::step] + # Detect normalised data (values near 1.0 vs millions) + normalised = ref_config.get("normalised", False) + if not normalised: + first_val = np.array(first_out["value"][0]) + normalised = first_val < 100 # heuristic: normalised data starts near 1.0 + + val_scale = 1.0 if normalised else 1e-6 + val_ylabel = "Normalised Value" if normalised else "Pool Value ($M USD)" + fee_scale = 1.0 if normalised else 1e-3 + fee_ylabel = ("Cum. Fee Revenue (fraction of TVL)" if normalised + else "Cumulative Fee Revenue ($K)") + has_fee_revenue = any( "fee_revenue" in time_series[n] and time_series[n]["fee_revenue"] is not None for n in time_series ) - n_panels = 2 if has_fee_revenue else 1 + has_reserves = any( + "reserves" in time_series[n] and "prices" in time_series[n] + for n in time_series + ) + n_panels = 1 + int(has_fee_revenue) + int(has_reserves) + ratios = [3] + [1.5] * (n_panels - 1) fig, axes = plt.subplots( - n_panels, 1, figsize=(14, 5 * n_panels), - sharex=True, gridspec_kw={"height_ratios": [3, 1] if n_panels == 2 else [1]}, + n_panels, 1, figsize=(14, 3.5 + 3 * n_panels), + sharex=True, gridspec_kw={"height_ratios": ratios, "hspace": 0.3}, ) if n_panels == 1: axes = [axes] @@ -189,7 +204,7 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): # ── Panel 1: Value over time ────────────────────────────────────── for name, meta, ci in _plot_order(configs): out = time_series[name] - vals = np.array(out["value"][::step]) / 1e6 + vals = np.array(out["value"][::step]) * val_scale label = f"{name}" if "test_objective" in meta: obj_name = meta.get("obj_name", "objective") @@ -200,22 +215,24 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): color=COLORS[ci % len(COLORS)], label=label, zorder=3 if is_optimized else 2) - hodl_daily = hodl_values[::step] / 1e6 + hodl_daily = hodl_values[::step] * val_scale ax_val.plot(dates_daily[:len(hodl_daily)], hodl_daily, linewidth=2, color="white", alpha=0.7, linestyle="--", label="HODL") - ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) - ylims = ax_val.get_ylim() - ax_val.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", - color="white", alpha=0.6, fontsize=11, ha="right", va="top") - ax_val.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", - color="white", alpha=0.6, fontsize=11, ha="left", va="top") + if train_end_dt > start_dt and train_end_dt < dates[-1]: + ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax_val.get_ylim() + ax_val.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax_val.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") _style_axis(ax_val) - ax_val.set_ylabel("Pool Value ($M USD)", color=TEXT_COLOR, fontsize=12) + ax_val.set_ylabel(val_ylabel, color=TEXT_COLOR, fontsize=12) tokens_str = "/".join(ref_config["tokens"]) + date_range_str = f"{start_dt.strftime('%b %Y')} — {dates[-1].strftime('%b %Y')}" ax_val.set_title( - f"reClAMM Optuna Comparison — {tokens_str}", + f"reCLAMM {tokens_str} — {date_range_str}", color=TEXT_COLOR, fontsize=13, pad=15, ) ax_val.legend(loc="upper left", fontsize=8, facecolor=BG, @@ -230,20 +247,60 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): if fr is None: continue fr = np.array(fr) - cumfee = np.cumsum(fr)[::step] / 1e3 + cumfee = np.cumsum(fr)[::step] * fee_scale is_optimized = "On-Chain" not in name ax_fee.plot(dates_daily[:len(cumfee)], cumfee, linewidth=2.5 if is_optimized else 1.8, color=COLORS[ci % len(COLORS)], label=name, zorder=3 if is_optimized else 2) - ax_fee.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + if train_end_dt > start_dt and train_end_dt < dates[-1]: + ax_fee.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) _style_axis(ax_fee) - ax_fee.set_ylabel("Cumulative Fee Revenue ($K)", color=TEXT_COLOR, fontsize=12) - ax_fee.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax_fee.set_ylabel(fee_ylabel, color=TEXT_COLOR, fontsize=12) ax_fee.legend(loc="upper left", fontsize=8, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) - else: + + # ── Panel 3: Cumulative volume ─────────────────────────────────── + if has_reserves: + ax_idx = 1 + int(has_fee_revenue) + ax_vol = axes[ax_idx] + vol_ylabel = ("Cum. Volume (multiple of TVL)" if normalised + else "Cum. Volume ($M)") + for name, _meta, ci in _plot_order(configs): + out = time_series[name] + res = np.array(out.get("reserves")) + pri = np.array(out.get("prices")) + if res is None or pri is None: + continue + # Volume ≈ 0.5 * sum(|Δreserves| * prices) per step + delta_r = np.diff(res, axis=0) + step_vol = 0.5 * np.sum(np.abs(delta_r) * pri[1:], axis=1) + cum_vol = np.cumsum(step_vol) + # Normalise: by initial TVL if normalised mode, else to $M + if normalised: + init_tvl = out.get("initial_tvl", 1.0) + vol_scale = 1.0 / max(init_tvl, 1.0) + else: + vol_scale = 1e-6 + # Pad to match dates_daily length + cum_vol_full = np.zeros(len(res)) + cum_vol_full[1:] = cum_vol + cum_vol_daily = cum_vol_full[::step] * vol_scale + is_optimized = "On-Chain" not in name + ax_vol.plot(dates_daily[:len(cum_vol_daily)], cum_vol_daily, + linewidth=2.5 if is_optimized else 1.8, + color=COLORS[ci % len(COLORS)], label=name, + zorder=3 if is_optimized else 2) + + if train_end_dt > start_dt and train_end_dt < dates[-1]: + ax_vol.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + _style_axis(ax_vol) + ax_vol.set_ylabel(vol_ylabel, color=TEXT_COLOR, fontsize=12) + ax_vol.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) + ax_vol.legend(loc="upper left", fontsize=8, facecolor=BG, + edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR) + elif not has_fee_revenue: ax_val.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) fig.patch.set_facecolor(BG) @@ -339,16 +396,18 @@ def plot_weights(configs, time_series, ref_config, args): zorder=3 if is_optimized else 2) ax.axhline(0.5, color="white", linestyle="--", alpha=0.3, linewidth=1) - ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) - ylims = ax.get_ylim() - ax.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", - color="white", alpha=0.6, fontsize=11, ha="right", va="top") - ax.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", - color="white", alpha=0.6, fontsize=11, ha="left", va="top") + if train_end_dt > start_dt and train_end_dt < dates[-1]: + ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ylims = ax.get_ylim() + ax.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", + color="white", alpha=0.6, fontsize=11, ha="right", va="top") + ax.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", + color="white", alpha=0.6, fontsize=11, ha="left", va="top") _style_axis(ax) tokens_str = "/".join(ref_config["tokens"]) - ax.set_title(f"Effective {token_name} Weight — {tokens_str}", + date_range_str = f"{start_dt.strftime('%b %Y')} — {dates[-1].strftime('%b %Y')}" + ax.set_title(f"Effective {token_name} Weight — reCLAMM {tokens_str} — {date_range_str}", color=TEXT_COLOR, fontsize=13, pad=15) ax.set_ylabel(f"{token_name} weight (value fraction)", color=TEXT_COLOR, fontsize=12) ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) From 4aa5623199d79dc7912b20037fde86cd6dbd468a Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:06:08 +0100 Subject: [PATCH 094/115] feat: extend period sweep with penalty + long_2021 axes; multi-pool plot script MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit run_period_sweep.sh: - New OVERFITTING_PENALTIES axis (defaults: ["" "1.0" "5.0"]); each value is passed as --overfitting-penalty to the tune script. - New ROBUST_TEMPS axis (defaults: ["" "0.5" "1.0" "0.1" "0.01"]); each value is passed as --robust-temperature. - New long_2021 period (2021-06-01 → 2025-01-01 train+val, → 2026-03-01 test) with optional --val-fraction parameter to give it a 40% val holdout matching its longer history. - Tags now combine objective + robust + penalty + period; existing result files are skipped on rerun. - Single-array overrides retained at the bottom of each axis block as runtime knobs. plot_sweep_results.py: - Refactor from single-pool BASE_FP/PERIODS into POOL_CONFIGS dict keyed by pool (aave_eth, cow_eth_mainnet, cow_eth_base) so the same script can plot sweep results for multiple pools. - Filenames now have suffix _mainnet/_base where needed. - Filename parsing in load_sweep_results recognises robust and penalty suffixes via _parse_variant; ALLOWED_ROBUST and ALLOWED_PENALTY constants control which combinations are loaded (defaults match the current run_period_sweep.sh active axes). - _short_label translates _robust0.5/_penalty5.0 to " r0.5"/" p5.0" for compact legend entries. - New --top N CLI flag ranks configs by test-period RoH and plots only the top N (plus on-chain baselines), saved with _topN suffix. - long_2021 added to aave_eth periods list. --- scripts/plot_sweep_results.py | 288 ++++++++++++++++++++++++++-------- scripts/run_period_sweep.sh | 77 +++++++-- 2 files changed, 285 insertions(+), 80 deletions(-) diff --git a/scripts/plot_sweep_results.py b/scripts/plot_sweep_results.py index b5a8778c..53d50ae5 100644 --- a/scripts/plot_sweep_results.py +++ b/scripts/plot_sweep_results.py @@ -27,41 +27,134 @@ from quantammsim.runners.jax_runners import do_run_on_historic_data -SWEEP_DIR = os.path.join( - os.path.dirname(os.path.dirname(__file__)), "results", "sweep", -) -OUTPUT_DIR = os.path.join( - os.path.dirname(os.path.dirname(__file__)), "results", "sweep", "plots", -) - -# Period definitions: (start, train_end, default_test_end) -# --end-test-date overrides default_test_end for all periods -PERIODS = { - "bull_2023": ("2023-06-01", "2024-06-01", "2026-03-01"), - "default_2024": ("2024-06-01", "2025-06-01", "2026-03-01"), - "recent_2025": ("2025-01-01", "2025-09-01", "2026-03-01"), +_RESULTS_ROOT = os.path.dirname(os.path.dirname(__file__)) + +# ── Pool configurations ───────────────────────────────────────────────── +POOL_CONFIGS = { + "aave_eth": { + "sweep_dir": os.path.join(_RESULTS_ROOT, "results", "sweep"), + "output_dir": os.path.join(_RESULTS_ROOT, "results", "sweep", "plots"), + "periods": { + "bull_2023": ("2023-06-01", "2024-06-01", "2026-03-01"), + "default_2024": ("2024-06-01", "2025-06-01", "2026-03-01"), + "recent_2025": ("2025-01-01", "2025-09-01", "2026-03-01"), + "long_2021": ("2021-06-01", "2025-01-01", "2026-03-01"), + }, + "base_fp": { + "rule": "reclamm", + "tokens": ["AAVE", "ETH"], + "do_arb": True, + "arb_frequency": 4, + "fees": 0.0025, + "gas_cost": 1.0, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": "results/mm_noise/_sim_arrays/" + "0x9d1fcf346ea1b0_2024-06-01_2026-03-01_mm.npz", + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "reclamm_learn_arc_length_speed": False, + "reclamm_use_shift_exponent": True, + "initial_pool_value": 20_000_000.0, + }, + "onchain_configs": { + "OnChain-launch": { + "price_ratio": 1.5, "centeredness_margin": 0.5, + "shift_exponent": 0.1, + }, + "OnChain-current": { + "price_ratio": 4.0, "centeredness_margin": 0.1, + "shift_exponent": 0.001, + }, + }, + "noise_builder": { + "token_a": "AAVE", "token_b": "ETH", + "pool_id": "0x9d1fcf346ea1b0", + }, + }, + "cow_eth_mainnet": { + "sweep_dir": os.path.join(_RESULTS_ROOT, "results", "cow_sweep"), + "output_dir": os.path.join(_RESULTS_ROOT, "results", "cow_sweep", "plots"), + "periods": { + "default_mainnet": ("2025-01-01", "2025-10-01", "2026-04-01"), + "recent_mainnet": ("2025-04-01", "2025-12-01", "2026-04-01"), + }, + "base_fp": { + "rule": "reclamm", + "tokens": ["COW", "ETH"], + "do_arb": True, + "arb_frequency": 3, + "fees": 0.003, + "gas_cost": 3.0, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": "", # built dynamically + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "reclamm_learn_arc_length_speed": False, + "reclamm_use_shift_exponent": True, + "initial_pool_value": 600_000.0, + }, + "onchain_configs": { + "OnChain-mainnet": { + "price_ratio": 2.02, "centeredness_margin": 0.5, + "shift_exponent": 0.1, + }, + }, + "noise_builder": { + "token_a": "COW", "token_b": "ETH", + "pool_id": "0xd321300ef77067", + }, + "fname_suffix": "_mainnet", + }, + "cow_eth_base": { + "sweep_dir": os.path.join(_RESULTS_ROOT, "results", "cow_sweep"), + "output_dir": os.path.join(_RESULTS_ROOT, "results", "cow_sweep", "plots"), + "periods": { + "default_base": ("2025-01-01", "2025-10-01", "2026-04-01"), + "recent_base": ("2025-04-01", "2025-12-01", "2026-04-01"), + }, + "base_fp": { + "rule": "reclamm", + "tokens": ["COW", "ETH"], + "do_arb": True, + "arb_frequency": 3, + "fees": 0.003, + "gas_cost": 0.01, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": "", + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "reclamm_learn_arc_length_speed": False, + "reclamm_use_shift_exponent": True, + "initial_pool_value": 500_000.0, + }, + "onchain_configs": { + "OnChain-base": { + "price_ratio": 3.30, "centeredness_margin": 0.5, + "shift_exponent": 0.1, + }, + }, + "noise_builder": { + "token_a": "COW", "token_b": "ETH", + "pool_id": "0xff028c1ec4559d", + }, + "fname_suffix": "_base", + }, } -# Base fingerprint for reClAMM AAVE/ETH with MM noise -BASE_FP = { - "rule": "reclamm", - "tokens": ["AAVE", "ETH"], - "do_arb": True, - "arb_frequency": 4, - "fees": 0.0025, - "gas_cost": 1.0, - "arb_fees": 0.0, - "protocol_fee_split": 0.25, - "noise_trader_ratio": 0.0, - "noise_model": "mm_observed", - "noise_arrays_path": "results/mm_noise/_sim_arrays/" - "0x9d1fcf346ea1b0_2024-06-01_2026-03-01_mm.npz", - "reclamm_interpolation_method": "geometric", - "reclamm_centeredness_scaling": False, - "reclamm_learn_arc_length_speed": False, - "reclamm_use_shift_exponent": True, - "initial_pool_value": 20_000_000.0, -} +# Active config — set by --pool arg in main() +SWEEP_DIR = POOL_CONFIGS["aave_eth"]["sweep_dir"] +OUTPUT_DIR = POOL_CONFIGS["aave_eth"]["output_dir"] +PERIODS = POOL_CONFIGS["aave_eth"]["periods"] +BASE_FP = POOL_CONFIGS["aave_eth"]["base_fp"] OBJ_SHORT = { "daily_log_sharpe": "sharpe", @@ -73,26 +166,62 @@ "weekly_rovar": "rovar", } + +def _short_label(obj_name): + """Convert obj_name (possibly with robust/penalty suffix) to short label.""" + for full, short in OBJ_SHORT.items(): + if obj_name.startswith(full): + suffix = obj_name[len(full):] + if suffix: + suffix = suffix.replace("_robust", " r") + suffix = suffix.replace("_penalty", " p") + return f"{short}{suffix}" + return obj_name + BG = "#162536" TEXT_COLOR = "#E6CE97" COLORS = [ "#3498db", "#2ecc71", "#e74c3c", "#f39c12", "#9b59b6", "#1abc9c", "#e67e22", "#2980b9", "#c0392b", "#8e44ad", + "#27ae60", "#d35400", "#16a085", "#f1c40f", "#7f8c8d", + "#e74c3c", "#3498db", "#2ecc71", "#f39c12", "#9b59b6", + "#1abc9c", "#e67e22", "#2980b9", "#c0392b", "#8e44ad", ] +ALLOWED_ROBUST = {""} +ALLOWED_PENALTY = {"", "5.0"} + + +def _parse_variant(obj_name): + """Extract (base_obj, robust, penalty) from an obj_name like 'calmar_robust0.5_penalty5.0'.""" + robust = "" + penalty = "" + rest = obj_name + m = re.search(r"_robust([\d.]+)", rest) + if m: + robust = m.group(1) + rest = rest[:m.start()] + rest[m.end():] + m = re.search(r"_penalty([\d.]+)", rest) + if m: + penalty = m.group(1) + rest = rest[:m.start()] + rest[m.end():] + return rest, robust, penalty + + def load_sweep_results(): - """Load all sweep result JSONs, grouped by period.""" - results = {} # period -> [(obj_name, params)] + """Load sweep result JSONs matching current sweep config, grouped by period.""" + results = {} for path in sorted(glob.glob(os.path.join(SWEEP_DIR, "*.json"))): fname = os.path.basename(path).replace(".json", "") - # Parse: {objective}_{period_name} for period_name in PERIODS: if fname.endswith(f"_{period_name}"): obj_name = fname[: -(len(period_name) + 1)] + _, robust, penalty = _parse_variant(obj_name) + if robust not in ALLOWED_ROBUST or penalty not in ALLOWED_PENALTY: + break with open(path) as f: data = json.load(f) - # Extract params from the first (only) objective key if isinstance(data, dict): params_raw = list(data.values())[0] if isinstance(params_raw, dict): @@ -111,6 +240,10 @@ def load_sweep_results(): _arrays_cache = {} +_active_noise_builder = POOL_CONFIGS["aave_eth"]["noise_builder"] +_active_onchain_configs = POOL_CONFIGS["aave_eth"]["onchain_configs"] + + def _get_noise_arrays_path(start_date, end_date): """Build or retrieve MM noise arrays for this date range.""" key = (start_date, end_date) @@ -119,22 +252,23 @@ def _get_noise_arrays_path(start_date, end_date): from quantammsim.calibration.noise_model_arrays import build_mm_simulator_arrays + nb = _active_noise_builder cache_dir = os.path.join( os.path.dirname(os.path.dirname(__file__)), "results", "mm_noise", "_sim_arrays") os.makedirs(cache_dir, exist_ok=True) arrays_path = os.path.join( - cache_dir, f"0x9d1fcf346ea1b0_{start_date}_{end_date}_mm.npz") + cache_dir, f"{nb['pool_id']}_{start_date}_{end_date}_mm.npz") if not os.path.exists(arrays_path): - print(f"\n Building noise arrays for {start_date}→{end_date}...", + print(f"\n Building noise arrays for {nb['pool_id']} {start_date}→{end_date}...", end=" ", flush=True) arrays = build_mm_simulator_arrays( - token_a="AAVE", token_b="ETH", + token_a=nb["token_a"], token_b=nb["token_b"], start_date=start_date, end_date=end_date, mm_artifact_dir="results/mm_noise", competitor_tvl_path="results/competitor_tvl/competitor_tvl.npz", - pool_id="0x9d1fcf346ea1b0", + pool_id=nb["pool_id"], ) np.savez(arrays_path, noise_base=arrays["noise_base"], @@ -169,7 +303,7 @@ def _style_axis(ax): ax.grid(True, alpha=0.15, color=TEXT_COLOR) -def plot_period(period_name, obj_results, end_test_date, output_dir): +def plot_period(period_name, obj_results, end_test_date, output_dir, top_n=None): """Plot all objectives for one period.""" start, train_end, default_test_end = PERIODS[period_name] test_end = end_test_date or default_test_end @@ -178,24 +312,15 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): print(f"Period: {period_name} ({start} → {train_end} → {test_end})") print(f"{'='*80}") - # On-chain baselines - ONCHAIN_CONFIGS = { - "OnChain-launch": { - "price_ratio": 1.5, "centeredness_margin": 0.5, - "shift_exponent": 0.1, - }, - "OnChain-current": { - "price_ratio": 4.0, "centeredness_margin": 0.1, - "shift_exponent": 0.001, - }, - } + # On-chain baselines (from active pool config) + ONCHAIN_CONFIGS = _active_onchain_configs # Run all configs + baselines runs = {} # label -> forward pass output run_params = {} # label -> pool params dict all_configs = ( [(name, params) for name, params in ONCHAIN_CONFIGS.items()] - + [(f"{OBJ_SHORT.get(obj, obj)} (pr={p.get('price_ratio', 0):.2f})", p) + + [(f"{_short_label(obj)} (pr={p.get('price_ratio', 0):.2f})", p) for obj, p in obj_results] ) for label, params in all_configs: @@ -214,7 +339,7 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): print(" No successful runs!") return - # HODL baseline + # HODL baseline (compute before filtering so test-period RoH is available) first_out = next(iter(runs.values())) hodl_reserves = first_out["reserves"][0] hodl_values = np.sum( @@ -223,6 +348,23 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): n_minutes = len(first_out["value"]) start_dt = datetime.strptime(f"{start} 00:00:00", "%Y-%m-%d %H:%M:%S") train_end_dt = datetime.strptime(f"{train_end} 00:00:00", "%Y-%m-%d %H:%M:%S") + train_minutes = int((train_end_dt - start_dt).total_seconds() / 60) + test_start_idx = min(train_minutes, n_minutes - 1) + + # Filter to top N by test-period normalised return + if top_n is not None and top_n < len(runs): + def _test_return(label): + vals = np.array(runs[label]["value"]) + if test_start_idx >= len(vals) - 1: + return float("-inf") + return vals[-1] / vals[test_start_idx] - 1.0 + + ranked = sorted(runs.keys(), key=_test_return, reverse=True) + keep = set(ranked[:top_n]) + keep.update(l for l in runs if l.startswith("OnChain")) + runs = {l: runs[l] for l in runs if l in keep} + run_params = {l: run_params[l] for l in run_params if l in keep} + print(f" Filtered to top {top_n} (test-period return) + baselines ({len(runs)} configs)") dates = pd.date_range(start=start_dt, periods=n_minutes, freq="1min") step = 1440 dates_daily = dates[::step] @@ -248,7 +390,8 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5) _style_axis(ax) ax.set_ylabel("Pool Value ($M)", color=TEXT_COLOR) - ax.set_title(f"reClAMM AAVE/ETH — {period_name}", + tokens_str = "/".join(BASE_FP["tokens"]) + ax.set_title(f"reClAMM {tokens_str} — {period_name}", color=TEXT_COLOR, fontsize=14, pad=10) ax.legend(loc="upper left", fontsize=7, facecolor=BG, edgecolor=TEXT_COLOR, labelcolor=TEXT_COLOR, ncol=2) @@ -274,14 +417,14 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): fig.patch.set_facecolor(BG) plt.tight_layout() - out_path = os.path.join(output_dir, f"sweep_{period_name}_value.png") + top_suffix = f"_top{top_n}" if top_n else "" + out_path = os.path.join(output_dir, f"sweep_{period_name}_value{top_suffix}.png") fig.savefig(out_path, dpi=200, bbox_inches="tight", facecolor=BG) plt.close(fig) print(f" Saved: {out_path}") # ── Plot 2: Test-only normalised value + cumulative fee revenue ── - train_minutes = int((train_end_dt - start_dt).total_seconds() / 60) - test_start = min(train_minutes, n_minutes - 1) + test_start = test_start_idx fig, axes = plt.subplots(2, 1, figsize=(16, 10), sharex=True, gridspec_kw={"height_ratios": [3, 1]}) @@ -337,7 +480,7 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): fig.patch.set_facecolor(BG) plt.tight_layout() - out_path = os.path.join(output_dir, f"sweep_{period_name}_test.png") + out_path = os.path.join(output_dir, f"sweep_{period_name}_test{top_suffix}.png") fig.savefig(out_path, dpi=200, bbox_inches="tight", facecolor=BG) plt.close(fig) print(f" Saved: {out_path}") @@ -361,24 +504,41 @@ def plot_period(period_name, obj_results, end_test_date, output_dir): def main(): parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--pool", default="aave_eth", + choices=list(POOL_CONFIGS.keys()), + help="Pool config to use (default: aave_eth)") parser.add_argument("--end-test-date", default=None, help="Override test end date for all periods") - parser.add_argument("--output-dir", default=OUTPUT_DIR) + parser.add_argument("--output-dir", default=None) parser.add_argument("--periods", nargs="+", default=None, help="Only plot these periods (default: all)") + parser.add_argument("--top", type=int, default=None, + help="Only plot top N configs by test-period RoH") args = parser.parse_args() - os.makedirs(args.output_dir, exist_ok=True) + # Set active pool config + global SWEEP_DIR, OUTPUT_DIR, PERIODS, BASE_FP + global _active_noise_builder, _active_onchain_configs + pc = POOL_CONFIGS[args.pool] + SWEEP_DIR = pc["sweep_dir"] + OUTPUT_DIR = pc["output_dir"] + PERIODS = pc["periods"] + BASE_FP = pc["base_fp"] + _active_noise_builder = pc["noise_builder"] + _active_onchain_configs = pc["onchain_configs"] + + output_dir = args.output_dir or OUTPUT_DIR + os.makedirs(output_dir, exist_ok=True) results = load_sweep_results() print(f"Loaded {sum(len(v) for v in results.values())} results" - f" across {len(results)} periods") + f" across {len(results)} periods (pool={args.pool})") for period_name in sorted(results.keys()): if args.periods and period_name not in args.periods: continue plot_period(period_name, results[period_name], - args.end_test_date, args.output_dir) + args.end_test_date, output_dir, top_n=args.top) if __name__ == "__main__": diff --git a/scripts/run_period_sweep.sh b/scripts/run_period_sweep.sh index ea6931b4..8233b1ad 100644 --- a/scripts/run_period_sweep.sh +++ b/scripts/run_period_sweep.sh @@ -1,5 +1,5 @@ #!/bin/bash -# Full sweep: all objectives × all periods × 400 trials +# Full sweep: objectives × periods × robust temps × overfitting penalties × 400 trials # Usage: bash scripts/run_period_sweep.sh # Monitor: tail -5 /tmp/tune_*.log # Results: results/sweep/ @@ -21,13 +21,29 @@ OBJECTIVES=( weekly_rovar ) -# period_name start_date end_date(train) end_test_date +# period_name start_date end_date(train+val) end_test_date [val_fraction] PERIODS=( "bull_2023 2023-06-01 2024-06-01 2025-06-01" "default_2024 2024-06-01 2025-06-01 2026-03-01" "recent_2025 2025-01-01 2025-09-01 2026-03-01" + "long_2021 2021-06-01 2025-01-01 2026-03-01 0.4" ) +# Robust temperatures: "" = standard mean, otherwise --robust-temperature X +# Lower = more pessimistic (0.1 = very robust, 0.01 = near worst-case). +ROBUST_TEMPS=("" "0.5" "1.0" "0.1" "0.01") + +ROBUST_TEMPS=("") + + +# Overfitting penalty: "" = default (0.2), otherwise --overfitting-penalty X +# Higher = stronger train-vs-val regularization. penalty=1 means objective +# becomes val_score when train > val; penalty>1 rewards val > train. +OVERFITTING_PENALTIES=("" "1.0" "5.0") + +OVERFITTING_PENALTIES=("" "5.0") + + OUTDIR="results/sweep" mkdir -p "$OUTDIR" @@ -39,28 +55,57 @@ wait_for_slot() { N=0 for period_line in "${PERIODS[@]}"; do - read -r period_name start_date end_date end_test_date <<< "$period_line" + read -r period_name start_date end_date end_test_date val_fraction <<< "$period_line" + val_flag="" + if [ -n "$val_fraction" ]; then + val_flag="--val-fraction $val_fraction" + fi for obj in "${OBJECTIVES[@]}"; do - tag="${obj}_${period_name}" - logfile="/tmp/tune_${tag}.log" - outfile="${OUTDIR}/${tag}.json" + for temp in "${ROBUST_TEMPS[@]}"; do + for penalty in "${OVERFITTING_PENALTIES[@]}"; do + # Build tag and flags + tag="${obj}" + robust_flag="" + penalty_flag="" + if [ -n "$temp" ]; then + tag="${tag}_robust${temp}" + robust_flag="--robust-temperature $temp" + fi + if [ -n "$penalty" ]; then + tag="${tag}_penalty${penalty}" + penalty_flag="--overfitting-penalty $penalty" + fi + tag="${tag}_${period_name}" + + logfile="/tmp/tune_${tag}.log" + outfile="${OUTDIR}/${tag}.json" + + # Skip if result already exists + if [ -f "$outfile" ]; then + echo "[$N] Skipping (exists): ${tag}" + N=$((N + 1)) + continue + fi - wait_for_slot + wait_for_slot - echo "[$N] Launching: ${tag}" - $COMMON --objective "$obj" \ - --start-date "${start_date} 00:00:00" \ - --end-date "${end_date} 00:00:00" \ - --end-test-date "${end_test_date} 00:00:00" \ - --output "$outfile" \ - > "$logfile" 2>&1 & + echo "[$N] Launching: ${tag}" + $COMMON --objective "$obj" \ + --start-date "${start_date} 00:00:00" \ + --end-date "${end_date} 00:00:00" \ + --end-test-date "${end_test_date} 00:00:00" \ + $robust_flag $penalty_flag $val_flag \ + --output "$outfile" \ + > "$logfile" 2>&1 & - N=$((N + 1)) + N=$((N + 1)) + done + done done done echo "" -echo "$N jobs queued (${#OBJECTIVES[@]} objectives × ${#PERIODS[@]} periods × $TRIALS trials)" +echo "$N jobs total (existing results skipped)" echo "Max parallel: $MAX_PARALLEL" echo "" echo "Monitor: tail -5 /tmp/tune_*.log" From 366999ac8f4a6cf44dd558dd25471ee5b8fb2015 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:09:22 +0100 Subject: [PATCH 095/115] feat: end-to-end reClAMM training + evaluation pipeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Four scripts that together define the soup-to-nuts pipeline for a full reClAMM parameter sweep + selection + forward-simulation cycle. scripts/run_full_sweep.sh Bash entry point for the parameter sweep. Iterates over a CONFIGS list (pool × TVL × gas × fees), objectives, and overfitting penalties, with optional --method=optuna|cma_es and --pair= filters. Throttles to MAX_PARALLEL workers; CMA-ES defaults to 4, Optuna to 8 (or the MAX_WORKERS env override). PR_MAX env var caps the price-ratio search via --pr-max for Optuna runs. Existing result files are skipped so reruns are incremental. scripts/evaluate_trials.py General-purpose loader and analyser for the result hash files written to results/. Walks the trial JSONs, deserialises params, and exposes load_reclamm_results which returns a list of dicts with study_id, tokens, initial_pool_value, params, train/val metrics, return_val (objective), and timestamps. Used by select_best_params and run_final_sims to discover candidate trials without re-running. scripts/select_best_params.py Picks the winning param set per (tokens, TVL). Ranks candidates by validation RoH (best_val_roh) and prints the chosen trial along with PR/margin/shift values, train/val metrics, and the source study_id. --all summarises every TVL config in one go. scripts/run_final_sims.py Forward-simulation runner. For each pair × TVL, loads selected params via select_best_params, builds noise arrays for both train and test periods (calling build_mm_simulator_arrays when not cached), runs do_run_on_historic_data, and dispatches to plot_reclamm_optuna_result for value / fee_revenue / volume / weight panels. Normalises all output by initial TVL so different TVL levels are directly comparable on a single plot, and records the initial_tvl per config so the volume panel can express cumulative volume as a multiple of TVL. --- scripts/evaluate_trials.py | 2054 +++++++++++++++++++++++++++++++++ scripts/run_final_sims.py | 358 ++++++ scripts/run_full_sweep.sh | 142 +++ scripts/select_best_params.py | 192 +++ 4 files changed, 2746 insertions(+) create mode 100644 scripts/evaluate_trials.py create mode 100644 scripts/run_final_sims.py create mode 100755 scripts/run_full_sweep.sh create mode 100644 scripts/select_best_params.py diff --git a/scripts/evaluate_trials.py b/scripts/evaluate_trials.py new file mode 100644 index 00000000..96bb63e2 --- /dev/null +++ b/scripts/evaluate_trials.py @@ -0,0 +1,2054 @@ +import os +import argparse +import json +from pathlib import Path +import pandas as pd +import numpy as np +import itertools +import matplotlib.pyplot as plt +import matplotlib as mpl +import seaborn as sns +import seaborn.objects as so +from jax import clear_caches + +import gc +import ast +from datetime import datetime + +from quantammsim.runners.jax_runners import do_run_on_historic_data +from quantammsim.pools.G3M.balancer.balancer import BalancerPool +from quantammsim.runners.jax_runner_utils import NestedHashabledict +from quantammsim.utils.data_processing.historic_data_utils import get_historic_parquet_data + +from quantammsim.core_simulator.param_utils import retrieve_best, calc_lamb, lamb_to_memory_days_clipped, memory_days_to_logit_lamb +import jax.numpy as jnp +from jax.tree_util import tree_map +from jax import config + +config.update("jax_compilation_cache_dir", "/tmp/jax_cache") +config.update("jax_persistent_cache_min_entry_size_bytes", -1) +config.update("jax_persistent_cache_min_compile_time_secs", 0) + +import warnings +warnings.filterwarnings('ignore') + + +"""Configuration for SGD analysis.""" + + +from quantammsim.utils.post_train_analysis import ( + calculate_period_metrics, + calculate_continuous_test_metrics, +) + + +def _json_default(obj): + """Serialise JAX/NumPy arrays and scalars for json.dump.""" + if hasattr(obj, "tolist"): + return obj.tolist() + if isinstance(obj, (np.integer, np.floating, np.bool_)): + return obj.item() + raise TypeError(f"Object of type {type(obj).__name__} is not JSON serialisable") + + +# Keys at the top of optimisation_settings that only matter when +# method='gradient_descent'. For other methods (optuna/cma_es/bfgs) these +# are inherited defaults that were never actually used during training; +# surfacing them in output columns is misleading, so they get NaN'd. +_GRADIENT_DESCENT_ONLY_KEYS = ( + "sample_method", "use_gradient_clipping", "clip_norm", + "optimiser", "base_lr", "batch_size", "use_plateau_decay", + "lr_schedule_type", "warmup_steps", "weight_decay", "min_lr", + "decay_lr_ratio", "decay_lr_plateau", "n_iterations", + "early_stopping", "early_stopping_patience", "use_swa", + "swa_start_frac", "swa_freq", +) + +# For non-gradient-descent methods, the real hyperparams live in a nested +# sub-dict. Mapping method → sub-dict key. +_METHOD_SUBDICT = { + "optuna": "optuna_settings", + "cma_es": "cma_es_settings", + "bfgs": "bfgs_settings", +} + + +def _method_hyperparams(run_fingerprint): + """Return the set of hyperparams that were actually relevant to this run. + + Emitted as a single ``hyperparams`` column (JSON-encoded) so a reader can + see at a glance what each run's optimiser was actually configured with, + regardless of method. For gradient_descent: a curated flat slice of the + top-level optimisation_settings keys. For optuna/cma_es/bfgs: the + matching ``*_settings`` sub-dict (with noisy fields like ``parameter_config`` + and ``storage`` dropped). + """ + opt = run_fingerprint.get("optimisation_settings", {}) + method = opt.get("method") + if method == "gradient_descent": + return {k: opt.get(k) for k in _GRADIENT_DESCENT_ONLY_KEYS if k in opt} + sub_key = _METHOD_SUBDICT.get(method) + if not sub_key: + return {} + sub = dict(opt.get(sub_key, {})) + sub.pop("parameter_config", None) # verbose; not useful as a column + sub.pop("storage", None) + return sub + + +def _gd_only_columns(run_fingerprint): + """Adam-family-only columns for the result dict. Returns every relevant + column populated for gradient_descent runs, or all-None for any other + method (stops inherited defaults from masquerading as actual config). + """ + opt = run_fingerprint["optimisation_settings"] + is_gd = opt.get("method") == "gradient_descent" + fetch = (lambda k: opt.get(k)) if is_gd else (lambda k: None) + return { + "sample_method": fetch("sample_method"), + "use_gradient_clipping": fetch("use_gradient_clipping"), + "clip_norm": fetch("clip_norm"), + "optimiser": fetch("optimiser"), + "learning_rate": fetch("base_lr"), + "batch_size": fetch("batch_size"), + "use_plateau_decay": fetch("use_plateau_decay"), + "lr_schedule_type": fetch("lr_schedule_type"), + "warmup_steps": fetch("warmup_steps"), + } +# from quantammsim.utils.plot_utils import name_to_latex_name, plot_weights + +# Environment setup +ENV_VARS = { + "XLA_FLAGS": "--xla_cpu_multi_thread_eigen=false intra_op_parallelism_threads=1", + "OPENBLAS_NUM_THREADS": "1", + "MKL_NUM_THREADS": "1", + "OMP_NUM_THREAD": "1", +} + +# Directory configuration +BASE_DIR = Path("./") +PLOT_DIR = BASE_DIR / "product_training_runs/plots" +RESULTS_DIR = BASE_DIR / "analysis_results" + +# Plot styling +PLOT_STYLE = { + "COLOR": "#E6CE97", + "BACKGROUND": "#162536", + "SNS_CONFIG": { + "text.color": "#E6CE97", + "axes.labelcolor": "#E6CE97", + "xtick.color": "#E6CE97", + "ytick.color": "#E6CE97", + "figure.facecolor": "#162536", + "axes.facecolor": "#162536", + "text.usetex": True, + "axes.grid": False, + }, +} + + +# Create required directories +for directory in [PLOT_DIR, RESULTS_DIR]: + directory.mkdir(parents=True, exist_ok=True) + +# # Set environment variables +# for key, value in ENV_VARS.items(): +# os.environ[key] = value + + +def make_run_period(name, start_date, end_date, end_test_date): + """Construct a run-period dict used throughout the analysis. + + The ``name`` is used both as a filename-safe identifier AND as the + display label in simplified-CSV column headers (e.g. + ``Returns test (baseline)``). Pick something short and readable. + """ + return { + "name": name, + "start_date": start_date, + "end_date": end_date, + "end_test_date": end_test_date, + } + + +def default_run_period_for_trial(trial): + """Synthesise a run period from a trial's own fingerprint dates. + + Used when the caller passes no explicit run periods — equivalent to the + legacy ``period_from_rf`` behaviour but named ``trained`` for clearer + column headers. + """ + return make_run_period( + name="trained", + start_date=trial.get("start_date") or trial.get("run_fingerprint", {}).get("startDateString"), + end_date=trial.get("end_date") or trial.get("run_fingerprint", {}).get("endDateString"), + end_test_date=trial.get("end_test_date") or trial.get("run_fingerprint", {}).get("endTestDateString"), + ) + + +def parse_run_period_cli(values): + """argparse helper: turn a 4-token ``--run-period NAME START END TEST_END`` + invocation into a run-period dict.""" + if len(values) != 4: + raise argparse.ArgumentTypeError( + "--run-period expects 4 values: NAME START END TEST_END" + ) + name, start, end, test_end = values + return make_run_period(name, start, end, test_end) + +def name_to_latex_name_OG(name): + """Convert run name to LaTeX formatted name. + + Parameters + ---------- + name : str + Name of the run (e.g. 'index_market_cap', 'momentum') + + Returns + ------- + str + LaTeX formatted name + """ + # Special case for index_market_cap since we want to shorten it + if name == "index_market_cap": + return "\\mathrm{Index}" + + # Split name into words + words = name.split("_") + + # Capitalize first letter of each word + words = [word.capitalize() for word in words] + + # Join with escaped spaces and wrap in \mathrm{} + latex_name = "\\ ".join(words) + return f"$\\mathrm{{{latex_name}}}$" + + +def name_to_latex_name(name): + """Convert run name to clean LaTeX formatted name. + + Parameters + ---------- + name : str + Name of the run (e.g. 'Current_index_BTC-ETH_min_0.1_index_memory_day_30.0') + + Returns + ------- + str + LaTeX formatted name + """ + # Handle different strategy types + if name.startswith("Current_index"): + return "$\\mathrm{Current\\ Index\\ Product}$" + elif name.startswith("HODL"): + return "$\\mathrm{HODL}$" + elif name.startswith("QuantAMM_index"): + return "$\\mathrm{QuantAMM\\ Index}$" + elif name.startswith("Optimized_QuantAMM"): + # Extract rule name from pattern "..._rule_RULENAME" + rule = name.split("rule_")[-1] + # Clean up rule name + rule = rule.replace("_", " ").title() + # Special case for specific rules + if rule == "Mean Reversion Channel": + rule = "Mean-Reversion\\ Channel" + elif rule == "Anti Momentum": + rule = "Anti-Momentum" + elif rule == "Power Channel": + rule = "Power-Channel" + elif rule == "Difference Momentum": + rule = "Difference\\ Momentum" + elif rule == "Triple Threat Mean Reversion Channel": + rule = "Triple\\ Threat\\ Mean\\ Reversion\\ Channel" + elif rule == "Bounded Mean Reversion Channel": + rule = "Bounded\\ Mean\\ Reversion\\ Channel" + else: + rule = rule.replace(" ", "\\ ") + return f"$\\mathrm{{QuantAMM\\ {rule}}}$" + + return name_to_latex_name_OG(name) + +def plot_weights(output_dict, run_fingerprint, plot_prefix="weights", verbose=True): + plot_path = Path("./plots/") + plot_path.mkdir(parents=True, exist_ok=True) + plot_prefix = "./plots/" + plot_prefix + + # Calculate weights from reserves and prices + total_value = np.sum(output_dict["reserves"] * output_dict["prices"], axis=1, keepdims=True) + weights = np.array(output_dict["reserves"] * output_dict["prices"] / total_value) + + weights = weights[::1440] + # Create DataFrame for plotting + df_list = [] + tokens = sorted(run_fingerprint["tokens"]) + for i, token in enumerate(tokens): + df_list.extend([ + { + "Time": t, + "Weight": w, + "Token": token + } + for t, w in enumerate(weights[:, i]) + ]) + + df = pd.DataFrame(df_list) + start_date = datetime.strptime(run_fingerprint["startDateString"], "%Y-%m-%d %H:%M:%S") + end_date = datetime.strptime(run_fingerprint["endDateString"], "%Y-%m-%d %H:%M:%S") + + # Create date range for x-axis + date_range = pd.date_range(start=start_date, end=end_date, periods=len(df["Time"].unique())) + df["Time"] = np.tile(date_range, weights.shape[1]) + + # fig, ax = plt.subplots(figsize=(10, 6)) + f = mpl.figure.Figure() + + # Create stacked area plot + pl = ( + so.Plot(df, "Time", "Weight", color="Token") + .add(so.Area(alpha=0.7), so.Stack()) + .limit(y=(0, 1)) + .scale(color=sns.color_palette()) + .label(y="$\\mathrm{Weight}$", x="$\\mathrm{Date}$") + ) + + # Render the plot on our axis + res = pl.on(f).plot() + ax = f.axes[0] + # Select sparse dates for x-axis (4 evenly spaced dates) + unique_dates = df["Time"].unique() + date_indices = np.linspace(0, len(unique_dates)-1, 4, dtype=int) + selected_dates = unique_dates[date_indices] + + # Format dates as LaTeX strings + date_labels = [f"$$\\mathrm{{{pd.Timestamp(date).strftime('%Y-%m-%d')}}}$$" for date in selected_dates] + # Set the ticks and labels + ax.set_xticks(date_indices,date_labels, rotation=45) + # plt.xticks(date_indices, date_labels, rotation=45) + # Adjust layout to prevent label cutoff + plt.tight_layout() + # Save plot + pl.save( + plot_prefix + "_weights_over_time.png", + dpi=700, + bbox_inches="tight" + ) + plt.close() + +def load_reclamm_results(base_dir, load_method="best_train_objective", metric_key="returns_over_hodl"): + """Load reClAMM training results from hash-named JSON files. + + Uses load_manually to read the training trajectory and pick the best + checkpoint. Handles scalar params (n_parameter_sets=1) correctly. + + Returns list of trial dicts compatible with analyze_best_trials. + """ + from quantammsim.core_simulator.param_utils import load_manually + trials_info = [] + base_path = Path(base_dir) + + for file_path in sorted(base_path.glob("run_*.json")): + try: + param_data, context = load_manually( + str(file_path), load_method=load_method, recalc_hess=False, + min_test=-1.0, return_as_iterables=False, + metric_key=metric_key, + ) + except Exception as e: + print(f" Skipping {file_path.name}: {e}") + continue + + # Load run fingerprint from first entry + with open(file_path, encoding="utf-8") as f: + data = json.load(f) + data = json.loads(data) + run_fingerprint = data[0] + + # Skip non-reClAMM runs + if run_fingerprint.get("rule") != "reclamm": + continue + + # Extract train/test metrics + train_obj = param_data.get("train_objective", {}) + test_obj = param_data.get("test_objective", {}) + cont_test = param_data.get("continuous_test_metrics", {}) + + def _extract_metric(obj, key): + """Extract a scalar metric from train/test objective (handles dict, list, scalar).""" + if obj is None: + return float("-inf") + if isinstance(obj, dict): + return float(obj.get(key, obj.get("returns_over_hodl", float("-inf")))) + if isinstance(obj, (list, np.ndarray)): + # Multi-param-set: index by context, or take max + if isinstance(context, (int, np.integer)) and len(obj) > context: + entry = obj[context] + else: + entry = obj[0] if len(obj) > 0 else float("-inf") + if isinstance(entry, dict): + return float(entry.get(key, float("-inf"))) + return float(entry) + return float(obj) + + train_val = _extract_metric(train_obj, metric_key) + test_val = _extract_metric(test_obj, metric_key) + + # Build param dict — handle scalar vs array params + param_fields = {} + for k, v in param_data.items(): + if k in ("train_objective", "test_objective", "objective", + "subsidary_params", "continuous_test_metrics", + "step", "hessian_trace", "local_learning_rate", + "iterations_since_improvement"): + continue + if isinstance(v, dict): + continue + try: + arr = jnp.array(v) + # For scalar reClAMM params with context index + if arr.ndim >= 1 and isinstance(context, (int, np.integer)): + arr = arr[context] + param_fields[k] = arr + except Exception: + continue + + # Zero out initial_weights_logits + n_assets = len(run_fingerprint.get("tokens", [])) + if "initial_weights_logits" in param_fields: + param_fields["initial_weights_logits"] = jnp.zeros_like(param_fields["initial_weights_logits"]) + else: + param_fields["initial_weights_logits"] = jnp.zeros(n_assets) + + # Overall objective (matches load_sgd_results format) + obj_raw = param_data.get("objective", train_val) + if isinstance(obj_raw, (list, np.ndarray)): + obj_val = float(obj_raw[context]) if isinstance(context, (int, np.integer)) else float(obj_raw[0]) + elif isinstance(obj_raw, dict): + obj_val = float(obj_raw.get(metric_key, train_val)) + else: + obj_val = float(obj_raw) if obj_raw is not None else float(train_val) + + trial_info = { + "study_id": file_path.stem, + "trial_number": param_data.get("step", 0), + "train_value": float(train_val), + "test_value": float(test_val), + "objective": obj_val, + "train_objective": train_obj, + "test_objective": test_obj, + "continuous_test_metrics": cont_test, + "rule": run_fingerprint.get("rule"), + "tokens": tuple(sorted(run_fingerprint.get("tokens", []))), + "start_date": run_fingerprint.get("startDateString"), + "end_date": run_fingerprint.get("endDateString"), + "end_test_date": run_fingerprint.get("endTestDateString"), + "return_val": run_fingerprint.get("return_val", "sharpe"), + "chunk_period": run_fingerprint.get("chunk_period"), + "weight_interpolation_period": run_fingerprint.get("weight_interpolation_period"), + "minimum_weight": run_fingerprint.get("minimum_weight"), + "bout_offset": int(run_fingerprint.get("bout_offset", 0)), + "initial_pool_value": float(run_fingerprint.get("initial_pool_value", 0.0)), + "datetime": datetime.now().strftime("%Y-%m-%d %H:%M:%S"), + "params": param_fields, + "run_fingerprint": run_fingerprint, + } + trials_info.append(trial_info) + + return trials_info + + +def load_sgd_results(base_dir, load_method="best_objective", min_test=None): + """Load and parse SGD results files. + + Parameters + ---------- + base_dir : str or Path + Directory containing SGD result JSON files + load_method : str, optional + Method for selecting parameter sets. One of: + 'last', 'best_objective', 'best_train_objective', 'best_test_objective', + 'best_train_min_test_objective' + min_test : float, optional + Minimum test objective threshold for methods that use it + + Returns + ------- + list + List of trial info dictionaries matching Optuna format + """ + trials_info = [] + base_path = Path(base_dir) + + for file_path in base_path.glob("run_*"): + # Use retrieve_best to get cleaned parameters + # try: + params, steps = retrieve_best( + str(file_path), load_method, re_calc_hess=False, min_alt_obj=min_test, return_as_iterables=True + ) + # Load run fingerprint from first entry + with open(file_path, encoding="utf-8") as f: + data = json.load(f) + data = json.loads(data) + run_fingerprint = data[0] + # Format trial info to match Optuna structure + for param, step in zip(params, steps): + trial_info = { + "study_id": file_path.stem, + "trial_number": step, + # Parameters are already indexed by retrieve_best + "train_value": float(param["train_objective"]["returns_over_uniform_hodl"]) if isinstance(param["train_objective"], dict) else float(param["train_objective"]), + "test_value": float(param["test_objective"]["returns_over_uniform_hodl"]) if isinstance(param["test_objective"], dict) else float(param["test_objective"]), + "objective": float(param["objective"]), + # Configuration info from run fingerprint + "rule": run_fingerprint.get("rule"), + "tokens": tuple(sorted(run_fingerprint.get("tokens", []))), + "start_date": run_fingerprint.get("startDateString"), + "end_date": run_fingerprint.get("endDateString"), + "end_test_date": run_fingerprint.get("endTestDateString"), + "return_val": run_fingerprint.get("return_val", "sharpe"), + "chunk_period": run_fingerprint.get("chunk_period"), + "weight_interpolation_period": run_fingerprint.get("weight_interpolation_period"), + "minimum_weight": run_fingerprint.get("minimum_weight"), + "bout_offset": run_fingerprint.get("bout_offset"), + "datetime": datetime.now().strftime("%Y-%m-%d %H:%M:%S"), + "initial_pool_value": float( + run_fingerprint.get("initial_pool_value", 0.0) + ), + "bout_offset": int(run_fingerprint.get("bout_offset", 0)), + } + + # Add parameters with param_ prefix + # retrieve_best has already cleaned metadata and indexed parameters + param_fields = { + k: jnp.array(v) + for k, v in param.items() + if k + not in [ + "train_objective", + "test_objective", + "objective", + "subsidary_params", + "continuous_test_metrics", + ] + and not isinstance(v, dict) + } + # Set logit_delta_lamb to zeros if present + if "logit_delta_lamb" in param_fields: + param_fields["logit_delta_lamb"] = jnp.zeros_like(param_fields["logit_delta_lamb"]) + if "initial_weights_logits" in param_fields: + param_fields["initial_weights_logits"] = jnp.zeros_like(param_fields["initial_weights_logits"]) + trial_info.update({"params": param_fields}) + trial_info.update({"run_fingerprint": run_fingerprint}) + # Validate required fields + required_fields = ["rule", "tokens", "train_value", "test_value"] + if all(trial_info.get(f) is not None for f in required_fields): + trials_info.append(trial_info) + else: + print(f"Skipping {file_path}: missing required fields") + # except Exception as e: + # print(f"Skipping {file_path}: {e}") + # continue + + return trials_info + + +def convert_daily_to_hourly_params(params, run_fingerprint, scale_k_by_frequency=False): + """Convert parameters from daily (1440min) to hourly (60min) updates. + + Parameters + ---------- + params : dict + Original parameter dictionary + run_fingerprint : dict + Run fingerprint containing chunk_period info + scale_k_by_frequency : bool + If True, scale k down by 24 to maintain similar daily aggressiveness. + If False, keep k unchanged (resulting in 24x more aggressive daily behavior). + + Returns + ------- + tuple + (converted_params, converted_fingerprint) + """ + + # Only convert if original chunk_period is 1440 + if run_fingerprint.get("chunk_period", 60) != 1440: + return params, run_fingerprint + + converted_params = tree_map(lambda x: x.copy() if hasattr(x, 'copy') else x, params) + converted_fingerprint = run_fingerprint.copy() + + original_chunk_period = 1440 + new_chunk_period = 60 + + print(f"Converting daily to hourly parameters:") + print(f" Original chunk_period: {original_chunk_period}") + print(f" New chunk_period: {new_chunk_period}") + + # Adjust logit_lamb to preserve memory days + if "logit_lamb" in params: + current_lamb = calc_lamb(params) + current_memory_days = lamb_to_memory_days_clipped( + current_lamb, + original_chunk_period, + max_memory_days=365 + ) + + # Calculate new logit_lamb to achieve same memory days with new chunk_period + new_logit_lamb = memory_days_to_logit_lamb(current_memory_days, new_chunk_period) + converted_params["logit_lamb"] = new_logit_lamb + + print(f" Memory days preserved: {current_memory_days}") + print(f" Original logit_lamb: {params['logit_lamb']}") + print(f" New logit_lamb: {new_logit_lamb}") + else: + print(" No logit_lamb found - memory days adjustment skipped") + + # Handle k parameter scaling + if scale_k_by_frequency: + frequency_ratio = original_chunk_period / new_chunk_period # 24 + + if "log_k" in params: + converted_params["log_k"] = params["log_k"] - jnp.log2(frequency_ratio) + print(f" Scaling log_k: {params['log_k']} -> {converted_params['log_k']} (reduction by log2({frequency_ratio}))") + elif "k" in params: + converted_params["k"] = params["k"] / frequency_ratio + print(f" Scaling k: {params['k']} -> {converted_params['k']} (divided by {frequency_ratio})") + else: + print(" No k or log_k found - k scaling skipped") + + print(" K scaling: ON (maintaining daily aggressiveness)") + else: + print(" K scaling: OFF (24x more aggressive daily behavior)") + + # Update chunk_period in fingerprint + converted_fingerprint["chunk_period"] = new_chunk_period + + # Scale other timing-related parameters + frequency_ratio = original_chunk_period / new_chunk_period # 24 + + # Scale weight_interpolation_period if present + if "weight_interpolation_period" in converted_fingerprint: + original_wip = converted_fingerprint["weight_interpolation_period"] + converted_fingerprint["weight_interpolation_period"] = int(original_wip / frequency_ratio) + print(f" Scaling weight_interpolation_period: {original_wip} -> {converted_fingerprint['weight_interpolation_period']}") + + # Scale bout_length if present (this controls the simulation length) + if "bout_length" in converted_fingerprint: + original_bout_length = converted_fingerprint["bout_length"] + converted_fingerprint["bout_length"] = int(original_bout_length / frequency_ratio) + print(f" Scaling bout_length: {original_bout_length} -> {converted_fingerprint['bout_length']}") + + print(f" Conversion complete!") + + return converted_params, converted_fingerprint + +def analyze_specific_trials( + base_dir, + trials_to_analyze, + return_val="sharpe", + load_method="best_objective", + use_pareto_frontier=True, + do_plots=False, + keep_top=1.0, + tokens='', + convert_daily_to_hourly=False, + scale_k_by_frequency=False, + force_reload=False, + run_periods=None, +): + """Analyze specific trials from run files. + + Parameters + ---------- + base_dir : str + Directory containing the run files + trials_to_analyze : list[dict] + List of dicts containing study_id and trial_number to analyze + Example: [{"study_id": "run_XXXXX", "trial_number": 42}, ...] + return_val : str, optional + Metric to optimize for, by default "sharpe" + load_method : str, optional + Method for selecting parameter sets, by default "best_objective" + use_pareto_frontier : bool, optional + Whether to use pareto frontier analysis, by default True + do_plots : bool, optional + Whether to generate plots, by default False + keep_top : float, optional + Fraction of top trials to keep, by default 1.0 + tokens : str, optional + Token filter string, by default '' + convert_daily_to_hourly : bool, optional + Whether to convert daily (1440min) runs to hourly (60min), by default False + scale_k_by_frequency : bool, optional + If convert_daily_to_hourly=True, whether to scale k by frequency ratio, by default True + run_periods : list of dict, optional + Run periods to evaluate on. Each dict: ``{name, start_date, end_date, + end_test_date}`` (see :func:`make_run_period`). If None/empty, each + trial is evaluated over its own trained window as a period named + ``"trained"``. + """ + base_path = Path(base_dir) + all_trials = [] + + filename = "sgd_analysis_result_" + load_method + "_keeptop_" + str(keep_top) + "_" + "_".join(tokens) + ".csv" + + if (Path(base_dir) / filename).exists() and not force_reload: + df = pd.read_csv(Path(base_dir) / filename) + # Convert string columns to dicts + if "params" in df.columns: + df["params"] = df["params"].str.replace(", dtype=float64", "") + df["params"] = df["params"].str.replace(", dtype=float64", "") + df["params"] = df["params"].str.replace(",\s*dtype=float64", "") + df["params"] = df["params"].str.replace("Array(", "") + df["params"] = df["params"].str.replace(")", "") + # Drop rows where params contains 'nan' + df = df[~df["params"].astype(str).str.contains('nan')] + df["params"] = df["params"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + df["params"] = df["params"].apply(lambda x: {k: jnp.array(v) for k, v in x.items()}) + if "tokens" in df.columns: + df["tokens"] = df["tokens"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + if "run_fingerprint" in df.columns: + df["run_fingerprint"] = df["run_fingerprint"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + else: + # Load all study results — try standard loader first, fall back to reClAMM loader + try: + trials_data = load_sgd_results(base_path, load_method=load_method, min_test=0.0) + except (TypeError, IndexError): + print(" Standard loader failed, trying reClAMM loader...") + trials_data = load_reclamm_results(base_path, load_method=load_method) + if tokens != '': + trials_data = [trial for trial in trials_data if set(trial["tokens"]) == set(tokens)] + # Group by tokens and keep only top fraction within each group + if keep_top != 1.0: + # Convert trials_data to DataFrame for easier filtering + df_temp = pd.DataFrame(trials_data) + + # Determine which column to sort by based on load_method + if load_method == "best_objective": + sort_col = "objective" + elif load_method == "best_train_objective": + sort_col = "train_value" + elif load_method == "best_test_objective": + sort_col = "test_value" + else: + sort_col = "objective" # Default to objective + + # Group by tokens and keep top fraction within each group + filtered_trials = [] + groupby_cols = ["tokens", "return_val"] if load_method == "best_objective" else ["tokens"] + for _, group in df_temp.groupby(groupby_cols): + # Sort descending since we want highest values + sorted_group = group.sort_values(sort_col, ascending=False) + # Calculate number of rows to keep + # Determine number of rows to keep based on keep_top value + if keep_top > 1: + # If keep_top is > 1, use it as absolute number of rows + n_keep = int(keep_top) + else: + # If keep_top <= 1, interpret as fraction + n_keep = max(1, int(len(sorted_group) * keep_top)) + # Keep top rows + filtered_trials.extend(sorted_group.head(n_keep).to_dict('records')) + + trials_data = filtered_trials + if trials_data is not None: + all_trials.extend(trials_data) + # Convert to DataFrame + df = pd.DataFrame(all_trials) + # Sort DataFrame by tokens column + df = df.sort_values( + by=["bout_offset", "chunk_period", "tokens", "return_val", "start_date", "end_date", "minimum_weight", "rule"], + ascending=[False, True, True, True, True, False, True, True], + na_position="last", + ) + # Reorder columns to put specified columns at the front + column_order = [ + "chunk_period", + "bout_offset", + "tokens", + "rule", + "return_val", + "minimum_weight", + "start_date", + "end_date", + "study_id", + "trial_number", + "objective", + "test_value", + "train_value" + ] + # Add remaining columns after the specified ones + column_order.extend([col for col in df.columns if col not in column_order]) + df = df[column_order] + # Save DataFrame to disk + + df.to_csv(Path(base_dir) / filename, index=False) + + if len(trials_to_analyze) > 0: + # Filter DataFrame to only include specified trials + mask = pd.DataFrame(False, index=df.index, columns=["match"]) + for trial_info in trials_to_analyze: + mask["match"] |= (df["study_id"] == trial_info["study_id"]) & ( + df["trial_number"] == trial_info["trial_number"] + ) + filtered_df = df[mask["match"]] + if filtered_df.empty: + raise ValueError("No matching trials found") + else: + filtered_df = df + print(filtered_df) + # Create visualizations and analyze results + simplified_df = analyze_best_trials( + filtered_df, + base_dir=base_dir, + output_string=load_method + "__keeptop_" + str(keep_top) + "_" + str(len(trials_to_analyze)) + "_" + "_".join(tokens), + do_plots=do_plots, + convert_daily_to_hourly=convert_daily_to_hourly, + scale_k_by_frequency=scale_k_by_frequency, + run_periods=run_periods, + ) + + df = df.merge( + simplified_df, + left_on='study_id', + right_on='Run ID', + how='left' + ) + + # Then apply the same sorting + # Convert return_val to categorical with custom order before sorting + df['return_val'] = pd.Categorical( + df['return_val'], + categories=['sharpe', 'calmar', 'ulcer', 'sterling', 'daily_log_sharpe'], + ordered=True + ) + + + # Effective period names for column-building: explicit run_periods if + # provided, otherwise the auto-generated default used inside + # analyze_best_trials (a single "trained" period derived from each + # trial's own fingerprint dates). + effective_period_names = [rp["name"] for rp in (run_periods or [])] or ["trained"] + primary_period = effective_period_names[0] + + df = df.sort_values( + by=[ + "tokens", + f"Returns over HODL train ({primary_period})", + "start_date", + "end_date", + "rule", + ], + ascending=[True, False, True, False, True], + na_position="last", + ) + + def _per_period_cols(template, periods): + """Expand a one-line template ``'Metric {side} ({period})'`` into a + flat list over the cartesian product of sides × periods. + """ + return [template.format(period=p) for p in periods] + + # Per-period metric column names — train and test, for each period. + train_metric_templates = [ + "Returns over HODL train ({period})", + "Returns train ({period})", + "Sharpe train ({period})", + "Annualized Ulcer Index [M] train ({period})", + "Annualized Calmer Ratio [M] train ({period})", + ] + test_metric_templates = [ + "Returns over HODL test ({period})", + "Returns test ({period})", + "Sharpe test ({period})", + "Annualized Ulcer Index [M] test ({period})", + "Annualized Calmer Ratio [M] test ({period})", + ] + + per_period_train_cols = [ + c for t in train_metric_templates for c in _per_period_cols(t, effective_period_names) + ] + per_period_test_cols = [ + c for t in test_metric_templates for c in _per_period_cols(t, effective_period_names) + ] + + CONFIG_TAIL = [ + "bout_offset", "chunk_period", + "noise_trader_ratio", "analysis_pool_value", "analysis_fees", "analysis_gas_cost", + "optimisation_method", "hyperparams", + "sample_method", "use_gradient_clipping", "clip_norm", + "optimiser", "learning_rate", "batch_size", "use_plateau_decay", + "lr_schedule_type", "warmup_steps", "ste_max_change", "ste_min_max_weight", + ] + + columns_to_keep = ( + [ + "tokens", "rule", "return_val", "minimum_weight", "start_date", "end_date", + "study_id", "trial_number", "objective", "test_value", "train_value", + "Comments", "Params", + ] + + per_period_train_cols + + per_period_test_cols + + CONFIG_TAIL + ) + + df_filtered = df[columns_to_keep] + + # Reorder: headline columns first (study ID + tokens + rule), then the + # most-useful per-period headline metrics grouped across all periods + # (Returns-over-HODL train/test, Sharpe train/test), then objective + # context, then detailed train + detailed test, then params/metadata. + headline_train = [f"Returns over HODL train ({p})" for p in effective_period_names] + headline_test = [f"Returns over HODL test ({p})" for p in effective_period_names] + sharpe_train = [f"Sharpe train ({p})" for p in effective_period_names] + sharpe_test = [f"Sharpe test ({p})" for p in effective_period_names] + + detailed_train = [ + c for t in ( + "Returns train ({period})", + "Annualized Ulcer Index [M] train ({period})", + "Annualized Calmer Ratio [M] train ({period})", + ) for c in _per_period_cols(t, effective_period_names) + ] + detailed_test = [ + c for t in ( + "Returns test ({period})", + "Annualized Ulcer Index [M] test ({period})", + "Annualized Calmer Ratio [M] test ({period})", + ) for c in _per_period_cols(t, effective_period_names) + ] + + column_order = ( + ["study_id", "trial_number", "tokens", "rule"] + + headline_train + headline_test + sharpe_train + sharpe_test + + ["return_val", "objective", "train_value"] + + detailed_train + + ["test_value"] + + detailed_test + + ["bout_offset", "chunk_period", "Params", "minimum_weight", + "start_date", "end_date", "Comments"] + + CONFIG_TAIL + ) + column_order = list(dict.fromkeys(column_order)) + df_filtered = df_filtered[column_order] + df_filtered = df_filtered.loc[:, ~df_filtered.columns.duplicated(keep='first')] + # Add empty rows after each change in minimum_weight + empty_row = pd.Series([None] * len(df_filtered.columns), index=df_filtered.columns) + + # Create new dataframe with empty rows inserted + df_with_breaks = pd.DataFrame() + + # Iterate through rows and add empty row after token changes + for i in range(len(df_filtered)): + df_with_breaks = pd.concat([df_with_breaks, df_filtered.iloc[[i]]]) + if i < len(df_filtered)-1 and df_filtered.iloc[i]['tokens'] != df_filtered.iloc[i+1]['tokens']: + df_with_breaks = pd.concat([df_with_breaks, empty_row.to_frame().T]) + + filename = f"filled_analysis_{load_method}_keeptop_{keep_top}_{len(trials_to_analyze)}_{'_'.join(tokens)}.csv" + df_with_breaks.to_csv(Path(base_dir) / filename, index=False) + + +def plot_values(results, tokens, suffix="", plot_start_end=None, plot_white_line=False, white_line_date=None, plot_dir=None, initial_hodl_weights="same"): + """Plot value over time for all runs on the same graph.""" + suffix = suffix + "_balancer" + if plot_dir is None: + plot_dir = "./plots/" + plot_path = Path(plot_dir) + plot_path.mkdir(parents=True, exist_ok=True) + + plt.figure(figsize=(12, 6)) + + # Create DataFrame for plotting + df_list = [] + start_date = datetime.strptime( + next(iter(results.values()))["fingerprint"]["startDateString"], + "%Y-%m-%d %H:%M:%S", + ) + if len(results) == 1: + # For single run case, add a HODL baseline + if initial_hodl_weights == "same": + hodl_reserves = next(iter(results.values()))["reserves"][0] + elif initial_hodl_weights == "uniform": + initial_hodl_value = next(iter(results.values()))["value"][0].sum() + initial_hodl_prices = next(iter(results.values()))["prices"][0] + n_assets = len(initial_hodl_prices) + hodl_reserves = (1.0/n_assets) * initial_hodl_value / initial_hodl_prices + prices = next(iter(results.values()))["prices"] + # hodl_values = np.sum(hodl_reserves * prices, axis=1) + # results["HODL"] = { + # "value": hodl_values, + # "fingerprint": next(iter(results.values()))["fingerprint"].copy() + # } + # results["HODL"]["fingerprint"]["rule"] = "HODL" + hodl_fingerprint = next(iter(results.values()))["fingerprint"].copy() + hodl_fingerprint["rule"] = "HODL" + hodl_fingerprint["bout_length"] = len(prices) + 1 + hodl_fingerprint["n_assets"] = len(tokens) + hodl_fingerprint["initial_pool_value"] = float(next( + iter(results.values()) + )["value"][0].sum()) + # print("initial_balancer_pool_value", hodl_fingerprint["initial_pool_value"]) + # hodl_params = {"initial_weights": jnp.ones(len(tokens)) / len(tokens)} + # balancer_pool = BalancerPool() + # hodl_reserves = balancer_pool.calculate_reserves_zero_fees( + # params=hodl_params, + # run_fingerprint=NestedHashabledict(hodl_fingerprint), + # prices=prices, + # start_index=jnp.array([0,0]), + # ) + hodl_values = (hodl_reserves * prices).sum(axis=1) + results["HODL"] = { + "value": hodl_values, + "fingerprint": hodl_fingerprint.copy() + } + for run_name, result in results.items(): + values = result["value"] + dates = pd.date_range( + start=start_date, + end=datetime.strptime( + result["fingerprint"]["endDateString"], "%Y-%m-%d %H:%M:%S" + ), + freq="1min", + )[:-1] + + df_list.extend( + [ + { + "Date": date, + "Value": float(value), # Ensure values are float + "Strategy": str(run_name), # Ensure strategy names are strings + } + for date, value in zip( + dates[::1440], values[::1440] + ) # Take every 1440th value (daily) + ] + ) + + # Create DataFrame with explicit types + df = pd.DataFrame(df_list) + df["Date"] = pd.to_datetime(df["Date"]) + df["Value"] = df["Value"].astype(float) / 1e6 + df["Strategy"] = df["Strategy"].astype("category") + df["Strategy"] = df["Strategy"].apply(name_to_latex_name) + # Create plot + + if plot_start_end is not None: + # Filter df to plot_start_end range if provided + if isinstance(plot_start_end, tuple) and len(plot_start_end) == 2: + start_str, end_str = plot_start_end + start_date = pd.to_datetime(start_str) + end_date = pd.to_datetime(end_str) + df = df[(df['Date'] >= start_date) & (df['Date'] <= end_date)] + + # sns.set_style("darkgrid") + # Get default color palette + default_palette = sns.color_palette() + + # Modify palette to use COLOR for 4th item if needed + strategies = df["Strategy"].unique() + n_strategies = len(strategies) + palette = list(default_palette[:3]) + if n_strategies > 3: + palette = [COLOR] + palette + if n_strategies > 4: + palette.extend(default_palette[4:n_strategies]) + print("n_strategies", n_strategies) + # Sort strategies to put QuantAMM rules (except QuantAMM Index) first + # df['Strategy_order'] = df['Strategy'].apply(lambda x: + # 0 if (x.startswith('QuantAMM') and x != 'QuantAMM Index') + # else 1) + # df = df.sort_values('Strategy_order') + # Define explicit order for strategies + strategy_order = [ + "$\\mathrm{QuantAMM\\ Mean-Reversion\\ Channel}$", + "$\\mathrm{QuantAMM\\ Triple\\ Threat\\ Mean\\ Reversion\\ Channel}$", + "$\\mathrm{QuantAMM\\ Momentum}$", + "$\\mathrm{QuantAMM\\ Anti-Momentum}$", + "$\\mathrm{QuantAMM\\ Power-Channel}$", + "$\\mathrm{QuantAMM\\ Difference\\ Momentum}$", + "$\\mathrm{QuantAMM\\ Index}$", + "$\\mathrm{HODL}$", + "$\\mathrm{Balancer}$", + "$\\mathrm{Traditional\\ DEX}$", + "$\\mathrm{Current\\ Index\\ Product}$", + ] + # Filter strategy_order to only include strategies that exist in the data + strategy_order = [s for s in strategy_order if s in df["Strategy"].unique()] + sns.lineplot(data=df, x="Date", y="Value", hue="Strategy", linewidth=2, palette=palette, hue_order=strategy_order) + + # plt.title("$\\mathrm{Value\\ Over\\ Time}$", pad=20) + plt.xlabel("$\\mathrm{Date}$") + plt.ylabel("$\\mathrm{Value\\ (\\$M\\ USD)}$") + + # Format x-axis + ax = plt.gca() + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.spines["left"].set_color(PLOT_STYLE["COLOR"]) + ax.spines["bottom"].set_color(PLOT_STYLE["COLOR"]) + ax.yaxis.set_ticks_position("left") + ax.xaxis.set_ticks_position("bottom") + + if plot_white_line: + + # Add vertical line and shading for train/test split + plt.axvline(x=pd.Timestamp(white_line_date), color='white', linestyle='--', alpha=0.5) + + # Add "Train" and "Test" labels in LaTeX near the red line + plt.text(pd.Timestamp(white_line_date) - pd.Timedelta(days=15), plt.ylim()[1]*0.95, + "$\\mathrm{Train}$", + horizontalalignment='right', + verticalalignment='top', + fontsize=10, + color='white', + alpha=0.6) + plt.text(pd.Timestamp(white_line_date) + pd.Timedelta(days=15), plt.ylim()[1]*0.95, + "$\\mathrm{Test}$", + horizontalalignment='left', + verticalalignment='top', + fontsize=10, + color='white', + alpha=0.6) + + # Remove legend title + # Reorder legend to put QuantAMM rules (except QuantAMM Index) last + handles, labels = ax.get_legend_handles_labels() + # Find indices of QuantAMM rules that aren't QuantAMM Index + quantamm_indices = [i for i, label in enumerate(labels) + if label.startswith("QuantAMM") and label != "QuantAMM Index"] + if quantamm_indices: + # Move each QuantAMM rule to the end, preserving their relative order + for idx in sorted(quantamm_indices, reverse=True): + handles.append(handles.pop(idx)) + labels.append(labels.pop(idx)) + # Replace legend + ax.legend(handles, labels) + + ax.get_legend().set_title(None) + # Save plot + plt.tight_layout() + plt.savefig( + plot_path / (f"pool_values_comparison_{len(results)}_{'-'.join(tokens)}_{suffix}.png"), + dpi=700, + bbox_inches="tight", + ) + return_over_hodl = df[df["Strategy"]!= name_to_latex_name("HODL")]["Value"].iloc[-1] * 1e6 / hodl_values[-1] - 1.0 + plt.close('all') + del df + del hodl_values + del hodl_reserves + del df_list + gc.collect() + gc.collect() + gc.collect() + if initial_hodl_weights == "uniform": + return return_over_hodl + +def analyze_best_trials( + df, + base_dir="./sgd_studies", + output_string = None, + do_plots=False, + convert_daily_to_hourly=False, + scale_k_by_frequency=False, + run_periods=None, +): + """Analyze best trials with different parameters and generate detailed performance metrics. + + Parameters + ---------- + df : pd.DataFrame + DataFrame containing all trial results + base_dir : str, default="./sgd_studies" + Directory containing Optuna study results + output_string : str, default=None + String to append to output file name + do_plots : bool, default=False + If True, generate plots for each trial + run_periods : list of dict, optional + Run periods to evaluate each trial over (format: see ``make_run_period``). + If None/empty, a single "trained" period is auto-generated per trial from + that trial's own fingerprint dates — matches the old period_from_rf + behaviour but without the duplication that happened when user periods + had the same dates. + """ + output_path = Path(base_dir) + output_path.mkdir(parents=True, exist_ok=True) + + # Parameters to test + initial_pool_values = [20000.0, 50000.0,100000.0, 1000000.0, 10000000.0] + fee_values = [0.0, 0.001, 0.003, 0.01] # 0 to 1% + gas_costs = [0.0,0.005,0.5] # USD per trade + noise_trader_ratios = [0.0, 0.5, 1.0] # 0 to 1 + # initial_pool_values = [1000000.0] + fee_values = [0.0, 0.003] + gas_costs = [0.0, 1.0, 2.0] + noise_trader_ratios = [0.0, 0.5] + + initial_pool_values = [100000.0] + noise_trader_ratios = [0.0] # 0 to 1 + # initial_pool_values = [1000000.0] + fee_values = [0.0] + gas_costs = [0.0] + noise_trader_ratios = [0.0] + + # Run periods: use what the caller supplied, or fall back to a single + # per-trial "trained" period synthesised below from each trial's own dates. + run_periods = list(run_periods) if run_periods else [] + + + # Print number of rows for each rule + rule_counts = df['rule'].value_counts() + print("\nNumber of rows per rule:") + for rule, count in rule_counts.items(): + print(f"{rule}: {count}") + + grouped = df.groupby(["tokens", "initial_pool_value", "end_date", "bout_offset", "chunk_period", "rule"]) + + results = [] + + base_path = Path(base_dir) + + # Create results directory within base_dir + results_dir = base_path / "analysis_results" + results_dir.mkdir(parents=True, exist_ok=True) + + plot_dir = base_path / "plots" + plot_dir.mkdir(parents=True, exist_ok=True) + # Extract study name from base_dir + study_name = base_path.name + + # Construct results filename + results_filename = ( + f"analysis_results" + f"_{study_name}" + f"_{output_string}" + f".json" + ) + + # results_filename = "analysis_unified_results_all_runs_uptodate_pareto_20250405_033520_best.json" + + results_path = results_dir / results_filename + # Check if analysis file already exists + if do_plots or os.path.exists(results_path)==False: + price_data = get_historic_parquet_data(df.iloc[0]["run_fingerprint"]["tokens"], cols=["close"]) + for _, trial in df.iterrows(): + params = trial["params"] + run_fingerprint = trial["run_fingerprint"] + run_fingerprint["startTestDateString"] = run_fingerprint["endDateString"] + # params["raw_exponents"] = jnp.array( + # [-0.48537339, 1.44890609, -0.03770628, 0.28857155] + # ) + # params["raw_exponents"] = jnp.array( + # [-0.5935307, 1.49911745, -0.03770628, 0.28857155] + # ) + for pool_val, fee, gas, noise_trader_ratio in itertools.product(initial_pool_values, fee_values, gas_costs, noise_trader_ratios): + run_period_results = [] + if run_periods: + local_run_periods = [rp.copy() for rp in run_periods] + else: + # No explicit periods: just evaluate on the trial's own training window. + local_run_periods = [default_run_period_for_trial(trial)] + if 'ARB' in run_fingerprint["tokens"]: + for i in range(len(local_run_periods)): + local_run_periods[i]["start_date"] = local_run_periods[0]["start_date"] + for run_period in local_run_periods: + # run_period["end_test_date"] = "2025-07-12 00:00:00" + run_period = run_period.copy() + prices = {"train_prices": None, "test_prices": None, "continuous_test_prices": None} + print("tokens", run_fingerprint["tokens"]) + print("study_id", trial["study_id"]) + print("trial_number", trial["trial_number"]) + print("rule", trial["rule"]) + print("pool_val", pool_val) + print("fee", fee) + print("gas", gas) + print("noise_trader_ratio", noise_trader_ratio) + + # Update run_fingerprint with new values + local_fingerprint = run_fingerprint.copy() + local_fingerprint["startDateString"] = run_period["start_date"] + if "OM" in local_fingerprint["tokens"]: + local_fingerprint["startDateString"] = trial["start_date"] + run_period["start_date"] = trial["start_date"] + if run_period["name"].startswith("March"): + run_period["name"] = "May" + run_period["name"][5:] + if "PEPE" in local_fingerprint["tokens"]: + local_fingerprint["startDateString"] = trial["start_date"] + run_period["start_date"] = trial["start_date"] + if run_period["name"].startswith("March"): + run_period["name"] = "May2023" + run_period["name"][9:] + run_period["end_test_date"] = "2025-03-16 00:00:00" + print("run_period", run_period["name"]) + local_fingerprint["endDateString"] = run_period["end_date"] + local_fingerprint["startTestDateString"] = run_period["end_date"] + local_fingerprint["endTestDateString"] = run_period["end_test_date"] + local_fingerprint["initial_pool_value"] = pool_val + local_fingerprint["fees"] = fee + local_fingerprint["gas_cost"] = gas + local_fingerprint["noise_trader_ratio"] = noise_trader_ratio + + end_date = run_period["end_date"] + tokens = trial["tokens"] + bout_offset = trial["bout_offset"] + chunk_period = trial["chunk_period"] + init_pool_val = trial["initial_pool_value"] + rule = trial["rule"] + # Broadcast scalar params to n_assets shape. Optuna runs + # don't save initial_weights_logits, so synthesise a zero + # template from tokens. + n_assets = len(tokens) + template = params.get( + "initial_weights_logits", jnp.zeros(n_assets) + ) + if "log_k" in params: + params["log_k"] = params["log_k"] * jnp.ones_like(template) + if "k" in params: + params["k"] = params["k"] * jnp.ones_like(template) + if "logit_lamb" in params: + params["logit_lamb"] = params["logit_lamb"] * jnp.ones_like(template) + if "logit_delta_lamb" in params: + params["logit_delta_lamb"] = params["logit_delta_lamb"] * jnp.ones_like(template) + if "initial_weights_logits" not in params: + params["initial_weights_logits"] = jnp.zeros(n_assets) + + # Apply daily-to-hourly conversion if requested + if convert_daily_to_hourly: + params, local_fingerprint = convert_daily_to_hourly_params( + params, local_fingerprint, scale_k_by_frequency + ) + # Update chunk_period for plot prefix after conversion + chunk_period = local_fingerprint["chunk_period"] + + base_plot_prefix = f"{tokens}_{rule}_{pool_val}_{fee}_{gas}_{noise_trader_ratio}_chunk_{chunk_period}_param_{trial['study_id'][4:][:5]}_trial_{trial['trial_number']}_runperiod_{run_period['name']}" + + # Add suffix to indicate conversion type if applied + if convert_daily_to_hourly: + conversion_suffix = "_hourly_k_scaled" if scale_k_by_frequency else "_hourly_k_unscaled" + base_plot_prefix += conversion_suffix + + run_filename = f"analysis_{base_plot_prefix}.json" + if os.path.exists(output_path / run_filename) and do_plots==False: + print(f"Skipping {run_filename} because it already exists") + with open(output_path / run_filename, "r") as f: + result = json.load(f) + results.append(result) + continue + # else: + # print(f"Skipping {run_filename} because we dont have time") + # continue + + train_dict, test_dict = do_run_on_historic_data( + local_fingerprint, + params=params, + do_test_period=True, + verbose=False, + price_data=price_data, + ) + + # Run continuous test simulation + continuous_fingerprint = local_fingerprint.copy() + continuous_fingerprint["endDateString"] = local_fingerprint["endTestDateString"] + shifted_end = pd.to_datetime(continuous_fingerprint["endDateString"]) - pd.Timedelta(minutes=1) + continuous_fingerprint["endDateString"] = str(shifted_end.strftime("%Y-%m-%d %H:%M:%S")) + continuous_dict = do_run_on_historic_data( + continuous_fingerprint, + params=params, + do_test_period=False, + verbose=False, + price_data=price_data, + ) + # Store prices and remove from result dicts + if prices["train_prices"] is None: + prices["train_prices"] = train_dict.pop("prices") + prices["test_prices"] = test_dict.pop("prices") + prices["continuous_test_prices"] = continuous_dict.pop("prices") + else: + train_dict.pop("prices", None) + test_dict.pop("prices", None) + continuous_dict.pop("prices", None) + + # (train_dict["reserves"][0]*train_dict["prices"][-1]).sum() + if do_plots: + print("top of plots") + train_dict["prices"] = prices["train_prices"] + plot_weights( + train_dict, + local_fingerprint, + plot_prefix=f"train_{base_plot_prefix}", + # plot_dir=plot_dir, + ) + # do_weight_change_as_rebalances_plots( + # train_dict, + # run_fingerprint, + # plot_prefix=f"train_{tokens}_{rule}_{pool_val}_{fee}_{gas}_end_{end_date}_param_{i}", + # ) + train_dict["fingerprint"] = local_fingerprint + plot_values( + {"Optimized_QuantAMM_pool_"+'-'.join(tokens)+"_rule_"+rule: train_dict}, + tokens, + suffix=f"train_{base_plot_prefix}", + plot_dir=plot_dir, + ) + del train_dict["prices"] + test_dict["prices"] = prices["test_prices"] + test_dict["fingerprint"] = local_fingerprint.copy() + test_dict["fingerprint"]["startDateString"] = local_fingerprint["startTestDateString"] + test_dict["fingerprint"]["endDateString"] = local_fingerprint["endTestDateString"] + plot_values( + { + "Optimized_QuantAMM_pool_" + + "-".join(tokens) + + "_rule_" + + rule: test_dict + }, + tokens, + suffix=f"test_{base_plot_prefix}", + plot_dir=plot_dir, + ) + del test_dict["prices"] + continuous_dict["prices"] = prices["continuous_test_prices"] + continuous_dict["fingerprint"] = continuous_fingerprint + plot_values( + { + "Optimized_QuantAMM_pool_" + + "-".join(tokens) + + "_rule_" + + rule: continuous_dict + }, + tokens, + suffix=f"continuous_run_{base_plot_prefix}", + plot_white_line=True, + white_line_date=test_dict["fingerprint"]["startDateString"], + plot_dir=plot_dir, + ) + print("run_period: ", run_period) + print("white line date", test_dict["fingerprint"]["startDateString"]) + del continuous_dict["prices"] + continuous_test_results = { + "value": continuous_dict["value"][ + len(train_dict["value"]) : len(train_dict["value"]) + + len(test_dict["value"]) + ], + "reserves": continuous_dict["reserves"][ + len(train_dict["reserves"]) : len( + train_dict["reserves"] + ) + + len(test_dict["reserves"]) + ], + "prices": prices["continuous_test_prices"][ + len(train_dict["value"]) : len( + train_dict["value"] + ) + + len(test_dict["value"]) + ], + "fingerprint": test_dict["fingerprint"], + } + def calculate_period_returns(results_dict, period_length=30 * 24 * 60): + """Calculate returns over different periods and strategies. + + Args: + results_dict: Dictionary containing 'value', 'prices', 'reserves' arrays + period_length: Period length in minutes (default 30 days) + + Returns: + List of dictionaries containing return metrics for each period + """ + # Convert inputs to numpy arrays + test_values = np.array(results_dict["value"]) + test_prices = np.array(results_dict["prices"]) + test_reserves = np.array(results_dict["reserves"]) + + period_returns = [] + n_tokens = len(test_prices[0]) + uniform_weights = np.ones(n_tokens) / n_tokens + + # Iterate over each period + for period_start in range(0, len(test_values), period_length): + period_end = min(period_start + period_length, len(test_values)) + + # print("period_start", period_start) + # print("period_end", period_end) + # print("period length", period_end - period_start) + + # Get strategy values for this period + start_value = test_values[period_start] + end_value = test_values[period_end - 1] + strategy_return = end_value / start_value - 1 + + # Calculate initial reserves and prices + initial_period_reserves = test_reserves[period_start] + start_prices = test_prices[period_start] + end_prices = test_prices[period_end - 1] + + # Calculate weights and price ratios + initial_period_weights = (initial_period_reserves * start_prices) / np.sum((initial_period_reserves * start_prices)) + price_ratios = end_prices / start_prices + + # Calculate end values for different strategies + end_balance_pool_value = start_value * np.prod((price_ratios) ** initial_period_weights) + end_uniform_balance_pool_value = start_value * np.prod((price_ratios) ** uniform_weights) + hodl_end_value = np.sum(initial_period_reserves * end_prices) + + # Calculate returns + hodl_return = hodl_end_value / start_value - 1 + returns_over_hodl = (end_value) / (hodl_end_value) - 1 + returns_over_balancer = (end_value) / (end_balance_pool_value) - 1 + returns_over_uniform_balancer = (end_value) / (end_uniform_balance_pool_value) - 1 + + # Set returns_over_balancer to 0 if within 10^-12 of 0 + if abs(returns_over_balancer) < 1e-12: + returns_over_balancer = 0.0 + period_returns.append({ + "period": period_start // period_length, + "strategy_return": strategy_return, + "returns_over_hodl": returns_over_hodl, + "hodl_return": hodl_return, + "returns_over_balancer": returns_over_balancer, + "returns_over_uniform_balancer": returns_over_uniform_balancer, + "annualised_strategy_return": (1 + strategy_return) ** (365 * 24 * 60 / period_length) - 1, + "annualised_returns_over_hodl": (1 + returns_over_hodl) ** (365 * 24 * 60 / period_length) - 1, + "annualised_returns_over_balancer": (1 + returns_over_balancer) ** (365 * 24 * 60 / period_length) - 1, + "annualised_returns_over_uniform_balancer": (1 + returns_over_uniform_balancer) ** (365 * 24 * 60 / period_length) - 1, + }) + + print("\nPeriod Annualised Returns Over HODL:") + for period_data in period_returns: + print(f"Period {period_data['period']}: {100.0 * period_data['annualised_returns_over_hodl']}") + print("--------------------------------") + print("\nPeriod Annualised Returns:") + for period_data in period_returns: + print( + f"Period {period_data['period']}: {100.0 * period_data['annualised_strategy_return']}" + ) + print("--------------------------------") + print("\nPeriod Annualised Returns Over Balancer:") + for period_data in period_returns: + print(f"Period {period_data['period']}: {100.0 * period_data['annualised_returns_over_balancer']}") + print("\nPeriod Annualised Returns Over Uniform Balancer:") + for period_data in period_returns: + print(f"Period {period_data['period']}: {100.0 * period_data['annualised_returns_over_uniform_balancer']}") + + return period_returns + + # Calculate monthly returns over HODL (using fixed 30-day periods) + print("="*100) + print("calculating monthly returns test") + calculate_period_returns(continuous_test_results.copy(), period_length=30 * 24 * 60) + # print("calculating weekly returns test") + # calculate_period_returns(continuous_test_results.copy(), period_length=7 * 24 * 60) + print("calculating monthly returns train") + train_dict["prices"] = prices["train_prices"] + calculate_period_returns(train_dict.copy(), period_length=30 * 24 * 60) + # print("calculating weekly returns train") + # calculate_period_returns(train_dict.copy(), period_length=7 * 24 * 60) + print("="*100) + # if "2024" in run_period["name"]: + # raise Exception("Stop here") + # Print period returns + + plot_values( + { + "Optimized_QuantAMM_pool_" + + "-".join(tokens) + + "_rule_" + + rule: continuous_test_results + }, + tokens, + suffix=f"continuous_test_{base_plot_prefix}", + plot_dir=plot_dir, + ) + plot_weights( + continuous_test_results, + continuous_test_results["fingerprint"], + plot_prefix=f"continuous_test_{base_plot_prefix}", + # plot_dir=plot_dir, + ) + uniform_hodl_return = plot_values( + { + "Optimized_QuantAMM_pool_" + + "-".join(tokens) + + "_rule_" + + rule: continuous_test_results + }, + tokens, + suffix=f"continuous_test_uniformhodl_{base_plot_prefix}", + initial_hodl_weights="uniform", + plot_dir=plot_dir, + ) + del continuous_test_results + else: + # Calculate uniform HODL return for continuous test period + initial_value = continuous_dict["value"][ + len(train_dict["value"]) : len(train_dict["value"]) + + len(test_dict["value"]) + ][0] + initial_prices = prices["continuous_test_prices"][ + len(train_dict["value"]) : len( + train_dict["value"] + ) + + len(test_dict["value"]) + ][0] + final_prices = prices["continuous_test_prices"][::1440][-1] + # Calculate uniform weights + n_tokens = len(tokens) + uniform_weights = np.ones(n_tokens) / n_tokens + uniform_reserves = initial_value * uniform_weights / initial_prices + # Calculate return from uniform HODL strategy + price_ratios = final_prices / initial_prices + uniform_hodl_return = continuous_dict["value"][::1440][-1] / np.sum(uniform_reserves * final_prices) - 1 + # Helper: calculate_period_metrics returns JAX scalars (0-d + # ArrayImpl) for most keys plus a 1-d daily_returns. Cast + # scalars to Python floats here so downstream pandas sorts + # / categoricals don't choke on unhashable arrays. + def _py_scalarise(d): + out = {} + for k, v in d.items(): + if hasattr(v, "ndim") and v.ndim == 0: + out[k] = float(v) + else: + out[k] = v + return out + + # Calculate metrics for training period + train_metrics = _py_scalarise(calculate_period_metrics(train_dict, prices["train_prices"])) + # Prefix each key with "train_" + train_metrics = {f"train_{k}": v for k, v in train_metrics.items()} + # Calculate metrics for test period + test_metrics = _py_scalarise(calculate_period_metrics(test_dict, prices["test_prices"])) + # Prefix each key with "test_" + test_metrics = {f"test_{k}": v for k, v in test_metrics.items()} + + # Calculate metrics for continuous test period + continuous_test_metrics = _py_scalarise(calculate_continuous_test_metrics( + continuous_dict, + len(train_dict["value"]), + len(test_dict["value"]), + prices["continuous_test_prices"] + )) + continuous_test_metrics = {f"continuous_test_{k}": v for k, v in continuous_test_metrics.items()} + continuous_metrics = _py_scalarise(calculate_period_metrics(continuous_dict, prices["continuous_test_prices"])) + + continuous_metrics = {f"continuous_{k}": v for k, v in continuous_metrics.items()} + # Store results + result = { + "study_id": trial["study_id"], + "trial_number": trial["trial_number"], + "tokens": tokens, + "rule": rule, + "run_period": run_period["name"], + "original_pool_value": init_pool_val, + "original_fees": run_fingerprint["fees"], + "original_gas_cost": run_fingerprint["gas_cost"], + "original_objective": trial["objective"], + "original_return_val": trial["return_val"], + "original_train_value": trial["train_value"], + "original_test_value": trial["test_value"], + "noise_trader_ratio": noise_trader_ratio, + "analysis_pool_value": pool_val, + "analysis_fees": fee, + "analysis_gas_cost": gas, + "bout_offset": int(trial["bout_offset"]), + "chunk_period": int(trial["chunk_period"]), + "run_period_start_date": run_period["start_date"], + "start_date": trial.get("start_date"), + "end_date": trial.get("end_date"), + "end_test_date": trial.get("end_test_date"), + "run_period_start_date": run_period["start_date"], + "run_period_end_date": run_period["end_date"], + "run_period_end_test_date": run_period["end_test_date"], + "weight_interpolation_period": trial.get( + "weight_interpolation_period" + ), + "minimum_weight": trial.get("minimum_weight"), + **train_metrics, + **test_metrics, + **continuous_test_metrics, + **continuous_metrics, + "uniform_continuous_test_hodl_return": float( + uniform_hodl_return + ), + "params": tree_map( + lambda x: (x.tolist() if isinstance(x, jnp.ndarray) else x), + params, + ), + "optimisation_method": run_fingerprint["optimisation_settings"]["method"], + # One column with ALL relevant hyperparams for this + # run's method (adam-flat-keys OR optuna/cma_es/bfgs + # sub-dict). JSON-encoded so it round-trips through CSV. + "hyperparams": json.dumps( + _method_hyperparams(run_fingerprint), + default=_json_default, + ), + # Adam-family-only columns. NaN for non-gradient_descent + # runs so we don't show inherited defaults the run never + # actually used. + **_gd_only_columns(run_fingerprint), + "ste_max_change": run_fingerprint.get( + "ste_max_change" + ), + "ste_min_max_weight": run_fingerprint.get( + "ste_min_max_weight" + ), + } + print(result) + results.append(result) + # Save individual result files + for filename in [ + f"analysis_{base_plot_prefix}.json", + # f"analysis_{base_plot_prefix}_studyid_{trial['study_id']}_trialno_{trial['trial_number']}_poolval_{pool_val:.0f}_fees_{fee:.4f}_gas_{gas:.1f}_noise_{noise_trader_ratio:.2f}_bout_{trial['bout_offset']}_chunk_{trial['chunk_period']}_end_{trial['end_date']}_param_{0}_usePF_{use_pareto_frontier}.json", + ]: + with open(output_path / filename, "w") as f: + json.dump(result, f, indent=2, default=_json_default) + gc.collect() + + del result + del train_dict + del test_dict + del continuous_dict + del prices + clear_caches() + gc.collect() + gc.collect() + # Save all results to a single file + with open(results_path, "w") as f: + json.dump(results, f, indent=2, default=_json_default) + else: + with open(results_path, "r") as f: + results = json.load(f) + print(f"Loading existing results from {results_path}") + # After collecting all results, but before saving: + results_df = pd.DataFrame(results) + + # Identify non-varying columns (keys that uniquely identify a trial) + id_columns = [ + "study_id", + "trial_number", + "tokens", + "rule", + "start_date", + "end_date", + "original_pool_value", + "original_fees", + "original_gas_cost", + "original_objective", + "original_return_val", + "original_train_value", + "original_test_value", + "noise_trader_ratio", + "analysis_pool_value", + "analysis_fees", + "analysis_gas_cost", + "bout_offset", + "chunk_period", + "weight_interpolation_period", + "minimum_weight", + "optimisation_method", + "hyperparams", + "sample_method", + "use_gradient_clipping", + "clip_norm", + "optimiser", + "learning_rate", + "batch_size", + "use_plateau_decay", + "lr_schedule_type", + "warmup_steps", + "ste_max_change", + "ste_min_max_weight", + ] + + # Create DataFrame with explicit types + df = pd.DataFrame(results) + df["study_id"] = df["study_id"].astype(str) + df["trial_number"] = df["trial_number"].astype(float) + df["Strategy"] = ( + df["Strategy"].astype("category") if "Strategy" in df.columns else None + ) + # Get all columns that need period prefix + metric_columns = [ + col + for col in results_df.columns + if col not in id_columns and col != "run_period" and col != "params" and col != "noise_trader_ratio" and col != "analysis_pool_value" and col != "analysis_fees" and col != "analysis_gas_cost" + ] + + # Create a mapping of (study_id, trial_number) to params + params_dict = ( + results_df.groupby(["study_id", "trial_number"])["params"].first().to_dict() + ) + + if 'tokens' in results_df.columns: + results_df['tokens'] = results_df['tokens'].apply(tuple) + # Set index to id columns for proper pivoting + results_df = results_df.set_index(id_columns) + + # Pivot all period-specific columns at once + period_data = [] + for col in metric_columns: + pivoted = results_df.pivot(columns="run_period", values=col) + cols = pivoted.columns + cols = [col[:-4] if col.endswith("test") else col for col in cols] + pivoted.columns = [f"{period}_{col}" for period in cols] + period_data.append(pivoted) + + # Combine all pivoted data + transformed_df = pd.concat(period_data, axis=1) + + # Reset index to get id columns back + transformed_df = transformed_df.reset_index() + + # Add params column + transformed_df["params"] = transformed_df.apply( + lambda row: params_dict[(row["study_id"], row["trial_number"])], axis=1 + ) + + # Convert dates to datetime for sorting + transformed_df["start_date"] = pd.to_datetime(transformed_df["start_date"]) + transformed_df["end_date"] = pd.to_datetime(transformed_df["end_date"]) + + # Sort the DataFrame + transformed_df = transformed_df.sort_values( + by=[ + "bout_offset", + "chunk_period", + "tokens", + "original_return_val", + "start_date", + "end_date", + "minimum_weight", + "rule" + ], + ascending=[ + False, + True, + True, # tokens ascending + True, # return_val ascending + True, # start date ascending + False, # end date descending + True, # minimum weight ascending + True # rule alphabetically + ] + ) + + # Reset index after sorting + transformed_df = transformed_df.reset_index(drop=True) + + # Add conversion suffix to output filenames if conversion was applied + conversion_suffix = "" + if convert_daily_to_hourly: + conversion_suffix = "_hourly_k_scaled" if scale_k_by_frequency else "_hourly_k_unscaled" + + # Save transformed DataFrame + filename = f"analysis_results_{Path(base_dir).name}_{output_string}{conversion_suffix}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv" + transformed_df.to_csv(Path(base_dir) / filename, index=False) + + # Create simplified DataFrame matching BTF format + simplified_df = pd.DataFrame() + + # Map basic columns + simplified_df["Tokens"] = transformed_df["tokens"] + simplified_df["Rule"] = transformed_df["rule"].str.replace("_", " ").str.title() + simplified_df["Objective"] = transformed_df["original_return_val"] + simplified_df["Min weight"] = transformed_df["minimum_weight"].apply(lambda x: f"{float(x)*100:.0f}%") + simplified_df["Start date"] = pd.to_datetime(transformed_df["start_date"]).dt.strftime("%Y") + simplified_df["start test"] = pd.to_datetime(transformed_df["end_date"]).dt.strftime("%Y-%m-%d %H:%M:%S") + simplified_df["Run ID"] = transformed_df["study_id"] + simplified_df["Iteration no"] = transformed_df["trial_number"] + simplified_df["Best obj"] = transformed_df["original_objective"] + simplified_df["Test Returns Over HODL"] = transformed_df["original_test_value"] + simplified_df["Train Returns Over HODL"] = transformed_df["original_train_value"] + + # Add conversion metadata to Comments + conversion_info = "" + if convert_daily_to_hourly: + k_scaling = "with k scaling" if scale_k_by_frequency else "without k scaling" + conversion_info = f"Converted daily->hourly ({k_scaling})" + simplified_df["Comments"] = conversion_info + + simplified_df["Params"] = transformed_df["params"] + + # Add metrics for each period + periods = np.unique(results_df["run_period"]).tolist() + for period in periods: + if period.endswith("test"): + period = period[:-4] + period_prefix = f"{period}_" + # simplified_df[f"Returns over HODL test ({period})"] = transformed_df[f"{period_prefix}continuous_test_returns_over_hodl"] + simplified_df[f"Returns test ({period})"] = transformed_df[f"{period_prefix}continuous_test_return"] + simplified_df[f"Sharpe test ({period})"] = transformed_df[f"{period_prefix}continuous_test_sharpe"] + simplified_df[f"Annualized Ulcer Index [M] test ({period})"] = transformed_df[f"{period_prefix}continuous_test_ulcer"] + simplified_df[f"Annualized Calmer Ratio [M] test ({period})"] = transformed_df[f"{period_prefix}continuous_test_calmar"] + simplified_df[f"Returns over HODL train ({period})"] = transformed_df[ + f"{period_prefix}train_returns_over_hodl" + ] + simplified_df[f"Returns train ({period})"] = transformed_df[ + f"{period_prefix}train_return" + ] + simplified_df[f"Sharpe train ({period})"] = transformed_df[ + f"{period_prefix}train_sharpe" + ] + simplified_df[f"Annualized Ulcer Index [M] train ({period})"] = transformed_df[ + f"{period_prefix}continuous_test_ulcer" + ] + simplified_df[f"Annualized Calmer Ratio [M] train ({period})"] = transformed_df[ + f"{period_prefix}train_calmar" + ] + simplified_df[f"Returns over HODL test ({period})"] = transformed_df[ + f"{period_prefix}uniform_continuous_test_hodl_return" + ] + simplified_df["noise_trader_ratio"] = transformed_df["noise_trader_ratio"] + simplified_df["analysis_pool_value"] = transformed_df["analysis_pool_value"] + simplified_df["analysis_fees"] = transformed_df["analysis_fees"] + simplified_df["analysis_gas_cost"] = transformed_df["analysis_gas_cost"] + simplified_df["optimisation_method"] = transformed_df["optimisation_method"] + simplified_df["hyperparams"] = transformed_df["hyperparams"] + simplified_df["sample_method"] = transformed_df["sample_method"] + simplified_df["use_gradient_clipping"] = transformed_df["use_gradient_clipping"] + simplified_df["clip_norm"] = transformed_df["clip_norm"] + simplified_df["optimiser"] = transformed_df["optimiser"] + simplified_df["learning_rate"] = transformed_df["learning_rate"] + simplified_df["batch_size"] = transformed_df["batch_size"] + simplified_df["use_plateau_decay"] = transformed_df["use_plateau_decay"] + simplified_df["lr_schedule_type"] = transformed_df["lr_schedule_type"] + simplified_df["warmup_steps"] = transformed_df["warmup_steps"] + simplified_df["ste_max_change"] = transformed_df["ste_max_change"] + simplified_df["ste_min_max_weight"] = transformed_df["ste_min_max_weight"] + + # Save simplified CSV + simplified_filename = f"simplified_analysis_{Path(base_dir).name}_{output_string}{conversion_suffix}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv" + simplified_df.to_csv(Path(base_dir) / simplified_filename, index=False, float_format='%.10f') + + # Continue with existing JSON save for compatibility + results_path = ( + Path(base_dir) + / "analysis_results" + / f"analysis_unified_results_{Path(base_dir).name}_{output_string}{conversion_suffix}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json" + ) + with open(results_path, "w") as f: + json.dump(results, f, indent=2, default=_json_default) + + return simplified_df + + +def load_run_fingerprints( + base_dir, + trials_to_analyze, + load_method="best_objective", + tokens='' +): + """Analyze specific trials from run files. + + Parameters + ---------- + base_dir : str + Directory containing the run files + trials_to_analyze : list[dict] + List of dicts containing study_id and trial_number to analyze + Example: [{"study_id": "run_XXXXX", "trial_number": 42}, ...] + return_val : str, optional + Metric to optimize for, by default "sharpe" + load_method : str, optional + Method for selecting parameter sets, by default "best_objective" + use_pareto_frontier : bool, optional + Whether to use pareto frontier analysis, by default True + """ + base_path = Path(base_dir) + all_trials = [] + + keep_top = 1.0 + + filename = "sgd_analysis_result_" + load_method + "_keeptop_" + str(keep_top) + "_" + "_".join(tokens) + ".csv" + + if (Path(base_dir) / filename).exists(): + df = pd.read_csv(Path(base_dir) / filename) + # Convert string columns to dicts + if "params" in df.columns: + df["params"] = df["params"].str.replace(", dtype=float64", "") + df["params"] = df["params"].str.replace(", dtype=float64", "") + df["params"] = df["params"].str.replace(",\s*dtype=float64", "") + df["params"] = df["params"].str.replace("Array(", "") + df["params"] = df["params"].str.replace(")", "") + # Drop rows where params contains 'nan' + df = df[~df["params"].astype(str).str.contains('nan')] + df["params"] = df["params"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + df["params"] = df["params"].apply(lambda x: {k: jnp.array(v) for k, v in x.items()}) + if "tokens" in df.columns: + df["tokens"] = df["tokens"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + if "run_fingerprint" in df.columns: + df["run_fingerprint"] = df["run_fingerprint"].apply(lambda x: ast.literal_eval(x) if isinstance(x, str) else x) + + if len(trials_to_analyze) > 0: + # Filter DataFrame to only include specified trials + mask = pd.DataFrame(False, index=df.index, columns=["match"]) + for trial_info in trials_to_analyze: + mask["match"] |= (df["study_id"] == trial_info["study_id"]) & ( + df["trial_number"] == trial_info["trial_number"] + ) + filtered_df = df[mask["match"]] + if filtered_df.empty: + raise ValueError("No matching trials found") + else: + filtered_df = df + # Write run fingerprints as JSONL file + output_path = Path(base_dir) / f"run_fingerprints_{datetime.now().strftime('%Y%m%d_%H%M%S')}.jsonl" + with open(output_path, "w", encoding="utf-8") as f: + for _, row in filtered_df.iterrows(): + json.dump(row["run_fingerprint"], f) + f.write("\n") + + return all_trials + +def build_cli_parser() -> argparse.ArgumentParser: + p = argparse.ArgumentParser( + description="Post-train analysis: walk a directory of run_*.json training " + "results, run the trained params on one or more evaluation " + "windows, and emit comparison CSVs + plots.", + formatter_class=argparse.ArgumentDefaultsHelpFormatter, + ) + p.add_argument("--base-dir", required=True, + help="Directory containing run_*.json training result files.") + p.add_argument("--tokens", nargs="+", default=["ETH", "USDC"], + help="Token filter — only trials matching this exact token set are kept.") + p.add_argument("--load-method", default="best_train_min_test_objective", + choices=["last", "best_objective", "best_train_objective", + "best_test_objective", "best_train_min_test_objective"]) + p.add_argument("--force-reload", action="store_true", + help="Ignore any cached sgd_analysis_result_*.csv and re-scan all run files.") + p.add_argument("--no-plots", action="store_true", + help="Skip plot generation (analysis CSVs still written).") + p.add_argument("--daily-to-hourly", action="store_true", + help="Convert daily (chunk_period=1440) runs to hourly (60min) for analysis.") + p.add_argument("--scale-k", action="store_true", + help="If --daily-to-hourly, also scale k by the frequency ratio.") + p.add_argument("--run-period", action="append", nargs=4, + metavar=("NAME", "START", "END", "TEST_END"), + default=[], + help="Evaluation window. NAME is the short label that appears " + "in CSV column headers (e.g. 'baseline'). Dates as " + "'YYYY-MM-DD HH:MM:SS'. Repeatable.") + return p + + +if __name__ == "__main__": + args = build_cli_parser().parse_args() + + run_periods = [parse_run_period_cli(tokens) for tokens in args.run_period] or None + + print("Starting analysis...") + print("=" * 100) + print(f"base_dir: {args.base_dir}") + print(f"tokens: {args.tokens}") + print(f"periods: {[rp['name'] for rp in run_periods] if run_periods else ['trained (from each trial)']}") + print("=" * 100) + + analyze_specific_trials( + base_dir=args.base_dir, + trials_to_analyze=[], + load_method=args.load_method, + tokens=args.tokens, + force_reload=args.force_reload, + do_plots=not args.no_plots, + convert_daily_to_hourly=args.daily_to_hourly, + scale_k_by_frequency=args.scale_k, + run_periods=run_periods, + ) diff --git a/scripts/run_final_sims.py b/scripts/run_final_sims.py new file mode 100644 index 00000000..7f3b6119 --- /dev/null +++ b/scripts/run_final_sims.py @@ -0,0 +1,358 @@ +#!/usr/bin/env python3 +"""Run final forward simulations for best sweep params at multiple TVLs. + +For each token pair, loads the best params from the sweep, runs train and +test period forward passes at each TVL level, and generates comparison plots. + +Usage: + python scripts/run_final_sims.py --pair aave + python scripts/run_final_sims.py --pair cow + python scripts/run_final_sims.py --all + python scripts/run_final_sims.py --all --method cma_es +""" + +import argparse +import json +import os +import pickle +from collections import defaultdict +from pathlib import Path + +import jax +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +from datetime import datetime + +from quantammsim.runners.jax_runners import do_run_on_historic_data +from scripts.evaluate_trials import load_reclamm_results +from scripts.plot_reclamm_optuna_result import ( + plot_results as plot_optuna_results, + plot_test_only, + plot_weights as plot_optuna_weights, +) + +BG = "#162536" +TC = "#E6CE97" + +# Train/test split around the Oct 10 flash crash +TRAIN_START = "2025-01-01 00:00:00" +TRAIN_END = "2025-10-05 00:00:00" +TEST_START = "2025-10-25 00:00:00" +TEST_END = "2026-05-01 00:00:00" + +PAIR_CONFIGS = { + "aave": { + "tokens": ["AAVE", "ETH"], + "pool_id": "0x9d1fcf346ea1b0", + "gas_cost": 1.0, + "fees": 0.0025, + "tvls": { + "1m": 1_000_000, + "5m": 5_000_000, + "20m": 20_000_000, + }, + }, + "cow": { + "tokens": ["COW", "ETH"], + "pool_id": "0xd321300ef77067", + "gas_cost": 3.0, + "fees": 0.003, + "tvls": { + "500k": 500_000, + "2m": 2_000_000, + "20m": 20_000_000, + }, + }, +} + + +def select_best_params(trials, tokens_set, tvl, metric_key="returns_over_hodl"): + """From loaded trials, pick the best for a given token pair and TVL. + + Ranks by val (OOS) returns_over_hodl regardless of what objective the + trial was trained on — this is the consistent comparison metric. + """ + matching = [ + t for t in trials + if tuple(sorted(t["tokens"])) == tokens_set + and abs(t["initial_pool_value"] - tvl) < 1.0 + ] + if not matching: + return None + + def _get_val_roh(t): + """Extract OOS returns_over_hodl from test_objective or continuous_test_metrics.""" + # Try continuous_test_metrics first (more reliable) + ct = t.get("continuous_test_metrics", {}) + if isinstance(ct, dict) and "returns_over_hodl" in ct: + return float(ct["returns_over_hodl"]) + # Fall back to test_objective + to = t.get("test_objective", {}) + if isinstance(to, dict) and "returns_over_hodl" in to: + return float(to["returns_over_hodl"]) + if isinstance(to, list): + for entry in to: + if isinstance(entry, dict) and "returns_over_hodl" in entry: + return float(entry["returns_over_hodl"]) + # If the trial's own objective IS returns_over_hodl, use test_value + if t.get("return_val") == "returns_over_hodl": + return float(t.get("test_value", float("-inf"))) + return float("-inf") + + matching.sort(key=_get_val_roh, reverse=True) + best = matching[0] + best_roh = _get_val_roh(best) + print(f" Selection: {len(matching)} candidates, best val_roh={best_roh:+.4f} " + f"(trained on {best['return_val']})") + return best + + +def build_fingerprint(pair_cfg, tvl, start, end, noise_path=None): + """Build a run_fingerprint for a forward pass.""" + fp = { + "rule": "reclamm", + "tokens": pair_cfg["tokens"], + "startDateString": start, + "endDateString": end, + "initial_pool_value": float(tvl), + "do_arb": True, + "arb_frequency": 4, + "fees": pair_cfg["fees"], + "gas_cost": pair_cfg["gas_cost"], + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "noise_model": "mm_observed", + "noise_arrays_path": noise_path, + } + return fp + + +def build_noise_arrays(pair_cfg, start, end): + """Build or load cached noise arrays for the period.""" + from quantammsim.calibration.noise_model_arrays import build_mm_simulator_arrays + tok_a, tok_b = pair_cfg["tokens"] + pool_id = pair_cfg["pool_id"] + tag = f"{pool_id}_{start.split()[0]}_{end.split()[0]}_mm" + cache_path = f"results/mm_noise/_sim_arrays/{tag}.npz" + + if not os.path.exists(cache_path): + print(f" Building noise arrays: {tok_a}/{tok_b} {start} → {end}") + arrays = build_mm_simulator_arrays( + token_a=tok_a, token_b=tok_b, + start_date=start.split()[0], end_date=end.split()[0], + mm_artifact_dir="results/mm_noise", + competitor_tvl_path="results/competitor_tvl/competitor_tvl.npz", + pool_id=pool_id, + ) + os.makedirs(os.path.dirname(cache_path), exist_ok=True) + np.savez(cache_path, + noise_base=arrays["noise_base"], + competitor_tvl=arrays["competitor_tvl"]) + return cache_path + + +def run_forward(pair_cfg, tvl, params, start, end, noise_path): + """Run a single forward pass and return results dict.""" + fp = build_fingerprint(pair_cfg, tvl, start, end, noise_path) + result = do_run_on_historic_data( + run_fingerprint=fp, params=params, verbose=False, + ) + return result + + +def make_plot_data(all_results, pair_cfg, start, end): + """Convert our results into the format expected by plot_reclamm_optuna_result functions. + + All values are normalised to start at 1.0 for cross-TVL comparability. + Fee revenue is expressed as fraction of initial pool value. + Weights are computed from reserves × prices (effective weight of token 0). + + Returns (configs, time_series, hodl_values, ref_config). + """ + from collections import OrderedDict + configs = OrderedDict() + time_series = {} + + start_str = start.split()[0] + end_str = end.split()[0] + ref_config = { + "tokens": pair_cfg["tokens"], + "startDateString": start, + "endDateString": end, + } + + # Normalised HODL: use the first result's initial reserves + first_result = next(iter(all_results.values()))["result"] + prices = np.array(first_result["prices"]) + reserves_0 = np.array(first_result["reserves"][0]) + hodl_raw = np.sum(reserves_0 * prices, axis=1) + hodl_values = hodl_raw / hodl_raw[0] # normalise to 1.0 + + for tvl_label, data in all_results.items(): + result = data["result"] + val = np.array(result["value"]) + initial_val = val[0] + reserves = np.array(result["reserves"]) + res_prices = np.array(result["prices"]) + + # Compute effective weights from reserves × prices + token_values = reserves * res_prices # (T, 2) + total_value = token_values.sum(axis=1, keepdims=True) + weights = token_values / np.maximum(total_value, 1e-30) + + # Normalise fee revenue as fraction of initial TVL + fee_rev = np.array(result.get("fee_revenue", np.zeros(len(val)))) + fee_rev_normalised = fee_rev / initial_val + + p = data["params"] + pr = float(jnp.asarray(p.get("price_ratio", 0)).flatten()[0]) + margin = float(jnp.asarray(p.get("centeredness_margin", 0)).flatten()[0]) + shift = float(jnp.asarray(p.get("shift_exponent", 0)).flatten()[0]) + name = f"reCLAMM ${tvl_label} (PR={pr:.2f}, m={margin:.2f}, s={shift:.3f})" + configs[name] = { + "tvl_label": tvl_label, + "params": p, + } + time_series[name] = { + "value": val / initial_val, # normalise to 1.0 + "fee_revenue": fee_rev_normalised, + "reserves": reserves, + "prices": res_prices, + "initial_tvl": initial_val, # for volume normalisation + "weights": weights, + } + + return configs, time_series, hodl_values, ref_config + + +def run_pair(pair_name, pair_cfg, trials, output_dir): + """Run all TVL variants for a pair, for train and test periods.""" + tokens = pair_cfg["tokens"] + tokens_set = tuple(sorted(tokens)) + print(f"\n{'='*60}") + print(f" {'/'.join(tokens)} — {pair_name}") + print(f"{'='*60}") + + # Build noise arrays for both periods + train_noise = build_noise_arrays(pair_cfg, TRAIN_START, TRAIN_END) + test_noise = build_noise_arrays(pair_cfg, TEST_START, TEST_END) + + train_results = {} + test_results = {} + + for tvl_label, tvl in pair_cfg["tvls"].items(): + full_label = f"{pair_name}_{tvl_label}" + print(f"\n --- {full_label} (TVL=${tvl:,.0f}) ---") + + # Select best params for this TVL + best = select_best_params(trials, tokens_set, tvl) + if best is None: + print(f" No results found for {tokens_set} TVL={tvl}") + continue + + params = dict(best["params"]) + pr = float(jnp.asarray(params.get("price_ratio", 0)).flatten()[0]) + margin = float(jnp.asarray(params.get("centeredness_margin", 0)).flatten()[0]) + shift = float(jnp.asarray(params.get("shift_exponent", 0)).flatten()[0]) + print(f" Best: obj={best['return_val']} train={best['train_value']:+.4f} test={best['test_value']:+.4f}") + print(f" Params: PR={pr:.3f} margin={margin:.4f} shift={shift:.6f}") + print(f" Source: {best['study_id']}") + + # Ensure initial_weights_logits exists + if "initial_weights_logits" not in params: + params["initial_weights_logits"] = jnp.zeros(len(tokens)) + + # Train period forward pass + print(f" Running train period...") + train_result = run_forward(pair_cfg, tvl, params, TRAIN_START, TRAIN_END, train_noise) + train_results[tvl_label] = {"result": train_result, "params": params, "best": best} + + # Clear JIT caches between runs to manage memory + jax.clear_caches() + + # Test period forward pass + print(f" Running test period...") + test_result = run_forward(pair_cfg, tvl, params, TEST_START, TEST_END, test_noise) + test_results[tvl_label] = {"result": test_result, "params": params, "best": best} + + jax.clear_caches() + + # Print summary + train_val = np.array(train_result["value"]) + test_val = np.array(test_result["value"]) + train_hodl = np.sum(np.array(train_result["reserves"][0]) * np.array(train_result["prices"]), axis=1) + test_hodl = np.sum(np.array(test_result["reserves"][0]) * np.array(test_result["prices"]), axis=1) + train_roh = train_val[-1] / train_hodl[-1] - 1 + test_roh = test_val[-1] / test_hodl[-1] - 1 + train_fee = float(np.array(train_result.get("fee_revenue", np.zeros(1))).sum()) + test_fee = float(np.array(test_result.get("fee_revenue", np.zeros(1))).sum()) + print(f" Train RoH: {train_roh:+.2%} Fee: ${train_fee:,.0f}") + print(f" Test RoH: {test_roh:+.2%} Fee: ${test_fee:,.0f}") + + # Generate plots using existing plotting functions + for period_name, results_dict, start, end in [ + ("train", train_results, TRAIN_START, TRAIN_END), + ("test", test_results, TEST_START, TEST_END), + ]: + if not results_dict: + continue + configs, ts, hodl, ref_cfg = make_plot_data(results_dict, pair_cfg, start, end) + # Create a simple args object for the plot functions + class PlotArgs: + output = os.path.join(output_dir, f"{pair_name}_{period_name}.png") + plot_args = PlotArgs() + plot_optuna_results(configs, ts, hodl, ref_cfg, plot_args) + plot_optuna_weights(configs, ts, ref_cfg, plot_args) + + # Save results for later use + cache_path = os.path.join(output_dir, f"{pair_name}_sim_results.pkl") + with open(cache_path, "wb") as f: + pickle.dump({"train": train_results, "test": test_results}, f) + print(f" Saved cache: {cache_path}") + + return train_results, test_results + + +def main(): + p = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--pair", choices=["aave", "cow"], default=None) + p.add_argument("--all", action="store_true") + p.add_argument("--method", default="optuna", choices=["optuna", "cma_es"], + help="Which sweep method results to use") + p.add_argument("--output-dir", default="results/final_sims") + p.add_argument("--metric", default="returns_over_hodl", + help="Metric to rank trials by for param selection") + args = p.parse_args() + + os.makedirs(args.output_dir, exist_ok=True) + + pairs = list(PAIR_CONFIGS.keys()) if args.all else [args.pair] + if not args.all and not args.pair: + p.error("Specify --pair or --all") + + # Load all trials + print("Loading sweep results...") + trials = load_reclamm_results("./results/", metric_key=args.metric) + + # Filter to our training window + trials = [ + t for t in trials + if t.get("start_date", "").startswith("2025-01-01") + and "2025-10-05" in t.get("end_date", "") + ] + print(f" {len(trials)} trials from Jan-Oct 2025 sweep") + + for pair_name in pairs: + pair_cfg = PAIR_CONFIGS[pair_name] + run_pair(pair_name, pair_cfg, trials, args.output_dir) + + print("\nDone.") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_full_sweep.sh b/scripts/run_full_sweep.sh new file mode 100755 index 00000000..47497007 --- /dev/null +++ b/scripts/run_full_sweep.sh @@ -0,0 +1,142 @@ +#!/bin/bash +# Full reClAMM parameter sweep: AAVE/ETH + COW/ETH +# Train: 2025-01-01 → 2025-10-05 (pre flash crash) +# Test: 2025-10-25 → 2026-05-01 (post flash crash, separate run) +# +# Usage: bash scripts/run_full_sweep.sh [--method optuna|cma_es] +# Monitor: tail -5 /tmp/tune_full_*.log +# Results: results/full_sweep/ + +source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public + +# Parse arguments +METHOD="optuna" +PAIR_FILTER="" # empty = all pairs +while [ $# -gt 0 ]; do + case $1 in + --method=*) METHOD="${1#*=}" ;; + --method) shift; METHOD="$1" ;; + --pair=*) PAIR_FILTER="${1#*=}" ;; + --pair) shift; PAIR_FILTER="$1" ;; + esac + shift +done + +TRIALS=300 +MAX_PARALLEL="${MAX_WORKERS:-8}" +if [ "$METHOD" = "cma_es" ] && [ "$MAX_PARALLEL" = "8" ]; then + MAX_PARALLEL=4 +fi + +OBJECTIVES=( + returns_over_hodl + fee_revenue_over_value + calmar + daily_log_sharpe_excess +) + +OVERFITTING_PENALTIES=("" "1.0" "5.0") + +# Train period: pre flash crash +TRAIN_START="2025-01-01 00:00:00" +TRAIN_END="2025-10-05 00:00:00" + +# token_a token_b pool_id gas_cost fees tvl_label initial_tvl +CONFIGS=( + "AAVE ETH 0x9d1fcf346ea1b0 1.0 0.0025 aave_1m 1000000" + "AAVE ETH 0x9d1fcf346ea1b0 1.0 0.0025 aave_5m 5000000" + "AAVE ETH 0x9d1fcf346ea1b0 1.0 0.0025 aave_20m 20000000" + "COW ETH 0xd321300ef77067 3.0 0.003 cow_500k 500000" + "COW ETH 0xd321300ef77067 3.0 0.003 cow_2m 2000000" + "COW ETH 0xd321300ef77067 3.0 0.003 cow_20m 20000000" +) + +OUTDIR="results/full_sweep" +mkdir -p "$OUTDIR" + +# Build COMMON command based on method +if [ "$METHOD" = "cma_es" ]; then + CMA_GENS="${CMA_GENERATIONS:-500}" + COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --method cma_es --cma-generations $CMA_GENS" + METHOD_TAG="_cmaes" + echo "=== CMA-ES mode ($CMA_GENS generations) ===" +else + PR_MAX_FLAG="" + if [ -n "${PR_MAX:-}" ]; then + PR_MAX_FLAG="--pr-max $PR_MAX" + METHOD_TAG="_prmax${PR_MAX}" + else + METHOD_TAG="" + fi + COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS $PR_MAX_FLAG" + echo "=== Optuna mode ($TRIALS trials${PR_MAX:+, PR max=$PR_MAX}) ===" +fi + +wait_for_slot() { + while [ "$(jobs -rp | wc -l)" -ge "$MAX_PARALLEL" ]; do + sleep 10 + done +} + +N=0 +SKIPPED=0 +for config_line in "${CONFIGS[@]}"; do + read -r tok_a tok_b pool_id gas_cost fees tvl_label initial_tvl <<< "$config_line" + + # Filter by pair if specified (e.g. --pair cow matches cow_500k, cow_2m, cow_20m) + if [ -n "$PAIR_FILTER" ] && [[ "$tvl_label" != ${PAIR_FILTER}* ]]; then + continue + fi + + for obj in "${OBJECTIVES[@]}"; do + for penalty in "${OVERFITTING_PENALTIES[@]}"; do + tag="${obj}" + penalty_flag="" + if [ -n "$penalty" ]; then + tag="${tag}_penalty${penalty}" + penalty_flag="--overfitting-penalty $penalty" + fi + tag="${tag}${METHOD_TAG}_${tvl_label}" + + logfile="/tmp/tune_full_${tag}.log" + outfile="${OUTDIR}/${tag}.json" + + # Skip if result already exists + if [ -f "$outfile" ]; then + echo "[$N] Skipping (exists): ${tag}" + N=$((N + 1)) + SKIPPED=$((SKIPPED + 1)) + continue + fi + + wait_for_slot + + echo "[$N] Launching: ${tag}" + $COMMON --tokens "$tok_a" "$tok_b" \ + --pool-id "$pool_id" \ + --gas-cost "$gas_cost" \ + --fees "$fees" \ + --initial-pool-value "$initial_tvl" \ + --objective "$obj" \ + --start-date "$TRAIN_START" \ + --end-date "$TRAIN_END" \ + $penalty_flag \ + --output "$outfile" \ + > "$logfile" 2>&1 & + + N=$((N + 1)) + done + done +done + +echo "" +echo "$N jobs total ($SKIPPED skipped, $((N - SKIPPED)) launched)" +echo "Max parallel: $MAX_PARALLEL" +echo "" +echo "Monitor: tail -5 /tmp/tune_full_*.log" +echo "Results: ls $OUTDIR/" +echo "Summary: grep -A3 'Best trial' /tmp/tune_full_*.log" +echo "" +echo "Waiting for all jobs to finish..." +wait +echo "Done." diff --git a/scripts/select_best_params.py b/scripts/select_best_params.py new file mode 100644 index 00000000..d67f9e68 --- /dev/null +++ b/scripts/select_best_params.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +"""Select best reClAMM params from sweep results. + +Reads all result JSONs and corresponding log files for a given TVL config, +extracts train/val metrics, and picks the best param set. + +Usage: + python scripts/select_best_params.py --config aave_1m + python scripts/select_best_params.py --config cow_500k --method cma_es + python scripts/select_best_params.py --all # summarise all configs +""" + +import argparse +import json +import glob +import os +import re + + +SWEEP_DIR = "results/full_sweep" +LOG_DIR = "/tmp" + +# TVL configs matching run_full_sweep.sh +CONFIGS = { + "aave_1m": {"tokens": ["AAVE", "ETH"], "pool_id": "0x9d1fcf346ea1b0", "gas_cost": 1.0, "fees": 0.0025, "initial_pool_value": 1_000_000}, + "aave_5m": {"tokens": ["AAVE", "ETH"], "pool_id": "0x9d1fcf346ea1b0", "gas_cost": 1.0, "fees": 0.0025, "initial_pool_value": 5_000_000}, + "aave_20m": {"tokens": ["AAVE", "ETH"], "pool_id": "0x9d1fcf346ea1b0", "gas_cost": 1.0, "fees": 0.0025, "initial_pool_value": 20_000_000}, + "cow_500k": {"tokens": ["COW", "ETH"], "pool_id": "0xd321300ef77067", "gas_cost": 3.0, "fees": 0.003, "initial_pool_value": 500_000}, + "cow_2m": {"tokens": ["COW", "ETH"], "pool_id": "0xd321300ef77067", "gas_cost": 3.0, "fees": 0.003, "initial_pool_value": 2_000_000}, + "cow_20m": {"tokens": ["COW", "ETH"], "pool_id": "0xd321300ef77067", "gas_cost": 3.0, "fees": 0.003, "initial_pool_value": 20_000_000}, +} + + +def parse_params(param_dict): + """Normalise params from JSON — handles both Optuna (float) and CMA-ES (string array) formats.""" + out = {} + for key in ("price_ratio", "centeredness_margin", "shift_exponent"): + val = param_dict.get(key) + if val is None: + continue + if isinstance(val, str): + # CMA-ES stores as "[1.234]" + val = float(val.strip("[]")) + out[key] = float(val) + return out + + +def parse_log_metrics(log_path): + """Extract train and val metrics from a tuning log file.""" + metrics = {} + if not os.path.exists(log_path): + return metrics + with open(log_path) as f: + text = f.read() + + # Find the final "Best trial" block + m = re.search( + r"Train \(IS\):\s*(.*?)\n.*?Val \(OOS\):\s*(.*?)(?:\n|$)", + text, re.DOTALL, + ) + if not m: + return metrics + + for prefix, line in [("train_", m.group(1)), ("val_", m.group(2))]: + for pair in re.findall(r"(\w+)=([+-]?\d+\.?\d*|[+-]?inf)", line): + try: + metrics[prefix + pair[0]] = float(pair[1]) + except ValueError: + pass + + # Extract completed/failed counts + cm = re.search(r"(\d+) completed.*?(\d+) failed", text) + if cm: + metrics["n_completed"] = int(cm.group(1)) + metrics["n_failed"] = int(cm.group(2)) + + return metrics + + +def find_results(config_name, method="optuna"): + """Find all result files for a given config and method.""" + method_tag = "_cmaes" if method == "cma_es" else "" + pattern = os.path.join(SWEEP_DIR, f"*{method_tag}_{config_name}.json") + results = [] + for json_path in sorted(glob.glob(pattern)): + fname = os.path.basename(json_path).replace(".json", "") + # Derive the log file name + log_name = f"tune_full_{fname}.log" + log_path = os.path.join(LOG_DIR, log_name) + + with open(json_path) as f: + data = json.load(f) + + # The JSON has a single key = objective name + obj_name = list(data.keys())[0] + params = parse_params(data[obj_name]) + metrics = parse_log_metrics(log_path) + + results.append({ + "file": fname, + "objective": obj_name, + "params": params, + "metrics": metrics, + }) + return results + + +def rank_results(results, rank_by="val_returns_over_hodl"): + """Rank results by a metric, handling missing/inf values.""" + def sort_key(r): + v = r["metrics"].get(rank_by, float("-inf")) + if v != v or v == float("-inf"): # nan or -inf + return float("-inf") + return v + return sorted(results, key=sort_key, reverse=True) + + +def print_summary(config_name, results, top_n=5): + """Print a ranked summary table.""" + ranked = rank_results(results) + print(f"\n{'='*80}") + print(f" {config_name} — {len(results)} results (ranked by val RoH)") + print(f"{'='*80}") + print(f" {'Tag':<55s} {'PR':>6s} {'margin':>7s} {'shift':>8s} {'train_roh':>10s} {'val_roh':>10s}") + print(f" {'-'*55} {'-'*6} {'-'*7} {'-'*8} {'-'*10} {'-'*10}") + for r in ranked[:top_n]: + p = r["params"] + m = r["metrics"] + pr = f"{p.get('price_ratio', 0):.2f}" + margin = f"{p.get('centeredness_margin', 0):.3f}" + shift = f"{p.get('shift_exponent', 0):.4f}" + train = m.get("train_ret_over_hodl", m.get("train_returns_over_hodl", float("nan"))) + val = m.get("val_returns_over_hodl", float("nan")) + train_s = f"{train:+.4f}" if train == train else "N/A" + val_s = f"{val:+.4f}" if val == val else "N/A" + print(f" {r['file']:<55s} {pr:>6s} {margin:>7s} {shift:>8s} {train_s:>10s} {val_s:>10s}") + + if ranked: + best = ranked[0] + print(f"\n BEST: {best['file']}") + print(f" PR={best['params'].get('price_ratio', '?'):.4f} " + f"margin={best['params'].get('centeredness_margin', '?'):.4f} " + f"shift={best['params'].get('shift_exponent', '?'):.6f}") + return ranked + + +def export_best(config_name, ranked, method="optuna"): + """Write the best params to a consolidated JSON for downstream scripts.""" + if not ranked: + return + best = ranked[0] + method_tag = f"_{method}" if method != "optuna" else "" + out = { + "config": config_name, + "method": method, + "source": best["file"], + "params": best["params"], + "metrics": best["metrics"], + **CONFIGS[config_name], + } + out_path = os.path.join(SWEEP_DIR, f"best{method_tag}_{config_name}.json") + with open(out_path, "w") as f: + json.dump(out, f, indent=2, default=str) + print(f" Saved: {out_path}") + + +def main(): + p = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--config", default=None, help="Config name (e.g. aave_1m)") + p.add_argument("--all", action="store_true", help="Summarise all configs") + p.add_argument("--method", default="optuna", choices=["optuna", "cma_es"]) + p.add_argument("--top", type=int, default=8, help="Show top N results") + p.add_argument("--export", action="store_true", help="Export best params to JSON") + args = p.parse_args() + + configs = list(CONFIGS.keys()) if args.all else [args.config] + if not args.all and not args.config: + p.error("Specify --config or --all") + + for config_name in configs: + results = find_results(config_name, args.method) + if not results: + print(f"\n {config_name}: no results found for method={args.method}") + continue + ranked = print_summary(config_name, results, top_n=args.top) + if args.export: + export_best(config_name, ranked, args.method) + + +if __name__ == "__main__": + main() From 9fdffa809dd140449c09c37eafd812273dc1e27b Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:11:12 +0100 Subject: [PATCH 096/115] refactor: ReClammPool uses noise_arrays dict and forwards mm_observed keys _prepare_noise_arrays now returns a dict instead of a (vol, dow_sin, dow_cos) tuple; ReClammPool unpacks it via .get(...) to forward whatever keys the chosen noise model needs: - "ratio" / legacy paths: volatility, dow_sin, dow_cos - "market_linear": adds noise_base, noise_tvl_coeff - "mm_observed": adds noise_base, competitor_tvl This matches the dict-based API used downstream by _jax_calc_reclamm_reserves_with_dynamic_inputs and avoids forcing the caller to know which scalar arrays the active noise model expects. --- quantammsim/pools/reCLAMM/reclamm.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/quantammsim/pools/reCLAMM/reclamm.py b/quantammsim/pools/reCLAMM/reclamm.py index c9186045..4a0b0891 100644 --- a/quantammsim/pools/reCLAMM/reclamm.py +++ b/quantammsim/pools/reCLAMM/reclamm.py @@ -478,7 +478,7 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( if noise_params is not None and type(noise_params) is not dict: noise_params = dict(noise_params) - arb_vol, dow_sin, dow_cos = self._prepare_noise_arrays( + noise_arrays = self._prepare_noise_arrays( prices, run_fingerprint, start_index, bout_length, run_fingerprint["arb_frequency"], max_len, ) @@ -503,9 +503,12 @@ def calculate_reserves_and_fee_revenue_with_dynamic_inputs( lp_supply_array=materialized_inputs.lp_supply, noise_model=noise_model, noise_params=noise_params, - volatility_array=arb_vol, - dow_sin_array=dow_sin, - dow_cos_array=dow_cos, + volatility_array=noise_arrays.get("volatility"), + dow_sin_array=noise_arrays.get("dow_sin"), + dow_cos_array=noise_arrays.get("dow_cos"), + noise_base_array=noise_arrays.get("noise_base"), + noise_tvl_coeff_array=noise_arrays.get("noise_tvl_coeff"), + competitor_tvl_array=noise_arrays.get("competitor_tvl"), ) @partial(jit, static_argnums=(2,)) From 4c92743de2c096b15484eafbcb1fb1f350d72817 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:13:26 +0100 Subject: [PATCH 097/115] fix: subsample lp_supply_array to match arb_frequency prepare_dynamic_inputs builds lp_supply_array at minute resolution from the LP supply timeseries, but the pool scan loop iterates at arb_frequency-minute steps. materialize_dynamic_inputs then expects arrays whose length matches the scan_len, so a minute-resolution lp_supply_array silently mismatches for arb_frequency > 1. Fix by stride-subsampling lp_supply_array (and its test-period counterpart) by arb_frequency before they leave prepare_dynamic_inputs. No effect when arb_frequency == 1. --- quantammsim/runners/jax_runner_utils.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index 2f931903..a16d432c 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -1508,6 +1508,11 @@ def prepare_dynamic_inputs( if lp_supply_df is not None else None ) + # Subsample to match arb_frequency so materialize_dynamic_inputs sees + # the same scan_len that the pool's scan loop uses. + arb_freq = run_fingerprint.get("arb_frequency", 1) + if lp_supply_array is not None and arb_freq > 1: + lp_supply_array = lp_supply_array[::arb_freq] if do_test_period: test_lp_supply_array = ( raw_fee_like_amounts_to_fee_like_array( @@ -1520,6 +1525,8 @@ def prepare_dynamic_inputs( if lp_supply_df is not None else None ) + if test_lp_supply_array is not None and arb_freq > 1: + test_lp_supply_array = test_lp_supply_array[::arb_freq] reclamm_price_ratio_updates_array = ( _normalize_reclamm_price_ratio_updates_for_window( From a2424045a3f1ba15de689006e1afd1a389b5e271 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:28:54 +0100 Subject: [PATCH 098/115] fix: subsample all minute-res dynamic arrays by arb_frequency MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously only lp_supply_array was subsampled. fees_array, gas_cost_array, and arb_fees_array share the same shape contract — materialize_dynamic_inputs requires scan_len = (bout_length - 1) // arb_frequency, so any non-None minute-resolution array would have raised in _broadcast_dynamic_input_leaf once a caller populated those DataFrames with arb_frequency > 1. The bug never tripped because current callers use scalar fees/gas/arb_fees. Apply the same [::arb_freq] slice to all four (train + test) so the path is consistent and future callers can pass per-minute series safely. --- quantammsim/runners/jax_runner_utils.py | 30 +++++++++++++++++++------ 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/quantammsim/runners/jax_runner_utils.py b/quantammsim/runners/jax_runner_utils.py index a16d432c..7b857545 100644 --- a/quantammsim/runners/jax_runner_utils.py +++ b/quantammsim/runners/jax_runner_utils.py @@ -1508,11 +1508,6 @@ def prepare_dynamic_inputs( if lp_supply_df is not None else None ) - # Subsample to match arb_frequency so materialize_dynamic_inputs sees - # the same scan_len that the pool's scan loop uses. - arb_freq = run_fingerprint.get("arb_frequency", 1) - if lp_supply_array is not None and arb_freq > 1: - lp_supply_array = lp_supply_array[::arb_freq] if do_test_period: test_lp_supply_array = ( raw_fee_like_amounts_to_fee_like_array( @@ -1525,8 +1520,29 @@ def prepare_dynamic_inputs( if lp_supply_df is not None else None ) - if test_lp_supply_array is not None and arb_freq > 1: - test_lp_supply_array = test_lp_supply_array[::arb_freq] + + # Subsample minute-resolution dynamic inputs to match arb_frequency so + # materialize_dynamic_inputs sees the same scan_len the pool's scan loop + # uses (scan_len = (bout_length - 1) // arb_frequency). + arb_freq = run_fingerprint.get("arb_frequency", 1) + if arb_freq > 1: + if fees_array is not None: + fees_array = fees_array[::arb_freq] + if gas_cost_array is not None: + gas_cost_array = gas_cost_array[::arb_freq] + if arb_fees_array is not None: + arb_fees_array = arb_fees_array[::arb_freq] + if lp_supply_array is not None: + lp_supply_array = lp_supply_array[::arb_freq] + if do_test_period: + if test_fees_array is not None: + test_fees_array = test_fees_array[::arb_freq] + if test_gas_cost_array is not None: + test_gas_cost_array = test_gas_cost_array[::arb_freq] + if test_arb_fees_array is not None: + test_arb_fees_array = test_arb_fees_array[::arb_freq] + if test_lp_supply_array is not None: + test_lp_supply_array = test_lp_supply_array[::arb_freq] reclamm_price_ratio_updates_array = ( _normalize_reclamm_price_ratio_updates_for_window( From 7f5ca92b36ecb24e6249ccb909d6ff90791a2e02 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 16:30:36 +0100 Subject: [PATCH 099/115] feat: noise model support for Balancer pools MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend the noise-model dispatch already used in reCLAMM to BalancerPool's two reserve solvers (with_fees_using_precalcs and with_dynamic_inputs). Supports ratio, tsoukalas_sqrt / tsoukalas_log / loglinear, calibrated, market_linear, and mm_observed — the same set the reCLAMM path handles. Noise arrays are appended to the scan inputs at the same positions as in reclamm_reserves.py so the dispatch logic mirrors line-for-line, and noise volume is converted to noise-fee income (minus protocol_fee_split) and rebated to LPs via a uniform reserves scale, matching reCLAMM's treatment. _prepare_noise_arrays lives on BalancerPool but is a verbatim port of the reCLAMM helper — the noise model describes market-level organic volume, so the same arrays apply regardless of pool mechanics. --- quantammsim/pools/G3M/balancer/balancer.py | 128 ++++++++ .../pools/G3M/balancer/balancer_reserves.py | 274 ++++++++++++++++-- 2 files changed, 380 insertions(+), 22 deletions(-) diff --git a/quantammsim/pools/G3M/balancer/balancer.py b/quantammsim/pools/G3M/balancer/balancer.py index 89cf46dc..9a819792 100644 --- a/quantammsim/pools/G3M/balancer/balancer.py +++ b/quantammsim/pools/G3M/balancer/balancer.py @@ -15,6 +15,7 @@ _jax_calc_balancer_reserves_with_fees_using_precalcs, _jax_calc_balancer_reserves_with_dynamic_inputs, ) +from quantammsim.pools.reCLAMM.reclamm import _prepare_dynamic_array DEFAULT_BACKEND = default_backend() CPU_DEVICE = devices("cpu")[0] @@ -153,6 +154,18 @@ def calculate_reserves_with_fees( initial_value_per_token = weights * initial_pool_value initial_reserves = initial_value_per_token / local_prices[0] + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + + arb_freq = run_fingerprint["arb_frequency"] + max_len = arb_acted_upon_local_prices.shape[0] + _na = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, arb_freq, max_len, + ) + if run_fingerprint["do_arb"]: reserves = _jax_calc_balancer_reserves_with_fees_using_precalcs( initial_reserves, @@ -162,6 +175,17 @@ def calculate_reserves_with_fees( arb_thresh=run_fingerprint["gas_cost"], arb_fees=run_fingerprint["arb_fees"], all_sig_variations=jnp.array(run_fingerprint["all_sig_variations"]), + noise_model=noise_model, + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + seconds_per_step=float(arb_freq) * 60.0, + noise_params=noise_params, + volatility_array=_na.get("volatility"), + dow_sin_array=_na.get("dow_sin"), + dow_cos_array=_na.get("dow_cos"), + noise_base_array=_na.get("noise_base"), + noise_tvl_coeff_array=_na.get("noise_tvl_coeff"), + competitor_tvl_array=_na.get("competitor_tvl"), ) else: reserves = jnp.broadcast_to( @@ -320,6 +344,16 @@ def calculate_reserves_with_dynamic_inputs( do_trades=run_fingerprint["do_trades"], dtype=arb_acted_upon_local_prices.dtype, ) + noise_model = run_fingerprint.get("noise_model", "ratio") + noise_arrays = self._prepare_noise_arrays( + prices, run_fingerprint, start_index, + bout_length, run_fingerprint["arb_frequency"], max_len, + ) + + noise_params = run_fingerprint.get("reclamm_noise_params", None) + if noise_params is not None and type(noise_params) is not dict: + noise_params = dict(noise_params) + reserves = _jax_calc_balancer_reserves_with_dynamic_inputs( initial_reserves, weights, @@ -331,10 +365,104 @@ def calculate_reserves_with_dynamic_inputs( materialized_inputs.trades, run_fingerprint["do_trades"], run_fingerprint["do_arb"], + noise_model, materialized_inputs.lp_supply, + protocol_fee_split=run_fingerprint.get("protocol_fee_split", 0.0), + noise_trader_ratio=run_fingerprint.get("noise_trader_ratio", 0.0), + seconds_per_step=float(run_fingerprint.get("arb_frequency", 1)) * 60.0, + noise_params=noise_params, + volatility_array=noise_arrays.get("volatility"), + dow_sin_array=noise_arrays.get("dow_sin"), + dow_cos_array=noise_arrays.get("dow_cos"), + noise_base_array=noise_arrays.get("noise_base"), + noise_tvl_coeff_array=noise_arrays.get("noise_tvl_coeff"), + competitor_tvl_array=noise_arrays.get("competitor_tvl"), ) return reserves + def _prepare_noise_arrays(self, prices, run_fingerprint, start_index, + bout_length, arb_freq, max_len): + """Prepare noise arrays — identical to ReClammPool._prepare_noise_arrays. + + The noise model describes market-level organic volume, not pool + mechanics, so the same implementation applies to any pool type. + """ + import numpy as np + noise_model = run_fingerprint.get("noise_model", "ratio") + result = {"volatility": None, "dow_sin": None, "dow_cos": None, + "noise_base": None, "noise_tvl_coeff": None, + "competitor_tvl": None} + + if noise_model == "mm_observed": + nb = run_fingerprint.get("noise_base_array") + ct = run_fingerprint.get("competitor_tvl_array") + if nb is None and "noise_arrays_path" in run_fingerprint: + path = run_fingerprint["noise_arrays_path"] + if not hasattr(self, "_mm_observed_cache") or self._mm_observed_cache[0] != path: + arrays = np.load(path) + self._mm_observed_cache = ( + path, arrays["noise_base"], arrays["competitor_tvl"]) + nb = self._mm_observed_cache[1] + ct = self._mm_observed_cache[2] + if nb is not None: + result["noise_base"] = _prepare_dynamic_array( + jnp.array(nb), start_index, bout_length, arb_freq, max_len) + if ct is not None: + result["competitor_tvl"] = _prepare_dynamic_array( + jnp.array(ct), start_index, bout_length, arb_freq, max_len) + return result + + if noise_model == "market_linear": + nb = run_fingerprint.get("noise_base_array") + ntc = run_fingerprint.get("noise_tvl_coeff_array") + if nb is None and "noise_arrays_path" in run_fingerprint: + path = run_fingerprint["noise_arrays_path"] + if not hasattr(self, "_market_linear_cache") or self._market_linear_cache[0] != path: + arrays = np.load(path) + self._market_linear_cache = (path, arrays["noise_base"], arrays["noise_tvl_coeff"]) + nb = self._market_linear_cache[1] + ntc = self._market_linear_cache[2] + if nb is not None: + result["noise_base"] = _prepare_dynamic_array( + jnp.array(nb), start_index, bout_length, arb_freq, max_len) + if ntc is not None: + result["noise_tvl_coeff"] = _prepare_dynamic_array( + jnp.array(ntc), start_index, bout_length, arb_freq, max_len) + return result + + needs_vol = noise_model in ( + "tsoukalas_sqrt", "tsoukalas_log", "loglinear", "calibrated", + ) + if not needs_vol: + return result + + volatility_array = self.calculate_volatility_array( + prices, run_fingerprint, + ) + result["volatility"] = _prepare_dynamic_array( + volatility_array, start_index, bout_length, arb_freq, max_len, + ) + + if noise_model != "calibrated": + return result + + # Day-of-week sin/cos arrays for the calibrated noise model. + import pandas as pd + start_dt = pd.Timestamp(run_fingerprint["startDateString"]) + n_minutes = prices.shape[0] + day_indices = np.arange(n_minutes) // 1440 + start_weekday = start_dt.weekday() + weekdays = ((start_weekday + day_indices) % 7).astype(np.float64) + dow_sin_full = jnp.array(np.sin(2.0 * np.pi * weekdays / 7.0)) + dow_cos_full = jnp.array(np.cos(2.0 * np.pi * weekdays / 7.0)) + result["dow_sin"] = _prepare_dynamic_array( + dow_sin_full, start_index, bout_length, arb_freq, max_len, + ) + result["dow_cos"] = _prepare_dynamic_array( + dow_cos_full, start_index, bout_length, arb_freq, max_len, + ) + return result + def init_base_parameters( self, initial_values_dict: Dict[str, Any], diff --git a/quantammsim/pools/G3M/balancer/balancer_reserves.py b/quantammsim/pools/G3M/balancer/balancer_reserves.py index e36463b2..01003a39 100644 --- a/quantammsim/pools/G3M/balancer/balancer_reserves.py +++ b/quantammsim/pools/G3M/balancer/balancer_reserves.py @@ -15,6 +15,12 @@ parallelised_optimal_trade_sifter, ) from quantammsim.pools.G3M.G3M_trades import jitted_G3M_cond_trade +from quantammsim.pools.noise_trades import calculate_reserves_after_noise_trade +from quantammsim.pools.reCLAMM.reclamm_reserves import ( + reclamm_mm_observed_noise_volume, + reclamm_market_linear_noise_volume, + reclamm_calibrated_noise_volume, +) DEFAULT_BACKEND = default_backend() @@ -62,7 +68,7 @@ def _jax_calc_balancer_reserve_ratio(prev_prices, weights, prices): ) -@partial(jit, static_argnums=(5,)) +@partial(jit, static_argnums=(5, 6)) def _jax_calc_balancer_reserves_with_fees_scan_function_using_precalcs( carry_list, prices_and_precalcs, @@ -70,9 +76,14 @@ def _jax_calc_balancer_reserves_with_fees_scan_function_using_precalcs( tokens_to_drop, active_trade_directions, n, + noise_model="ratio", gamma=0.997, arb_thresh=0.0, arb_fees=0.0, + noise_trader_ratio=0.0, + protocol_fee_split=0.0, + seconds_per_step=60.0, + noise_params=None, ): """ Calculate changes in AMM reserves considering fees @@ -174,6 +185,77 @@ def _jax_calc_balancer_reserves_with_fees_scan_function_using_precalcs( reserves = jnp.where(do_price_arb_trade, post_price_reserves, prev_reserves) + # --- Noise model dispatch (non-dynamic path) --- + if noise_model == "ratio": + if_noise = noise_trader_ratio > 0 + applied_trade = reserves - prev_reserves + noisy = calculate_reserves_after_noise_trade( + applied_trade, reserves, prices, noise_trader_ratio, gamma, + ) + reserves = jnp.where(if_noise, noisy, reserves) + elif noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + from quantammsim.pools.reCLAMM.reclamm_reserves import ( + reclamm_tsoukalas_sqrt_noise_volume, + reclamm_tsoukalas_log_noise_volume, + reclamm_loglinear_noise_volume, + ) + volatility = prices_and_precalcs[4] + arb_volume = 0.5 * jnp.sum(jnp.abs(reserves - prev_reserves) * prices) + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + if noise_model == "tsoukalas_sqrt": + noise_vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + elif noise_model == "tsoukalas_log": + noise_vol = reclamm_tsoukalas_log_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + else: + noise_vol = reclamm_loglinear_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "calibrated": + volatility = prices_and_precalcs[4] + dow_sin = prices_and_precalcs[5] + dow_cos = prices_and_precalcs[6] + arb_volume = 0.5 * jnp.sum(jnp.abs(reserves - prev_reserves) * prices) + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + noise_vol = reclamm_calibrated_noise_volume( + effective_value, gamma, volatility, arb_volume, dow_sin, dow_cos, _np) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "market_linear": + noise_base = prices_and_precalcs[4] + noise_tvl_coeff = prices_and_precalcs[5] + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + noise_vol = reclamm_market_linear_noise_volume( + effective_value, noise_base, noise_tvl_coeff, + tvl_mean=_np.get("tvl_mean", 0.0), tvl_std=_np.get("tvl_std", 1.0)) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "mm_observed": + noise_base = prices_and_precalcs[4] + competitor_tvl = prices_and_precalcs[5] + effective_value = (reserves * prices).sum() + noise_vol = reclamm_mm_observed_noise_volume( + effective_value, noise_base, competitor_tvl) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + counter += 1 return [ prices, @@ -182,7 +264,7 @@ def _jax_calc_balancer_reserves_with_fees_scan_function_using_precalcs( ], reserves -@jit +@partial(jit, static_argnums=(), static_argnames=("noise_model",)) def _jax_calc_balancer_reserves_with_fees_using_precalcs( initial_reserves, weights, @@ -191,6 +273,17 @@ def _jax_calc_balancer_reserves_with_fees_using_precalcs( arb_thresh=0.0, arb_fees=0.0, all_sig_variations=None, + noise_model="ratio", + noise_trader_ratio=0.0, + protocol_fee_split=0.0, + seconds_per_step=60.0, + noise_params=None, + volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """ Calculate AMM reserves considering fees and arbitrage opportunities using signature variations, @@ -260,6 +353,11 @@ def _jax_calc_balancer_reserves_with_fees_using_precalcs( n=n_assets, tokens_to_drop=tokens_to_drop, active_trade_directions=active_trade_directions, + noise_model=noise_model, + noise_trader_ratio=noise_trader_ratio, + protocol_fee_split=protocol_fee_split, + seconds_per_step=seconds_per_step, + noise_params=noise_params, ) carry_list_init = [ @@ -267,21 +365,38 @@ def _jax_calc_balancer_reserves_with_fees_using_precalcs( initial_reserves, 0, ] + + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + ] + + # Append noise arrays at position 4+ in prices_and_precalcs + if noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) + _, reserves = scan( scan_fn, carry_list_init, - [ - prices, - active_initial_weights, - per_asset_ratios, - all_other_assets_ratios, - ], + scan_inputs, ) return reserves -@partial(jit, static_argnums=(6, 7, 8)) +@partial(jit, static_argnums=(6, 7, 8, 9)) def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using_precalcs( carry_list, input_list, @@ -291,7 +406,11 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using active_trade_directions, n, do_trades, - do_arb + do_arb, + noise_model="ratio", + protocol_fee_split=0.0, + seconds_per_step=60.0, + noise_params=None, ): """ Calculate changes in AMM reserves considering fees @@ -422,6 +541,81 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using reserves = jnp.where(do_price_arb_trade, post_price_reserves, prev_reserves) + # --- Noise model dispatch --- + # Same pattern as reclamm_reserves.py: input_list[9+] are noise arrays, + # protocol_fee_split / seconds_per_step / noise_params are passed via Partial. + noise_fee_income = jnp.asarray(0.0, dtype=prices.dtype) + if noise_model == "ratio": + noise_trader_ratio = input_list[9] + if_noise = noise_trader_ratio > 0 + applied_trade = reserves - prev_reserves + noisy = calculate_reserves_after_noise_trade( + applied_trade, reserves, prices, noise_trader_ratio, gamma, + ) + reserves = jnp.where(if_noise, noisy, reserves) + elif noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + from quantammsim.pools.reCLAMM.reclamm_reserves import ( + reclamm_tsoukalas_sqrt_noise_volume, + reclamm_tsoukalas_log_noise_volume, + reclamm_loglinear_noise_volume, + ) + volatility = input_list[9] + arb_volume = 0.5 * jnp.sum(jnp.abs(reserves - prev_reserves) * prices) + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + if noise_model == "tsoukalas_sqrt": + noise_vol = reclamm_tsoukalas_sqrt_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + elif noise_model == "tsoukalas_log": + noise_vol = reclamm_tsoukalas_log_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + else: + noise_vol = reclamm_loglinear_noise_volume( + effective_value, gamma, volatility, arb_volume, _np) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "calibrated": + volatility = input_list[9] + dow_sin = input_list[10] + dow_cos = input_list[11] + arb_volume = 0.5 * jnp.sum(jnp.abs(reserves - prev_reserves) * prices) + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + noise_vol = reclamm_calibrated_noise_volume( + effective_value, gamma, volatility, arb_volume, dow_sin, dow_cos, _np) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "market_linear": + noise_base = input_list[9] + noise_tvl_coeff = input_list[10] + effective_value = (reserves * prices).sum() + _np = noise_params if noise_params is not None else {} + noise_vol = reclamm_market_linear_noise_volume( + effective_value, noise_base, noise_tvl_coeff, + tvl_mean=_np.get("tvl_mean", 0.0), tvl_std=_np.get("tvl_std", 1.0)) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + elif noise_model == "mm_observed": + noise_base = input_list[9] + competitor_tvl = input_list[10] + effective_value = (reserves * prices).sum() + noise_vol = reclamm_mm_observed_noise_volume( + effective_value, noise_base, competitor_tvl) + minutes_per_step = seconds_per_step / 60.0 + noise_fee_total = (1.0 - gamma) * noise_vol * minutes_per_step + noise_fee_income = noise_fee_total * (1.0 - protocol_fee_split) + scale = 1.0 + noise_fee_income / jnp.maximum(effective_value, 1e-8) + reserves = reserves * scale + # apply trade if trade is present if do_trades: reserves += jitted_G3M_cond_trade(do_trades, reserves, weights, trade, gamma) @@ -435,7 +629,7 @@ def _jax_calc_balancer_reserves_with_dynamic_fees_and_trades_scan_function_using ], reserves -@partial(jit, static_argnums=(8,9,)) +@partial(jit, static_argnums=(8, 9, 10)) def _jax_calc_balancer_reserves_with_dynamic_inputs( initial_reserves, weights, @@ -447,7 +641,18 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( trades=None, do_trades=False, do_arb=True, + noise_model="ratio", lp_supply_array=None, + protocol_fee_split=0.0, + noise_trader_ratio=0.0, + seconds_per_step=60.0, + noise_params=None, + volatility_array=None, + dow_sin_array=None, + dow_cos_array=None, + noise_base_array=None, + noise_tvl_coeff_array=None, + competitor_tvl_array=None, ): """ Calculate AMM reserves considering fees and arbitrage opportunities using signature variations, @@ -543,6 +748,10 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( active_trade_directions=active_trade_directions, do_trades=do_trades, do_arb=do_arb, + noise_model=noise_model, + protocol_fee_split=protocol_fee_split, + seconds_per_step=seconds_per_step, + noise_params=noise_params, ) carry_list_init = [ @@ -551,20 +760,41 @@ def _jax_calc_balancer_reserves_with_dynamic_inputs( 0, lp_supply_array[0], ] + + scan_inputs = [ + prices, + active_initial_weights, + per_asset_ratios, + all_other_assets_ratios, + gamma, + arb_thresh, + arb_fees, + trades, + lp_supply_array, + ] + + # Append noise-model-specific arrays at position 9+ + # (same ordering as reclamm_reserves.py scan inputs) + if noise_model == "ratio": + ntr = jnp.broadcast_to(jnp.asarray(noise_trader_ratio), (prices.shape[0],)) + scan_inputs.append(ntr) + elif noise_model in ("tsoukalas_sqrt", "tsoukalas_log", "loglinear"): + scan_inputs.append(volatility_array) + elif noise_model == "calibrated": + scan_inputs.append(volatility_array) + scan_inputs.append(dow_sin_array) + scan_inputs.append(dow_cos_array) + elif noise_model == "market_linear": + scan_inputs.append(noise_base_array) + scan_inputs.append(noise_tvl_coeff_array) + elif noise_model == "mm_observed": + scan_inputs.append(noise_base_array) + scan_inputs.append(competitor_tvl_array) + _, reserves = scan( scan_fn, carry_list_init, - [ - prices, - active_initial_weights, - per_asset_ratios, - all_other_assets_ratios, - gamma, - arb_thresh, - arb_fees, - trades, - lp_supply_array, - ], + scan_inputs, ) return reserves From b5d8d1b98729fe395c40aea49d12e8e5b4180136 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 18:41:19 +0100 Subject: [PATCH 100/115] test: configure mm_observed noise model in lp_fee_revenue tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit lp_fee_revenue_usd is noise-only by design (so optimisers can't monetise sloshing arb volume through a decaying pool). The five tests below were written when lp_fee_revenue_usd reported total inbound fees, and stopped passing once the semantic changed. Add a small _mm_observed_noise_kwargs helper that supplies constant noise_base and competitor_tvl arrays (the simplest noise model to wire — two scalars vs. tsoukalas' volatility + noise_params). With noise_base=13.8 and K=1e7, a $1M pool produces ~$0.20/step of noise fee income at 0.3% fees — positive and detectable without dominating the pool dynamics. Tests updated: - test_fee_revenue_positive_on_price_jump - test_higher_fees_more_revenue - test_protocol_split_reduces_lp_revenue - test_dynamic_inputs_fee_revenue - test_lp_supply_with_fee_revenue (uses K=1e10 so noise volume scales ~linearly with pool TVL, preserving the doubling check) --- .../pools/reCLAMM/test_reclamm_fee_revenue.py | 18 ++++++++++++++++++ tests/pools/reCLAMM/test_reclamm_reserves.py | 11 ++++++++++- 2 files changed, 28 insertions(+), 1 deletion(-) diff --git a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py index 106f8a3d..2becdc73 100644 --- a/tests/pools/reCLAMM/test_reclamm_fee_revenue.py +++ b/tests/pools/reCLAMM/test_reclamm_fee_revenue.py @@ -49,6 +49,18 @@ def _init_pool(initial_pool_value=1_000_000.0, price_a=2500.0, price_b=1.0, return reserves, Va, Vb +def _mm_observed_noise_kwargs(prices, noise_base=13.8, competitor_tvl=1e7): + # lp_fee_revenue_usd is noise-only by design, so arb-only configs report 0. + # mm_observed is the simplest noise model to wire (two constants vs. e.g. + # tsoukalas which needs volatility + noise_params). + n = prices.shape[0] + return { + "noise_model": "mm_observed", + "noise_base_array": jnp.full(n, noise_base), + "competitor_tvl_array": jnp.full(n, competitor_tvl), + } + + class TestFeeRevenueShape: """_jax_calc_reclamm_reserves_and_fee_revenue_with_fees returns correct shapes.""" @@ -112,6 +124,7 @@ def test_fee_revenue_positive_on_price_jump(self): arb_thresh=0.0, arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, + **_mm_observed_noise_kwargs(prices), ) assert float(fee_revenue.sum()) > 0, ( f"Expected positive total fee revenue on trending prices, got {float(fee_revenue.sum())}" @@ -136,6 +149,7 @@ def test_higher_fees_more_revenue(self): arb_thresh=0.0, arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, + **_mm_observed_noise_kwargs(prices), ) _, fee_revenue_high = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -147,6 +161,7 @@ def test_higher_fees_more_revenue(self): arb_thresh=0.0, arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, + **_mm_observed_noise_kwargs(prices), ) assert float(fee_revenue_high.sum()) > float(fee_revenue_low.sum()), ( @@ -173,6 +188,7 @@ def test_protocol_split_reduces_lp_revenue(self): arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, protocol_fee_split=0.0, + **_mm_observed_noise_kwargs(prices), ) _, fee_revenue_half_split = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -185,6 +201,7 @@ def test_protocol_split_reduces_lp_revenue(self): arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, protocol_fee_split=0.5, + **_mm_observed_noise_kwargs(prices), ) total_no_split = float(fee_revenue_no_split.sum()) @@ -256,6 +273,7 @@ def test_dynamic_inputs_fee_revenue(self): arb_thresh=arb_thresh, arb_fees=arb_fees, all_sig_variations=ALL_SIG_VARIATIONS_2, + **_mm_observed_noise_kwargs(prices), ) assert result_reserves.shape == (n_steps, 2) diff --git a/tests/pools/reCLAMM/test_reclamm_reserves.py b/tests/pools/reCLAMM/test_reclamm_reserves.py index b9664aa1..d3347e75 100644 --- a/tests/pools/reCLAMM/test_reclamm_reserves.py +++ b/tests/pools/reCLAMM/test_reclamm_reserves.py @@ -1334,11 +1334,18 @@ def test_lp_supply_through_pool_class(self): ) def test_lp_supply_with_fee_revenue(self): - """Doubling LP supply → fee revenue increases (bigger pool → bigger arb trades).""" + """Doubling LP supply → fee revenue increases (bigger pool → bigger noise trades).""" reserves, Va, Vb = _init_pool() n_steps = 40 half = n_steps // 2 prices = _make_trending_prices(2500.0, 3500.0, 1.0, n_steps) + # lp_fee_revenue_usd is noise-only; configure mm_observed with a large + # K so noise volume scales ~linearly with pool TVL. + noise_kwargs = { + "noise_model": "mm_observed", + "noise_base_array": jnp.full(n_steps, 13.8), + "competitor_tvl_array": jnp.full(n_steps, 1e10), + } # Baseline: no supply change _, rev_base = _jax_calc_reclamm_reserves_and_fee_revenue_with_fees( @@ -1350,6 +1357,7 @@ def test_lp_supply_with_fee_revenue(self): arb_thresh=0.0, arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, + **noise_kwargs, ) # Supply doubles halfway @@ -1364,6 +1372,7 @@ def test_lp_supply_with_fee_revenue(self): arb_fees=0.0, all_sig_variations=ALL_SIG_VARIATIONS_2, lp_supply_array=lp_supply, + **noise_kwargs, ) # After doubling, fee revenue per step should be larger From 99ea8b37d168f385c166f1f05c67d47112eb659e Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 11 May 2026 18:49:38 +0100 Subject: [PATCH 101/115] chore: remove accidental file --- .codex | 0 1 file changed, 0 insertions(+), 0 deletions(-) delete mode 100644 .codex diff --git a/.codex b/.codex deleted file mode 100644 index e69de29b..00000000 From 20dd4efdc120619f281222b2731261a994371e73 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 15 May 2026 14:18:01 +0100 Subject: [PATCH 102/115] fix: unify optuna trial schema with cma_es/bfgs and wire cma_es overfitting penalty Optuna's writer now persists train_objective, test_objective, and continuous_test_metrics as list[dict] with the full metric set, matching the shape save_multi_params produces for cma_es and bfgs. Rich metric dicts are captured in the optuna objective callback via set_user_attr. --overfitting-penalty is wired into cma_es_settings and applied inside eval_single using the same train-vs-val gap formula optuna uses, with val evaluation points drawn from the validation period. Penalty now factors into the run_fingerprint, so the 3 penalty variants of each (token, objective, TVL) produce distinct run_*.json files. --- experiments/tune_reclamm_calibrated_noise.py | 2 + quantammsim/core_simulator/result_exporter.py | 23 ++++++- quantammsim/runners/jax_runners.py | 61 ++++++++++++++++++- 3 files changed, 81 insertions(+), 5 deletions(-) diff --git a/experiments/tune_reclamm_calibrated_noise.py b/experiments/tune_reclamm_calibrated_noise.py index fa9fd198..2c9946be 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/experiments/tune_reclamm_calibrated_noise.py @@ -203,6 +203,8 @@ def _build_opt_settings(args): "tol": 1e-8, "n_evaluation_points": args.cma_eval_points, "compute_dtype": "float32", + **({"overfitting_penalty": args.overfitting_penalty} + if args.overfitting_penalty is not None else {}), }, } else: diff --git a/quantammsim/core_simulator/result_exporter.py b/quantammsim/core_simulator/result_exporter.py index 1c0fd27a..d460c2f4 100644 --- a/quantammsim/core_simulator/result_exporter.py +++ b/quantammsim/core_simulator/result_exporter.py @@ -222,8 +222,27 @@ def save_optuna_results_sgd_format( # Add metadata in SGD format param_dict["step"] = trial.number - param_dict["test_objective"] = float(trial.user_attrs.get("validation_value", float("-inf"))) - param_dict["train_objective"] = float(trial.user_attrs.get("train_value", float("-inf"))) + + # train_objective / test_objective / continuous_test_metrics in the + # same list-of-dict shape produced by save_multi_params (BFGS, + # CMA-ES). test_objective and continuous_test_metrics carry the + # same dict — the test-period metrics extracted from the continuous + # train→test forward pass — mirroring how save_multi_params is + # called in the BFGS/CMA-ES branches. + train_metrics_dict = trial.user_attrs.get("train_metrics_dict") + cont_test_metrics_dict = trial.user_attrs.get("continuous_test_metrics_dict") + if train_metrics_dict: + param_dict["train_objective"] = [train_metrics_dict] + else: + # Back-compat for trials saved before the rich dicts were + # captured: fall back to the scalar training-objective value. + param_dict["train_objective"] = float(trial.user_attrs.get("train_value", float("-inf"))) + if cont_test_metrics_dict: + param_dict["test_objective"] = [cont_test_metrics_dict] + param_dict["continuous_test_metrics"] = [cont_test_metrics_dict] + else: + param_dict["test_objective"] = float(trial.user_attrs.get("validation_value", float("-inf"))) + param_dict["objective"] = float(trial.value) if trial.value is not None else float("-inf") param_dict["hessian_trace"] = 0 # Not applicable for optuna param_dict["local_learning_rate"] = 0 # Not applicable for optuna diff --git a/quantammsim/runners/jax_runners.py b/quantammsim/runners/jax_runners.py index d8cddd32..d7671f92 100644 --- a/quantammsim/runners/jax_runners.py +++ b/quantammsim/runners/jax_runners.py @@ -1499,6 +1499,20 @@ def objective(trial): continuous_prices, ) + # Full train-period metric dict, parallel to the BFGS/CMA-ES + # save_multi_params path. Persisted on the trial so + # save_optuna_results_sgd_format can write the same + # list-of-dict schema other methods use. + train_dict_for_metrics = { + "value": train_outputs["value"], + "reserves": train_outputs["reserves"], + } + if "fee_revenue" in train_outputs: + train_dict_for_metrics["fee_revenue"] = train_outputs["fee_revenue"] + train_metrics_dict = calculate_period_metrics( + train_dict_for_metrics, train_outputs["prices"], + ) + # Calculate validation metrics train_length = data_dict["bout_length"] if val_fraction > 0: @@ -1608,6 +1622,14 @@ def objective(trial): trial.set_user_attr("continuous_test_return", continuous_test_metrics["return"]) trial.set_user_attr("continuous_test_returns_over_hodl", continuous_test_metrics["returns_over_hodl"]) trial.set_user_attr("continuous_test_returns_over_uniform_hodl", continuous_test_metrics["returns_over_uniform_hodl"]) + # Full metric dicts for save_optuna_results_sgd_format — + # match the list-of-dict schema produced by save_multi_params + # (BFGS / CMA-ES). Plain floats so optuna can persist them. + # `.item()` handles 0-d and (1,) JAX arrays alike. + def _scalarise(d): + return {k: float(np.asarray(v).reshape(-1)[0]) for k, v in d.items()} + trial.set_user_attr("train_metrics_dict", _scalarise(train_metrics_dict)) + trial.set_user_attr("continuous_test_metrics_dict", _scalarise(continuous_test_metrics)) if run_fingerprint["optimisation_settings"]["optuna_settings"][ "multi_objective" @@ -2208,6 +2230,9 @@ def solve_single(flat_x0): tol = cma_settings["tol"] n_eval_points = cma_settings["n_evaluation_points"] population_size_override = cma_settings.get("population_size") + overfitting_penalty = float(cma_settings.get("overfitting_penalty", 0.0)) + # Penalty only meaningful when there's a held-out validation period. + apply_penalty = overfitting_penalty > 0.0 and val_fraction > 0 # Generate fixed evaluation points (same as BFGS/optuna) min_spacing = data_dict["bout_length"] // 2 @@ -2219,6 +2244,23 @@ def solve_single(flat_x0): min_spacing, run_fingerprint["optimisation_settings"]["initial_random_key"], ) + n_train_eval = len(evaluation_starts) + if apply_penalty: + # Sample evaluation points from the validation period and append + # them to the fixed start indexes. The split point n_train_eval + # separates train-period objectives from val-period objectives. + val_eval_starts = generate_evaluation_points( + val_start_idx, + data_dict["end_idx"], + bout_length_window, + n_eval_points, + min_spacing, + run_fingerprint["optimisation_settings"]["initial_random_key"] + 1, + ) + evaluation_starts = list(evaluation_starts) + list(val_eval_starts) + if verbose: + print(f"[CMA-ES] Overfitting penalty {overfitting_penalty} active: " + f"{n_train_eval} train + {len(val_eval_starts)} val eval points") fixed_start_indexes = jnp.array( [(s, 0) for s in evaluation_starts], dtype=jnp.int32 ) @@ -2281,9 +2323,22 @@ def solve_single(flat_x0): # Build eval function: population (lam, n_flat) -> fitness (lam,) # Each individual is evaluated as -objective (we minimise, objective is maximised) - def eval_single(flat_x): - p = unravel_fn(flat_x) - return -batched_obj(p, fixed_start_indexes) + if apply_penalty: + # Mirror the optuna penalty: penalised = mean_train - α·max(0, mean_train - mean_val) + _alpha = jnp.asarray(overfitting_penalty, dtype=flat_x0_template.dtype) + _split = n_train_eval + def eval_single(flat_x): + p = unravel_fn(flat_x) + per_pt = batched_pts(p, fixed_start_indexes) + mean_train = jnp.mean(per_pt[:_split]) + mean_val = jnp.mean(per_pt[_split:]) + gap = mean_train - mean_val + penalised = mean_train - _alpha * jnp.maximum(0.0, gap) + return -penalised + else: + def eval_single(flat_x): + p = unravel_fn(flat_x) + return -batched_obj(p, fixed_start_indexes) # Un-jitted vmap for fusion into lax.while_loop's XLA program eval_fn_raw = vmap(eval_single) From 856eacb031bac790d8a0a58de0cd70d39ebdc53c Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 15 May 2026 15:09:18 +0100 Subject: [PATCH 103/115] chore: move noise pipeline scripts to scripts/ and add training guide MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Move experiments/{fetch_competitor_tvl,run_mm_noise,tune_reclamm_calibrated_noise}.py into scripts/ to live alongside run_full_sweep.sh and run_final_sims.py. Update scripts/run_full_sweep.sh and scripts/run_period_sweep.sh to the new paths. Add RECLAMM_TRAINING.md at repo root — end-to-end guide covering data pull, grid build, MM noise model, training sweep, selection and plot stages, plus extending the calibration set to a new pair. --- RECLAMM_TRAINING.md | 430 ++++++++++++++++++ .../fetch_competitor_tvl.py | 4 +- scripts/run_full_sweep.sh | 4 +- {experiments => scripts}/run_mm_noise.py | 6 +- scripts/run_period_sweep.sh | 2 +- .../tune_reclamm_calibrated_noise.py | 8 +- 6 files changed, 442 insertions(+), 12 deletions(-) create mode 100644 RECLAMM_TRAINING.md rename {experiments => scripts}/fetch_competitor_tvl.py (99%) rename {experiments => scripts}/run_mm_noise.py (99%) rename {experiments => scripts}/tune_reclamm_calibrated_noise.py (98%) diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md new file mode 100644 index 00000000..7d8db964 --- /dev/null +++ b/RECLAMM_TRAINING.md @@ -0,0 +1,430 @@ +# reCLAMM training: end-to-end + +Soup-to-nuts guide for going from no data on disk to a set of trained reCLAMM +params and the heatmap / weight / fee-revenue plots used in reports. + +All commands assume the working directory is the repo root and the conda env +`qsim_reclamm_public` is active: + +``` +cd /Users/matthew/Projects/quantammsim-reclamm-public/quantammsim +source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public +``` + +The pipeline has five stages. Outputs of each stage feed the next, so order +matters. + +``` +[1] Data pull → quantammsim/data/*.parquet, local_data/..., + results/competitor_tvl/, results/pool_grids_v2/ +[2] Grid build → results/pool_grids_v2/*_daily.parquet +[3] MM noise model → results/mm_noise/model.npz + meta.json +[4] Training sweep → results/full_sweep/*.json + results/run_*.json +[5] Selection + plots → results/final_sims/{aave,cow}*.png + .pkl + results/final_sims/for_fabio//price_ratio_sweep/* +``` + +### What depends on what + +The pipeline has two distinct dependency surfaces, separated by the MM model +artifact. + +Training the MM noise model (stages 1+2+3) needs: + +- Balancer V3 API (`api-v3.balancer.fi`) — panel snapshots +- Binance minute parquets — token price series +- DeFi Llama — competitor TVL +- Pool grids (`results/pool_grids_v2/`) — built locally from the panel + +Training a reCLAMM (stages 4+5) only needs: + +- Binance minute parquets (`quantammsim/data/_USD.parquet`) +- The MM artifact (`results/mm_noise/model.npz` + `meta.json`) +- The competitor-TVL bundle (`results/competitor_tvl/competitor_tvl.npz`) + +`quantammsim/runners/jax_runners.py` does not import from +`quantammsim.noise_calibration.*`, and +`quantammsim/calibration/noise_model_arrays.build_mm_simulator_arrays` reads +only Binance data plus the two artifacts (standardisation stats are stored +inside `model.npz` as `x_mean` / `x_std`). Once the MM model exists you can +train reCLAMMs on a different machine by carrying `results/mm_noise/`, +`results/competitor_tvl/`, and the Binance parquets for the pair. + +--- + +## 1. Pull the data + +Three data sources, three scripts. + +### 1a. Token price history (Binance minute bars) + +``` +python scripts/download_data.py BTC ETH AAVE COW USDC USDT WBTC +``` + +Reads from `scripts/ticker_list.txt` if no tickers are passed. Output: +`quantammsim/data/_USD.parquet` (minute resolution) and +`_USD_daily.csv`. Used by all simulator runs and noise calibration. + +### 1b. Balancer pool snapshots (volume + TVL) + +``` +python -m quantammsim.noise_calibration --fetch --chain ethereum +python -m quantammsim.noise_calibration --fetch --chain base +python -m quantammsim.noise_calibration --fetch --chain gnosis +``` + +Fetches the V3 API (`api-v3.balancer.fi`) for WEIGHTED and RECLAMM pools, then +pulls daily snapshots and assembles a panel. Outputs the panel parquet under +`local_data/noise_calibration/panel.parquet`. + +#### How the pool list is determined + +`--fetch` does not take a pool or pair argument. It calls +`enumerate_balancer_pools(min_tvl=args.min_tvl)`, which asks the Balancer V3 +API for every WEIGHTED + RECLAMM pool currently above `--min-tvl` (default +$10k) on the specified chain. The threshold is checked against the pool's +*current* TVL at fetch time, so a pool that once held large TVL but is now +below the floor will be excluded — even if its earlier history would have +been useful. + +The calibration set the MM model is trained on is the survivor set after +downstream filters: pools with both tokens matched to Binance data and with +enough clean daily snapshots. The survivors are cached in +`local_data/noise_calibration/_cache/stage1.pkl` (the `matched_clean` dict). +`scripts/fetch_competitor_tvl.py` and `scripts/run_mm_noise.py` both read +that file for their pool list and do not expose a per-pool selector. + +To get a specific pool included: + +1. Confirm its current TVL is above `--min-tvl`, or lower the threshold. +2. Make sure both tokens have Binance parquets in `quantammsim/data/` + (`scripts/download_data.py `). +3. Re-run `--fetch` → `fetch_competitor_tvl.py` → `run_mm_noise.py`. Any + pool that passes the threshold and Binance-match check ends up in the + trained MM artifact. + +#### How the pool address propagates downstream + +`scripts/tune_reclamm_calibrated_noise.py` takes `--pool-id`. It is used +only as a lookup key into the frozen MM artifact: +`_find_pool_index(pool_id, meta["pool_ids"])` returns the per-pool +`log_alpha`, `gamma`, `log_K`, and `log_cadence`. The address is not used +to fetch price data — prices come from Binance via `--tokens `. So +`--tokens AAVE ETH --pool-id 0xnotreal` will run reCLAMM training with +AAVE/ETH price feeds but with the median MM parameters and `K = $10M` +fallback documented in §6. + +`scripts/run_full_sweep.sh` hard-codes the 6 (token_a, token_b, pool_id, +gas, fees, tvl_label, initial_tvl) tuples it sweeps, which is where the +pool address is pinned per-job. + +### 1c. Competitor TVL (per-pair, per-chain) — DeFi Llama + +``` +python scripts/fetch_competitor_tvl.py +# optional: python scripts/fetch_competitor_tvl.py --cache-dir results/competitor_tvl +``` + +For each of the calibration pools, finds all other DEX pools trading the same +token pair and sums their daily TVL. Output: + +- `results/competitor_tvl/competitor_tvl.npz` — `(n_dates, n_pools)` array of + daily competitor TVL in USD, used as `K_i(t) = sum_{j ≠ i} TVL_j(t)` in the + MM noise model. +- `results/competitor_tvl/___history.pkl` — per-pair cache + for re-runs. + +Optional token-mcaps cache (used by some plotting / classification code): + +``` +python scripts/fetch_token_mcaps.py +``` + +writes `local_data/noise_calibration/token_mcaps.json`. + +--- + +## 2. Build the per-pool arb-volume grids + +``` +python scripts/build_pool_grids.py --workers 6 --train-days 90 +``` + +Sweeps `(cadence, gas)` per real Balancer pool to produce PCHIP grids of daily +arb volume. Output: `results/pool_grids_v2/_daily.parquet` per +pool, plus a summary CSV. The grids are what the joint calibration model +interpolates over to attribute total observed volume to arb vs noise. + +This step is slow; it forward-simulates each pool across `cadence × gas` for +the chosen training window. Use a non-default `--workers` to parallelise. + +--- + +## 3. Train the Michaelis-Menten noise model + +Fits the per-pool MM model that the simulator uses to predict noise volume at +run-time: + +``` +log(V_noise) = log_alpha_i + x_market @ gamma + log(TVL) − log(K_i + TVL) +V_total = V_arb(cadence_i) + exp(log_V_noise) +Loss = Huber(log(V_total) − log(V_obs)) +``` + +``` +python scripts/run_mm_noise.py \ + --per-pool-gamma --epochs 5000 --lr 1e-4 \ + --huber-delta 0.5 --observed-K --no-split +``` + +The hparams the existing artifact was fit with are stored in +`results/mm_noise/meta.json` under `hparams`; reproduce by matching them on +the command line. + +Loads the panel from `local_data/noise_calibration/panel.parquet`, matches +each row to the right per-day grid in `results/pool_grids_v2/`, and fits the +MM model jointly across all pools. Output: + +- `results/mm_noise/model.npz` — fitted parameters (per-pool `log_alpha`, + `log_K`, `log_cadence`, shared `gamma`). +- `results/mm_noise/meta.json` — `pool_ids`, `n_market_feat`, + `per_pool_gamma`, `hparams`. +- `results/mm_noise/trials/trial_NNNN/` — checkpoint per trial during a + hyperparameter sweep. + +`meta.json` has the canonical schema: + +```json +{ + "model": "michaelis_menten", + "pool_ids": ["0x9d1fcf346ea1b0", "0xd321300ef77067", ...], + "n_market_feat": 18, + "per_pool_gamma": true, + "hparams": {"epochs": 5000, "lr": 1e-4, "l2_alpha": 1e-3, + "huber_delta": 0.5, "init_log_K": 17.0} +} +``` + +Diagnostic plot: + +``` +python scripts/plot_mm_noise_fit.py +``` + +writes per-pool fit overlays to `results/mm_noise/plots/`. + +--- + +## 4. Training sweep + +`scripts/run_full_sweep.sh` is the orchestrator; it shells out to +`scripts/tune_reclamm_calibrated_noise.py` per job. Each (pair, TVL tier, +objective, penalty) combination is one job. With the default config there are +72 jobs per method: + +- 2 pairs (AAVE/ETH, COW/ETH) × 3 TVL tiers = 6 configs +- 4 objectives: `returns_over_hodl`, `fee_revenue_over_value`, `calmar`, + `daily_log_sharpe_excess` +- 3 penalty variants: no penalty, `--overfitting-penalty 1.0`, `5.0` + +``` +# Optuna (default, 300 trials per job) +MAX_WORKERS=6 bash scripts/run_full_sweep.sh + +# CMA-ES (500 generations per job) +MAX_WORKERS=6 bash scripts/run_full_sweep.sh --method cma_es + +# Single pair only +bash scripts/run_full_sweep.sh --pair aave +bash scripts/run_full_sweep.sh --pair cow +``` + +Each job writes two files: + +- `results/full_sweep/.json` — per-job summary with chosen best params +- `results/run_.json` — full trial trajectory used by the + selection step. Hash is computed from the run_fingerprint, so changes to + any fingerprint field (method, penalty, TVL, objective, token pair, …) + produce a fresh file. + +Per-job stdout / stderr go to `/tmp/tune_full_.log` — useful for spot +checks while a sweep runs. + +`MAX_WORKERS` defaults to 8 for optuna and 4 for CMA-ES (CMA-ES uses more RAM). +Override either with the env var. + +### What each method optimises + +| Method | Optimiser | Per-job time (300 trials / 500 gens) | Notes | +|---|---|---|---| +| optuna | TPE sampler, median pruner | ~20–40 min | Single-objective; multi-objective optional via `--multi-objective`. | +| cma_es | CMA-ES with box constraints from `parameter_config` | ~30–60 min | One restart per `n_parameter_sets`. | +| bfgs | `jax.scipy.optimize.minimize(method="BFGS")` | depends on `bfgs_maxiter` | Unconstrained; needs sp_/logit_ reparametrisation to stay in bounds. | + +--- + +## 5. Selection, final sims, and plots + +### 5a. Pick the best params per (pair, TVL tier) and run train+test forward sims + +``` +python scripts/run_final_sims.py --all +# or +python scripts/run_final_sims.py --pair aave +python scripts/run_final_sims.py --pair cow +``` + +`run_final_sims.py` loads every `results/run_*.json`, filters to its +hardcoded train period (see `TRAIN_START` / `TRAIN_END` at the top of the +script), ranks candidates by val-RoH pulled from +`continuous_test_metrics["returns_over_hodl"]`, and picks the best for each +`(token, TVL)` combination. Then runs two independent forward passes (one +each for the train and test windows) at the picked params, and writes: + +- `results/final_sims/_sim_results.pkl` — the full per-tier results (large; ~200 MB per pair). +- `results/final_sims/_{train,test}.png` — share-price / fee revenue / cumulative volume per TVL tier. +- `results/final_sims/_{train,test}_weights.png` — effective weight trajectories. + +The printed `Best:` / `Params:` lines are the per-tier selections — that's +where `(PR, margin, shift)` for the next step come from. + +### 5b. PR-sweep heatmaps (PR × period grid) + +Once a tier's `(margin, shift, initial_pool_value)` is chosen, sweep PR across +a multi-period window: + +``` +python scripts/run_pr_sweep.py --tokens COW ETH \ + --pool-id 0xd321300ef77067 --gas-cost 3.0 --fees 0.003 \ + --noise-model mm_observed \ + --multi-period --period-months 3 \ + --onchain-pr 2.02 \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 5.0 10.0 \ + 30.0 50.0 80.0 100.0 120.0 150.0 200.0 \ + --output-dir results/final_sims/for_fabio/cow/price_ratio_sweep \ + --margin 0.8235 --shift 0.1418 --initial-pool-value 500000 \ + --selected-pr 99.665 +``` + +The `--prs` list should bracket the selected PR for the tier. The +`run_final_sims.py` output tells you the selected PR; extend the upper bound +of `--prs` past it so the heatmap shows the surrounding landscape. Output: +`pr_heatmap___m_s.png` in the output dir. + +The script tries each period from `2024-01-01` onwards in 1-month steps; +periods that start before the pool's data is available raise inside +`run_single_period`, the script catches the exception, and the final heatmap +only contains the surviving rows. + +### 5c. Other diagnostic plots + +| Script | What it produces | +|---|---| +| `scripts/plot_reclamm_optuna_result.py` | per-config train/test panels from a single sweep result | +| `scripts/plot_mm_noise_fit.py` | per-pool MM-model fit overlay | +| `scripts/plot_calibrated_vs_real.py` | total volume: modelled vs observed | +| `scripts/compare_modelled_vs_real.py` | noise volume of model pool A vs real volume of pool B | +| `scripts/select_best_params.py` | ranked summary of sweep results, optional `--export` of best params | + +--- + +## 6. Extending to a new pair + +The frozen artifacts in `results/mm_noise/` and `results/competitor_tvl/` +cover the set of pools that survived calibration filtering at training time +(see §1b). The list lives in `results/mm_noise/meta.json` under `pool_ids`, +and the same set is the column index of `results/competitor_tvl/competitor_tvl.npz`. + +If the pair you want to train on is in that set, training picks up the +per-pool `log_alpha` / `gamma` / `log_K` automatically. + +If the pair is not in the set, `build_mm_simulator_arrays` falls back: + +- MM noise level → cross-pool median `log_alpha` and `gamma`. Logs + `MM model: pool not found, using median alpha=...`. +- Competitor TVL → constant `K = $10M`. Logs + `WARNING: pool not in competitor TVL data, using K=$10M`. + +The simulator still runs, with cross-pool averages replacing the pool-specific +saturation curve and intercept. For pairs close to the calibration set this is +a reasonable approximation; for very small / illiquid / very large pairs the +median is likely well off and the constant `K` flattens the TVL→noise response. + +To get proper per-pool parameters for a new pair: + +``` +# 1. Make sure the pool is in the panel +python -m quantammsim.noise_calibration --fetch --chain ethereum + +# 2. Refresh the competitor-TVL bundle (DeFi Llama) +python scripts/fetch_competitor_tvl.py + +# 3. Re-train the MM model so the new pool gets its own entry +python scripts/run_mm_noise.py \ + --per-pool-gamma --epochs 5000 --lr 1e-4 \ + --huber-delta 0.5 --observed-K --no-split +``` + +Step 1 requires the pool to have TVL above `--min-tvl` on the Balancer API. +Step 2 caches a fresh `___history.pkl` and rebuilds +`competitor_tvl.npz`. Step 3 produces fresh `model.npz` + `meta.json` whose +`pool_ids` list now includes the new pool. Existing reCLAMM training results +for other pairs are unaffected (they're keyed off the run_fingerprint hash; +re-running step 3 doesn't change those hashes). + +After this, any `run_full_sweep.sh` / `run_final_sims.py` / `run_pr_sweep.py` +call against the new pair reads its per-pool MM parameters and per-day +competitor TVL series instead of falling back to medians. + +--- + +## Caveats / things to watch + +- **The `results/run_*.json` cache is fingerprint-keyed**. Changing any field + in `run_fingerprint` (method, penalty, TVL, ste_temperature, …) produces a + fresh hash. Identical fingerprints will reload from cache instead of + re-training. Delete the matching `run_.json` to force a fresh run. + +- **CMA-ES penalty** factors into the run_fingerprint. If you have older + `run_*.json` files from a version where the penalty was not in the + fingerprint, they share a hash across penalty variants and should be + deleted before a fresh sweep, otherwise the candidate pool mixes + configurations. + +- **Stale data ranges**. `run_pr_sweep.py --multi-period` starts at + `2024-01-01`; pools that didn't exist that early will fail the early + periods. The script catches the exception and produces a heatmap of only + the surviving rows. + +- **`MAX_WORKERS=N`** in `run_full_sweep.sh` controls shell parallelism only — + each job is a separate Python process. Memory is the binding constraint; + 4–6 workers is typical for a laptop. + +- **PR axis on heatmaps** is what you pass via `--prs`. The default upper + bound is 10; pass higher values when the selected PR for the pair exceeds + that. + +--- + +## Quick re-run shortcut + +If the data + MM model + grids are already on disk and you just want fresh +trained params and plots: + +``` +# (1) sweep +MAX_WORKERS=6 bash scripts/run_full_sweep.sh +MAX_WORKERS=6 bash scripts/run_full_sweep.sh --method cma_es + +# (2) select + final sims +python scripts/run_final_sims.py --all + +# (3) PR heatmaps per pair × tier — adjust margin/shift/TVL/selected-PR +# from the run_final_sims.py output +python scripts/run_pr_sweep.py --tokens AAVE ETH ... +python scripts/run_pr_sweep.py --tokens COW ETH ... +``` + +The whole chain from a clean cache is ~half a day on 6 workers for AAVE+COW +at the canonical 6-tier sweep. diff --git a/experiments/fetch_competitor_tvl.py b/scripts/fetch_competitor_tvl.py similarity index 99% rename from experiments/fetch_competitor_tvl.py rename to scripts/fetch_competitor_tvl.py index 7ad3cb3b..5f003932 100644 --- a/experiments/fetch_competitor_tvl.py +++ b/scripts/fetch_competitor_tvl.py @@ -11,8 +11,8 @@ - competitor_tvl: (n_dates, n_pools) array of daily competitor TVL in USD Usage: - python experiments/fetch_competitor_tvl.py - python experiments/fetch_competitor_tvl.py --cache-dir results/competitor_tvl + python scripts/fetch_competitor_tvl.py + python scripts/fetch_competitor_tvl.py --cache-dir results/competitor_tvl """ import argparse diff --git a/scripts/run_full_sweep.sh b/scripts/run_full_sweep.sh index 47497007..5b11353a 100755 --- a/scripts/run_full_sweep.sh +++ b/scripts/run_full_sweep.sh @@ -57,7 +57,7 @@ mkdir -p "$OUTDIR" # Build COMMON command based on method if [ "$METHOD" = "cma_es" ]; then CMA_GENS="${CMA_GENERATIONS:-500}" - COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --method cma_es --cma-generations $CMA_GENS" + COMMON="python scripts/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --method cma_es --cma-generations $CMA_GENS" METHOD_TAG="_cmaes" echo "=== CMA-ES mode ($CMA_GENS generations) ===" else @@ -68,7 +68,7 @@ else else METHOD_TAG="" fi - COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS $PR_MAX_FLAG" + COMMON="python scripts/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS $PR_MAX_FLAG" echo "=== Optuna mode ($TRIALS trials${PR_MAX:+, PR max=$PR_MAX}) ===" fi diff --git a/experiments/run_mm_noise.py b/scripts/run_mm_noise.py similarity index 99% rename from experiments/run_mm_noise.py rename to scripts/run_mm_noise.py index 45323a67..7153635d 100644 --- a/experiments/run_mm_noise.py +++ b/scripts/run_mm_noise.py @@ -20,9 +20,9 @@ log_cadence_i: per-pool arb frequency (via PCHIP) Usage: - python experiments/run_mm_noise.py - python experiments/run_mm_noise.py --epochs 5000 --lr 3e-4 - python experiments/run_mm_noise.py --per-pool-gamma # per-pool market coeffs + python scripts/run_mm_noise.py + python scripts/run_mm_noise.py --epochs 5000 --lr 3e-4 + python scripts/run_mm_noise.py --per-pool-gamma # per-pool market coeffs """ import argparse diff --git a/scripts/run_period_sweep.sh b/scripts/run_period_sweep.sh index 8233b1ad..0cfdfa78 100644 --- a/scripts/run_period_sweep.sh +++ b/scripts/run_period_sweep.sh @@ -9,7 +9,7 @@ source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public TRIALS=400 MAX_PARALLEL=8 -COMMON="python experiments/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS" +COMMON="python scripts/tune_reclamm_calibrated_noise.py --noise-model mm_observed --artifact-dir results/mm_noise --n-trials $TRIALS" OBJECTIVES=( daily_log_sharpe diff --git a/experiments/tune_reclamm_calibrated_noise.py b/scripts/tune_reclamm_calibrated_noise.py similarity index 98% rename from experiments/tune_reclamm_calibrated_noise.py rename to scripts/tune_reclamm_calibrated_noise.py index 2c9946be..e0d6ed9b 100644 --- a/experiments/tune_reclamm_calibrated_noise.py +++ b/scripts/tune_reclamm_calibrated_noise.py @@ -22,16 +22,16 @@ source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public # AAVE/ETH with market_linear noise (default) - python experiments/tune_reclamm_calibrated_noise.py + python scripts/tune_reclamm_calibrated_noise.py # COW/ETH with no noise model - python experiments/tune_reclamm_calibrated_noise.py --tokens COW ETH --noise-model none + python scripts/tune_reclamm_calibrated_noise.py --tokens COW ETH --noise-model none # All objectives - python experiments/tune_reclamm_calibrated_noise.py --all-objectives + python scripts/tune_reclamm_calibrated_noise.py --all-objectives # More trials - python experiments/tune_reclamm_calibrated_noise.py --n-trials 200 + python scripts/tune_reclamm_calibrated_noise.py --n-trials 200 """ import argparse From 1d7228ef9063ed86cecf67ea32b053b2dfd1e29b Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 15 May 2026 15:12:18 +0100 Subject: [PATCH 104/115] docs: use canonical qsim env name in training guide --- RECLAMM_TRAINING.md | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md index 7d8db964..5723d4c5 100644 --- a/RECLAMM_TRAINING.md +++ b/RECLAMM_TRAINING.md @@ -3,12 +3,11 @@ Soup-to-nuts guide for going from no data on disk to a set of trained reCLAMM params and the heatmap / weight / fee-revenue plots used in reports. -All commands assume the working directory is the repo root and the conda env -`qsim_reclamm_public` is active: +All commands assume the working directory is the repo root and the canonical +conda env from the README (`qsim`) is active: ``` -cd /Users/matthew/Projects/quantammsim-reclamm-public/quantammsim -source ~/miniconda3/etc/profile.d/conda.sh && conda activate qsim_reclamm_public +conda activate qsim ``` The pipeline has five stages. Outputs of each stage feed the next, so order From 6a857df023b03dafd812b1c47df6e925e8b63ccc Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 15 May 2026 15:14:19 +0100 Subject: [PATCH 105/115] docs: cover extending run_full_sweep.sh CONFIGS for a new pair --- RECLAMM_TRAINING.md | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md index 5723d4c5..d73db624 100644 --- a/RECLAMM_TRAINING.md +++ b/RECLAMM_TRAINING.md @@ -376,6 +376,28 @@ After this, any `run_full_sweep.sh` / `run_final_sims.py` / `run_pr_sweep.py` call against the new pair reads its per-pool MM parameters and per-day competitor TVL series instead of falling back to medians. +### Adding the new pair to `run_full_sweep.sh` + +The sweep orchestrator iterates over a hard-coded `CONFIGS` array near the +top of `scripts/run_full_sweep.sh`. Each row is a whitespace-separated tuple: + +``` +"token_a token_b pool_id gas_cost fees tvl_label initial_tvl" +``` + +For example, the existing AAVE/ETH and COW/ETH rows: + +``` +"AAVE ETH 0x9d1fcf346ea1b0 1.0 0.0025 aave_1m 1000000" +"COW ETH 0xd321300ef77067 3.0 0.003 cow_500k 500000" +``` + +To sweep a new pair, append one row per TVL tier with the pool's address +prefix, an appropriate gas cost (chain dependent) and fees, a unique +`tvl_label`, and the starting pool value. The same change in +`scripts/run_final_sims.py` (`PAIR_CONFIGS`) makes the new pair available to +the final-sims selector. `--pair ` filters to a single pair when needed. + --- ## Caveats / things to watch From 6d494d8cda0bc7818f7ebcf0a7ed9a579171f02b Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Fri, 15 May 2026 15:23:08 +0100 Subject: [PATCH 106/115] docs: add smoke test, objective/penalty definitions, cache and CPU notes --- RECLAMM_TRAINING.md | 84 ++++++++++++++++++++++++++++++++++++++++++--- 1 file changed, 80 insertions(+), 4 deletions(-) diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md index d73db624..895cb27c 100644 --- a/RECLAMM_TRAINING.md +++ b/RECLAMM_TRAINING.md @@ -215,6 +215,33 @@ writes per-pool fit overlays to `results/mm_noise/plots/`. --- +## 3b. Smoke test before sweeping + +Before kicking off a multi-hour sweep, run a tiny single-config job to verify +the data, the MM artifact, and the competitor-TVL bundle are all in place and +the simulator returns sensible numbers: + +``` +python scripts/tune_reclamm_calibrated_noise.py \ + --noise-model mm_observed --artifact-dir results/mm_noise \ + --n-trials 5 \ + --tokens AAVE ETH --pool-id 0x9d1fcf346ea1b0 \ + --gas-cost 1.0 --fees 0.0025 \ + --initial-pool-value 1000000 \ + --objective calmar \ + --start-date "2025-01-01 00:00:00" --end-date "2025-10-05 00:00:00" \ + --output /tmp/smoke.json +``` + +Completes in a couple of minutes. A non-empty `/tmp/smoke.json` and a new +`results/run_.json` with five trial entries means the pipeline is wired +correctly end-to-end. `scripts/demo_run_reclamm.py` is a complementary smoke +test that runs reCLAMM and Balancer-50/50 forward passes side-by-side without +training, useful for sanity-checking the simulator independently of any +optimiser. + +--- + ## 4. Training sweep `scripts/run_full_sweep.sh` is the orchestrator; it shells out to @@ -261,6 +288,41 @@ Override either with the env var. | cma_es | CMA-ES with box constraints from `parameter_config` | ~30–60 min | One restart per `n_parameter_sets`. | | bfgs | `jax.scipy.optimize.minimize(method="BFGS")` | depends on `bfgs_maxiter` | Unconstrained; needs sp_/logit_ reparametrisation to stay in bounds. | +### What each objective measures + +`--objective` picks which scalar the optimiser maximises: + +- `returns_over_hodl` — final pool value minus the HODL counterfactual, + divided by initial. Direct profitability against holding the constituent + tokens. +- `fee_revenue_over_value` — cumulative LP fee revenue as a fraction of + initial pool value. +- `calmar` — annualised return divided by max drawdown. +- `daily_log_sharpe_excess` — pool's daily log-Sharpe minus the HODL daily + log-Sharpe; risk-adjusted excess return vs. holding. + +### Overfitting penalty + +`--overfitting-penalty α` adds a generalisation penalty. Optuna and CMA-ES +both reserve `val_fraction` of the training window (default 0.2; see +`default_run_fingerprint.py`) as a held-out validation window. The +optimised objective becomes: + +``` +penalised = mean_train_obj − α · max(0, mean_train_obj − mean_val_obj) +``` + +When train looks better than val, the gap is subtracted. `α = 0` recovers +the raw training objective. `run_full_sweep.sh` sweeps three variants per +(pair, TVL, objective): `α = 0`, `α = 1.0`, `α = 5.0`. + +### Single-process alternative + +`scripts/tune_reclamm_calibrated_noise.py --all-objectives` runs the four +objectives sequentially in a single Python process, without the +penalty-variant axis. Useful for quick single-machine exploration without +spawning the orchestrator's parallel processes. + --- ## 5. Selection, final sims, and plots @@ -402,10 +464,17 @@ the final-sims selector. `--pair ` filters to a single pair when needed. ## Caveats / things to watch -- **The `results/run_*.json` cache is fingerprint-keyed**. Changing any field - in `run_fingerprint` (method, penalty, TVL, ste_temperature, …) produces a - fresh hash. Identical fingerprints will reload from cache instead of - re-training. Delete the matching `run_.json` to force a fresh run. +- **The `results/run_*.json` cache is fingerprint-keyed**. The hash is + `SHA256(json.dumps(run_fingerprint, sort_keys=True))`, computed in + `quantammsim/core_simulator/result_exporter.py:get_run_location`. Every + field of `run_fingerprint` participates: token pair, pool_id, TVL, fees, + gas_cost, objective, method-specific settings (`n_trials`, `n_generations`, + `overfitting_penalty`, `val_fraction`), noise-model config, and simulator + config (`arb_frequency`, `ste_temperature`, `max_memory_days`, …). + Identical fingerprints reload from cache instead of re-training; delete + the matching `run_.json` to force a fresh run. To map a hash back + to its fingerprint, read the first entry of the JSON: + `json.loads(json.load(open(path)))[0]`. - **CMA-ES penalty** factors into the run_fingerprint. If you have older `run_*.json` files from a version where the penalty was not in the @@ -426,6 +495,13 @@ the final-sims selector. `--pair ` filters to a single pair when needed. bound is 10; pass higher values when the selected PR for the pair exceeds that. +- **CPU vs GPU**. The simulator runs on whichever device JAX picks. + `scripts/build_pool_grids.py` forces `JAX_PLATFORMS=cpu` so the grid build + is deterministic. The update-rule estimator backend (scan vs conv/FFT) is + selected by `DEFAULT_BACKEND` in + `quantammsim/pools/G3M/quantamm/update_rule_estimators/estimators.py` and + defaults to scan on CPU; it is not exposed as a CLI flag. + --- ## Quick re-run shortcut From c71eec064ea30d06c03c8dffd581f832fe20834c Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 15 May 2026 17:19:50 +0100 Subject: [PATCH 107/115] fixes --- RECLAMM_TRAINING.md | 2 + quantammsim/noise_calibration/__main__.py | 5 + quantammsim/noise_calibration/cli.py | 106 ++++++++++++++-- .../data_processing/historic_data_utils.py | 119 ++++++++++++++++-- scripts/replace_usdt_with_usdc_data.py | 74 +++++++++++ 5 files changed, 286 insertions(+), 20 deletions(-) create mode 100644 quantammsim/noise_calibration/__main__.py create mode 100644 scripts/replace_usdt_with_usdc_data.py diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md index 895cb27c..f9303fe1 100644 --- a/RECLAMM_TRAINING.md +++ b/RECLAMM_TRAINING.md @@ -64,6 +64,8 @@ python scripts/download_data.py BTC ETH AAVE COW USDC USDT WBTC Reads from `scripts/ticker_list.txt` if no tickers are passed. Output: `quantammsim/data/_USD.parquet` (minute resolution) and `_USD_daily.csv`. Used by all simulator runs and noise calibration. +`USDT` is fetched against `USD` when needed, since `USDT/USDT` is not a real +market. ### 1b. Balancer pool snapshots (volume + TVL) diff --git a/quantammsim/noise_calibration/__main__.py b/quantammsim/noise_calibration/__main__.py new file mode 100644 index 00000000..2f05ddc2 --- /dev/null +++ b/quantammsim/noise_calibration/__main__.py @@ -0,0 +1,5 @@ +from .cli import main + + +if __name__ == "__main__": + main() diff --git a/quantammsim/noise_calibration/cli.py b/quantammsim/noise_calibration/cli.py index 3e169ac3..e7f0233a 100644 --- a/quantammsim/noise_calibration/cli.py +++ b/quantammsim/noise_calibration/cli.py @@ -9,7 +9,7 @@ import numpy as np import pandas as pd -from .constants import CACHE_DIR +from .constants import CACHE_DIR, BALANCER_API_CHAINS from .data_pipeline import ( enumerate_balancer_pools, fetch_all_snapshots, fetch_token_prices, assemble_panel, @@ -25,6 +25,60 @@ from .output import generate_output_json, _save_sample_cache +CHAIN_NAME_ALIASES = { + "ethereum": "MAINNET", + "mainnet": "MAINNET", + "base": "BASE", + "gnosis": "GNOSIS", + "polygon": "POLYGON", + "arbitrum": "ARBITRUM", + "optimism": "OPTIMISM", + "avalanche": "AVALANCHE", + "sonic": "SONIC", +} + + +def normalize_chain_name(chain: str | None) -> str | None: + if chain is None: + return None + + normalized = chain.strip() + if not normalized: + return None + + upper = normalized.upper() + if upper in BALANCER_API_CHAINS: + return upper + + aliased = CHAIN_NAME_ALIASES.get(normalized.lower()) + if aliased is not None: + return aliased + + accepted = sorted(set(BALANCER_API_CHAINS) | set(CHAIN_NAME_ALIASES)) + raise ValueError( + f"Unknown chain '{chain}'. Expected one of: {', '.join(accepted)}" + ) + + +def merge_chain_scoped_cache( + existing_df: pd.DataFrame | None, + new_df: pd.DataFrame, + chains_to_replace: list[str], + sort_columns: list[str], +) -> pd.DataFrame: + if existing_df is None or existing_df.empty: + merged = new_df.copy() + else: + merged = existing_df[~existing_df["chain"].isin(chains_to_replace)].copy() + merged = pd.concat([merged, new_df], ignore_index=True) + + if sort_columns: + available_sort_columns = [col for col in sort_columns if col in merged.columns] + if available_sort_columns: + merged = merged.sort_values(available_sort_columns).reset_index(drop=True) + return merged + + def _parse_args(): parser = argparse.ArgumentParser( description="Unified Bayesian hierarchical noise volume model " @@ -56,8 +110,12 @@ def _parse_args(): help="Plot output directory (default: results)") # Predict args - parser.add_argument("--chain", default=None, - help="Chain for --predict") + parser.add_argument( + "--chain", + default=None, + help="Chain for --predict or to limit --fetch " + "(e.g. ethereum/base/gnosis or MAINNET/BASE/GNOSIS)", + ) parser.add_argument("--tokens", nargs="+", default=None, help="Tokens for --predict") parser.add_argument("--fee", type=float, default=0.003, @@ -123,6 +181,12 @@ def main(): "is required", file=sys.stderr) sys.exit(1) + try: + normalized_chain = normalize_chain_name(args.chain) + except ValueError as exc: + print(f"ERROR: {exc}", file=sys.stderr) + sys.exit(2) + cache_dir = args.cache_dir or CACHE_DIR # --- JAX setup (BEFORE any JAX ops / imports) --- @@ -157,17 +221,33 @@ def main(): print("=" * 60) print("\n1. Enumerating pools...") - pools_df = enumerate_balancer_pools(min_tvl=args.min_tvl) + fetch_chains = [normalized_chain] if normalized_chain else None + if fetch_chains: + print(f" Limiting fetch to chain: {fetch_chains[0]}") + fetched_pools_df = enumerate_balancer_pools( + chains=fetch_chains, + min_tvl=args.min_tvl, + ) + if fetch_chains and os.path.exists(pools_cache): + existing_pools_df = pd.read_parquet(pools_cache) + pools_df = merge_chain_scoped_cache( + existing_pools_df, + fetched_pools_df, + fetch_chains, + sort_columns=["chain", "pool_id"], + ) + else: + pools_df = fetched_pools_df os.makedirs(cache_dir, exist_ok=True) pools_df.to_parquet(pools_cache, index=False) print(f" Saved {len(pools_df)} pools -> {pools_cache}") print("\n2. Fetching daily snapshots...") - snapshots_df = fetch_all_snapshots(pools_df, cache_path=snaps_cache) + snapshots_df = fetch_all_snapshots(fetched_pools_df, cache_path=snaps_cache) print("\n3. Fetching token prices...") token_addr_by_chain = {} - for _, pool in pools_df.iterrows(): + for _, pool in fetched_pools_df.iterrows(): chain = pool["chain"] tokens = pool["tokens"] addresses = pool["token_addresses"] @@ -182,7 +262,17 @@ def main(): ) print("\n4. Assembling panel (with lagged TVL)...") - panel = assemble_panel(pools_df, snapshots_df, token_prices) + fetched_panel = assemble_panel(fetched_pools_df, snapshots_df, token_prices) + if fetch_chains and os.path.exists(panel_cache): + existing_panel = pd.read_parquet(panel_cache) + panel = merge_chain_scoped_cache( + existing_panel, + fetched_panel, + fetch_chains, + sort_columns=["chain", "pool_id", "date"], + ) + else: + panel = fetched_panel panel.to_parquet(panel_cache, index=False) print(f" Saved panel -> {panel_cache}") @@ -412,6 +502,6 @@ def main(): data_meta = json.load(f) result = predict_new_pool( - sample_dict, data_meta, args.chain, args.tokens, args.fee + sample_dict, data_meta, normalized_chain, args.tokens, args.fee ) print(json.dumps(result, indent=2)) diff --git a/quantammsim/utils/data_processing/historic_data_utils.py b/quantammsim/utils/data_processing/historic_data_utils.py index b83a97b7..3010bf09 100644 --- a/quantammsim/utils/data_processing/historic_data_utils.py +++ b/quantammsim/utils/data_processing/historic_data_utils.py @@ -726,6 +726,71 @@ def get_binance_vision_data(token, numeraire, root): return result_df +def get_quote_currency_candidates(token): + """Return quote currencies to try when bootstrapping minute data.""" + if token == "USDT": + return ["USD", "USDC"] + return ["USDT"] + + +def get_coinbase_live_data(token, numeraire): + """Download minute OHLCV data from Coinbase and standardize the schema.""" + market = f"{token}-{numeraire}" + start_time = "2016-01-01-00-00" + end_time = datetime.utcnow().strftime("%Y-%m-%d-%H-%M") + + try: + retrieved_data = HistoricalData( + market, 60, start_time, end_time + ).retrieve_data() + except Exception as exc: + print(f"No Coinbase live data found for {market}: {exc}") + return None + + if retrieved_data is None or retrieved_data.empty: + print(f"No Coinbase live data returned for {market}") + return None + + standardized_df = retrieved_data.reset_index() + date_column = standardized_df.columns[0] + standardized_df = standardized_df.rename( + columns={ + date_column: "date", + "volume": f"Volume {token}", + } + ) + standardized_df["date"] = pd.to_datetime( + standardized_df["date"], utc=True + ).dt.tz_localize(None) + standardized_df["unix"] = pddatetime_to_unixtimestamp(standardized_df["date"]) + standardized_df["symbol"] = f"{token}/{numeraire}" + standardized_df["Volume USD"] = ( + standardized_df[f"Volume {token}"] * standardized_df["close"] + ) + + standardized_df = standardized_df[ + [ + "unix", + "date", + "symbol", + "open", + "high", + "low", + "close", + "Volume USD", + f"Volume {token}", + ] + ] + standardized_df = standardized_df.sort_values("unix") + standardized_df = standardized_df.drop_duplicates( + subset="unix", keep="last" + ).reset_index(drop=True) + standardized_df["date"] = standardized_df["date"].dt.strftime( + "%Y-%m-%d %H:%M:%S" + ) + return standardized_df + + def update_historic_data(token, root): """Update historic data for a given token, handling reruns gracefully. @@ -743,18 +808,42 @@ def update_historic_data(token, root): os.makedirs(outputPath, exist_ok=True) os.makedirs(root + "concat_binance_data/", exist_ok=True) - # Try binance.vision data first - print(f"Attempting to get Binance vision data for {token}") filled_timestamps = {} - concated_df = get_binance_vision_data(token, "USDT", root) - if concated_df is not None: - print(f"Binance vision data available for {token}") + quote_currency_candidates = get_quote_currency_candidates(token) + primary_numeraire = quote_currency_candidates[0] + concated_df = None + filled_timestamps["Binance Vision"] = [] + + # Try binance.vision data first, using a USD quote for assets like USDT. + for numeraire in quote_currency_candidates: + print(f"Attempting to get Binance vision data for {token} quoted in {numeraire}") + candidate_df = get_binance_vision_data(token, numeraire, root) + if candidate_df is None: + continue + concated_df = candidate_df + primary_numeraire = numeraire + print(f"Binance vision data available for {token} quoted in {numeraire}") if concated_df.index.name != "unix": concated_df.set_index("unix", inplace=True) - filled_binance_vision_unix_values = concated_df.index.tolist() - filled_timestamps["Binance Vision"] = filled_binance_vision_unix_values - else: - print(f"No Binance vision data available for {token}") + filled_timestamps["Binance Vision"] = concated_df.index.tolist() + break + + if concated_df is None: + for numeraire in quote_currency_candidates: + print(f"Attempting to get Coinbase live data for {token} quoted in {numeraire}") + candidate_df = get_coinbase_live_data(token, numeraire) + if candidate_df is None: + continue + concated_df = candidate_df + primary_numeraire = numeraire + print(f"Coinbase live data available for {token} quoted in {numeraire}") + if concated_df.index.name != "unix": + concated_df.set_index("unix", inplace=True) + filled_timestamps["Coinbase Live"] = concated_df.index.tolist() + break + + if concated_df is None: + print(f"No live market data available for {token}") concated_df = pd.DataFrame( columns=[ "date", @@ -767,11 +856,13 @@ def update_historic_data(token, root): f"Volume {token}", ] ) - filled_timestamps["Binance Vision"] = [] + concated_df.index.name = "unix" # Fill gaps with cryptodatadownload Binance data - print("Filling gaps with cryptodatadownload Binance data") + print( + f"Filling gaps with cryptodatadownload Binance data quoted in {primary_numeraire}" + ) concated_df, filled_binance_unix_values = fill_in_missing_rows_with_exchange_data( - concated_df, token, "USDT", root, "raw_binance_data/", "Binance_" + concated_df, token, primary_numeraire, root, "raw_binance_data/", "Binance_" ) filled_timestamps["Binance CDD"] = filled_binance_unix_values if concated_df is None: @@ -897,6 +988,10 @@ def update_historic_data(token, root): filled_timestamps["Aerodrome"] = filled_aerodrome_unix_values print("Filled aerodrome data") print(len(filled_aerodrome_unix_values)) + if concated_df.empty: + raise FileNotFoundError( + f"No market data found for {token} using quote currencies {quote_currency_candidates}" + ) # Ensure data is properly sorted and has no duplicates concated_df = concated_df.sort_index() concated_df = concated_df[~concated_df.index.duplicated(keep="first")] diff --git a/scripts/replace_usdt_with_usdc_data.py b/scripts/replace_usdt_with_usdc_data.py new file mode 100644 index 00000000..2eaa185a --- /dev/null +++ b/scripts/replace_usdt_with_usdc_data.py @@ -0,0 +1,74 @@ +import argparse +from pathlib import Path + +import pandas as pd + + +FILE_PAIRS = [ + ("USDC_USD.parquet", "USDT_USD.parquet"), + ("USDC_USD_daily.csv", "USDT_USD_daily.csv"), + ("combined_data/USDC_USD_hourly.csv", "combined_data/USDT_USD_hourly.csv"), +] + + +def rewrite_usdc_frame_as_usdt(df: pd.DataFrame) -> pd.DataFrame: + rewritten = df.copy() + + if "Volume USDC" in rewritten.columns: + rewritten = rewritten.rename(columns={"Volume USDC": "Volume USDT"}) + + if "symbol" in rewritten.columns: + rewritten["symbol"] = "USDT/USD" + + return rewritten + + +def process_file(data_dir: Path, source_rel: str, target_rel: str, dry_run: bool) -> None: + source_path = data_dir / source_rel + target_path = data_dir / target_rel + + if not source_path.exists(): + raise FileNotFoundError(f"Missing source file: {source_path}") + + if source_path.suffix == ".parquet": + df = pd.read_parquet(source_path) + else: + df = pd.read_csv(source_path) + + rewritten = rewrite_usdc_frame_as_usdt(df) + + print(f"{source_path} -> {target_path} ({len(rewritten)} rows)") + + if dry_run: + return + + target_path.parent.mkdir(parents=True, exist_ok=True) + if target_path.suffix == ".parquet": + rewritten.to_parquet(target_path, index=False) + else: + rewritten.to_csv(target_path, index=False) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Overwrite USDT data files with USDC data rewritten as USDT/USD." + ) + parser.add_argument( + "--data-dir", + default=Path(__file__).resolve().parent.parent / "quantammsim" / "data", + type=Path, + help="Base data directory containing the USDC/USDT files.", + ) + parser.add_argument( + "--dry-run", + action="store_true", + help="Show the files that would be rewritten without modifying them.", + ) + args = parser.parse_args() + + for source_rel, target_rel in FILE_PAIRS: + process_file(args.data_dir, source_rel, target_rel, args.dry_run) + + +if __name__ == "__main__": + main() From fc3b93c1d20dbd70b11803e56e2127206964b772 Mon Sep 17 00:00:00 2001 From: christian harrington Date: Fri, 15 May 2026 19:26:18 +0100 Subject: [PATCH 108/115] fixes --- RECLAMM_TRAINING.md | 16 +++++++++++++-- pyproject.toml | 1 + scripts/build_pool_grids.py | 9 +++++++-- scripts/run_token_factored_calibration.py | 24 +++++++++++++++++++++++ setup.py | 1 + 5 files changed, 47 insertions(+), 4 deletions(-) diff --git a/RECLAMM_TRAINING.md b/RECLAMM_TRAINING.md index f9303fe1..658e9256 100644 --- a/RECLAMM_TRAINING.md +++ b/RECLAMM_TRAINING.md @@ -92,7 +92,8 @@ been useful. The calibration set the MM model is trained on is the survivor set after downstream filters: pools with both tokens matched to Binance data and with enough clean daily snapshots. The survivors are cached in -`local_data/noise_calibration/_cache/stage1.pkl` (the `matched_clean` dict). +`results/token_factored_calibration/_cache/stage1.pkl` (the `matched_clean` +dict). `scripts/fetch_competitor_tvl.py` and `scripts/run_mm_noise.py` both read that file for their pool list and do not expose a per-pool selector. @@ -149,9 +150,11 @@ writes `local_data/noise_calibration/token_mcaps.json`. ## 2. Build the per-pool arb-volume grids ``` -python scripts/build_pool_grids.py --workers 6 --train-days 90 +python scripts/build_pool_grids.py --workers 1 --train-days 90 ``` +Increase the worker count based on the memory available on your machine. + Sweeps `(cadence, gas)` per real Balancer pool to produce PCHIP grids of daily arb volume. Output: `results/pool_grids_v2/_daily.parquet` per pool, plus a summary CSV. The grids are what the joint calibration model @@ -173,6 +176,15 @@ V_total = V_arb(cadence_i) + exp(log_V_noise) Loss = Huber(log(V_total) − log(V_obs)) ``` +First generate the shared stage-1 cache used by the competitor-TVL and MM +scripts: + +``` +python scripts/run_token_factored_calibration.py --stage1-only +``` + +This writes `results/token_factored_calibration/_cache/stage1.pkl`. + ``` python scripts/run_mm_noise.py \ --per-pool-gamma --epochs 5000 --lr 1e-4 \ diff --git a/pyproject.toml b/pyproject.toml index c6b25c6a..b5ce069c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -15,6 +15,7 @@ dependencies = [ "flask", "flask-jwt-extended", "scipy", + "scikit-learn", "seaborn", "cvxpy", "matplotlib", diff --git a/scripts/build_pool_grids.py b/scripts/build_pool_grids.py index ecf3152b..d056908b 100644 --- a/scripts/build_pool_grids.py +++ b/scripts/build_pool_grids.py @@ -34,6 +34,7 @@ import numpy as np import pandas as pd +from quantammsim.core_simulator.dynamic_inputs import DynamicInputFrames from quantammsim.runners.jax_runners import do_run_on_historic_data from quantammsim.utils.data_processing.historic_data_utils import get_historic_parquet_data @@ -365,9 +366,13 @@ def run_arb_sim(tokens, fee, initial_tvl, start, end, cadence, gas_cost, else: params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + dynamic_input_frames = ( + DynamicInputFrames(lp_supply=lp_supply_df) + if lp_supply_df is not None else None + ) result = do_run_on_historic_data( - fp, params, lp_supply_df=lp_supply_df, verbose=False, - price_data=price_data, preslice_burnin=False, + fp, params, verbose=False, price_data=price_data, + dynamic_input_frames=dynamic_input_frames, preslice_burnin=False, ) reserves = np.array(result["reserves"]) diff --git a/scripts/run_token_factored_calibration.py b/scripts/run_token_factored_calibration.py index 1f1cbf58..ad6d7b0c 100644 --- a/scripts/run_token_factored_calibration.py +++ b/scripts/run_token_factored_calibration.py @@ -65,6 +65,22 @@ def load_and_match(): matched = match_grids_to_panel(GRID_DIR, panel) print(f"Matched: {len(matched)} pools with grids") + if not matched: + daily_grids = [ + f for f in os.listdir(GRID_DIR) + if f.endswith("_daily.parquet") + ] if os.path.isdir(GRID_DIR) else [] + if not daily_grids: + raise RuntimeError( + "No daily pool grids found. Build them first with " + "`python scripts/build_pool_grids.py --workers 6 --train-days 90`; " + f"expected files like {GRID_DIR}/_daily.parquet." + ) + raise RuntimeError( + "No panel pools matched the available daily grid prefixes. Rebuild " + "the grids from the current panel with " + "`python scripts/build_pool_grids.py --workers 6 --train-days 90`." + ) return panel, matched @@ -691,6 +707,10 @@ def main(): "--cross-pool-only", action="store_true", help="Skip baseline ablation, load from cache, run only cross-pool", ) + parser.add_argument( + "--stage1-only", action="store_true", + help="Build stage1.pkl and exit before ablation fits", + ) args = parser.parse_args() os.environ.setdefault("JAX_PLATFORMS", "cpu") @@ -721,6 +741,10 @@ def main(): diag = run_phase0_diagnostic(matched_clean, option_c_clean) _save_stage1(matched_clean, option_c_clean, diag) + if args.stage1_only: + print("\nStage 1 cache generated; exiting before ablation fits.") + return + # ---- Ablation 1: Baseline ---- if args.cross_pool_only: cached_bl = _load_baseline() diff --git a/setup.py b/setup.py index 6134e36b..de29c34b 100644 --- a/setup.py +++ b/setup.py @@ -12,6 +12,7 @@ "flask", "flask-jwt-extended", "scipy", + "scikit-learn", "seaborn", "cvxpy", "matplotlib", From f0b2770537f3bb75a41921ca33d780d91ff9b729 Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Mon, 18 May 2026 16:42:22 +0100 Subject: [PATCH 109/115] data: add MM artifact, competitor TVL, and 6 winning sweep results Carries the minimum data to run scripts/run_final_sims.py and scripts/run_pr_sweep.py for AAVE/ETH and COW/ETH without re-running the training sweep. Binance price parquets are not included; pull them with scripts/download_data.py. - results/mm_noise/{model.npz,meta.json}: frozen MM noise model artifact - results/competitor_tvl/competitor_tvl.npz: DeFi Llama competitor TVL - results/full_sweep/*_penalty1.0_*.json: per-config sweep summaries for the 6 winning (pair, TVL) configurations - results/run_.json: trial trajectories for those 6 winners, read by load_reclamm_results in run_final_sims.py --- results/competitor_tvl/competitor_tvl.npz | Bin 0 -> 1326202 bytes ..._log_sharpe_excess_penalty1.0_aave_1m.json | 9 + ...log_sharpe_excess_penalty1.0_cow_500k.json | 9 + ...s_over_hodl_penalty1.0_cmaes_aave_20m.json | 8 + ...ns_over_hodl_penalty1.0_cmaes_aave_5m.json | 8 + .../returns_over_hodl_penalty1.0_cow_20m.json | 9 + .../returns_over_hodl_penalty1.0_cow_2m.json | 9 + results/mm_noise/meta.json | 227 ++++++++++++++++++ results/mm_noise/model.npz | Bin 0 -> 7534 bytes ...d7757eb7b8267493a30b2f12a3dbd83e85922.json | 1 + ...e15f90d406c94406139d1ea9af59decc3c4d2.json | 1 + ...04bf44992034ba79b905bf4b1ac4e67ac3155.json | 1 + ...2bed8255ea3f608707054ffc306e77e008d76.json | 1 + ...78a5629d0cbdc9df7f15b86448fcb516266fb.json | 1 + ...b4960d250393ba49dbe0637b1bbe7eb7ef348.json | 1 + 15 files changed, 285 insertions(+) create mode 100644 results/competitor_tvl/competitor_tvl.npz create mode 100644 results/full_sweep/daily_log_sharpe_excess_penalty1.0_aave_1m.json create mode 100644 results/full_sweep/daily_log_sharpe_excess_penalty1.0_cow_500k.json create mode 100644 results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_20m.json create mode 100644 results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_5m.json create mode 100644 results/full_sweep/returns_over_hodl_penalty1.0_cow_20m.json create mode 100644 results/full_sweep/returns_over_hodl_penalty1.0_cow_2m.json create mode 100644 results/mm_noise/meta.json create mode 100644 results/mm_noise/model.npz create mode 100644 results/run_0bd5282d0820f9de3f3a83bea60d7757eb7b8267493a30b2f12a3dbd83e85922.json create mode 100644 results/run_0e471ee46490550772bc381f10ae15f90d406c94406139d1ea9af59decc3c4d2.json create mode 100644 results/run_121c8f61d369929f8544d6b51b304bf44992034ba79b905bf4b1ac4e67ac3155.json create mode 100644 results/run_88bd93d37f514232b86396678742bed8255ea3f608707054ffc306e77e008d76.json create mode 100644 results/run_dfb59b57a096e931ffaf48617ba78a5629d0cbdc9df7f15b86448fcb516266fb.json create mode 100644 results/run_fc93de2085bafce3a9c0330ff03b4960d250393ba49dbe0637b1bbe7eb7ef348.json diff --git a/results/competitor_tvl/competitor_tvl.npz b/results/competitor_tvl/competitor_tvl.npz new file mode 100644 index 0000000000000000000000000000000000000000..8a96f48ab979bb1483e93772ec4a7186045bcbfd GIT binary patch literal 1326202 zcmeFabyyVLD832e4aFu`w_a3l$p+TTu~PP!a6zL{#kV?oRCX%l-Bl z=KR=af3weL;r5^5y7t;(*_qd#d(L^yIWuRLMs*V;%xLxJC$Cj8=b~4vl1Kd(^+;)z z(W+1H-X86{ckW-(v(MmoR%!qD`v2aG_eA~7YS65Hqn7ck23id*)Y-j%r+$ShSr>9> zR;GNR;?{+_^zP@?uY+g%-u*he;}11Dc=UIVdcA+Q4t?CCelJ|szIc)1*2As;&p*ja zM|}qGs0W@aL_KVy9$lgyWuhK-QIC#M5BsP`xu{2ns7L329_6E+%SJuq_3>W(UNP!{ z_qj(sunx9`_hb8b#`eI#dzJO^9xUT^dcWjlBcJn*qTwi$DM{0*N2@5Q#T4SZI74m^}{;Ik_ng|Fm- zb+HeyzOpTpNy!V_!+Y>Iye_XVGx7QGUgW1_hR==Pm3^nw7vzuaW4l-f+g8@WXTf@S z4a?Xre#b+}OIaW5A}?hfu<$$90TX{$J_BBtKR@!p_V7M@7Q7eFcpYWLdiWcb@tRUE zl=9;>Y!79?dzI|*K5Pr`Lz(a%{2klI-{kf2xv?G|crSQ(#(S_`Wt(^n+fe@I9kq<| z;CGZ&c^}?`b-@Qq`7C%(Oumu_GQj(l@?(9JO?f}Ip?nrRE1w_hEBgn}SO?p{>v#`7 z6Fv`q2OsOopI`ax_#57bZ7A=>@8IGy;CZi+KREbIcwk%jJXi)o7Wv~dpnQ$~KfTAtA4h&}OPjSx2<4N2QoQ|suUSTE%DFqD5AXV^3UsdZzU7UC8gK#K2r~KnAlxRCd!V_ z%VvVhW+Hz!6FOis?fwE2?S@@WD2~ni8|@QHW|tG1XEGHxvYBiTp`2_Ffyed`n09}m z4zWFiaWP1oqwuiuEdk9RnhtM|J9s(0(Wgb_+WP1oawuiuEdk9Rnhrnce2u!wzz+`&} zOty#q#<&bjmWQwp*<(SOiG9lOt;}S(PiC^*Co@^@lbI~{$xN2}WG0r`^_Jy6rJQgO zs}CkK+3o|6?IAGP9s-l?Au!n<0+a0_FxegglkFie*&agMWP1oql$GgtFxeggkL@8a z*&YIu?IAGP9s-l?Au!n<0+a0_v=g?6z+`y{d9ufXG86lh;ai!>a-Ym(xld-Y+$S?x z?vt4;_sL9_`(!4TnQa9QV!2N#Czx#afyed`m~0P$$@UPKY!89S_7IqC4}r<{5SVNa zp>47~1SZPL^evcd4}r(_5SVNafywp|m~0P$$@UPKY!89S_7K_$!$a^6ILvWbOeXS; zYxl)uGCUNM$?#B2Cc{H9nG6rbWHLMylgaQ_lLklS(zRJlkFk!*d79t?IAGP9s-l?Au!n<0+a0_ zFxejZ8{;xCSsnt1JrsC zQ%tArts!a;G_2Q$TFyAM2k z7N&>5WP1oqwuiuEdk9Rnhrnce2u!wzz+`&}ZIkUGFi}>f6ToD92t2ljz+`&}Oty!> zWP1oqwuiuEdk9RnhtN*g9s-l)A!N!P3(8C^GkhyES?-gWEceMwmiuHT%Y8DF#y@9s-Z;Au!n<0+a0_FxegglkFieSsp@NWsg^6Cd$fig3M&KPnpSTpE8rxK4m7W zeacK$`;?h~qkXda2}(I}yk_+iWG1VhAT!w>LK(qlx(`gYhrnce2u!wzz+`&}Oty!> zWP1p0lkFieQC4RA1e5I{@Yo&#lkFie*&YIu?IAGP9s-l?Au!n>;%w)R{Jhq3xWP1oq zn)gU4}r(_5SVNafywp|m~0P$$@UPKY!89S_R!xL zmx0Oh5b7#>EGRRvPZ_?InJo9oOqTm(Cd+*?6U*%V$#S1uPL}&*Cd+*a6Aoha!DJ@e zec-V@1SZ=CRL99NQ%w)R{Jhq3xWP1oqwuiuEdk9Rn zhrnce2u!wzz+`&}ZIkUGFi}>fWP1oqwuiuEdk9RnhtN*g z9s-l)A>_#(3(8FFQ-*J4Cd+*?ljS~{$#S2}WVugfvfL*#S?-gWSZ1~rIEdvwrJP{0 z-3K1qLtwH!1SZ=WP1p0lieQz z6J=$32u!wzz+-y|Oty!>WP1oqwuiuEdk9Rnhrnce=x>b6z+`y{9QIgHW@4W*d@D0q z?vt4;_sL9_`(!4|eKM2fKAFjKpUh;rPhsMj)d!QASY|o_Jhq3xWP1oqwuiuEdk9Rn zhrnce2u!wzz+`&}ZIkUGFi}>fhrnce2t2ljz+`&}Oty!>WP1oqwuiuEdk9RnhtN(~ z9zyxp<1(3vWrq7?Cd)%IljR|q$?}lQWO+ztvOFX+Sss#^EDtG6^h?+t0+ZDrlFNx@ zriV~QwuiuEdk9Rnhrnce2u!wzz+`&}Oty#6HrXBm6J=#O0Zg`sz+-y|Oty!>WP1oq zwuiuEdk9Rnhrnce=x>b6z>G@|Vf*sWP1oqwuiuEdkF1>?IAE(9zv$< zv7pSvGQ+nrljS~{$#S2}WVugfvfL*#S?-gWEceMwmirVY9K>>;%w)R{Jn)$w0+a0_ zFxegglkFie*&YIu?IAGP9s-l?A+$}lhrmQxnN9$c?IG~k9s-l?Au!n<0+a0_Fxegg zljR}QRrYvAW}>VNC&)}z`;?ii_9-)2?Nes5+NaE9wNIJ(H`*twpP-Zz$7@zUL1wc0 z2{M!IA(Rn(ru)ETdk9Rnhrnce2u!wzz+`&}Oty#6HrXBm6J=$#PcYdY0*~z>Fxegg zlkFie*&YIu?IAGP9s-l?A+!^=hrncc2yK!*7L=LTrwrf9OqTm(CYIUdWVuf+C(C^@ zljS~{$#S2}WVuga!a*$e$xOEUz+-y|Oty!>M9wZz5A0VwDw$X=DZQ?|_urn88(zmV z%A*`hzcwuewowuiuE zdk9Rnhrnce2<-&NI!3)kT~&^yEDy;{Mkm6kyc14}r(_5SVNafywp|m~0P$$@UPKY!89S_7K_$%R|_Q z>~WdQM7|97$xN1qWG2f)GLz*YnaT2y%)~Og4zWBWmy_iog^7L%+e2Wo`a^O#*&ad} z*&YIu?IAGP9s-l?Au!n<0+a0_FxehL+hltPOq7**Tmh5qA@JB90+a0_FxegglkFie z*&YIu?IAGP9{L;OGB8;l!aih=1!X4oDZ{riljS~{$#S2}WVugfvfL*#S?-gWSZ3E- zmiv@)!a=M)n9O9m4?MPqz+`&}Oty!>WP1oqwuiuEdk9Rnhrnce2yK(?Auv%^rsKh6 zdk8$Xhrnce2u!wzz+`&}Oty!>WP1oqwujJ8*d79t{EtsWhTpgGLz*# znaOgW%w)MwX0qHTGgh$@UPKC@a&qV6r^~9@|4;vONSQ+e2WoJp?A(LtwH!1SZ= zWP1oqwuiuEd+2YB%fMuL2pslUP-bGEGJGpDS?-gWEceMwmiuHT%Y8DFWP1oqwuiuEdk9Rnhrnce2yK(?Auv%^riZ{} zdk8$Xhrnce2u!wzz+`&}Oty!>WP1oqwujJ8SRO+8*yA#piDicSWG2f)GLz*YnaT2y z%w%~;X0kjaGg%&znJf<}O!Q0G9s-lqACk+7Wu}KvMz)8*WP1oqwuiuEdk9Rnhrnce z2u!wz&^Fl~0uyCrIsr_!hrnZd2u!wzz+`&}Oty!>WP1oqwuiuEd+2YB%fO6F51}2( z<3D*huH0veiG9ju#-;mAl@qVC%NdvMGgZ#Gbe}0^T)NK`GcMg{iW!&g`!^;W6qkK4 zQ%tt|z+-y|Oty!>MCLA05A0VwV)K8Ko7$gj522iF4}r<{5SVNafywp|+9umWV4|$d zwgM*GL*TJJ1SZ=~gZ)Czq4uKAFjKpUh;rPiC^* zr!e6lmiuHT+kN1%Jp?A(Ltr9jm#7E!D;{w*cT(P;Y!9KFY!89S_7IqC4}r<{5ZWf& zLtvt;%(en1+e6^7Jp?A(LtwH!1SZ=dk9RFm054WWP1oawuiuEdk9Rnhrnce2u!wzz+`&}Oty#q z#<&bjmWNPR*<(SOiG9lOt;}S(PiC^*Co@^@lbKj%?@yNdWP1oqwujJ8SRTSYWRJ^aCh}#tPiC?_Br{nal9?l+e6^7Jp?A(LtwH!1SZ=GT{lbI~{$xN2}WG2gfGLz*#naOgW%)~OYt-wJn_bKHB zlkGn6*d79t?IAGP9s-l?Au!n<0+a0_FxegglkFk2O}2-?L|K`>1(WR|@Yo&#lkFie z*&YIu?IAGP9s-l?Au!n-xLtwH!1SZ=-xLtwH!1SZ=< zV6r^~Cfh?`vONSQ+e3e2Tn1)bdI)Vo9{~Jhq3xWP1oqwuiuEdk9Rnhrnce2u!wz&`#JM0+Zz- zWXc{3%1kUXd@D0q?vt4;_sL9_`(!4|eKM2fKAFjKpUh;rPhrAAEceMww)?;XpXnhm z*&YIu?IAGP9s-l?Au!n<0+a0_FxehL+hltPOq7-B1Tfhi0*~z>FxegglkFie*&YIu z?IAE(9ztDZk5^WP1oqwuiuEdkAfl?IAExR%ZJIlkFk!*d79t z?IAGP9s-l?Au!n<0+a0_FxehLJ7IeWOqPexCfQ>_nTdVM@U6^bxld+dnO#nn`{Z)6 z+$S?x?vt4;_sL9_`xGV|#B!g^WV;VMwuiuEdk9SA>=O0Be#N7biRF^g>&koo?HRe@ zbv(m2%CSV=pKK4IoNNz)$@UPKY!89S_K;G?nDrJ+l$Ff{m(65*2xWvH*yUt<2<2pZ z2u!wzz+`&}Oty#6PH?Pa)LZ#j%JPuRM852DvOFZ0ljR|qiDh=SEDy=$WO+ztvOFX+ zSsqfD@B!OHV6r^~9@|4;vONSQ+e2WYJ~2H6Cfh?`vONSQ+e2WoJ%qN&_7IpTE3@8$ z$@UO~;ajdk9RFm3dqNlkFk!*d79t?IAGP9s-l? zAu!n<0+a0_FxejZ8{;xCSsubZWRC@9CiW@Aw=$FEKAFjKpUh;rPiC^*Co@^@lbKj% z*ISnRlybsBtUj2`WV;VMwuiuEdk9Rnhrnce2u!wzz+`&}Oty!>WP1p0lkFieQC6no z!DM>~Jhq3xWP1oqwuiuEdk9Rnhrnce2u!wz&`#JM0+Zz--xLtwH!1SZ=WP1oqwuiuEdkAfl-5&xIWo3E@Oty!>V|xfpwuiuEdk9Rn zhrnce2u!wzz+`*qZ;Z>pWO)c2_E=D6VxKa6D>GT{lbI~{$xN2}WG2gfGLz*#naOgW z%w)MwVd9z92a}mtW;y{pwuiuEdk9Rnhrnce2u!wzz+`&}Oty!>WP1p0lkFieQC6mh zz+`&}Jhq3xWP1oqwuiuEdk9Rnhrnce2u!wz&`ww$LiyO^GMR~GhWlhD%R@4g zWP1oqwujI**&YHDWo0@6Oty!>V|xfpwuiuEdk9Rnhrnce2u!wzz+`*qZ;Z>pj7txp ze#+xNc{#4!XNrk^%4Wu;`%INHF5PE}8JF%e#l-9E{TY|;GgZ#Gbe}0^T)OYym~c>B z_Q6as+3o|6?IAGP9s-l?Au!n<0uyzM=^-%L9s-l?Au!nWP1oqwujJ8*d79tdk9RFmFWa9*&YIq?IAGP9s-l?Au!n<0+a0_Fj*c#U1g6~WG2eWaDvQawNIJJ zYM(Nb)jnk=t9{B$R{NBhf1`b}`Uy%oalB^r6J#c|F zcj{NDl64`6F7}0rTNmomyPsFT4xa6M_v`GA*K2g}=z;=XgpqO~DHSVcY?{d=^( z;eQ|2>dVLkV{VL$?|vcrcdKJ<_6;w3)!;LC$$%6eHX4$*yXDgJTu0%#yuzqrpJoYV zz2i0w7SBxj#BkVu^!b&a2SmRfRM{$G*3ecW?YqIl7o5DPZufsHei@l`eUb*Nx|I-y z3p0WfPu^rOj9V5mbIWRRwfXX1Pb2yYzg+{96}dN0jCRcB8_#*5D4S`2)BUMEqPKr( zZHcA%dX@a=_o#EeRWo{Ce&i<%@o(?E)9Okz&w5qaHh~u#i}Du(>kZpnR_qBZ{%e1x z_F`Gc<<%LdH;rCj&kH-|%Sb)1KRM|2OK*?5zjWWj2JoVOeAN3B`teckf9d_zKR(p^ z_xgB9A3y5jM}54lpPxX#NI##Wpa0U&7wYF%_4BL$J7J)okJiuM|L?rfpW2`wANA`) z`t>yZ`lEjR@sAJn>$UpzZPbJ>BlY8>K7ZnW^+2C*qR-FK=jTMvN7Cmr>GQwz;zGub|(r zpqx+F?^n?8SHSrr{rIRKANBidzl_xHSJ3ZQ(C=5!?^n?8SJ3ZQK)*=8UqQcLLBC%? zzh6PWU*XRg1AV@UKHo&2Z=%mPQO+Og-~Xk5Uz+}XY5Moe>EFkve}AF=eUJM0f9c)%iQr$3-yPt&ib>DSZr>uLJ+v_EIi z_3LT+^)&r@ntna)j}4-}=SIJOTfcu>zkgf5e_OwQTfcu>zkgf5e_OwQJF07>-=C)R z5B2-E_4~K=`?vM`w-q1g_iyX>Z|lcL{r}DB|8EXP)aRS%^G)>mCi;96eZGl4-{jBn zK%Z}-&o|NMo9OdR^!X-#&KT(PP4xLD`g{|8zR4dO=<`h~57zJB*6-ie@88z%-`4Nn z*6-ie@88z%-~MynNWXvkPY(Kg6MepkKHo&2Z=#GB_4y|Hd=q`Xi9X*%pKtP~f29Av z0R8_3=>IQ3|9=7c{|nInUjX_=`g{|8zKK5HM4xY>&o}vV#z3EMqR%(c=bPyBO_cM8 z`uCgY^G)>mCi;96eZGl4-$b8pqR%(c=bQYwj-bys`ICb_-$b8pqR%(c=bI?wMSZ@B zKHo&2Z=%mP(dV1|=^yFyP4xLD`g{|8zKK5HM4xY>&o_BuIP9;_H__*t=<`kV`6hqP z80hm&^!XpKqeiH__*t=<`kV`6l{&6MepkKHo&2Z}R6lfmCi;96eZGl4-$b8p zqF+zbucztP)AZ|U`t>yZdYXPcO~0O|Ur*Dor|nOzpD)zU7wYE=Q+(KH(EtCo{{Of2 z|G%yO|84#MZ~r+S=--E|-=C)6pQhiRrr)0iA6i-I_owOir|I{n>G!Aov4Q^oxAp(O z9avAlo~B<<)32xL*VFXtY5Mgv{dyXDMqfti_ow~ILBBsuzdudCKTW?sO&Krh_owOi zr|HK>eZGl4-{epKNWVW#zdudCKTW?sO}{@)zdudCKTW?sO}{@)zdudCo~B<<)32xL z*VFXtY5Mgv{d(G;^CbH9H2r#7;>nu~hH?7!H2r#-emzaUp7tjJ{rIRKANBjUO9=gX z+MjlyUr*Dor|H+z^y_K*^)&r@ntnY^zn-RFPt&ib>DSZr>uLJ+H2r#-emzaUo~EBK z)Xx{{=L_}og@3Ldd>N^qFVxQ$>gNmf^M(J{^M&@|W$#}&<{<3f&2kM(;VTlHh&)#P zm*wA=mLszBk-BO9h2p8>HVqcfOltZ0oOItacJ~QcEtK)l=-l@W-YFLd&F_26cB9+x zlb;QXzh}+tc1-*GF+WV&oB!}$H?b__@@o0}M7U@s(C&AuqR!T1=@Mtow+#QoV3b!*k>}XA#+nV*t9d2dHEa5{PK&kc*Pn`RmH$h+9YeFt7Fz9b*U082j-EEoYe$~joj%TY z6ke5IKec^4UL@ML)VXAU<$0y=QKQb**|DEj_hU}23wo`IZjZL#cipskw@5RupyLn1 zj_bJ z7GGSc^5pt6Vm+^teI_Pun`C8lUOk$nj%+({u4esG^18UFb&{D0{WaTTtuz*WHyhMp_i)Ses&{L4h74t5KQF`5V|OwyUlE;G()KB? zT)k|5Ua$Qolnq!sOEa&Ot1hOTwxyj>UUzdp9rVoGPAtB;H2a7(_wgV#bGA#{QZxqu!@NRUJgF^zqKN2pT8OI=xy}v$5rQc_emC z=ldEBImvvC&Z$eTMoXoJg@c74mhrk1ApCgZ$H7Rty;f=oo{=04b{%eaJ=KDa|LWf z(9YYNJzcfy*WQ=jOQ-EN&dbg|q;e9zE`Ie3HfEq4&AR&l_+q z_UqU5`+J=2uZKkYzI4qQXFG}6X4kLwFZShmSbc`3zp{VcdoD%t7DoPZaQBR?x}d5s zRQZ*sS&*;r^$yrYO#);sxwMsXOs;`!gx91Pb zP_@L6*w1V8`HR7IGOO*;qkVmsM`yKtZ&m5p^QGyZ85TwsuAZcYc3zEK&u#Cp*LeK+ zHf4>y)>A8qu+DZbuFfAPYS(?_`=q<&dHwv7B3+&Fv7c8w&+^}k+o*X>>fQTU)8*QE zg;{s~dDYKKj7wG}*W0M^i`IDiSMG9Y?RFdIRk~LE>E~Mtk@;%ID<>C^7a4{=-F;)F z<#}z0*lBCCDfaV9wEySeyB=y@+maOe^ukqp{Ad{6Gx_x^NyVDH^=jVjsNKK+oGPqd z>wCuSae~XxZnd9P6D5z8i~lLRuZU>!aJADl%kz5E^ir0<+p(WlyDQT(Oy8~M75t?_ z!yl*2@Ye&UB_%WD6|3f4nb~2Fc3yXjyXH@O$#}edx5&iI@f<6Q)KAmYKC#hPej9Vf*P-g*=P2eXrzItm@EeYZqwyKCffNp^f7k z``-21t#bo%*^A~6zWws6I8NAQavXZAtL1rhyg7C7k4drLU!78CQo>17)Vyj$Yt3vO<)ij^Xzi!bUC)j&&g<=rZYyiIw-x=)JKV0Ae5|-RY}t#~_Lk@M{nS{y z_03{GuQpY(OxRUJ&C5QdQTP2f0yO)pO8sh(_iL7(W3}@dxa>*L;7!KozpOLv-nYM! zt*9}kOZE(n#)`qSGbLHl*YfdWt*;YjUC9;ud6iE2rBlBIYF^V4Jqij)Z~pn5VP}`j zcI%;?SMmbiYyDhmoY&liZQAd*kLph}O<&+auW_Pe=%Jn;c3PfS^7pq3ohudldHLoI zzj36Zn%B|o@s__%q&?oY>Rz?|g-gjr+AXgmexKLQYtxEDo2RxiZjaVs-YZ7cuP*!p zGFC43)>kYp;&ruecFX${<0t2Qd3Am4=e6r&a)%K=)x2^Cc8}L_nRb8G@BNUifg=kF zmxZ&7{_d=Oe0<&SOyer5+?aeqorY$hLrp z9KE8Q*V5Xhz4F#KKL2HXI;2v8+xEgMS=#QuZ;ch_Rt@}7)5Y?7~Sn=+JRmpn}mgiNmpj**=kEX}^`RH~S0$NS>QY&;q3oWXjpmLK zyHh>ibK=ZYi}T7gV|{{RER@7T}l?*1Zoa#c~=WA=-kTn{Y^)Ep1Jy;yhjQJa!taHKM?*tDoy?*Dq!KSSSBjm#`b!dA(S4@zlkCsb94UtN$o%%kcrJC1>(AL-Xy)wT&iUDbQG&G|< zdgS|AZvGIYKi94sM{3%4u z%VR*w>dim-YxXDlbPLTjb8J4*uDajk6w|fqSLdV0+;cbVYBV1pVdfJp&Wv{y<>$`( zVA$p>&QIMiBzu*zWSQ4=sKkP_1|eN&+F*yRviz-!Sytk zgSi(b`=#bpY5kWEhN;@;3$13)8d>Lfq@ll4Z_k8>wDW4$@%YW6myOToc+Ne4{n_&7 zqFk@4JB!y16#;!}EnRTL^7E_lZhzZ4%VB@4_rFfL`qq14cP@JTSlK1VqPO+KHQQrY z|9QXnHYy{|Z=Tk)a?d&1^Cwz-I`N_T7^C($GhpwH#Hm||-w9gIZ5JFWo|cc7;!dL3 z7WY@X>^>jA^TmCddAXfGl|HbAqe%PCyzd97+P+uq+y*tTes)`r9ony*SGOYlSFG{2 z6*W&^FB$YmJ1_66+ovqsW1Ls+5%oRt)^8!+4#<|@J<1C~KdMyj`quLLbbq1OrR zyp(Z@xp|d*(D8>&aW$`zL66U8%B7uGG-uw}E~7ST=d~yB{<+6BM;q0z%k`7CDc-HM zaR1=ncl4-WQFw$$&s_N|uU{J;hdFotq@9<~(n>{BX6R;i{VG4mA=}(fXQSI=giWV6 zMaO98<(i=1uvY&hXPV z*AUC=m(TW~&?GCh^NQZ@X12#Uu5S{j2vqYj?8{(%>&O93e<_^fy_)SByFvSYn5vs< zUa7IlxPILaFEa4t#rEQdOY6SFE`^GG(`L`j*}?Mq)uaBw5^fE&eQ(H^{Q9G{P0g-f z9agpIyD?JDYel1TSvpVFu3t)C>2C}Txqd*qey!?RBSY)?#`SC9j{Oa@A8aW`930ct zxlpL6G5Xk&-CUy$*;eT{rZ*j?eC4XQ>y;*xN>57?XKE+1tq@JWQkK+QRn^i zvjO9@^V;#ES&NygjPt5gvhJ+INn43eVPbHJouQ(t`_?X1`dgmYki$8i-YB76zpP$Y zO1k)|gW2_K_O=Oid}^zCr7pSkZjUwEd0Ce|Tc~E=?4s?xz!uZKXy?_+z0JgCt;ZVq ze&La}#eQUHDGVhIU(e476<_*4Z$GZ0<#`1-7QPVv%8dF|KCu4P^Lfn9%fFFJw}V2> z>v}@h3*XG>fBD?L<{e%ul}LW6?UbY2we#|NRi%IO5aaswDYH{r>&q^pZSaeO4Jw3+ zeaX*u{}weKvvhv+&c+L$FJ5CtUT0GHPDy99)a9u+gqxJ`JB8 zewa3ec$g;tt~Uj=^Q!7}>uqFPih4zZ+ez_5T<#2^!Zk(_04s8L@K{|3)ZL6 z9uGOb9b2(Y%27t`ao;Cft^}k+G>(|mL>4Or@*X~aQ4f4qA+9R{bSmN{0f)};(dRJ%U^3*+z^Q!v# znb+wgE+X}Y3zZ%n3lSUb8U>atYq`H(C*SMPJ*jqHR>8~d`u27=J1_e#hwZ!FQS+K+ z@H$*ATsyC@>PddB9-cy!&v1ONm~F=WFdJVr&oE`AQT<|k(4if7}u|M2RkLd)6Yd%PcPM@bU=u>7qH%Q+!f37I=H*DUFIsWpO;8`FGjDb6oyoX=UwCQr_7#`Z7aIOG zTAr8n(7p4!mo+1=9v*v#G)QE&zb+M?fAOt!ej@GRDIGWO*lTuPzBi_fi_EH>*SW=w zk2jj^W8|;!LGDHNhBOz256-XMJ7b8*wdp{s6ONYWRp)!}wGOX$NAG7--=E_)Bx8%T z<(iqD*M_EHg(B*!?XlaHlZ&$0sO`~I{R$fIlQ8}9tYU|qZP6XrFwwH*V#(KG)lX-&Jg>k! zzbY3P8vA)w%DHOj|$DxsQn33?a`;-(T4L=d^7AHk~L{Y5AD1>vyb(dd)zp$(|3N>-teNSn3Au~ zu(=nPi^N~1RsWWEhQ;H@%o$F+o_ExYy#C|y@kpRcqJ#<6`W4^ManGnH+Wu17eexBj zcE1Untl6IkI#zS_tDDBh$JE1`HNU$x>ig#AOiG@0$8vFMTE_lWN?V>+Cb%~kzKrf^{mG_U+uiSntY01dA#xQ(XYjVbIY!F5umRqKYXhi=k;OuoOB*@TZ$=Lm%iVaYq?mTV*iSG3oO6B9p0KexT{}Ezc`?#>n`+C&zwX zz0-ZTkngfuzvd+SH0!9F`FYv(Pt&MpL+!lo*4Vgc)_&u>?*0hQd31G4(O~Y_W-}fJ zi(GwIw0yP3^1L<&*gmcOWtZl7=)bnd`y#c#lhsmkZzf>pIqdXL(+=%Kq|nZlGPivOaj2EdHv_ zX8XR^^#wldtE$JxqSe+cbAFPk3KRQgl-M)*@b|OiyM{3>HN;i=Q{ch9~(Fxa?%@3G+L(s6}b zi2`Y!?VU9;SUj%cdFIg>%kxT{>Q?S);D1NW$TwLILU&x#-2 zOt!~j3(HMQ-^z^k*nG|N;meHu71qt8Xa(IasApd zYVhd^ubYTqn{U(J#S0cYS~r>VJlyiUd?L4a7N{KidA(jgq}9_Rs_%~;E%Wnfe6@Z> z*687I>U|fJdF|WpyK_JW?fRASe#vC7J`6YVSCHMQ`o6iFi){~beDSUxELPn0yOwRW z<$09~&L7tM(9Y=3ZR-5$f9+4qdv!i}gfzeK>0tNPlQNp0SB?VtCs-ZV&g-sc{}jy= z82jG&>S2#>-(AGLz3GBG4G0zu6Q1)3e`I-HQ(vWealMOn{W2slG&)+Utl9mENreXA zDJq>Ww8~z`?)d>T@^Vi1GfVYI!+=)R=QwWC9zQ1VDxP~}dgJ;Pmb27=eB)ijL+@tI zzJFgPZu*>glpxUZyh<2SeBYd0J1?Ictp=Rv(%kI4Mt+RHaaAd`J?71{I@{iNs=u__ z)Y`7Dr!gKk@B{3h8Wea@Eu!nt}EY6Tu)nPOGL#G5x?%b z9CO}Vo>!vp6}pEt)E*C6x0+LG-)krH+vC`_bu)fb^D1zn)sQI>+IjgT33tEM`lDe| znw_V8`)cR)W@+K2jSCJn$}8i(YAYtcXe2^2)pu_tP+w*@4N-&uiL&E`uF& zYUh=`bNk?`Lwz^7!W0$=hPR|Mkx9 zXobz?PesqaFW&gY(V2VI`4daWEgm~2XUAx6%<~h&nmP<`Us$_8@wvvX0iCOR8P%_= zZ?kkrl%$?W={IFl;|q&KHLoT06KAtLuUm=TgLe;#{k(pB+d6qjF*Prn2hM)Io2u8h zyIilaVg9cUCi9Bl>1)$G6}9tPHfHszL7xX2h^tHt{t6-CY`wMtFDuvpxQ@Lks> z-177H-`t9n&R#e6^ExmfvdjZBk+>d9|!#-7fFWHqqRe?Qvy^-S%|@Mriu}v{$jEC2I6D z%Ij#&9Q(Wcsv*jaYuMuX=7pkI`U~MNcTKVQ`gVt2{>|2fZi)Wfrt;cw>d9)S679_H zPb6-;?q~Y0YW><(@_s?b%BsJnoh))OVLG>HZggIIg@1+$Uk#l{6djgg#*4KQkto6{^m zU9V14F*fOkk7bK46fW`eb$Z}!dHtF(a)isi1KRcLzxG#qWxH$hu#1}4lLtQCI_^;G z*Mh{e@1>6Vp2k1h{W3DTJ^lka*8hH!F+H{G*Pt4S-i2i`&gnA@MUB8y~NN_XZd9{9Z zIuo3s@5W9h^GbKN?2z@xwDWTC2rN-!n{i%6T7R&wzO0cbUOlDX&SMKiXtnRPj?TBd zer>TnlzK>OGwN6Ewyh83w=+AhRjH>IoiahKUkM7h=i4$-tzWsve%-mei-*a)Y|`~_ za^H50W_uj*GRej!AC3DHE&8{vI_5?b@zHr{uQ^T&#Oa*p&$Vr7dHrfLb9JRvceV3! zi|@Yr@-1O@UQ2v`2Zi=k`x9$#xF+*itM*sBzVZ!Qd#tl@URgauTbx^NsF&HfW<(zC z`ZaCEn~EV1jeW1=)v*8Y$>|@@6K>B7+3$a2dHuTXQe{k{#oBopRuAoeHF0IL^Qt_p zXS@fq)w}}hPxiK5q_)S1gK3WR4m9nrkK;tj&}rKB>)en2`*S;*&daV~c*25}=8H57 z`={z13&+P=J?rK1J*1tN&)!q1T~jqNJFh%n*B|L`qxL5O1pv76$`c*l7zc!)0jPo)Cp7Ng4 zC#kq`Z`z=FW3=1jXrH1l2G=n5y^@!8*@wx~WSB43eaPN3;*jNeoqT(w)6XN?_3OX( zCth!U*JDCL)%On{<}XmdPxbxuHzyj*n`rv{#Nw8FyS*E#UB8Z4H7UM$rtx@4$;o+8$G0%|GFNC*%61 z?3X>mzb0PNaJ1&}apC?mr7Nd4K0YdWZ7gtp?Y0f`Mcq@kzfLG^`S+z2DjB$Z#yai1 zeC#)r-FGOT+5Sp-WN4dM+79t5J7lPFUWQ-Jox(#O8D@`p*LrnX z?Ys`oz5DxwpXvHl{!8n1+kOOy)PY_PC#Saj__%B7OGoPjv44Kz!uad!FO)hO-Cyn8 zzE9BfNcH^0fU=W!=85{h2&Ve?Ck}i0m#`hBS-*Ds^ctH<^fej}DfMe`#|{(TjI zxAs2!z2jty&riJCcA>Y8?WX9@ZEF0ev=ejB-{1e$@M6AeN22rEaldr)rvBTb$B$!j zxvq8XWjuaVIJ=H6c9?hDM>DUi$=;nfG{29@yz=L5+ub%eKul|u_fq>lmgiL}^S);t zytMN&B>9nL_?G%+x5x5Tr{B1~TCHEz;!U4dEQeaZyiZIyn9SDL_tx!Qwr^?o*)XB3 zU-|VFwCh(y(xp2#euiacVDS_UCLZO-r^N%{c>KG49$Ek8acYcy+r&}i+vo*pmiv+tVu*SGVvXrB6E6}A7Btaklc z(+8-&uQ>QX)%>}Q=jXW8tdV0~Mr$$U_@!t5HroA(MUlRDyssJ`A8pS?bX)VwMSL!^ zWNnQ;bHu3@wOxF#S?>GSHii@bUB4C{p1iNpS9LtJa#@WTv)8ES z?>)~=y=VQ@`1q*!D_z#E)e0ZcK0fYw6H+VUnQ>k{CgyXieY&ym>s#vglg$ys8}V^i$sHT?MSskIBKdF`sUG*xDIwSJ{cJ$UMjr^dcl+GB&= zYc9vjqkVm-^QjYirns1{UwU3;+XP;0EXrRDtT$|PS)r9z`M&$^r|U9ToJ&;hV!vFL zx5r((Z})0laHFQb&MpjlanPrW*}k_vyLH0nLh5+vTlRvZ;?GjsWBiA!vrRi=T)+H! z6!$n&DZNPH)M7;Hm)iAfL*a50g3p_7j~N}-UYUJizDVJ?v$0``<#~mqYO%Jnwf6DR z(ENA3emkm|omcP64?oO$q2{%0nQQW!-PPk`c%LNG+hj4GA8p7xWUf!17lw{^GTs?h zO1pm_m~B_Fn*~fCAM*@ay#Gs!1;Wt1aJ`TImgklHs^{fT<+aD#idW1X53LRPKK{9M zJ?++@9ljHqs`V@4V6v(WSC}3Teb~Mr{8pH@??Yw`wfR-abY5Q_+w}gNb%7|}_02f9 zsQa5O{r;~KuL@L1_uGu|w)f0~*U}X?yMAST>9{}XZMA+m{BD%kt(Tft(u-XJZVWS? z4-l3x*U_M23B|&v=f+Qp`rZ<)?}OQx%x0dqgXz2+D-=0>@aKHd^G$}-GiO=eUwzl1 zWU*P7%*gAX3TAHo@|e}*)#?jsUWq=2_Q~H-&CCAUj0vY|G>Z0S%)pe2UVRIOZVHV0zqtpF4SZ+c-SYb7k~`DQDxG3KFR}RN_%UJXc&M-A?E5)e zY>PfW(WTdu&+(G9HksFrw6{)dnW*ji!+rCfDf-rQe`3g*21OoTnK^S=`CtIs{Q+X1^av{ZNDWtFT>VWr(Qp=Z89&XjPJ_Mnc}V49(NsB z;&Il=_@4D zuqiq(yN8K}zN})peq~O(Ztm%fgEaG+H1@{t9&J61&VMO+4LaBKV&tJ2B6#BGKEG3r zvpBB|+mAQh`*~&bevYm4g)?^BtypA#DEfTirLXa8+g@4|otMX%H{L%#)QslF^w;M) zYi-A@@Y2j{RnwX2o!@mc%FAlqq@S*HY(zS{qD~h}_zT}e62=sd@J|w+CqEWtcO^c}`FdqrAco zBxqfCeogVFX_g)#cc+QZtF5x{UpKK1Mk$?cKWFyPvJe^GPNJ5nOBHcf`K-j zJ&f|YwrOE{`=vr`+CA`F{q}z1*s(*uZq*oWd0tyXA2jS58vVIV<@I0tt2eq&UUaCv znpariBQNf+R@>u=mm-_}y5`Z`=)C>`?fJN@(C8(7HS-!@y6A!P9lIOlwZ3w~`pdRC zi862O7P(~!5MD_(zOoImJTIGOtDUC~j{Uq^Jik}*R4Ub9Q_6O&62J4h==RtnT^-v# zKaJ~G>GY>u`%nC27(BCw-@WtwHS^j~tb3bYpN#Wblgx2Pvi{Y?zT2OzKi&2dySnsC z_iXG?i|d!0!)Q2e*uJOx7xyy$|1h~HWt_dKfwO39ow~z&zuBVA?{!voQTIt&Iv)D5B4oh% zZz0j2+f-gI_agGliEuO9UtNnmnB8=$TEF7$T6*?;hmFzoYuQAb>}B#BpI=Q{CdGij z-l;_E;YHkUbkaUPS}!;{Vd+j|e@!aelp2hRz$J^ICI1ne*(vt)ll+bp5J4{%0BA z2t&N&soME*1xygY%%+4xfhQ&T3)}ZFMe1g zYt;95|Iw`J|Ig>Pciohw14^5nSI*?0DH z#y*jT4R({>ey^=vzdp6FE#rO6IIryOdboK1a1z72O?upCt-lCM)H1l@l~|}>5rv#9 zbTy-X{m1ckvt^UN_fMkcRp`*Z3~qPT`sL97T=5faP4~YR4ECQmB87HdTimj?to6p& z_mBPh{CYOJx>)O#bXfM~v&F5ghkWw&_p#Vt55I`WS5g1ZNUMH1^iP!Esd9I->sOAC zNkbn+s(EF9pY=}GGiv=RXq%;H(r%5Txk`KV`RR~q$$`Y8YWC0@-u<-utKGYvcswl7 z*k292YhEAZT0^v*Q(iF3t^-p7W%x@dV`vmR8<;?-eA^yfC^`=q%uto^vHpV@gm zOLOeu{nLk|`>P$Yd%3=!qK+Rc4e401Ws{cCTqUo6sWcw^_}r z>0Y<^E|=82Y$~K#x8|vFUWWVMvSxV_Zdl+Fw0p)C?Yw3bN|kbP7vsDJey=`!*+nPu zDpj3}*y02C9T9cs7p$`pR zP1dg)4{pCWGgCXS@k1;33(eHa==j*Z^PsfJqUMi|Us!s1Sg*O_=8kbU;@z}7uW@eU z2IX%S`*~d)?z-)F0X45yRmN>ib5zZ%h5KiRwQr64_ewo3^}5X9&4v1D_U}`qn(qH= zh_Sz15*=QEb^h%g*KfN`k01S~PfS)Z>N>yH`TG-25hWKNFg`!AZS|-0W3o7krc*a{ z5x@P!<(0O(D|E6vujJjrBY%vJ{k-gZ7N|V%o;u#%)h=k-6=(JMI6vc<9H%Flu3xT` zoRVJ6r=8c_lJycdYiXRsxJfTp!)PfA3b};l-(KP5LX%uc;Z=I(ur?uW2>`X@iOypD%2Z zwL`#0J0aGN+7vK%tDm^Ipz)DHi7oH1R=zjD*Ef$D^~*1yTlz%iU$4#8;{8y!d1_vr zw(a$P_*?C-_WN=9ec8LlzIQvezE788cMRzVjF~+3RWHrFyxUqGAG*;vujZxeOu4ty zNlcBmYSZXleq#9XGJ7X?w!D5lSy?h7*F-b&vhO$}q+l7d`(J@xnKLIHuI9D2h||XD zebx5(s@(YWU&otnk9OBe@5($=JFn7$OxfutdKasb z$-a-}dA(dRb<=k<&QJVT-v|Gy)NSYlHLu5$Dy&Z7qUN3QCmH)bEOhL;#@43{ z)r#fG)TN7dUZZ#R-0?MuabE2wcXT|Hxw<&%()Rk<#xq29p9}GOOtHK@PCn3o*7tkb z^(*a=)`@D^xtLwQS~zxHcBYhieB5$>X2I*<)V#`H7}U|JmT`a8D)n^xCs&^u-q!7I zmC3fZW_!%D=Vzy8&5V7YzILg5_4_%Aq6en7x9aLIPVDZnv(a|T>zAjU;b=fl?Yw+e zENGsllELh}vb?YPCQofOFZZ=aM!Za@_9t$xO>t&mDdY1wVOQN+T%si8=uelapy^`ZX=w<#|O!p?0D`ke%5W0vZ90K=f5Ve^cue5gLYm)+l$?{>R|r; zq7TPTjO_Vc&Fg@B!fkgNZI13we7l~n-`nNJ=f8q_TYWlT`;(!OeMo|-m$ma6J@DFk z|8mCt`?u$P8@5>EBnsR*akb-Tf3dpC%hWs0S)SMP!AJaaPLBQYkh_(k!V$-a=<^eo z8#?<}@lf;n)o@grr_YT2bvB!C_Rg1I8pbxQ+w<#v?YtU!^$ZUDVO+nyExML2Nj+z= z%khN&;RLhAw#ZQ1QC1@@?yqJq_j`Kyz}V00(u>TQvgK6sD!q7U-#5-`UU&9)jJGnm zvA^!_*c)fGu@#zFt< zonyz6-wvwB$FQ~Tty)x2`xE)!J~_52(zySn?3bE8_N{uI_R#cwX!@5U8s0bV-}^c5 z-PmJAR9U)_Ub>DoUrRDvJ?XNG_t!o|ouodkeNeg` zR=I878_r%0w{LYsyMFn_LLjj$Qws{5CQ9UEHiFZb2O z=Djw8w_EP1W{bMd(lsk(XZi85((}UAy@Rxm zk3k*oeEN2=jE|2~&Wb#RM0K&T)P@(gH_R4cFZ$0aS>E#Vg^{CPt$olX_UC_Hxn8;B>Ot!K z!bykP6dZ6@%_~>?N@1na8`m!-uPfKL_8F2=yFby#Ke&0C6UKS%uTeAS=4#F&v~P!a zu7_reaft@4$=%fQy!^*xbN)VFJ1@6^4Z`oVHUIlpgR<5he0G7F*Whx`rXQH0<`vQH z)~hmCOxLfDdwd&J&aRzT?Q~0%KE7?7*XP=!y@rgaDOP7rvgh0201;-ra!-%$mS0b^ zOWkbH!}Z$pDDe`~C_IX!XfMz2EBS=PE--w=y0NolV{@ zd)x8f43oUQ$Gs_|eLmW$R+b#jnT_ZBc3V?qWVdOxgy+r4b^V3~h zd2*>`sCHgKfzHqFT{pje1>auQCv<|E*TnAE&U$B8?}sTqdCZJr_2Grrn+DdjNt z{6s0+0o8Z!SKH&ub(uE5E2h@38L8@D&-&Z+{VaXQ<%Z*2oC&#Tq0rIF|BY5U9SSD2^BZ~p(UP4V(@UN1*AuNJO- zt1cR;=H<5X?uB7njqev#_DiGBnQIn5sGV1n`0cK4@;A`1Cb9Ww&RoBK3 zUl~G?6f%z)Gtcw2lv##MMM+e~OhrthiJU6T3 z*?ISMzuWIQ-oKvz?(6B}d!GB;Yp=cb(@bB+JquAWH-W zv!BhRMJ{8`SLV@Itu-a+UW1`(#IhV_c+aa;M5)%NUc!8l#Zc6@1j2mf-^{E|usVr& z{#cwIM9$ms38(t@A^$EW^!*chOjEtf(#ZFR7)B2?nx9n#I@_r2w3ris_Su9(I8p5R zQohZ8o#E!j*CX$?^3uW_n6EdV>VAw9!hD(Vy`bN8WBq#EY+y7-(271!{K#|_&Np#J z>^Dh!Xk_^%c|Y*1l76Apu`=ItetNdMZv*;$07tN1#Bo@Uj~a9@_w>Q}t8`f>lK0gqL_OLLMBnpJ zBLImGxLDg0==FGLT5hUu9NE7(Zg8<*B+~_fF=y{cI3xkjZO_EPaqM0fsvf4j{fs_; zSvGijebdDM`?anftZs)V;k-IhsXXva2cDM}xoqB{LuZTd!l^ben3`M$eXmzSq;kv0$}d#w8_D+$DGv0Cg7!tTXtS8Hsdb!YAW_j>y=3H;6P z*d~$Uov%v!pd%Bz;Q14rII-{9E8k)ES78p?wBG}mWw5wT=h&_N?%ICLNlxtouOJ)? zf1OprLv*j3!DPgj<&p2#W`4E_pTDO8Ud8_IOkdeAH)~V#%Ekk`m-xV`Q41nGyds7? z(x;U0_VQZp;pUiw`8wkAv@7h7?<)X`{@RmYkoCCQzjjSh7JLmv_v-h{O}EiO=F7q6 zTKAH!2B17;VnZO81Vqmhw#8{<_mYbKZ4@_;UXQE)4t(dYGY`hKk34~1vfIsl3mjm+ z2=}IzPL{9l)iTBN(=QC&i>c$v+>1WsJTcHfaf(}16ZkkL*^rIK0qHD*yOvYf=Lwt1 z9%`2x8}Iej;->`Vd+7DmQ|Ej-HT2S0uyhUFh3sFee;2OB7Ry!~x|e=@n`U|f(um(-lXXHsB2=9A&xo!gJ>UzX|hZH%nlxDJnt z91QyC@55|yi3!bojl4cmJ&%4ePILfNv{o_-D#ZffoQ7?7yRmzHW9|DsoKP zR|sLQ2FC?$K%9L_@`7Y6NIA>+dST^!c&zV>nkPIHmc4`SwR+vccb+Kp(^BZjgkBG~ zWz92RfL@RE%$mZ;kzNE$KkdR&uj4)|2RzX_hVG^4QfF6k4mqzzB)%M7xTOPl>_z2b z5@LaIZ^hK32iW`9wXQPCj66KN_Em`WD&l`W`7&2|!)qPrRk%ZF`Mm)=pFAuiar|uI zU%U$HIe%wY;CNDEjkfWjpAS&(#!C4p0ePOgz{z(o3HQ2TCR#e;XJ{Z{VZzH2IlLrx#h1ugTLGl{qsA!H=PVM$DLAXoLB>80%*fxop4oyQ}w)mP!5lLTSfwnS(h;m&MWZmHykc zX*d^o{{4W*@b0ivy1?ImMor{!958COCp)wmyH|St>-?`w=w7uBvk5_BimOw6Jl%9(FGVdai~;^5|ar%EevjLTB*KS9{M3;x|n2{I6X<)d?ks zpx2So*M}9}BKsGqisJ2$j+=o-l1knjBf6K*m&oL~^XCxjtH%V4w!fLy1B`y1RLqKT zpiAdmipehQUPkqIcJ~CKd;Qz%VtigSkF;qTy^P&!td}XY#(Lwuq=R$ZT8}_4su_RLyIhliK{CYy`T`SI>w z1VJ*88D{yg&ITh49GNjzz!gp7h1wb6!FtNgCF>H7OkIs_6}m`c!}r(NDu zH~Cr=dH>jaGO**BwHkO6Wo_q39}CXCZg79G0sG}#<@g#KAEJBtZT)=V(Pl}!^L1P8 zT<>@y^s-?V8@f6JpKr(Z9+Y}dy#DiT)|wbqi3Rli6O_v|xm^TEFOE;+;~kqdL0kH@ zJWG#QFj3*5a()^6^X;W9@)Lf7=<9_pej}H?Osw(tI=tJY+WQ#1e@v+t1W}bRUm51Z zma!L*&$nxn_`>@FhHy3ahHtZ_qvxx3d-DZCDx_DIqsFW#k2?4&Axc7dJ{Cw??^uk_ z!S0pbMn+}z7Cm48wjL)g-6o~df%)1?A^Yp}AK!r9mf50DnP!pJzHS2kB#ev8A!3`;@*!x##d>!ruEqcBzk34y# z=U|U_z8;bnY)V#w`O@38*EeJ<+z*f>sTJ9H8d;BnUQUA#MTT*EdQJr_W})ZHu$19R zpa(Kvn|~DL@V04yhKzb^SLryQ`ml|^tO~o=8`rMNED&3{M)3aBcC2 zUa8xf=2JvBSzfi<`H)@9c^){O7twKtLgp zOKNyq1#pIaq$%=90BlpAmp(ON_v*Z3P`Z;5kNHa}qTidu1n+zm%b%A0)(ySxYfHXj z2GA>J&%WC=6K-qQ<<<3^)w<{Sm7%$}7Trt5VDLtN5Ynrb{^V8R_o_fqu9`6ASv>H5 zI?cwBh`oOejHomeyx;i##WQdHy)zm1FL33C%-c$MeLU`!vhOU%`tw9m4j;XMFM9tX zc}3SoF@StNv_Q#V7hk3ZEGS-%O*_VeSR?M4IAQEwgMOo&nMND$b<}|9SgQQ#RGUsIC!C|rB z&yqi|zu)mYRjl(g|HgZrZuOr}1<)((^Bw2eXVB~C?l+@(t;l@&mCIF%yWPdPzx!$< zw*~$A~mHB-0FZmUy^F=lFIq$ULuEY50IMt zi|>I+_t6(t!2YEcdv{RQ7|vfzf%)mQeaP#hh>xLOw9Pjhv+J_hr3Q2_nzx!Z=@ZEH zi6q7Ro}`uYn`2EPNj&2d!IT=ec6}gruWYTXeF~-MUbPOaX%n#)c=!9s0q>CqzVLqg zo?5tfuL;c8$-_M{5~fHm0v}zRm1j53LS7*0O*kH2{BNsRUm?BX-XA+8TdEEi8NQKt zD93?j+oMd%MA)y7HIocl3J-GEu2bvXKSo!SIUV`sg16V^^yV!D?9gjI>1rlRH}pz; z^ke+eHKdniPL*KsjuG7Jv|D^1Gts@`SY64}xRL9t!^xh3+6UCZ!bN@yfBQIKe>z!% z^b~flbXQm6z_;jL#emJ8#rQJbUfnY1oFlBD*Qq_RbC3V97V`G zg!{cnTgJNu{dsNF&MYyrS4gijJY>Cvdo=*4JH6NYWE^N@RXtInj@|17mwY^bH@esA zT!ioXM19Kf7*-SLwdqL1rq?YnU(v&|yJg*x`C9#*iSvg-Ydq2O<$Qg(L`&;DVqR@{ z&bisyR2{s2Q1_V9A`UF*RTP_SK)yWa2jVjRFPx7l-~4(%uN(B@Waqpp=mEXdlHRNL zWFfDQt6r($HmGvx(8dNfr5y?(m- z4GLU?UK%w=b+z9k`d{NqD_V|!MuYM-W#_A6EJsp;<%*=Ji*RRJU^;kym z8~uIcha#I#Ox{PX=agKcer#5)3a*l8%1C~Q1%?Ui0(vXolf=4yU+`wnbK}Q3Yjv^S zJh9qW@a^{xvadV#|1qx?3Q@cIb-?$xr=NAyi~ObE*RmPYyp!p{ksVd%j^{&vA7*E- z(+iHDNH6A*+oCK=+~6yj*#7exmcXr1wbgfv3udoC3*XtAUGAt}(RW&c^sR02o+sSo zG~ds;My;LCbk$+`pAx=Gab`UBCn6QxaZSuF`YP59-mGs74&*kny{1mT7dK_j^BK5XNxvaT5dl6 zWMw@%E0XfVcP;E*oiZnWMkt_r{p;5iJT~59A*Q3Yg7SDV+i|&<7McQ-X6LP=Y85`A8b_AN>dC~ie z7RfOnTNf7=%#8iKp=2C;V++&9&lA@BE+;=^g6m(mYK+{N{y5)vm%Ezcqzdx9j@7@5 zv%;{f;_l<9{oW$AY-bfQ^7*zb^|Nl_{mS5>X-9xrQ4FXXOs*RA#eSZZJIideo$1DV znW$~yo{fg*S&jeqD@yHD3(sSoCkii%x7Da@{QUKTltXYx8NQFai`R}*>LWZa z&2NA2SD)bZ*Q4Jig|u9_V2PS9_qWnt%sr7_xKEdzLfQ8N1KcgaP2q7M_ex%qNVgYe zuSX>!kA@63-s_p>&0c+3=oS1c-{&DK%-3E6>L7}x_2+wqW~uE!3q4TX z+gc1Lcd7z%3){S(RdL|bF-~*Z&)B~QR<>O7P(B6St2Sa$Ui!%ayyuCS!v;>D=b+b@ zcPZvAvCxb6k(U1J-^lyN+NGuf>ZHdwvzmB<6f-N-d@c1!M!)SvdL{o1{K(&^0yNc& zoni-L!Hc#kVrOpbUL(8&q_{`7*5>Sux*u{I_;CEH0`%H_%3yhiH>^iDkIWbK%Ilvu zl-a}?bRCDDui>L)@4W+&UY`gbesGN32R7%tS65ev2aM|Zaj8$Sd-?Uuhx5~+_xoCI z`JIvZ8hGa`pj&`Nh!9>MFOt*Po^gcvnoJMzS?WjDBd+N?N%Gs+%TF71;=;?qpw<@!`#2_kj@)W)TC~BzP_KmrM6J~Xzl(n zr-6J6Nm{|${i9GfhljiF`uUoa|7Wy8Isv%iF&(SEq*<*RAU`RpY8UgnAZ!MQ}A_0Qw0(kXj? ztNs{j|0GzJ?BOb`AcFp0CNVtpTTc ztnjYKE#AGOQc6{8UZT&#C%rA<_c9+{SZp$iL7q=erRIG}#rzG9PfJVjI30TbVhO*w zv-2MEdrh4l-A@x=DS;wdF*2t!p+K)a%IS^(_Wq^xDe7aL({sx!%J6JihHI z8d3KWpqIHT4~|6e&+nZOGtBG%r5>dYj@Eca<0{mC5F4nXd%fdTJX7!nx&BpjFwTZb zT?riC!R^-U849{)dbbw%V)xo{y16p^AiCG;eG0zog=CFtJqqg8Yx$DRBqz1s3cXB~ z!`+OP)<0kF<9p(!!z<^hpnZSnO>xSrl3?V108^fzOe#A?u(Qa&;)7l&kP+o_eYOv~ z*Pu|Jr!(Egd(DLWs8sZUUNru>S*KVzk>w9@FObTfMbgy6r8LJ^%q!+R8 zxfZhwMR0vIwBO}MD45ajdrjSq{r+)?8Rsw>i|*yu{eJpkvIO4!D~|3`KBZ^PTE3Wg zx}<9>VZM@Tzbk#J`b#}JlMRsXs>j*Ra*N4aMxQ6d$7^all#uhp(zHEop0*;WbS{s) z9})_Ri4IQwq{Dtbxq?XTh3Pgtylj&{BwQ88+iOZ)_WTq6+BL777BThr6>wg)>24ol zW?DaAxtUQM5{%ZU^=RQ+Gyd)`^VcFX{FemJj6kQP`RQe(NJy%ZXuR>KN;4nbZ+j=P!d{U(G&A5x*%dnXH!jtGU)a&EF?e~`lQeMna zKrh3o7d73}Fkg18ShYg|;X6-6Jr0}-`2_p@KG8`7=^22kBR^oJWO^l2?ZBsf+prRWA|$DvIleR8}Ajd*T>G+{hA(8N}z)*!ys&Bz8}`vslKm#1J>^^ zp5Z=JhhC5Wwtsmn^f%W*-uqXW>9h@fu#yD>A`2VFIcZp|>%xB?5kGh#;d!c&` zHpZO7Ua1TU(zp8#Mu&oaA&Dui9PH1B9xQbl*5J}n`+e=c2ehB4@IMbm z_4Svc0cGfA#@iQcEeGpyr(T2cl|1Bn&R|LFf{bYkE>p?uk*FnlJ*M3KwlL9yeBQC@ zmEeErXjNt?;6HSeuk0LlFH=r!`|gkoRIh)VulcAd*1(6b9trC12bD@eFR`R171zJy zYxQ>&_a6+S;6?ZHnp2qk;f3^yua@CjeyarZ6O8STWrPBuQ691P7qNROUu|wZMtNhy zy<$jD7e};0udhKTZ*uH~UYC6pZ=Fj=*5j&I0Rt~3zaqL<;Ux3%qr=F0T=jBPH7p}^ zT3MgCTHh6)js1S;_lsa(AJUEY+M4)ku5l82nN188JYs-eD$cwNHz@w%)t#B2_A3PU z#fW3Lf&+b?U>hitR4`sYU)SbdafA(ofZd6=rP!OWKOb@q*i%yQc;o9)nepoPYdO%X zW*Ik9Q3m_HV?SH(eA!>Fj~U+KmPV>L-e)J}!U)jwMPx)%>hgR2d>yNocwsRW0#4qN z$UDM<-Rmm1Qk>5_bT2>4UFrmOGI-w)eQUhqKg|LAJ-ONo^07te)#J8}==NXMzx*VQ zZPA=6##QF@zfe7neti`0D>A)dgUpxBMc4RWQwo4+o15coeh47G@;cOK2K)0`e7#ot z3cmB|(H}-^*7-1BreA!8o(n@SDhlphANl^0uYZ21CrJLe$6NIKp)FP;sqUMQUSr?J z$)BdkgF++6P^IV)(3y)90^ZoYPT5tw$MK@)>)(3GasJ$FCX@LHPVY$L4;}7c13Vx?w(#|K?n#l=*iDd@WlLlh+6+D!95c^y#D#a z<6Do7ri5MsB5-|mW-t;*UvH>d`jjx@_w85n829m zNFQ;x9k{t^G7{?cOA?gx(>bGCwf)xCtly{ zL%pkP^Hp@OxbJ+A#GfI(Iu8;*BRwh)wx+Td-SY_nvvW2YHC@=Br@6Cpo1IqMcrTG1 ztb-pMV7?xm4g$Mc~+Z|TIj8^_k|U+>I3$K6!Xy_$kwB}JAay@*>?sy~e>0#CAS z(oFuLpz>kI#Z)Hj`PzSLamF+P-OEqobBA`19Nzs)&BxfDek<%>F?2_)D*w12a+|mz zm3kXFuhxp+@i;d63TLaq-b|W??seD8r_w^|cLczkoyHwE& z*zMbLiVq(XI#m|D;2=_agHCl(xE_{ zW{8o49{YZi>%QhVrPJtMxSBI=y)*cKFO&G~@BA+}px3b=4b_W(c!`fZlP4BLdaaza z!A|Yqg^O3x8aq9Q{=7EXsUXsj66wXbb*q5j3k6V`ctdx{J`_ltiXnfghTZFYtKP4= zd+1)Pa}mDtmyxt`h=&TS$Kr1NEv(frUoP{{%c#Qt;`O>rlP<{zXFF7s(3Fe5pUKm| zXlKP~q}TYK$ejQvg011-&dnr+f_DE~n9u?L?}0rcbZJ&0gzFQ6GHgdrCc^6@ z1D92~>>@H>wbD+8lMlOa)RY!JvZCnqNI#Zwp5ia(Sut&1j!PX-1nf1}r)j^0g6aLy z+Hb?LdlhP_lrnOld;QybB&k@syHo?~@!DA#kt$c{MZ}f$!)Rr{_<#Ppf1LOD&kyzP zXQk+RgPt$jDM>+vhsb;p9ZCC2bzd3yJkI@jcWW4+)z7`Nj|;n3ccj%B+ZuE)Hu`cc z_m}vemll)obD%N+=F24DxqQxT_`KFOKVy=k6IqYXH^se>D<8py$<6-?GeXZ7&BsN8 z(~ijRTWw8CO@1+e1C+O#o9HqA*^gW7q z|B4{n^8|?3t?f6NZA`mbXa&E&9bq>*pG1m$K4j^0+CoL<6^;jx&_{hk_v)mfJg!`U z^h)2Plg+Jw1Cs}fG(s}Nz?u06dbj4W=j-5hSD{A6o2Xv@wjOV(6N=i=Y{%^-{)geFFj+Liv!syL7WYr!T@nle)KzH z?D^W-MK?iz0uQh9E@vBQ{GUG(eI)Ir1<-4c_JbT%Gng-OhG0du8%QrdN@geTizT=& z?aNv{XVKSJ?YUY)q*f4 z++kC^>#@zKg#4l|^kSRxn+dUjUZ*Hj7a~WIUY2#D0Ren3an7PY2jd3Oy;#CU+r|Wt z`wQ*I&m7@p7X#^1;f-zA13`_Y^ucij>|UHAbCq(z8^0eQQKnOn=Lpq*K_1O4o{iXpnGjKF~y~6A@@gHOtifp@KXU^yXq)zy$lC5&6{3v(qQ-6 zIX(6A)B*H-{oDS<*LMB&xI4T)Y7BW9l<>lQX^96k9b`o2iy({h_TK6o98KUAy<~E9 zFR?R1+rAkh^Y!_MiM&s+3MgDOKS26594sIHKB)5pyO)XPuFbyo8}BtJcp|uc8?49v z_TZPZve4_o_+h=@R)6vO=MOXTq5hlaHgqrj*%ss5N0I9j)zqfU$5ys81$K1U^I5E1 zA5T8#JbfCwSL&el9>tv-@5N9%&nuV+^VR(^eb?e2^VhvAs^0@w&V^X-`Z&g4El0_X zzCKZ|YLvD=1v#%Wdkvb;EdyZmut~iwJ`4~jaBMSN!0vTjjiLO|GWz`WZ|jl5C}Fc| zJ@g`t6?`l{2K$#tgR`1b#9#8Yi&;-YHxMWGSWBEr5#1}Ghp|`E9htBGv{zEzpZ9?s zTASG|qryS=9gUB&5!j!nHL`pfOdChf7jAk-dg=@O&+E`lnVL&D3hU9C#(>-M2z=i? zd4Bd~ayxQfeNOGrM0srl=O*}EmD>&7D|*MSjJw&$_uWta{H1$k<$Yv+$^8dMO~S#* ziwFs(L)g8x4d?1vl%sq7+kBm7#+{P$gIqG-|~nJpjYqf zUo~$epqCz@6ZO8q_4`*<%e<_SHM-Zi{pKxJCy@IAJm-FocZH|`{cEQ^B~!w{oqMXb z6l5Fle8~C65sluB?_Wd8l2tMC(Chv~#=)d(@cP*Jb+2~T$-j8bd}rTZnuN;~dmWV1 zj($HBmw8`}I}hn)D5vrzIZhFXhU`pa2nq$a)fpJelFnj&zO4oblINB()~-|Q-EXhf z3BK#A3$z2rNW$U!X^&&ctAqLBdQL9!BO>h-v$fy-@BI^tLqTs&{IEdX?+bWLRK_kM z=P%O1$Bn%ml7OPL+u;RWUmv$q0VK|xpCvL028V0t z$9j$3Fz3r%@r52OC;EPJUel&HZw(*3?;o!TEe`0I!}n`@j~7(V?1cI9QS)Q3BSF?< zNSRB$rrR9ub|u}VgW2nt@3gr)^TvCHWbj`XaDZO#AFA^E z%KVWpb+gmLtn2Up($aqUIp@j|RIi&Yg?8Kckn1@{5#wVuwbJ0+*$N%S&>*1Pbi_;f zH}-r9=o%-#e1b>+!qMf3ZCb$L+wZ##+xPyKgI@V9250Ag!uLB)?iRcC>(ToC-frT6 z$J}mouaX7n>NppqS4GqB?>`#k0rmI0d|o3#K)v-&%ULq)`QrI`aZGMGd9ClRcYX8| z>~!E!H^;kwfs>*Idz_({x>>8pnY-|L+DlH)n3BH#7q1E>^Tw0rsQDTRyP=UDkMzlm+Dvd}SjJ1%oi@((t`2-1y=p#JgZjnZjHxRaj)XcL-hfAzB*&X4iW1j`~AC*AM}*o$$&vF zzN=NkL4aVG=spQO_I!=|J}9y?L-+c({ob`bXPeS*n6DQDi6qU#upamML{rL`ub(e+ zv#nN5rs(-nKGSxlsuJ1nZ4AnWEDp;670=|U2<9O0{)SUVyEt~Qm<|G`9%FQ`=<1W& z?^eD?g>U~l88%b(YvudVuT8Q~k2Sf1vL7W^calJ_yKSJB^~K?}KCn8kiZB})S5Y+J z6xqbP$Bogw=+zCO81T=3HPOaD43{``mcCEV_v)`6+#(EFE42!Nh@1O6RQ2tE?P>BV87p% zO>W(rern^rtSYa?@7IT3H)yR~XOG=kyMN?J@O*j8XMHb@%kQjIebDb8#SVZGvr=R| zChbek&P-? zUm6Im&01Y!jl%9VKX%}F-z#)4za0g*X6m$7?|T@~eLl-hVNs~#4| zucFZFvrcHJmngj77H!X>CuLuMp70JNq{`?;uSZY8^tO}lko{ib=*JB9gEAo5-MrCV zAPAVvJ}|P8#P0RmhT|%K6S|k>)TZ}=PmS>I_hf~cKPPSB{v4y}(^5|e;Ci6~BSFb$ zI%L213pq)jDqfG1F-T1jHbSq*_a1p~azc>b)2UcwPpKc11w)&}Oco-8L5|`c;cv9A znCE+{C!#g6>^IixV!idR+Hs~!vEwRud$loSzkXK(pAYS?o^+B{f?h|`4;`McL3-78 zTwSajdWL(IO&TEc3jO-%NLZW5(ddQf_j@Z%q-+Ca!CVrbNSSLe@L{$vceD`0=*pxGBt-KHKd5=?ET)G+dRI06)(-y4Uk4*;Ec{WWK0^dv&vK z%YZDcJbhh*HD75s(pPhFmNOm?&H9oFHgA}H|f7^yw@i# z?H=AL=!LVf3-?rj`C`6rtin8o^zwT%tUbQ14Clf8OO$;A-OG^gQp2O$$a);0kA5cq zTn^aQKG{l{9t`q_DEKut;QDx>mZ_2&kL%;V&DZT|?T|}vp;!0!_5jjYn6E>5%%$o! zNUzo3q026*uq;ORO5Kt7Ei4=9b>&*m+u39}kYFV_rX3y(M7Pw9v28&A`oHCM(V{zO znGe3trz?F}p)d{Ri>ur5G}qDfy#%uZ-;_?GdlicrGHPx{dYwD(AK@V^2h81ZbZ@nT zfWtO9<76J}^?1Rbt2W`t#@C~dalN>TC-lc$|(}xl{u%&{Mr=}z5|C~3}_}&k@7Y}iyUULAqDf(<0Y2pUfni;OO6VeuT`)7-x;S;zXt(HRlnyg->`cX z_85uJ3T(XBSV;dlV=3r0Sfckj`7QKvVHmMY5<%vR;0g!s&`>1qoVTziuM4^td8ze$ z5kPufx!9^UU@ZsKb&5KVcCS1i^8PS1p7sAE|T7Zg!_z=2R($*nv zf^4LhSBR#Fwy7N8FRb#d^$i4k9ugnO+_8IkIbG#gynyayDIXkIvhOh7_d{xRGvzV6 z;Qsrh&&*F5FTnlxuWwX*a^pqbZ~JMy%1r92$K5>|X}CCu?!`rOfx$f&+3#;t?;_Du zl>>oN*B2t)0>S&ft2*1MuzTIMdZU)_gzi<_FspjNLkn-OZ67*vpOZqbo6oD4Yy6td$e;xO5{+*tR3#R_5Hv1ag%j>(+^@LQU*Ng3Ma@Cb&fo9>=nR{G;pkw!F z!_0l`{j0HE9BdOt_xiWjM~x_B7DjehkAhL>1&AhL|Dw-U?>xHl{yd@{|M{W8$jQ;n zcj)t1&QWQ@CyB`Y_p@Zuq^iMEz%K39d-wJL@YeI(v)BoH%>B#b=M9ng(|9}|`sd$) z?>zDTs4mg;ZaA;T*Rsbb3qh~&%fy$HoQ|*SMRmdXMI{;fdZD6+tL>K(K7>nm2oDmwixa9?^k;EK_j+Pd1gyo7`zKbtx}0p^#g+tsRyiidW4zeCYB~>d zd!I!2^2^}=aJEPh@BS5X{qSzjKi(G&T&5X$`Nw+UmaXN_^jGF9L_NAUi&;b^;UY|? zJz0j)y=YWUx9{19^fK1+Q^-)40xy;}6EX(`fGp{L!L)nW>ruUJF8?k$y4SzWR}QI+ z>CLb3c}G*(+k*%0!1W2TZ{|@NX2|akt^UrIF*{WeJ#?=pUVbz}GRXayM6(^mQaRE< zTDa&8;%LsbE_wVpO;niV-xPR2B5tv#Y zl>#4m;?ggF2>_Dd<=xUy>|W=O7?s)YK)>H6aM`KIc=aRR`=jq){Z#ig2d;nFirhTF za2)op>F1x69$rTFFM{KHS#t~HaJ7Nl#Dg*DUY&uxbECS*em_3da#CGD8gMk`FLKQW z0Am^kI?a37z3AmHO;UZ__Tu4ShF)gj8nZJ@$nTvHeEUFXKOT$o zl~Es+bU^pg`bA$Q`ViT_l-gAX6bPh2WDdzelI4~6hf1ndi9i1@y}EAkf6Ku;UzPjx zr1dl4_0c8EC+zNZ=oRtdC98Zl(yNwzr+0AR4IJ&+huiwMqtB}ecdphYKR|l*zC20j z`auc=DLJz8ZwUnKoEfU~eAwqzyz^Bc!5<$fk9Yr4)iPHWxCOn!P83U*U4mZZbt0;7 zFZ`w7YYh+j7GJ=P%4>v}EuiO%ejsH!q5$co8Xx()(@qL}t=aL_hAt2Ua%u>T4Pwui zZPD-jRl7HSUOj3gM_KSkz8tle!+Srqni}ixm^0OgE^eu5`TT$AB@_N~N*Vw2khPNP>~}T7e5rG~xCgpJuir^& zrMA|8@w)!SPTTMyE=;qJ)_fV=t0ORrrf2}^^?RIG&c9F!>>Z?j{AMNqjHoYG@}*+; zqNWb5qvXOPUl~V>^OouHu16C8=y1&`=%u?j5@D(hy?TR{D7@K_UVbLRA==tWxO1*y zbai+aJHfdcH=aPECm2LCzDBMW-&8lS_dQ-A~EOuYO?WO77sdh^YlP3k-|6Xyb{o>f2;T9&F7K%s@*^EdM{xX?zH-`tG_$Z zy<)qhH1A$SdX4sc$jxPu0k~r0#Dett#^c|2&`AW0M~)8XvIc_f-OKxNo=caEr-iyAlJaT(3^g3l8PAu}r_jHz1 zmha`dAoJzNwrz+*Diw#X*U@vO<%UGa_dDF?b}r9OKWyBA5d z6+@N)9@j_5HxB9j0(f5^O&+J-F^z^^TuN-Uy4EmXm9oUkLx#wCl}&F~f#tD#xa(&( zeW2k-&sUqpsIxM``d&7>B235U0>Fg@vGZvFdp*iK-}K$fgziQWyb*L$vCvJs}DLiR~@55+2zGJqu`OErOS7Wbq=uKS0OfrKO9`C0~8sE=z63WpF&B zdi(PF`6?`ZTzB3TJzod=E+78273sx6x?hFsiWrz^f0`cRxbl9QP=n=E@&C|kn4>LF zKp5}-RX4G8j*}gF{j3$54lRUU?`RkH&N?Fd7eNHowhOPaagMJjbH?!4|26U8FNe;~d14Wn-0k^3idpR-E+N{|436xJz$J6GO+9QgzcNwDuXAtW=4$$fX`Cgu2cphI7fzIM~2ju>o+6P(|-(~LOIHX!97QdjMXLW+OU`+iH((Bp8-U=#y zX%J7mKPsd#5R_f{?f#MoyH_NU*(bgc^!sgG&!sVm21mTTN^~iDbOPY@F^O{OhSfEA z-ed7__Dj0G7KrDyIJ0eOHxEC@9h?Xfkt;oh`o5?d%dW1lN67UYPx2*Nt#i^KoX(}l zrX&!A=IdoGl3~B!b_CzXi8ayh9|J08BU>l&5j3-%<3cO36)MlJ0q%<{Qx6B=Bv&I1I@})@;l0H zG4Ic@RF>c!vq$e=1Q|Ya1iy9h&esXf+!kpCxSuJu*2cB^HuTEPFj6S6{EL_SGEY>{ zeOzUN>idH~(7ibG$UkxDBE58kPkh+-P8u9LRbCTr7YyhlF3w5~V)yd!%|XsdtJ4~e3ccGAKqN3RCFndkCs@C&z#>4A*c?1Jk|NV|E$)h}9gE-c_ z9x1iwNi)Fxm>k~ooxd%h*O{<}n<7KVeBp|m=c*(g<0eb@%?r_@=WAxdh3e@D(ra6e zp;Gf}c~H!7;^gVyEAKy+-Eci|7Q5HwD_1^SbfbH@6rDJx(7pcmbV9&!owRM@YhGM0 zb!;z-L9eD17UkcA_twtOdHmzjZet=##PxA-;_+|F@0~cy2Eqe52hh*=o%Ug^*?uAy53p?Ff4r%OM^Ac5_KNtQG zdKqfl&~xrBTk~q`wR9^RKZfvH{T;b?)y|8F=w4>Jji^l zdfj2rj5w~2?q%y)%p7Zl^b)`QR^&9FJTM=%BDcL70uH8~9~v3J?p1T*yP(W09$u4= zl)q%GzkfBl&&4ah5a#QXldj&u7})P`btWCMpM+i-n|mWpw<7a}3kq2hTe^qSAU;02 z_ZYetO_N4FZwAt9di;Qj!g*PcNE{|2BNYOi15}$UEU|mZ*b~oP$iu_ypFjM1>l3TF zSiL?5zk7OL-WqxV5+*C+L|Bi7iJ}?&`N)31`roPcrl|MpLGSl6`f*XuDr+eU*2HRTy`WkOQGx#0mtCdH^ z$a(;J^~ADsCrcyik>En*sPdQVIFCaqSKPLv*JFpNa_7U9b7=p2ULBS_VBrts!4=9& zvsM8i;ARn9%%NfI`J%S<^FLRQ-HU*Gl>NSN18%h*NBFGy?=(WM82k4>6;z>D$ZblV zM>moATK&859ef|)WR30>B@2GY%^|(iE}jaXJt7C#3hg+~8-#%J?@M*Q53%Pfu$B3s zK?8QLtjVPG<9#q+ZedsLH3p#9omjq(2h7k*PL{7LRB(MSGqz{amye@+wKWr*tn5d6 zReudI{^ckSvdX#1HCE;?EuRK;hj8rcIoixbec4Rt-=ko&jgA^|T>m`L6Lnrd(O>zV+<)=}-+!zfnCNJG(T?NLvAcfB5&ioe z;rl^l*96kbvEk&S$Z{D_v(NE31!D-{XQA8ES%E!Yv_GlbR0^@@t9aC;twDR!n%53# zn=su!z9)UweDe19F*r}eOr3kZY1jbawOWtNmlr*L#-ryeTD*oRS0CwR1NehdD`Y_( zz2L7Z=@9VgE@4FUW9(kF@lWETZ(#S5xgfnMw~cGfEA-hCP9YN3BOBr712_U$kN)4{ z-LLIKuCEffd}t2Pi^4TO7}_M%kM6~D|2^AS0@AB-atlpNg#z&KI$_TDe~vo3-tA^# zi`|P`Qi6AMCwe`Kmb-flkI>Ra{VgGvmEsQa1KlD=i(kf(i9eSNLPQIyj2%Hu4+KRx+VK`8L;0h2tS~pNWhuzC{_>$!KE$m)?OI;Cf zX;|01XdTWv69DK%&l3(FOu&4Rge$WvcdtKB9MZJhqCtl4)fg>KQ=EwOVvXCMD$Jt* z2rT$ipWO)sg+7k8T_3P}eLSN(x8#W3i%KBcD$9J7iS&x?d1t!P)+Hg~TG1R`a z==C^ba!T*P$~njXeg7Ec+81Qe4Zy))5eeHGLczUgzw9qxvFB@Ia?1hZ4e0m9{MSx0 z#M7_k>vQYmq>32y3VL+E_4jS)WqaR2hROIZ`TD0IhsbNnKQsP+_o`R3dJ*+d2ABm( zDc#f$1;%ufW6H;|d%YsMF)gi%p0CxuvflM^@W~Z*?!${~_uKk(irk5^(Cg)ZV;!Fr z^y2yw-jv_Ee*fZA;OXPJfnJZBG6+4bqLKOHj-$KyOGFOX1XDEiw*-TyPoDVK^tDa(&g}B5f%BKcnU#%4m*D-7h4^`{C4%*zr)>?Ks%hSf zKCk+)=-y$JK=v~OkniJZGFto)0jSUyv{ltx^^G8-uv#g ziPwVl+YHz3U#t0BUC()bU!+WJI%Dm9_p7-i0xI^NYxmo^+EJIfe*)?t5S|jnpCb#6k{qsScTv4FgBjd0z%$L({yQiI! zb4KRNKK_g#Pq7?0(3Rz25*rNK@~%(WU&LOIL`F&7Qt8<1adP|S5y?{GwSKQ??QCA= z59_gAKAQg)G4v8soOXN~w*G$K-Fkef+jGVygLloV+?K}3X(!B=AXC&b z?vLkb_olw5&adpR|4$vm^{?o3%klXmO}O&6WcN-fqvy+fsiDxo1eq^t96L?@R~!(1 z^&!OEHv|*~$QU(#XzT34>p=AHBM;^>#Ox$q|NDGgULPl2KC`cREgxl^(m4XX0xPbi z67PiTtEGYZN7$H=`!QFq8>X92olQA|K2NOYl_^>j$i#{R37=@6$W4U+jkmg&e0Z_X z_f_GUWOl9SUaS9(_2#d9^J}uCpZM0iBFI|fZ{L7k>M`RFZ$v_`$m>=u&)tyye(-Eb zdcAuiZqnbs{17|(ygH!MDMFZl-2b(o?8;D#HV!26iT=>~6auPekCzGRVfV_*`c3>G z9J`kW<)K@ykEGYUa@w~0<-|a*W=%b^D<05`AxDp*)%`E)UrNU`9kcUrC4@azaRA-x zw+)GLQ3djR-vx0cv+!YMFuQ|A%4#+Qe0@UnZPo*OJx&=v6SSe|!lCy((?5WTov!_OI0(vjm)%AlF3q3i!od9DW|@b!Gd=PH}e}h>$*f zPvq-Lzc;%^(-8f?^Q!e10-PJ5ixoB$9TexO;NW(w^7Ry_$rzKDUM=y_yoJHm99e2JUUS*3WZSo!X`LVP+=h`uVDe*?r#ukNfSs zSewGspUC~>O>`lPUj-Gxa?06V|7SlFr${$5qa}7Pef`RZhL^F|qu-^TJ05`}YxmnG z-}_x32tluLUe4rAx1iVPHtJq-ek;WMwfc8)Ej&JAsD+-d;<-(aHZdXJhnaDSp)*dw zfx*b9rc`+$;ETA;6*V60UetP0_Z&0Py>OOVGnxMDe{b$JZS8#XIR9F{C?f8D337*C z2YGn+&}2fd3{&@TmJZ~4j^Bbd-S6o_oc1}DFODqeUfnl5iv=8z=OH)kiH!<=y$?KI za7qo$2?43QMsve=VE1~QnYq=+7u{?1@4j9=T0S8bc%uT>@9RcJ_Hsx=FBZ4TuB2_y zE6z~D_||FUe82iTVHxy#^?0mDzuYtOvtAQfkF>gjq2C&1z#*b>{VmAnhs zy;x@}W-{eBz-wTbtG6CrA2(ambG7|(zHdaPs=m}Pynmee#+lFBwEnz$^_ln#Z3p_i z$|%tAR!Ifv^}%c#by+F^lzgr_)vY1mf^fCAq%ijTAx&)>N^Yv`wLY@mdQS0KFgLf zGAI6VTJ16mzN=n->h5}1rQ!O$j84|!nQoY`W1O+V)V$D3aydGl^AmEOz^TMuKi-;& zyXgI5+I1Se9tpiGKjk?g>yee5yY%ZF01OsG3L?IT04rPf2c##l-w)yIMeyiA<@t6K z;Ac)sPNJTHTXkCfUlBH|%Y53*Yx&wc%e6aO0QP$`g|P?Of9zlVf0(=Tc&fJdao{Ip zh>9eXWGGRl%(K1BQ^pLLNl1fv$Pfw{qmmFAGZU3LL&=aiR3t?j6p6^tpr7xz&RKO& zzT17B`@5gdKX=D{y^eF<=Xsv}thLu(JMx~M>31&teif4{>e!%OhImBMJl5Nf9$zbY zwb;<{77B`kLQxX&viG4NoJ3E(`X6|n;i|oUz+jBhn*|cIdGPYFew?{gbBGS-<(=Ic zV#o=-KI&|ooN| zB{*!&1H!%P$tuO;eBk{3%apfr)@mzzi9bvVW;l)aRbM^5Bzq$YU z*UI%P>S0#HYi_t6lZw{gaL+F2RpGuMR?v;_g?)WS84rGmYv}X$Sg)_qhkdSB!T4gK zN^>kW7XznSuhMF|hk{Ktk)NcT39nz@Wozln0txpD5IH=+O1>7Ck6%c}9w$q{`1+JF zKs?O?y#$ZHVEQt;^7wJ-icwZW0J;~)7Rj-lS6qOb%lHl2i~2v~9)pp}O+VxDT^R5^ z*$Y(p&w`5WTNgQm315#n(!K3a>-p7>AF<#6kK-@#l|_~|dVd6$k8XVjJ5LqC{S^9( zJ_R;wV=aUKF#4I|GHl*HH@Qx;QU}V~$(; z-?0M~dcVzKlObUEQ-T`bS;E)**7>`db36YHFP)Aptsib7fAyNmaH`m~2Yj%CnGnMzd zL6ri}n;qjppp4tX%wf4668b#WRHw412nqUlXxL2r@vig(xbN?;UP#61M1F(@myi2- zPZ4u0uea|vxORN3$r+5DV-7HGfnJWG*-`2b@Vx?%@_U&rIfx;mpR=Tio{up%MsjON zUBK~NuU5{pdbsyDchaUwrdkSIwP5>JM1B^OM)sw9_Z}wLYx0KUs9{Rh&%d+a|4;R+ zm-Y5sDt~?x=f${jX1Got*00cy2Y7^3u;(-U__{}OhkrXGdVl}*fy+Z>#1$0tG9+%= zjm>WiM*e$j$tG_FuzT1ju>SGc<*mMRO`O&fzCQqC+dWynIP`oB*t?tevg#4s_xIQM za-zSWdhjMY&dc^_l(u#Se16AWuL|$Hbm-;Wf5-Bw8on3H zUbbx;Qg0wcV-BpzvglqFc2=d4b#CBf=)D5zHt2P}f`db6P8kT~hUoTSsB>GJ)~kM3j^ABH3rYb5T1`*nq3EPy&>F7 zJ8*|;(w&{S_@eF-pnoh2y-eAF*_0BDuMWeG?-*tL`H9qEOnj(x3F7(5(ke?7J-&7* z-nlA+(V}7?THeNB960bGN%s>t}s#3J&m4$(r z%*^^j#|V$F0GmCs4!N4WU<|yQEYoOEu1m zPgeC6eHVUwVe|UX^!l|-a_I5JDN`QovDE_{Q{~zC+6Oycz+k3Ewv-*+ArHc7rn4Pq z!$3;WL@bbi|*!o3c^)HoT$p7W3QUl$D`kJW#IUS02d3!)UNa9*$1^YoB^ z#?MD=d>wsbs(-T+z1}BsImN1Zdw`duHi7T0V7>2(le0Q;TM^{G_*|f~CLEO9Bc^q_ zOSsnsa8==!?<%}p1?a!3n1cGW`*8!t&+y`FN|8lId%rj?zBXmv@Rieq^*-$TGX|$O z&`aB5*1S;=|9YCG@lFPL#|p$v($lcR1Ums?VY3xC6Zs`6ZLYnqd9XQ+u_u=l06= zYj;Px!jLw4e3^QCY#l8BQUM@hds}b94AJEp|@{#md zjSinSetc2u9^~;XZbuBiS+;FZK##BRa=Ufwn%sees!{P76&PP-YsTwM`vibd{86KX zg)rcaiJ+bjB;2cK*w%SR6uK9-uKa%dWj^QU|G@UE#RjjIZV*@C`c*Q^e!Y5TSns1~ zZ1?VNz?6h{lY4|1fJT<*~_iiPdJf}9`$Nw#?>#UwNJCTFwo1uz1c=U0D8Sx zs8QE*#;^AQy|yk9QQ1iO#l_hYcXY4wsjkWj5^mt0&ebz<127*=S!uh+-Q>WDQvMmM zt>NHP(606U7ydmj{sZO#Q972OUhH5@D&qiNK2CkHvAFad=3`3RMCEOL=oPk$L~|}0 zdZ|{A$Etq7&&Oe2o_8J3su0UPt!ex6SMSAbK(9M10{e+#-=+&R+W`pN?C`{|JPeGU z6Lr|%Lbz9TSu_3K1Gxm(dzK=#nIuOtTzrkGpREdyhF+o>J}ehBp_gRbl6fly{`xO$ zzEkVn;@l^N?zN@snc?F^SMb?*0K86y@pXUd3-fbEIgsZeH}PhFMKxYrJ~ z%$K?+R_|3mDqmqQ}=d(X3|wM{YnfROsnU754nGJl>8BtnW4jKuAMJ zjPhw1I7E)IiA*QlYro21YM2SS7xr_AEUlN{D!mVvzdwGz;6J?9Md58KTyL^nY-4y% zDDjf9Lyaw;jY1=w7iiH%vl3-9R4Q(KU}X;rVeciJ_ja zMh-CAs<~9{3jYH1G{EpC(k+I~UlFVm_85@H5W8Rw z_s4XbJxo}&ZD-NELi zj7j}U=ylcodva2S76>n9C@t;^2XrpiczOSU{Y1HnyBBPgR^gS^{I1oHeZ_tPQ^%f1 z_vsz(dfK;56bHTp!s|mVuZL?3@}XCg@z(<(>VL>b9SNT6n}d;$`zE4N*P{2U^33m< z_!Zqj!>gskX-8l_64f4=v5C+Eh2OtR-24;{HYQ)~yw6Q|d_9!Dkm9my6<*jx`R%xJ zzp7$ABxTqfM>__nw!rIYAhmO>pPGA7IY^G=_V_c{0Q;;VQk zkXqBHw|c?*70SuK=3CQ1FM0l5uATQ`e7#!7kb9wM<^EpfR$HgGQ4pt zk7glHJro{NcL&^)9U-Pf zuzqnk-zMuzRsc3HS(WdKh5)uKg}j;%gnNm+E75c!TYY?`njg}bciDuCN1U5E-S^_b zXf38$0^A7^*) zi9?;z=@q`$L7}UgK3jwUp5o^5Gns^Yb;1Yz{ku;iepBiXi;)cBwR3bEw}Q1yc{4S z&m2jr3kIQw1@pqH37^l2*gBz@AH8}n?B~HcVUPSTClu!AnQ&f7IcAr0HKA7o=hEh` z<>#}WmfIe(M!^HDPfhQ-nzeHOWwdF3c0fIP{URT1S2SsI1L_TZ<#&r=J{}9u?wi~% z4~l&;jE6ZwfVidM*tlY&UuI~Vs2H-X#(H|h$idVnf;vbbsX6?gy zb$Ix7&o0LkcD>K&*2>XfQTIYp~CvbhvsqGR&@A;rtgIh#U{qp$|n~#TTF-Mg%|4_f`2NSaD zU6JTTRuY*qbgwh2cSma&J;4jifN0(F^Zc-HPb)1d)tXZguo|fAuos^JPlJ3Sc^C=z zB6qE3Wc$1dFQ%dIo*$OiJ7&)m=WiH8e!MT}6>z|Vafu7|s~`4;y}IWByW3TwY`M|H5c9FU0S=gnS6$u;hD(?o}8SS}y7C30@snq&lbsy+ZqYJZm|!;H^{Z??_j-0Z;q^XXZFa=tqvWuD5wq-d8i)0(x$sr?l2th_ zzV;ApX*4|Xhxqy#AQZC2Vz!p(UR3+tZM;8t0_M+Wo{9W&y>{QZwzCTKN}!?kgxAJ| zGhmKfSciPzzsHO7ySj1kbGSao+-`6Wcns$!sA{#9=9=Jox%h^qeovb}#Me(R$DB8z zODE91LK|k~==XR6T}$~LJ=`$9yiU{Y-o0GECikg_^pKwcZ+q%A+};v?eR~G8<>r%? zRjfDp`8j?+z`O8FHVDm@^=N^w1meOL+4t%)~kej1z1Yv zEhK-&jfXUqGq+O-!}~>Zv`(G&@rCh~f1x*-=MVdTVTA5#4ZAoaZ}0X~@mxmlze4<; zunEq)1CpB3rtb=HJ$Y`dWBpcBH8AX0ew5wwG`Kx}jMb!)@c6Rcw3K@~a~10WumSYj z5yn?Qg!?(|7`WfIk_8aV@`3jcP5SGl6HUYTy5jB>FK_yX`}uypzS7{L?a907_ooe= zF?+6&?*?4lF)p*;;r$dT4|DwA9aaaskF_Q|Yd;Ot+f*VZh6(pN^6IMDk%#DB$b*YB zUV1CfuVNBg2T$q2^_U6^JIu2G53e_e9V`Tu@cVm2t@zG*p917$5R&2A+e#?IZud`oh(EVZ$n*57A}HgyY{IzhCgTR5%ta zQ4aUp=1C9~AZvv2MV*xWC6Ny1V~nVY!~Q=#|B*;J@9%q2O%%5HMOmTY-`!uS+-i3j0cTZJ8OHm~CTp@0#RcO%I5zwg($1-%+s7pfj=$$_z?E5S^3!JyvzX>Z6?!o65q4Np7tqQ_TiTPIuB4!FJ@ z8&}vPi#4%e&v9AtNkwYexmbX_h|be zH}LGlN3EDZd@p)`-*>L0!Qg$M0hO!<;a*e0x$G>v(CdAG?&Z%jx+(w}o)kokKObAq7W|R5Z*!(ix8l4M z?WV1&Za}X@1LsU8KIkPw7pgXT3EwM#|5{+r$pWN&#wfU?7~She$0u!eqUGxy$*1=m z5Qbh8daw1p*QkKkE$8m%HwOckBVXf|uKxFUO*#3_#JIry3oklr1hUvcFPmod&phy+N1Z~G z@hbH5Bi4&S{n!cpKzKj0+pTHSQfd`2xqdqQYI`u4vZ^`56F~U=+y9hTvAWq~#Zz!R zwA5E@RjCf+%j)K7H-oRxD{4J?@@skg_(I-pDmihh9O2>C&UnIse*d=kwK8`VE_WdK zWDSjn28^$7=X$1lLRCOgS^k-VsbCQ7S;_M1ADGWMb;8~z_Aq+APd(euB*3!VPmr74 zC8fWE{D>zQUt;%!q^}jg=bd~zEj_?u1icuX z=Z~S>e(3p#^?Iuuk9asjuS{j_p@w&A%j@!Y5ZjdpgQusjiLpi#o{uJ!#8WX_(7iBs zy&8SHg>dt|fAxwysd+Y&;YFG5=-OVUk13y<&X*h&K!dTknsmaNjMFgkJis=WYvPpx3du*wHQ1e~7Q2 zFBBD;!z^Wi9$#(uDz`~zy8~5D#h&UxIGc8g2Cn1!N(<|gnKFS zv5z`1pnC(s?aiH;vHaOX2;^+dLL-fP(+%PTo-1fB9LS6QANi!2a|h zt@-8n+R*>--ptqk9KyDih6oyq~Q%eZ~`yk-O$S^sUi4sfh-7HSJtd~BpB$K)fb&UK)4r8!wCTjw$*zLH@r-~hnYZrxIwRa zS<&XyT)00|6osl1Ll5*i=v>02eF)C4_7B{9G1-XkRbQkqmD7-cM9hxxw-lp$t({a$ zN;|N;{+>F+oxB}-ZON9e8g&)~R@Rq}1-JWxBlkz&bUOcgUWi)9nGepIAi#VnTHoe9 zzE_dK^sdivej>p7>2{Bc@P6bYQc9iPw9v~{K3&f04fgu#FZb7eJixv~!XMfEprnz$ z6WvRVZgB9SstdS79MTc%2yk9hBd*X2T0T^E!KPi8B_he-+A=Vp= z4kqaJK3SadQg^ZosEIsKIIID^7%2H?g*%17+V>Wj8)kfg4xQfC@BhGle5>{P`L$!} zbCNt86Y=L)sjV5_ZIg!m>JZQ{%pZgCrP9?!1OCqndv$Q*k3KJ}go9s%L9q4JUAy1;f^g@h#iK9&y?n&R*Y8JoeoVcb(@pVmxzdN=%r>M)9lCuy*dYui>xbNIlk1391B>b(Y^BY4B{rwxd3x>Ke=IU-ie`;#kQ4vDD#gdbmf7nc^*tAl~% zef?yGy@Y#l6hsBBFDBfpJM>!lST*b?9*}#)ie6aRi=&SIyW#|X|AqZN{;pRg_n~3d zhj2g9XC99x&1R)RBKSOYus9eHRq@qr!Tcj$8@iJT_ZsG9r^%&<^Rm9$)){3wto&MqiqUG=Q zYp6o66Qanwhr%MDrZjFZ%RWEAu&%#2CW>$``THdMk8ehQUM6Pd%82h{UJ$TGzt1?j z9l<(bkHte$9kcUrJfx2zfFwSj`@f z?j?KF+_R0y6|h{~c+TW0oS(3oN^-3)7Y1k60(XHPU*KNE$v!VkxYxmOp*?qp(fidO zF3Z=q_mU^LooU4LO8vs+mbwP6|H_;GJVQDR&yU;6MmfI+!}z-YN{LoJ><|6K&p#y6 zz$))cw0f^y`_9SK~cS0|^I5*HQp_zHM`?ex?l_I!p7GU!FF@%D~u35>6^$A*KA zp2GSie3v5XnJ1iI4cVpf(Y6IYzET;yIJz3@5eb)OrDM_P*VBa7-MXw??gCsJSv*7* zVZXYA-rJ^%NdyQkmRAX|`2s0BN_#I_!t2*lduk$G0dz0U^V}{z#clu*6^yeG{D%BB zzNjx6X}o^}`+LWV>|T;K=+!hH6Oap_SAIcqQM&~GdH{@OD%ID6mB@yhVb7=%SMPPz z{}|7jI_UN0)|l=fg$Ss;*QKG&oY$0V-Fydipguev7pU7_s0Agy{QwLT!ceo5RQV(3yr&qoZ2 zFJl|uwiWAr>K(1{$2Z{pRz#H(UufN-7dyiEuv-Lrog*px$h;H3eqp-z1tk0DB3$c* zgLX`!&qwdFXb^5X>jH{%o~GSp!1ubT$dY*JMKG8wO}#y;M7WpLU9WGJ7xQuBo0aBQ zF++*cGe7Qz|F%S9PxQC@;7CpYkcbdL`JYK&&QH2DOA-{tGK@w`^EougkIVMw~Hq!x8S^vbzAJ- zc?ibW_RA-!S+>IZB^6!BwqD{7_3P*BbC#tJs%4OFzXF!xpDex629~X#i8kJ!Mt*!O z=w+m&SNFgKjvwQsOTH_XL9eby_Zy12V0_uS8Md?kVSgse4Eo0#I;)XPn%6#;TF~Rm ziQ)p;hAdZLLv_Do@A>^WugX__C5jEQpttd8#kp_6fYf?`@7rs_<1179HQBAa0#vWx z!V32T^g@Wwb+wnk_@aw=9!75ny|iXFR2GH6_zKmi|GcMj<@oZUXWjo?72PY^W*|6W z`FdJ+8^t;QRg`Z=ZlRq zB$h$TJ`T>Wx*Qv7+UKycmw+4frrq=SI%C4{XZ~<9` z0+!aX_+IO5eNVW$2LI3f?p13l2#>E~mZ)&SI>O_txbD%A%ni6c`q8jH^^@iOasyM9Lz2-S<_iOAsh9%Ak&CC?A23~6y_Z=?qkVrO?7!+ecD2fIO98z! zx4cKo_or1SO7R!X5WXH@*~_=~JbFKYjq~4+zvScEYu3@W@cQGP*A+Vy6ybUkvcuW} z^N-;Cgi*>;us0)i{Qk>)^v^$FYA1X7=6m#d&rjP($G61^2w1!jQX)Q zec-yq=X+-ge86m^pGt!g;nyEuSx;mrB>#=)>GWxm^6R7{KRy=p!c2B{w5r4FLyrfB zZ+%{Z&!do-5?sucfnINA`WQExt?XsN==+dXA3eUJFli^|wjBrL<6I3H=TvZB_NTPX z<;M4chl#Ww*4z7lfJ0qX**=7iA1ytNTcV@>hS$&U|F2&4N79_Fzry%B{Cze?y&rn@ zYsPgw+5^4n)B_vDd{_2TCK=&B`v5(@wzGvsm0vstZqXm6F;&sPd8r7zNFHt42U3iV zskxo=0m6NpJ?#I$c$=NCdR?6h;a;J>H$u2b;rhbaV@s#)sDGKC7*TORWN+g3>v+Z_ zVg8MDWiOX6m&;98aeu8et;>1-qsKwgn=`DcT(JN0u;07siugY8T1ZDjHrg9dvFSE% z9U%OEJ{c(!7vb2y@%*$E;>)RElfXtZIKOInz5eZX9r(OCN9920oFMo-u!O>a43^53 zy$ot4T@J3|`L)xp!-I~E90N{i#-TBfVLqyhi{`~y?E@Pn$Tnr=c!RNlf?Hij2(S0Y zI&4LSr&*Um)EhUG&q1!ZmR&3Z7t4t}Z_wv)uJ|Nf=k*KrQ?t-XMZ zU47Pb@hEye+Jl9;b%&0FowtfDHw>%d^08U_XrH;A03cWGd%&^F2auWFRb`m{N4x|- zXA$m&@NRGQtcLrq?p!CWqbC6C*Mq$6rM}b9i&9oxi6nC6_!|FoO)W(Zz1~+?L^GaI zIu5*Ic61+Qg7pjQbxIP@eNGJm{nHln!t;cCrC<`*4ZEU`hcFas-&4;oudfPuH4sJ8 zjQq9Uw?`g}%1h+NjUS21JI#{f;QZIfy|&t&rm)_ldG*L7m2?lGdr2~y-6C7#1gMH{ z`)!bcB%gEY{K{UQhc{p-ZB`x+wI62T7)eL>dT%-MB#P1zcFfxSk^(IcyUVT;y z1f0CTtMy(GXz;IEcbbjx{_9Zu8N-D`tM{t^UP>OT(tsCV46}yLo4)MRt zuMP_hI$j8e@fBs*v8l;nWv`mVY4)r~=w8glOqe5^9f4BFfQQ==cz(osDdm|M8L$L{ zuk_?p1=fUnN$S40pC>K)StnP1er)?E$Y`)aB z$Zs&dvOTMaB8cGpgmF;vTteZ>_gk&;JKr_sgg$?t)u3;-_RtAn09b1^6JUHP%{@~c z!2od8Yk%E~<@IvnK6doi|AF%(V-t@Wquq5>FU&Yk%*g!{z>;t28nHY6`upOM7yDi_ z!ujZw`&ndT)|I&NqcU=UnREI1tl0hdQdCMGtR2UnkJjd)qUhUm57Fl!_Zzu^etl?3 zs53p;+8NXy_7*Vx49DBD>_hY2Rx%)c>(@Zf8^Pd)+S>_pF2d{Am8LA^0@YP`{VZ7l zjKuOT23`2^^_;w+_5pl;2d^l#Mo|j%(sJz2AIO1Tvu53Thimb@hI6Vft1;d|hR>HV z-93fw#c4mt!>qXcylTlYD}`}b@6Y(WRUY~x10K45(2eK}29$oqdf&nc_tHDB_Ndu+ z^Ag=cA(laJre}0<3d{+CJWdUTe?q;$hU50k&y{m6i3ufQ)kY?T@mAd!5+w zT=}f!-|+ewXFm?`{D`5x)KxDCy`}>Xzq1Nm*{lCF@2uw-{(O#QQ8Y2;KqKO;={K{1 z8{Lbttc#N6jth7~$L^I@2fbW0r#D@^h=6-kr4pS(!9e|O0@Eua!o6%{CD)0kuEI-S zVza5B#Lqg1MT1^quikhbw%LX2zv_!Tghg&auVYJzgLh?Md(Ad>({8eKrhzt&WRTtaK7*!bLv5EHGHr7&4`8S z{c0qbAtgU?0==K8AlvqZ`l<^^4<}a8DS-8!^Lk-c{PO*eGM%~~_#{F=j`&H+$za00 zT=aSYndjf|5)3l&Zp457V`x3`Ex+44aq+c4pG(Oa4!sgy%}nl*g7G!`%`TOg6~Epi z#m~x#Zk8ZJ<2&EVs-wqO?2{hK=j5)y-rp$vfe!2^9O@8(w}LXDwSLpdyNkhqMl^eH zES&KEE0e*=_jWRR|Mgo~;eLSe)&C(b{1g}LSC1GTyzaFJddDJ)pum7A)sly91|r|2-x1vq2p6Y zxYvCt-NsMtf5Yoyigp+e{(c>nF5xQNCt*J!(8kPBs|~$cW=)>=jzF)EqPb3vBKY}; z_3EHWf9tghFAtYfOZrJJfF;rHRy7^;YR*5XWd0EWE|xEfl+T5L*cUSE>FEidUrpeA zpDxL*ZWlMg?FU(9f(h$)P#5?y4ThUjLO72SKzgwK4!fw zjITUbo_RVO1Q>5Sc!aVw1gulF)lb+%xYvzfewnsac%|ySrIoJG1_5lIg6wDU?^j^) z*q9q&3)h>t(3yao5$MHXseeE5BlHpqy0liY1K%s)LGHtLku>CmJ?XQ+I&?3##%x*# z8aE(7{LHp<8uk@C%V^eE^LoxneG?@s;!k?yrCvMAO9(@fJnN& zl33{F?ey59Vhi+gpQjq9|7HE^{&?G5rYL-`0I_Jw9jX zMgNQIJ1N+IVZD}O&+k&S4go5HTS(`>5*}Yh+h*F|tR*}jQ@_~XDWrho$Gn7^p@|6C zf9Z@pK+Mm>_~LX%Haj%p_Y>GUIDDbGacUL&ZDYL*PVTK$sfFVqwg<$!tX-r)w-aOc zQ~MB*HmELDI7+xzi^3jfFJ^Qvq~PYc##3@D_N$>6#sV_<;Ci`84edQ;SD;tR)aUg1 zNa!_Dd1}u8D85&~lT%d}Eh~`T{Wj@0+*Ti7UUxqxa(#gP#DdqIxfve< zlIq=v9z7)7>zrY?P2wf=^JD7fuy^VV@_+FT+_Jd;y8OTK`#4jik-HT>Qw4SWR4cwl8vY`;c_rlCh zKC?>*0SXzChwsc0?$tC;CmNQ4?uE^h-;dA>`+ICZ(I}<9^LZ@v3Lz4Xp>*ot?2@xj|uSfS16W+I*Wt}@<7)w2VnH%P#@r$G{nG@1rfjI6PZ|whi9W(d` z`iV*D_U#6K|GXFTwtXBq`Or)7BrB=FXLvo$w8&cE`5F9rKm54UZDzOu;a3gXv9OB$ zhOl0#dt#VKy5RYdKkan?^ha6XKEv63fd4EwEKV`0FGcwIaoOt{X%@Oych%N?Hs7p3 z3}5!XLnHY2TlKd~8lHIu`+I)cnmvLF@cdY)A}Dz1m;Jw9MlrXn8^G^Zrw$F|mB_Rq zFZ~}H%yppmUwJOtWJSx*YudyZnc&=|g{!ZbUzmuMwdDY5H_2<)m=NIfUO_u(lJI=Q ztdXQ0S4Q{3=36~YH<=4D{{6L<)1u{)dh2lW6N;bCJ-Db3~bPsKWc*Tb;fRojwV@mV_@YfK2F>u~1m4PdJiA-&;VcduBg?)X2=618#_2~6p<*oc}?mTy} zMQf7pNG0^**&9>Cs186(txpn%@v>KPvekx{gy-W&S=C~)hpX|bpm<*O1i^luA4eEp z{5Q%B=@Q}o_h!dy-25$}SAU0;+x=hm>-b_t79J}92d@*`O3%3mAo&>eJr=82k6G)% zt=4$M158+szpw+)Yx{-K#IpzhiDVA4ks2YuczXRe$L)m2*B6#;;ezC=@%mlruU?$H zz2_=B;q}LF8t!DQcF^l7cZVpM1kA@P8&5Lam|J*c;Qg|rqN^8mUt zyR`P}!1X5A)Ax0-mH^;#zHz{QdHm=wpReU|{J+Nw>Dx9iECKIV&@7d?eX9m~X;)iD z%JD+4bw>Q^Zrv-_`?K51l){G5y|$>&P`NBWPiKK7EB>Y?^h*Am_*FVU4#;!%k&^m_ zfNLpEkw>Bk_cERfl4s!g8}Vhfb)Hx7F7ns-vfSq7zb+QeM?W7vl$7@bdQmmS+TQ-f ztHCfSef^DhEJV-)yym$g6mS=MF*Oc*NBheGhA+FVx{oYBzt;P!%(V!@y@u^f z*`sg$jrdx@Ys#ZpRbnq(@9Q|rCH!&;dgVWlY2#*v`FOl>H>aCEzE^5-agxxfN<_J( zVaxY(=w9^}Ua>|P4}kE`IX^Xm`M4tqWUDC2fmGEm1E-IN0OoCh3$6b^Ke1{rX_09Y zT22r!V8N42)P($CgYi{gdj8waDMnnsT0fChLPZMud-*krt-`;=*IT!>GjhOqlEz9v#KJ{x^G{(N-mNz?p+<`QJzDWbEQ zV(4CE{G*j0*W!B#v6rNT-2s4Z>4YS)cL?~WcQEZbJ121wdQp$*rcqZyui4i}t2SBwp`R$b_wotxO{C=0 z2CFcAbT4JL_I(VS+(Dnm{oxbEaDH{pv~<*;Lk?^^z8Kh`6ABn6LTra62=`(wD0_5W zYZYF<)h+A~Fdt*eo4wwjg!fzR5y`qeeHVJY>6sNTeGI)yOFO=48sU2d80UGm6{jIP zhBVtByP|s;?F-91dEFhHR(7&qdIIZLP-IA@l!zR78&}c(XkRE;$f=tx+d#P2r~ua~ zS4qOX`W@}J?Cs{j)vx14tuw^epckt;4e(BdUKTEF%5>)V^&b08LoXtOF-huf=wSXm3NQ2*qQ;qeqr;Gng5+=o-w*t-bM$B2L$}qmHt|yZ= zk>K}V0lhWc;ni1=*FggA=`QGARM$@(_ThC0MNA*{wT0mQkJZ;21unKBz?MoZcR)7; zl#gFy{nSYK`ohh!ak^tngnPvkH5r@zvR_APuJQ@dU)D!sGAjdbi=lgA<7?5yt1v(w z-ODm4>2MyMJ2)o@+$sZC#N}}*CNrZiGSY8^=pl}+qL~d=w5GxZ@gdz?jWg7 zxK8W=%*V>k4^L)W5m4&#fwDy^1Q?os1X>gSJzm2+8Wa6t7+n8l%;jr7pAEg9cb@-{ zHVN++T?iC=FDQebkH`ZB&zM_zNS4x|f(y~=y;$b8-961FIqYf+65R8|V7>iiCT`21kE+{h{Dz z+^l^6<8Ms3AE4J$>$csJ{`+uVU*p(qq>G_f(d$PkkJ_OZHTnCp-6{C}D)yxV9?^TL zuOc723+WVXZQMa3Rb|69X6W^e=cCfRJphKEkBLdyg#c5Rr(}=C{sUeCcha=NERY|c z3wp&UFFvNRh4T}9M+(Gp7NOVs^Yk+J#-UdZuP)#94fyee^?GA18F`r-J-!&^n)9#H zdVm8PeC`D1Lobi*k0lam<-qYHPsutDhJbcovg@tq6bassFR{3zxaIHkUx;H#u6Q?| z7yo#&mwX|dk5+DBIb_8di@RSmDa~@g{UP*vuPWwtY5U6O$7AC?!yh*z03mu-`P}$1 z?(ebW37V$#=>UqS*h zH7HT!4hlqG=uC+Jl8-;Vl!euIep+7dn{6cApAba2SIm6tqS|WqpZF0-fB7%0U-c}i zpY4h1a9*TWK402%{vs|P|LVmnCvU#_56?S6wkEV2_hhWvD?gKgH6r|f^6}dM^Ahyp zpnRR#>9o9l!f@E_%!v@7ZnVua$c1pPM-g}SYU>joU){u8Y>jB)e2#0spLnMq^coYX zzMOFi#uq;mI2Ty=hxq#Wjj*sMoz}c7i(0?1Ub<#Fnbl3O|H69Zw!Bhkati^GXVm3C z1pfP8G5qh=p5eydAHeB)$!~A1yaQ=(?a@Rd=6FIgH_1i&LwXTg$N&+q; z-A$h@f%oJ<3EY?el?Ul;+BF4vo`z%AfltyPsYZhR{px|OmjNE)m$ zd6_~f9}GAtEmGb&67D6cMMoz0CYN9@%X2fw%}n9-+Ht2?;od;#wXIq$wM!0qZGR!7 zq@29+epX&@UeWq_NTGTOZ5}^8km(9?RH{YP`T@?%T}Wf0^NTcKqt+JQq!+A{}5PEN!A{@(a!bJq~)HCuJlH6;PvD?q_xZT?&mBI>g+ur3DO z>xEB&Zf>J1=*qi%E4KjFuhyPy9U~`c@PYj0E0>%Qz)+&aWO(6U^s-}r{j7u&_x=4y z8(8m=`|4z$<>7v5m%n*?45vY_W(!h-!dU3_nSH3zb8hAMBA#g9)Z4jwuh}d10)pqE z7f1GWmb{lzpiNw~U(qi(c3Q_1h78C2dnnj!=c`zq}5m9nH85y)s)q zu+Tq9=b+JRSzmk9oQ&r)UpJ0|o`MR7;m&z)wxm-f2pBd|h7S zoI5qV3NP1>D!i>)%g-qzGS5AZzyAKI!n?`{iF}?fMz|N-HOA!i*{ksSEf29jzDS!!2qU4UMB zenDPFu!%$ov(5kAA<%PRWGJ$Gz!I0q?wFeErqSVJ5zBLkRYH#`yiJmS|uQxgok& zZgcRf=PzAAm~)(DQvl4zOI(~)fKnRVaT1x9*9!(DLL@qmND24aM>Q5mqLlqJ@Bfea z`%$_9m(8kRC9lwJJbMN&?!bBRY~bHqb#7&^<;OeCulvLEr0Y9nw$f^4A%EA4B=qbf zHf?x*#CoL>y2-Yi0yy}ErG?BQ({t1b;uiRG_;lNg3YX z(Xx(%S+gAOhn#JHe{J_+xPNr4&Ap=sr0~bvF{-Hw$xQXg8(VI~boU|DenQKH{f1(# z8)!OV;kYRq)-SA=L!WI^!sTEPIp+57JZ=RYPc(LSK zvPlqnr9P}rI#wYJ*2`2Zw60m+A9LDLSh3tNpn27~p=liMED zJQ4&BpL?Zfltj1}k9Ri6ZXyLkh*8+eUuJlGdaxYyg=Ng|d?7jWP6 ziqD(-{R95|Y2|z*ro_abZ-ridLGmF@1D?3~!rxyG-z)6?T#f5;e~xy&cG2ML#;j;K zZp6N>S_&^+3cED8bcJ_-;(HJ~%GH;4~@c@68Aje}p~ z@pZ6%-Dk@g46MZO@3EhEwO;3>pJabuat3K8o`tMIaQzp%kZ`Y0w3>@LCV!)TX=k0Uk-muh_*^i)Viu&;OKRYH-%d<+k3u-cyU9eiY{}P$^8{?8kuV29@m`W?EoI&>zBZpZ9?C-~RM%;+)kOos*Tv_SX1%trJ zwzbzU{0F?S0r0;^=v8ssZB7%w@z9-Y9lhOJu-^aGOY72}llg)xdkM>&qg`u=?!{zp z>YyUv1g7}XG`nxY`RMK^L!~`jGN9?K=9>YrU@)T8dh*dQ;a*qc6*m+B^!P%a9bl|U zg#FE5dKH%^)*KKA~n0E zIW+7H%=&t67_Q~T`B;&6N zpC`2M2VH?mZrvG+LbQK{!QkdYQ?)9|f$0{dHPy9&AXKSO&{ww-GI)4Kd)Yo8{LNW+~`cg0!Vh~Rz)K}`w#*I_cIS7C8 z>=ffN()wZxs0`v)a0A0^hpj%(LVaDyb>W&DmIF$rqe^jQ*|6-p94In8qj!0E5U?q;HPyBypI24rajahJpYy6-uYD0KSVy3* z#@Vq+k<)O#NylCfmus?+m+RM4Y?`a#{2W)$xGlq`g!wPTZ|=!G*#bm6)hBP*iZri< zgBuLIxLv?fULWNBOAYSJgHy@RcBU`~fiE7OH6Q)R=k>lar=905XfqhYh090oc z#=0LOJWo1`O?t8d1M{o4X763+V<4{vr^81mv>-3WNv?^)z~AU=J=H4HyvxW2zU9xv=yb`{IB zS^&9n_EzVU-<%KgvtdKq$J;tHFYO$O^UC;E6B_=z`Cnc-v^!oKBViu2`;X+2=GBwR z&6UdM0v;!Nz8rOi`=!w(xgnfga=^FlMp^TlK=9PLQm!SZL!^LZ zySHp-A>n@RE~^yEJ@&!zT7y0+nPD!-%jBe|!wF@`%Qc6sKwy%P7dDh*;c;yt;_lY@ zTyugnukBxhcVBC80Xh%VZtnL(^N$z|%B#=Yz;Df1AXr{R5$PsIe*Qi`tnTi{_oR7Y zH_jv_o!No+4?lDO^_6gMG+O@|JdbtY(CAi~O^_Eu0Y>?zCgfGN1}yu!kI;W%C|pgC zDMcd(D8_>Jb(7{bcI&~2!FE^RktoH9*{_PPUx>}?RcvqMK*FmJ=P%*nuM^BnBTwka z=Vg2*=WKNuXrQc{#k`HI`;`1yQH1thPRa@x=NRJ*(FYIRDkwUL+fe&8zzC z^>J*z3$WvKL})+3cvabebE9&r9H84heb^^55FD6#q;1DSzP|YCwGFicNb^FJE@$Kp z$ph?q!73eB!t-<(FKuf4Mg{dX$Ncrq*d@rze#_gkXuNeR#^nDj-kDtEWpnF2Pq84El*qE4E{SPD65x%rf%zMQQ(eQAp%J;Dz4M|CW%~ z5w33>mk=0F)HV1%kC}zMv`#3PT)+t9 zuvp${xZk6{t5kPSk}Qb0G*-sciUpZ-HqzI*$mg{)GTV8CgM40$EsY=TR>F8<#c?Bf z#VW|l%gFIkz!k`Af@RM(Zvo=|-jZGLoYfrZ`~*!ohDFfI6-@fQdLdE`gil zSs=3OmV;#r7GPezlRPCzKCd38EPKC|qpd^Cl zBIG4^Pc_cP4f0y2SNQgY+i&c#Oz=>dS^{#!`FXhe0%=~=i{HC+1zmv`#W$bgLREY` zk-#Z)qU^COaBVN(u^hnu&-_BqvcF)DxO4Q5oLonm*FVe<{{q@$b@vm;O5acT{S;Kw zW=@&-L0$)>dEaOzLtZ00N2cWYf1@vsu*l@|ClFofw=<3Hq*ZA5F zlQZ{WfBVE2#zr;VfK88arqg)6E67m){+WIwj3*TDqzL<+Bh(j`%G#*%U@@Zp!KO{z zhcvHqQBl{od~gMBrq;zbPeOfxTW6bXgyaFgOT=V~gCDqHjeRd*{omt-$aiGWzQoP< zrTL&EdbkQ%$_vA>f0)J%wvX0x=afYOIe#Yqu~ZF`}ntOb;5WR<&`l<6THg94|o@g3%zh9pVzLN!(ZZ8U;C+> z|6_h3-UfgD2eikw*GJD&eTDvh{`FWPtt#a8aMc@sPAA9<`))wh$NV?*nRcv`6v}i) z!i42Jyt7I3sx!|@_?+YhJhltQW41$o|FEDd%d1oY2p+DO`d;G)3~SHCjWm+aYnyV6 z3XKD4Ua5*wQP)4>)%|Wb8~>ESxX5xyWky!#v1q{5J1Kido1jv|KV( z!Jlyddt`i1&W)0M@9-u~5xRS2lsG@Nr~vmat{A)~;S=DL59t4|^=1Kdh7+GcU!qBnZ z8vHF@`n6jupElFu`;VgO^>nLGKwikHgU?4NA+OqPv&iNC-?WcEd40&6udl8q&5PkI z^3uq~4HTHSYsTq8UTfdzVTMEzkoMW}V=S8=;LGNe1G~uQ^~RI7Eeu1xzNQ&Fy3}pp zc}<6P%M(8*VY#(NJNBjaG?*1pUQM8{fQ=a7W<$YvW!v$#(VgzZ|p!d2KG z)C^oIel||N|5|s%S%EEte0|L(n3T$l!FWPkj% zD`9^K<)y#iWMq5AlH~k}>)sXOPrFZmg$JjP)@^|K)%HAI#(q;d&`x_X_|P>h_{6rl zM9hMGUd>`R1ac3N&r9DTNKq9FdFed3p+{Q>dCf_LXl`qTyv`l#5L-k48-K5KRIAy? z9|`M}>k8K)&C7Bd6;-OB7YMCtq2pqN@kF-d5F-O_y;cu+eVQU03nKShC3zc=&uf_w zZ!gbT(!7xC(NnQ)5&*-0<^VUN3xPdGrEce;$cFuG8!VltS_S0Q`PIbyK{4dzqja;U zv5in)*yD7US)_uHXBASqd77koDa_J)E&JvPsxMMG=-8{_+oLDOvj+Z7vOvutT`0;E z3p(E?NGq-(pO-fmPr{mQqBk?O0Ft@t0+Iz#U;R3(kvp@pV3(}*tF5tE z;F56U)%8X4=YJVWe<+qv{{vogi%eYMq5$hyw&6HmJb}K7Y_FwWu7mb?BJD!E#vRD( zFl%>5e+1+usv(r4ZS)&^{0VIDTr9ArJxG$*m`7ZW|m$sYZaQkoe_r;`_ANS1mMI`4k+wY{1=H>N5g2vY010d%0*-zxu@%D)Fx^qh7 zisU6MK*XL4ykAYeJ>vBBN4zlg6lTvvWC2F!a7aaF8UZhHhvMR*)i7SAi))QM&<1(= zV-HTL-GRKmy5N#KTEycot+dN0owG>u(%9a8pBHyN--6g(&^He4(ULEiGBFtd{oAV+ zIwG;)(x!{T3!>!nVv;(_xWbjRzN*DJBG-%%&)*~EzNKHfV7_o|qXhrjEinFSzy12I z*5;jhvFZ-K`~;nch;+y=_NckZ z{rh1N$V-x`PApZ1c)V)lIAr&sk+i=L3!LXumiGb|3lj^ngJ8Ux<|?%6tv&$q?}fC( ztg(P&W&Hk7F7ov?Wl^Qs`xkiaOSb$nBns`(rZdxJx*ytOlJEL={^F3A`vTM9xBA3+ z*}OP|ky%ArUpoayE=w4Cfo0iKJI(8$J%%fZd}$U3z>!7!&2vjEkp5zF>&q7Mc{$P^ z`_^YjzP|LIr?FiOhW>uw>0{x2k0GzKs$W8?cR^k%oJUQvcK(JJmL>R#uo%M8v7zr! zKWTkYyz5{x+v)}O$vPs+=LmU8v6xSp7h*x{C+2(R8szh`33dgX*GTh96^bowc!2;c zQ)|+-Lj?r>9^2=t7L(PBA5WWD$g#mnLH|{>?(CT^4#?|P;BzOXNWy$HW=`wO8m>%a zFW3E>0zIVlMT2$l{ci3B6r4@g?dOC3YcMDhNOmfK_lb*?mvQ5@7P^-?<^IBYK88ED zb2U0i^BTCJDnr?)0eIarEcUOyLBOl0iH}n94&1*Glm9wOJ*_nV5IcU!f3f`Q=j#jZ*%+@m$>+8D z<_qhe;6yk5%FcelPnz$ABaTG1_k zV3wQHkVd%^XhlFDwu_BD`~?jrL+ z;0N=o^z4RqA>4&Q8m{YO`?lQt{8 z5Kw!9ZhhG)e-JDBor89se0}i?4;F-0kiWmlzF{hv{=+c;MN#Y4NHq(2mGC8Bo;+2E z&##^|+54I81#x?nI3{T(G(uWmFOG4(n6f$zo=Wz~E!zhDJ>~Kj*^;gZ2t46;@{X@R z*p4wQzVa913HikJ&Qu(vc_A6Wf|3ll`g}+`A$ad7fju%Z@42^f1uxznO~!Lqs0zdN zqBaL(A{RIyFF6hte<4M}^8&CoYdhq#GLctYJ5|)blIEpzsMT8K2ytE+r?S@L_V@+I zG#Qo)Y$V?vUlx={RYj8?@4%YuVdMR-N0b*ey{w`09zHy;`aXvU7GB6pYtwL-MI_`k z6xshVXWwt^k+*o`V4fS|`k4QMwHWF76RsPFg5_P#;MRqPvwURKz{g*xzN%8}x(*fi z|Ihy9EBF2dUeC^Zte;jUtuM?vaDQR76j5H5TE&V7gJAqs+jD8t?TZ!o{Azg$?X&kM zATP?N3J=CM{f5^Y?fVXD=aE{vlgnjPNb{Qa3VJB^<_uUCP%^ZmAI1|rSNGAq!>tc9 z9&tFE|J5HX=U3}Lv7LN@9I}3(bC){+n68b;-cn0```C1$MbuCB4|x52?*A-cJb_SjCp7WF{$q-E&4{)W za6e5^G;0M=y`^4<3FX096aK4FrH8m z7N*ZS2SBO2vRk)J0Ei8p7AP7fpVy5kF}*G&@_A*iXcEI7g8E{;kXG1;JAV}&AJTlw z_3BYG|_I{;=;^RpU$ z0l@OLPi-9o`Tk1}6tKU2N17M5hkw_jd^Lcih<|qBD=9MgkCISd^%qS|Ld9|RSh08Z z;0}1csQ-#|_6OUPfbSOZiNLD?AoGrKJ;PrZf266!J~dz=9Zz63r}D5+;q9JqhN5{H=Rllr9E}PvXmWM!IW3M@jd&N+DgZ)SBi?|)DjjkZFr=Gd% zWm$vRG~Y4Zy6t!!F7;o^=KF`Tn*9Lh$G7~WT~J@lc?N4W`IJCO;BE-wm?uTdRFwg8~N?yX)2YJ7+=!* zvOFR*C=h1~20DlI<+EFmANK`$Wvm_WyRlIe&+A&kxm%)TkQa}3%!AQ(n4dthuF)7+ zI^gZBD_`U)pOYwvQm%ic_Nf**V?V`k>3_~+RsOD+)^oZCu% zJi**Ye{63tX=<2%vfY-h|ms@uS;_Lr|->h$=;BcruFO0ulm!GG}{e{g8FO2!mvteaf2~tZ`QHBrywtUu!OJRN5O4c~Lf9xMX|AmVg&}zZ7?D1=5vB^P*5>_D(U5v;f1oe)hW@vNyoz-64{*4{^~li^7bTV@ zL0$@( zcN?niPyzDg`Mi&tLV#`{J?3l*`Mln&q)ZweC!ZI3N9g`JQ8RL9u(q7Hj7ZJR;xUKhfX#E$}S)yc!p4vhPg7`Ai%27MV*jATRY& zjyUxXF#bxj4IpzPPbUk2?_jasbK?*!x(`66cHSU2P~BFtHp zF^K91>I>y1l$f|HAb~V58WlPR^(Uu5$5iZLUKI_zzn_?kyyldv0s`od6wIs-0d3X- zIw|tx^HR}eaUE>@172QjZ;m%n;Om3h5n6g}Z5{Qif%)j2nWaY0Eg>%&Q@#y#^^g~r zj;`W>7I9vNwb*emYz?rp=MAemJ|pOLmc|qDgW@qeSNnh!ymK)XZMgpJ&!6y_iA7w> z2o!n*S>9I+0priu=O0#+&&ybl!(>#F^m!-MBWl?L&dT`up!G+~Ha@HK?;7BEM^8tL zpP?D#1zH(`s7E0$dGAc6^s(RY;@~$9an(o8n~ZJLr^mg&V%6q`EA0q+ou#}ycbn78 zQlAEvX`Dl!Gob(SS9@G@p;Qf2Na`7~Ylnal5tFJUaq|5=j#qg|;2-dM(B!RQpaN3w zPiiG+mf`D<-bc^1u!f^@{P_OR%dzUnuP0%A-z}A`e@h$s`}vjr9bd`_{XMqaCyYw| zB68y*JMSx7T)f(Q;dbTs1^o4v@(OH;q37Cv8r(krZLsU94*q#kFW(guaYz;HYD_UT zY6%9Lo(?fBUL;>%Uz+^HN;Z+^^@GdO-!PsSPkMZ^c`x)|kE}UAYCVJcQlN@kVO0Y4 zwW*dap%34mCiGt`*{eC4t4aH>hKDaakF=fwJPD2T)2rb4kRPYq_Qqi~U@g2n$1x-X z+}0k7Sos&$iw0bCy)?S+`p@S=dc8X$zD+@tuN42he-NDYP1RnLWU>g?r`4)XZ8g|m zj*lm@r9LgZN{9N2EROgZrbei*RF1o2mC5%J`j|0O@tdfDjf}p6vC=|0yrD<|rq$^fINnjwyiouS*AN z2C5vPJ<>3n&glJe{^JQc|MWFf#N&x3rWl!qOdPK@TV#f%CJ^*GOYJfJmGvh~7jC_= z^WNO*EwDcn_?}T|yQMlehBRKGYLHVP^YFJdre%KX|Vg>Z|I_k(sbVg!)3i`7!gI;IuXzuPWZ$x&5=qkN+ps zSAJgg276;a5c{3$`0aJjf5qon3V(=K2Z2M|a@_+%z>AZ*Z{DsXUtjHo+#A?z$oF55 zB(5b;`N8w~Mqh90<>`X{t9(_n*NeN57mfB#F|Qm#Uf2Xxt9Rbf2%Z-T@PA7}UP3%` zSJZC!fl1b9>-9om`xxo`a`X6iHLyCoYM;7i2uMDrvM6Hl{~um*3hb^^YhnJ&Lc3?f zulBe>&G1CVFZ(k+H&4#G=Ss*6#nUevHneS&wEx=s$xN7oJpd%Li*vo;fc~qZr?BJ6 zV^y#wEBjIvR|x3aQ5KiIyif4@jdzWi12#!>LkGZnT%el?aKRQ{kD*gGh+SA zc@(M=V(&C-P#%Qs<42$>tO#4wvS1dudwppPy^1K z`|0})LO^2jv!NTBo>1-NqK1s{*TQfP<&>pvB<}#dDRRf!A z&dtq8hkyf?=T-fq$md0$TVWiwl6+o>N89Q(m#6Xb@3&Lh&|q~TFV^;yD-HW0FB5^# zBlphz#vUg_dpff0kb4XYjd|%dz;c&w-(=DW0gpQL`nlZ5=M{47!tK_Kzrc%Covy5x5+C1p9@-_?%?Ei+7rqQp zJ_C6@p>cmTCPc^!n{LOr(K{Xq=R5ng>j7zd+|P)lZxr$e91CLoW)_gwx(ie`PP1yD z^eXlKeL*3h{o&or*<$i}r4G`EOS%66FNI1@-fPf5EVV~*`J$~Oi7>yK&?h;TD*}1# zwlFxU{TT8}*-CMeiIy<_8gMWJn-7*F4Lx^)jJDZ=)Q83C$qSrB^_92g&9l5xe;{S3 ziEyMF{rnsTbMec&1^-EPu-BbF_58CCK$XJC%FgRRcKr3pKutia^AC89)vRF`A-? z-D0reT51fkR9_f`Nz$+$uD4nyba3C1U-mC_dOS1wtO@dJxuV~khy4vNq2lf}H;NHY zajD3-Nz%Ofgojj~9PkIWO3~a(VQ_!)G8c(^v+ZiY?77C1_1i*05~h}0rIh@5V$BoE zz}`I4`l_ypi^~_DlPCEZ>b!Qv&XX{HWtH z{X1!Uyq%%rLv06wDEic^5r$3clfPu(>~&*)i}P{f{(hgyXrPX$KVVyyUY>nj8*h&x z)LpkmXf%L-f1Z=dzEB{W=dxNp|G&ozwFqLovhz=w$Wsd9dG)1#(G<3X_PD~sh;Hf{ z`0t`_I---q|g6b6CMlehY*waDisaqQKXwY#q4>;1RR=i8w^wb=B_dcIX(wNxv; zAg^jn=Sne2LVclp@{lKfv&^J<(ap=dZK4eTR5On5yCWg5FBL79MzHFD$#h4_?zO={ zTb@=?_#FAXjAu&LpZY}F9?|>2a?v07OmQFi@fyafUXj6X9~8p=P+eq8z$JO8FG1=k z>NnD*`1J}wLFUU+E*-%?C)6I%>zwd2@-ObP1IURd_W@QA|GB05N_hPWEK~;o_wl!G z7X%FOdK5P6?@VXV0Ih)x72BDELCa#Ud6zBuy!yj()T{31!+7WB({t3d6YeIZBTe3LedxVW*$P+zY<2zg=8Hza)UDMHwD9|=q@lGYc4 zN>Tr1u|V+f)u!lkRFId=hN3Y# zVg5@+Pm%KgJLDyZn9KU%_DA+t;tr-zISyXd(A{ZFAmoLyY{?`2fiy26 zZN)*cd7M28xDKCr2ir&XG((Qky}0!-g8qT;cHs6mIq|ffdYpV-dwk^?A5*82)Ym^b zp|2pXqNtZoTUFqC_lknhyMj)+`21>Yv*yk|QOJvBY`N(MX_Ut={q2&K$DWr5Aig{^ z4n{PG3G}?w9);5Hlr>TW00)eBSu~=L|9n=>80F_NMR3aQ?dPKzzQDBRSV5)%`MknZ z!(X1~xI#8BOOvRo>C9F5_EEXsJ>@hzF#h5^^yabXZ~8;%ldz(tw8*=Q z#{cO5685(}ecBF$2Ka%2{G7)U>aaiL5O|I9@+AfEHIX%%#@rWBzOblSwS|0Mv)`sS zk2hZ?n^)1Bo~SGY=5y%AGn-fOL0-AHf+wDg!u}Aa-6Q?DRJ4Eii@tuMc}G>>5@Sf3 zm%H5w$@b-bKp`PXF>x*IuO-w}?OOFh0r2h&TI_H11%gIbcL&9i&uiD2*=3v<()NfQ z((^h#1)txK+xo>Gm++di>$BQt4SAWrDUKP9hx&RP_PMU<5gNY}^0G_b*MiH6uWad#rub)ygH>m?l_?i*Kd0-YxnMwg}km?d=d_8fxMLG_9!!l6X$h+YQ@x^L%4V% z<5nPF$ws{Xm-;V}fOV>qxO|RN$aC99dFa0wX>Pxy_$UuN2IoHdMfrk{lp^b`4wKJI zp~O3uWjSeH=o9{D3H3FwFudV&7PQBUC$G_@7D8UzxTJR=29TE$d!nX=A8~u^K3#6) zI7-?cvz*`Gq`|EpVs^Qy<-!m3e@wT&VtBIxcu>^S9+2bIen`iriMl13Dt)k@fig%R=?am~hmw1@noGKs8f60Bh zyiURv^5VGv)^?8$M$Q8!Fh8+Ymz||2 zSPop@p*y(qB#zgqE@y9Q@_BV$Dm%0PBKf>N&23)sh9Bz7#=2H{?f~SaqrbMYZX4uv zr^O(BH6O}@uzjSn^j1t%N8cx|uP^7_di!zn0akFC@h?w=Zr*e@@8Dt*1GmYl|(Y*4Ce)6kJBse@UdzdLN-Z1EM8;*$>phcp@;Xv?yXw z9`K0PM{XVS1v@lqR{HXgzg{6k!gtr)71F$rusEh8x#<2=@Ol6E*Q{@JkL=NgNO638 zA8fpGodFfp*91M~NbfJ_BZs>W81_s5hF6yT=Ly-{h@)fT`U{b^VEWn7o1Vj4i1JDm zoE2Y-TW{5D%X5EHAI1~>wV$08aQok{Z!8z6!_9B?vJj*!{R``1*zE7i$0G!ifN?UI1J7_zCvkN$H0_NYQ3ZQvtYV!Kh59-@|G|mj6XbRHRPohoooGG~>Wfh@bgdzK z4U&Y&2sK0=0@beH_ULi26XnH}GDrWa#2+9I?+a@mYyA8i2D9y<-DSGhis0&MiML(_ zSa7m`_D1JD@_9MAbfkR=|8rg_H<(eHQD>%41=JO3Gdjb3M|J(j8jI?c!Xfjn&3!vzf(e@$8S_fF#a+aPZE z%6ktiD4g0S(sP1*UU4eNlgzcr=fxYpXVHHkj6WX)t4EF#Zd?bUIh&I>Yq}qnkWlOy@ve^nzxb+i?4_qT?NxIQho)TZ!w-JBw!z zZ7*qFPCz3k(Af{z#F?&m$PDLC=vWE0+@?|h)S0bvy3lMSGoecQ)w1REN8go_P@3qf5R0b}Q*x zN4y3p*?bj#Fe%sqdZ_dfqwr}hoil%mZa@bvQ@Y38}9tQ zZy%oIvsJ=;p}5FJuP0mNLFK~9neZ$uxb8Q)GSixTUYixRnl^4Dt*`3bmxrembOD!s z{j>5{&^~_L->h%ZfSC7pS!j=9_H?)P+90pfv?qfxw;->>YF7DMwiEiVqO^u4%>8O4 zUSfNWH*SAMlplJXrS^#OGA?X5q1N?FeuC@i!5!){pz5RU``2f%fWj%aokiil#|y1@ zs_ja4B$x2K^6I!(S9ck~`0KNPx|97r$jfikVMR_K9Dh_z7Gq!cns|OSi}}VQ!$Q*f zYAC<`!uK8)nAKOS%=p51V(PK2AH!#PkiGKu(R2A&z?{a^(t3t`UY|H-R8OxXtuK_< zKTBwjQIj@o9Lr!lvFB#QQC~;M3!~D*xz-r!>$*vk9(@3DUW$3GA7l5E=4I-Bn+egw z?Ps;9pZC2B#;ZB6A0L#)$bW-T-;g)IIucyaV!LetUhkVh3?v1CN_dJMSUQON}QcS@sQ%M{eUYPji@$KDtM2 zbFr}kXi(Pn@lo>wc4;ZeSSRv%nSQyV_sR9od0}~dOh3K4iTuzR)R%NxLRgR!)YmOe z@0{slkQcT0%$`Di7*8;^Yp!QLi25f&e}63VXn5rx^oN!gxp6x%_yd>6>7SZYVZM+* zi@hD=p$ej3h5LMU4gtXhDsjPzz>M$qytyek9Pn1@iK+ z+I8NGK@-fswfVr>76RzTIGN=b$mcaaS9zt7q2AG z5yl~5d_1AHiIXNy6s}LRHJsG75mH+-@&Q*gpPzESP}BBX%CR_32R!MTbN; z%Hiv`w0#^}e$w@lWdJA{Gxp3Bhy5YjB6lu}!}Qp*!Qked4>il2w@ynrZ;AyvmSVt8IpnFx|k#M|r-sIl+!GZuVn*DAcRgeij{z&@hrWISP4wTH; zwT`V02JU@T`koqAWb^vGA4rA?Qwf5_3O*40zlxZ_i=I|&>r(TGhat*Yk&)zcUG%w z27@Vu2J8L{ux~*6{~hw zXC-bwX*55vUgqQRunoWQU*pQ|i6^p=z5enCPSzd*h)0L)hTXjQ`YpA`U8xrg`bz?U zmv0)&ni4qPk-IhOojGnjvO|cr>XnVbAjy3?-`10SUTuk;L%DS1>r4ETbJ6ulxZd5# z_)x$JRycn(;^73XT_w~PQC>gwJIUg4TBwh-zH~J>-F-I&g3pz9*dbcT%URx^#Y;gQ z@IDh0P!J9VD@?pgjB)3AliS~3CL;2EKPUOT7%x}~dR~M3e`(7lY%yC7c`bVBmJIho zUS-2c(WMzEk6+@66D6$k_k)m0gVsbrUcB6lF#dV)fGnv`Tu|yzudd-W#^C|cL4oY>WT8#A(C4&(Bwq^v z+hdmBV-SG$SgLaWbk-MDK(8-3J^2zh@#9kQQ4E%RUayj)mOsak&#QvPM#m{0=D%dT zy&D^!KwkWrwzqU}{7`?t@^Ea${So5*A>W-d`m%dbdnL}Rp9$hw7j2eCBxx- zru<4B;qXyaP%#MP^yY&AXHEUJ%q!&cDg;8iO(zgC9 zH6Upw!qz$&1oRSFT&0 z2nKrhI9@#o=OeS8eHN7XNfl@oFBhix5(M^ZQ`qf`C0}2E%nPmZKT61}?M#Ot2NUdX zXB@i$V-Uv+i|Od zfk1ed?oc@k^k1KXxfzk>b+lt%JR>&%xLkUd zAcequ4#({lA&xoRexku&?KPeRf}2titY7>s$&SAk;}0C81myF=)EeGQje@*b3T47P zaqExK{QZNwQ^5k#F#mN;X#R}lIds3OU-b187)$?0z2qvCCrXh}U&~nTJoV`D2T>0K zbH(<-`2dXRZ*RR!QUmqR9Nx^g2Li5*z1)s5gQGNL>PdA+P(5Ff(NHz&?Zrdm+Q;t_6tla%8NCUP*p>J#^= zNo1=U2p5bKWqlL~Jh`tg^6dEU^BPD!$i9?>poNF+zO}mQg>;Sndx3X&Yr2G~jwZnuOFj znFSRvQ{!bfMHK{g8iv;j>XXmQYn8y(XMTUsKK^{fNW>|J1;JE&{n7itoXkV-GccA{ONd4*hk*~ke*93#W+N6ABY*F!zKo9!<&x>P-^IDh1 z`c8K@j3-cD0|z6=+NFX(N^imQ`By4r`>%P$d;FErf6fadAhYK29V`JaY3$B?``2*) z)%)9D(scfEp7h!D=CNMEKj8HfjXA4Ve=tx1*s4H@H``X@+oPrZp(3r|b7q_YV56m; z3_~W2C%XDx@8>8~0T15xi*IEQ0`Jugu1GeJ&#UCw$keAZf6fcb{i#{sjPSgs@!dJw z#x_C!b?=FXS2h#Wm%>=qX+=56OTIz)L_Tgl)(<>{{!1@3^K%-#3PHTLl-ItnkH;Ur z@dtJK=Wf%Ez!{6`t!- zUMp-?lmrw70B;-mwLN9f9b*(Kj1~3 zdF4I^)(7Q;mO7GFJL$}};^VKq3Lc%J>5$j4=GozK9vH6@(^vJ`B;_>2Y=o;zxA+#H z8fX(Oz+@cXLX_95la2Q;ss{q6zI)zvPhh-yP*f4qFs2If*0tUyRBAXuUi0-E=Q?X4FCEVztIBxdymB1WF83L$k>n-(b%FV6UJ&45TuGh% z5%Qu4YNI!*QUly+yJb(Q1_3N%Q(GQ4`MkUmi#PS9lFzH{U|N~)9L#@h88YY3^MJh0 z4>rGggUj!r@kG}thRTCfXn*Gye?Pe$<02A-R1GI89Cjwni=WjZPZ9(I1JRaCr7_Uo zpF65LcvBa*-_V6%m+c@3e93c^!~6w(@h#FVc%LPoSGCdJV_KAuS5-}XPt9J)D>w1u zgw>rt;Pun6?SiR3P@0nF)oEf~z=RO;QV`^6Z${L>;&lTy^KC((O#Rph%Qy1?N4`+{Wljcu|{RLh{yG*PiE6L}T zI;ytQNC4*VGjg0?KE(NVw7=ck2ih9mLS98waXMeG6X(@^DRo_$25Eg!UB1ru@u?rU zJ?)gZdl?w4%(UAke{`Jd@&0zW)mSBK+;NQZfGd5zF5Xyb0UWOHDlfVmWVp z(jybbUmbPz#{#<{uZS6b0gcNro*>4{GCFG6K8+G&XKU`Mz62bvi`UL{&(Y!Qx3qms z^EVez7{%?UFg175_t|cI`%uQYJWO9e4LlMEEFF9h1bQkK1UN2}&#PWY(k@RD1gMI^>6&f$>D>$y1W1mICGMbr~fj4yVB2^Rvmx;==F&4;^&*&Jst%0h4NbbdXR(71Q&m?`Ek$Wlh5l^WnKLJ z_oVwnmV4EGCggDOBN1L)k+quAZctx`Pdj`~6ovUh5ox=^>rQ{b%R$z9qnQ`7>B_-d zPW#mX@^XhxL<0p;eYxJuFIuI7+aEKIZq~0D`Y)6h-?`8Vk@z5B_Nete+jsJL4LkEw z(K?gng}F-+;*h_AC@=9(ezoHL@VtO0f%#Ha;ZR>mPi(^08<*qt#h4u=zw6U)#vgy` zYtBCQjk+|<{}I#Im|ya(qbqUq0Zd)`mOI1wIVdk@#w|3buLl84lgEg_U*I*JVdVFK zi?lsr%p>iOeBD8mS7~cZxN{=Z7k`TW=8#gz%U4E{V(m%D3q#4i=~FWyFRUx2x9iyi z*r&{42Ojp zb#3|sUO)e!`plk9Ok%eX^m(JDT-RMr<*RVLyRgU;7P~mei)wG(aH2cpb<2NBGBuHq z*U!sf9-j2yA8$mO7sH|ybt4r{UpCrz;{BliVoEqbt*oj60$DV~O>y&09PLCXKNqR} z>--J<>$0<_?*#+i+Ye$A?!x&1 zC@-5L=9y)<{WE8;$4OGe{d->MbNo|Yn5%h=d}BsLd5yGZ=I`AF?U8bATGymA%tx#0 zyj_gc{sUe=E7q}Fr2g<3xW5bfJeRhQTu z6TRvWK7B4RDPE`kul0rU8s?0*c0Nbi-&cF;1{>n?UtY&winbjmT#p?3zD)4ZR+!I8 z;-RCzlm~g;y>|Y9Z^Ki3Z6d~Q7>yK|Q(*miqJ2&w!5TF0` zfTCoAF)bK)d^NIfAAx_56?|EV9p6)3+J%>Dwaj%{jAk7P#B1RSQ3J}EG=zX|c zsJfg{0M`$FnsZU&e*$@BJW^iqwg>VmoIU5Q)%6=*SDU;KU@szi4~`C2{sFHWj^Q89 z*#rZ%kusi)Z0NsGUU%NJaz4I?+mHM7a@yiJGI;4fnu*_G1bS4PPFS&$ZXX9OpI2q( z!qunc{rNc6YP@}*ym;L=r{&MX`4fDG-p;oMA+N88Kb`D91MN{@J1bTwosbvyok3E8 zdf&SOtipoTU?Fg zJAZIBaqE@X5C5K5^~>4)niVRbB$apgpebp6Aru#0(OTp7Cnv(|$j0Xq=jUKNQEm1l zu6_jantbA3k}^Y@7h;ftJzISP`DSxkgs)s1SXLEMC%3T><@K?pj*a475a8cvD)A;9 z&L`g(aiWF#ye9BQ>V+^X0>Inul$TFeYmnVOa#(m@AK6cu*AE@xet!wB-N2_Q1)(5vu12kGaA z9kyotF`xfnJ~{da|0rR8BC0N?Z~Zrz|6+JHZ*1TRc|GGd;B~5o{>$uQ;^Eh_#Caip zR|Br;ljij*bd6jPbugGTZ#YQ59me-2>2%S(z8awFk;<;MiT*(Po|T)|b2514(teqA z+yVNwv4+$yn#}ysy!Qgh(esii$xc^G3 z@2;G8%QV4@@(tY69{vEcAYdMN^WXDA*!fw-=(NFA%ivBPgfy>z$O``g>I+-$tz~i; z&c`%~=~w!o1$kvZ%E&Pkh4x5{7iQVo>+9{CO9f1qf&k;l2L7T| zaDTEq<;oqvycZqN|-S_izMyZJJHlJx%Mytj9h z^HJ^v1N%PS3*@_x{6GeIQO~eQh<$_ceQ$htLP#~_byOl^Q~!0yYru^1S#-M{fxg7A z4J3LzxsPnHPs`Y1p#!RI*c=bhtiyk9Y5ORmsna<0H5e=`GtfM#3ERhL7lyT=JG6k4 zbH=)s&wk)EueVUK9r^zL*x|+G8lgY%Un{l0?YbilrUi!B=Qa?=6G)}?Y@i;r$NbM6 z=ilyyyw*W?I))hE#1T&^KVw1un8p6tgBGM0w?xhUmnt3kJq> ziP*e5uzkGvz1%W&xi)ws>lZMC+fQxLtB}tpk_=wk?wLkJNrTISUHyqyNb^GO>fLvZ z*8-6DkkEA`}> zM|(I*kge|OaY>naz_R{$M;3;eC@)2I8;U2>?al?+zSaB53B4}i{I3krZ*@iK!9aBTgVtVk zxc>yBlJDL8_gaAE zcM#P7$DPdjj&rCS+i?l57p;9CyIhvL2+vFU#UXLF`MY>~q`73G+s5XA*A3b}V%p=~ zcJ06-p|9JIf05QFxZYC#mAgIs-QKWZpeNdak<5nrVwiSFejkQwUuEoeH|qt1QrQe% zO%^hEF{;z~QEUTm>68-(PUhk3O)OuC0sS8$VzvORajM|OmzfC42`!5T1#ebcZ^qlB zV6N$HZP7wJFXk4E_s*47q|>-B&%FJ75wvqT#h52BpVGJ!>}C$Jc{k2iN# zx5j1O#PiZ;YFiYDfxLWH3m4TcI^mxas;_E!+7{9E*~k+H+RU&ZeNdf1v$s-{`1m8L zueP~O7jz0>JRuw+w}-)97wi`?K40n)44&(|zqdvHJ+JCL^&i8^4Z&rrvw}`cq~i&U zM&6X+AxZ-KKzT)#7z_=VKwhGeLlX)i(BD(Pm7dF)fV|A^(`SpyITP^0gnDc#(QrUa zTfQIePu2YWal(FGIx6~dVSt%To-5UyI6n1%yT!*}*QpF}{CvWBUPR?(n)C2l6 zuUo`dhJYSME{X0>Wbmr&5xu)y1`H(EWZI;Y))%&T?>V^z+<3cy(KWi~7YKM2$(kwZ z@WTCSmAR>WDxX7pEIW7c_Jya_c>l%b%xt4Fa}2K=)Eb_N~;Da1@SbVKNFgx+_+ktd4c>Vld zpAP%rxKh&o3$4dLOBhe2CU~3j189%)lmY&a_#m&_S|vr(TOqHZ*qdj45|0w)wa?dp z@!le7e?M~mPTK*RK)}N5SQ#`4{p)pMgO~Q^8h~pt^5YfX0APAFMsF4FJdJKz?w8ye7K>)LdeB;M>P)WSZ*h zCKyi~-*6;Muov>W`g8@4*d(Dn&SqT>O04NbP8@nG7+|>vSbiVSQeu?E*Keu5lG;)q zg`Erp`Y+_aP%^{#D`L_st7fGp`2Sct^LVO*@Bd$=M2Z$!QrXwYmR;_#Mv7G0h_Wwb zDYCm#wh&Q~B}%1|ecz%|Xi-@b$u6xVp;CVL_o?@ta=V}F%f}zhgWmVPXP)yqXU@!= z_q^{uVA;uYCICVT%r{uGZ<;YLLRQ0}?~M3q^TOTAxgx5y0b2A~9LH0ilJRP}8N|#T zi{dYPg4De7k0_qdUpt~IL{~+!$9h#$2~G}jdnD9?o!&g+*xaJWS` z2D}6tsOt;6o)WKJ86x;?V`zU!_ULxess_Ys$!bw7R6dr*u9ZgqYb{M@UsX5qUt>v<>o;!w2e02Z*k)Sb-K$Am zUxUH|-}*lUz^Rc?qiXOt5au5I!dHv@SLh0hQT2toz_hs3QF;Dx_`cKV`cm;( z@cP|Ay*o}N=FGF_Wgu`Pv$zZ8N9)Ed_c9{p8DPA$WegK1-4L(2mPe1cA9Eu0)71DY z;M^Oo=+8ubg+v|6F`}F=SGDHh{Vha$3|w0+a?=6DUvsy=N?J;Mzs@$omcR_>08kKq zxQ%Y-%z1IS9DDh+b1#H6C|td}gF3IuD=l36jCBEjBKV5ADsi>`KR-xc!Ol+l&v;OL z|530fpwbrcI`g>WLwF(LwV7~1Z(+tic&R?gJ5-sCw+_u@aM(g!Ut=a|u@8p>ATmAt zO2P!>G181a8onj8m)E!w9n7KVa_~rUccpqbM!TjY!sl*3%maB67dpBDS5s5 z5IQd{KW*L55DUa>SRlLFKn3x-x-@uop$5fWrh9?(Lo)7IfvQ(#ZJDT56;+3^pZ_UNc$KYFwp|$rF zbp7e&wofD7wV=dj%Y+MY|Ixg_;!DEPnd_^wf`7il0}EJAtGc}43w3+MzY~8eOx(|- z#LMicSwclW;38sq>0-c+-E>h`2vAG;&6zYb)TFF-zG;bCSk| zsqsC=%Y|RxN^u134_zr_e!f~%6LQmc&95Pzmn^Ql&eA=5_WD}yU79oTVD|R-WV`zF zisk5em}fLm1=>o8mw3;T!_Xu}UDuoqWFot5=;S!6tS@pG?nF_2uNk*C<*` zfbHGPgR*?Or0d1xl11Mv(|}$5onn$_{lS5bSC=kk=K8{an>2iPc|UkD^XTgBp}s%F zHIT-;U(pnjm#<2VWPgmu`upz^hr{C43v*(_-ZG$k-^y%7Nrl&Cr1|#4=)?DmuaomK zuwku^UwJ1(JaNw3K>As|Rb?r|4@zoyqw`bgHlz6Gzqh40! zS%J=-(Es|oXz(BVk6P>x2CFMIpm~S4Se$GKC@H9WloZb19*>OXsWF_(BGsD`uU|Tv z`Y(#VVzWQnC)T6<`^G}XcRkf8|I0D%xyg=Lw0>W|%V;s*4a)w@bGf!!uRZnk!aY?L zIS-xup;;3i9@M}yRP5$fk}6p`8iW;GfQG#x6)c-6d!Zlc|V)+f4Y zE)RHJK)gToCUEKI`Pe!$>bxp19|(%R zUZY!L*bcx zWBj%kGv|e)UC}uDd;=U^TcGo^JCoFZl>Ap^)?4%YI{=a|D0aIVK;g| zpPb|JWE0}^q1gI_&}iF*rYxkdmet>#7A3fl?u*$Yp*+S-fF%Pj{Vwb-%~~|x;r{bO zJ|D7b-EFbc?ijpvUAs(lJDLxzF|saJP}G98vfYU~ETJ&SF%oVTG7DZeb&qzAN713-v7IOziN*op8tcjgPAc%{#+RHU+wgPffA;O*ZGZE?>+yR4+Vq< z4OI*|lkvj7SvQE2-QpSb@e$*7H&jko?h#tg@N#-RVr;1mYnc^QJrqKr@SMz?*vwh* z`dzWih4}^YqaTMP(DlLE0!mkBJT|X|3u!Ta*<>njw z0kh|IyIsCUUefbts2xQ5x-oBykAVu}#S_G${lF28k0$TUAMd_KSzn@6taI-X z=Z{nSXX^Oq_H=Kr=*7|#>+<&X=0g!^{>Y|$B;sWbIWIHkg*?OD6?m_! zhi@v|Q}_4jFTd12D+qv$(pU;%gxfGHHVTZ8_%hrJ+E}QR^3&J@fPBehUQG z*A&ug^Yj0pFABVtYlIXReMI|dg1Q&0j|ZW60x!#1*Hwb_RjfYO-jbJ`7gjHaA|uJm zU$Fge?62hGW6Q?WNk#Smcvkgx^5kO_-wTcAb?9iT!~2#)S1RN~;C4#k(3_XDpAQM2 z6VI(XID1}|+;Kq*vR0DTC)`bUcrM<9^mURZig6AH%E#0?5hSf}+==7|uy_I^Q-UwB`a&jYDfpjwL+9I{3?TVU!kv%8tR${e{XLC93*ivkF;JFdzHu}V<-1x6 z_OGH>Q3q{Cz8_{fA)tM~sik~)=DY~)@2@UcDFd}{8WNR*roroX$K$;Y@F*}|#82gA zGf8}NU=W&b@0VqFSrdqOZFBRN%4vTz(!c8;`r59=~XR*^mc!p8qmMyXH6vPYribbzM7vi=k>e!IPTT^ zg?L=2&kHB**oAY9Amhcns=A{~2(3p8(?v&}I)!*0kUN#fl2ZyR?8;IFGhIR7`U9smKC)<_?QXd~}Mv5b(SJB>1;UI{wY5M*z&*UC+wOzugZHzU$MLD!Iw{2??yecNqVO= zf5drT8qHiK1eFWEp1*VTDw)0#nk@D1eM9+7%A-m*RJNgf%sUG^40pPf!3YOWKKqzr55NdKate5ygy5Cg)Rd+r2-li?bph9lH6`ukz zKkfI=u1l|v1ryJcTg$|vrQ{h39&5BOWm?aCf2guyp5{a1dDrJ-59zv?QRg+}@5cS; zuq|++t$VKl{X7GX#8u=hB7?l+OoHtDR%=fj$7{?5Pa8wQnDZ z*XX4*({M(02+p{Ee(#$QxID_ED?N+-p-RJ=655}}p!34T<@F$SUX_tA4tQr}4$ELC=e>JnqbVNeSu8Y}=}{>&MD~36yWqTDU>DOU$FA`^ok7TOK&R z6SvJdA8iG*NVme)cFOC8k1Ne?RC64IvP=8ZKkh*Bty|ATmvEUHY}H%h?)f7G`n00h zjULV19!x!iGJOJ zB?pecqZ3aG!k1F!WjX0#RYHf_|LeZw@hZmaW5UkF@pb5V1&r4hKOeUxoCZ2t9$q{x1K}*YE$p_S5pR=#P+Zy(waJ2L-;eLo@Gsm5l_RhJFse$+`3_X9JnY!{AVP15y-cZ~I=&nv-> zO1rPgTY>=JvEt>!SF?Ct^moMujz3|H_#dq2Y`iYSY*CBmkBtdyCrtOC`Qy-MeY1#@ zMez1h*}O!ZgTScnS}>P|{CPf%*Bz6v)#`Evr1t;%JGni+$$qzbW3LxH$>IK}@>ZU7 z{UhtP*WKKzfM>ZCyk8~|OxU;IhzOm%zVx;km-8;AZjV?y{w|TeYIIvUaSzab#{w>~ z1OrE;FIo31wL5c*L4&==%bnF3%wzkv-zz8IuO&Q)d2(Mk8?RGv`CVzV0SrmiAKNR# z{hI?hufl7qO!+c>A@=fhxUHI>&sYo(|w-iX)r1BV}!-7JR7mY+A7k2`^G7++%` z;UB!LUYl=Y55hm(;?c}%OSDJ)3i+Fd(R$ld|FyYOM9)~q2NHzF_vLTiOzO{8!<>^^ zK1wifIi3H)nm}k(l*g^8oxT5JoPR5uci-%J$yd7ZcQ~N)VDNfx!{nF{Fa5)3_ykyq z_E?%g7hvZE93Sn^sy(Ibzxw*OCDfGYliH8zcq%U~J}Wncv)=G=AZYE`dC1=h)sBi^ z^HKp>-^@sy1Mzv%&z&t%t7f4u*F1yDt-2st&=PK$M4eYB?t49-@^&!0YMXpbtsIZl zA1kNw;&51cQ!^Uz3T*HGnRBWbu1lU$9=+fMvIi8h8_db|Rk?Uaht$?|{6ay2ASG=B zAk=62(+N^O4@TfkP2AlfA24EUd)%#y?9qkbTruRm0Ukwlex^+f0?##yMFm{5;PtDa zI9vwjh+r>uUfAWyBi3$nG{||Wsam~FNJRdtozItb={v+ra;=9To@kG68pD(oBAg%< z$Nl*0Gjd+oeV(Z&9_4UEdcpc-s=iEp>+57U`NFrv^{e?7q5U-3`WjCE7*+6(5jz$= z90Zr5>Z3o}&72q4Xd>NGmE9oUO2cQlZu-2i_F(nLO79={Teg2dykzv+V_R9#^J$KU z0(c2hCEyWkkQlG;3>#uj&)256Cyjg9{IS#G-R;j6dH8K*-#sj|4Inbo+q8cT59zw8 zyfhE9j;0;)g|&Nr1TAMm>t6{0->xz~+W<1J^)wYGg27ltFtNsI=Dct%?YjGgJMqok zHx|7V&mq-^;`&6dsm0sF%zzgYI91hu5sz`gN{Nbz@H>)de@LW9|KV;&#Op|~`62qP zC15>lYU4lY1j=goikvLU@v6cKuNbK^qCNVrOM9F)pH#o8yl`pz0#;S{0&PQbqCzF& zHL6tHyX}%H1T+R$IjV%fRbGOCZ&jTx67TOTU>U|xE;B^DsN^4r&ng; zk5?;O?c_rK2Kb14_zI&lSe4fs4{H>i_2*}AQ)%~oV zIj>*$$Km|+Wmd;+o(3;_=VSA?ucOFI>I+*Hdl_1v;4K+@cBcgSFYdhZucymOz&G%T z+Zr)vXcUcbH&i0$Mfh+-+-Y?J-gcg>!|}z`dAaP93d>jagL~Vx82VQre^+$u^@n3G zlpu1l@RGiM2+-C%S>Nw9dtSr)=CS2+PlFc*_^%ShUxAP64PIm-eLb$ud~`Jp@uE=< z+Ija}F;qMkJeom_CrnFUT@tIO?C+ICtRJrlLhYxduif_olIAq|K$*xsG4WosKg6eS z<6&Nh3Oqd2c&&%9*b$r=$^z|1f>=(q#%+~AB%6I7eRb9Vz=P%ts zyt3Fr^O9g6SEQ5jg`s^% zSmyun{-H|0K!Ws5WjKjv*+Q=o0@=2ggT2jW&+GQ3AL<0*+3Ra6uZ9x?xyG4@SNfi) zt->dY!JOli*Nbz9U~Q=lv+*&?`oi}&zv=L!&MUR!NVm&-A9y)bprJpGt}k4DKc+ic z4Cdel4~tYBg?J-|&9ObR=cN>Jf%jnMCDQmqaeNGGX`WzY0{j_$4+VGhd@yDwnBuVY zoSHL=xG+AH?+f<|TGV2K;NiluF8OG+) z?>(V|Jgk+@FfB^`Dv)$jlF3Q@J$D`fclIvnop z4#g4uLmMDs$MYRmxI$rD%% zua;il`Fi&8YQp#3Uh_Ag{l|M95eo7tD4uwP|Mv6DcVc|c`F<_0wi6tqf4QQ-iLyOD zjErBC*GOGoTFkX-bh~_r=P)IHS$Z8^ZzdP--_kD+<_fpQjuG>7ZdV?DeXX1Fd^@R( z_L`+QbzV*$d0wm5%%0a3HknhIYRKO!FEvdYtwjDyP~@kpiaYVVZL!jY25THaTf{H4 zdokttwq{%Thw5zV^KJgHM+!f~yaR|-~}6c?f`MUvfJd)D!Vi?dt_kP{30p=u)f9sGS%1$pypGT=Jx|oho&3D*U-pf|nZ_~9%LzmI5SZOhwZ|nQ ztGRC|d%=iiZQ#F?+7`CoWkrzXQ+CW& z#}TyJZ;MN9|A)Tjj(pyelZtN`6s}?ABkF6sDZOPGdOyKbUedBLB5PiF!F=<3B5Nv< zzh~{wmt>CyKl}Z;abL zkSV}txLW1B)+hR-7aG~z+?4Zkv^->JMytJHT5*-!BQd|Y7V&h>ifVm z^JtC6a%7JrUPsEV#fgVNWNPZ9pAcnUmv6nY%;WzrynZ*lHadONivPj-J)Z|hFXvs9 zA3ewsCZGMs`^p;>H(9D4D~5#$wmbH3b%G3;3m?K9$$9;*7H-t;b^jwh6dzFH#d`3> zg-I_TsMxIY(8U^EKb%TS6Q!d7PoI{XSK&gyWqpoy!31T0PoKFqc%{{U;pH>WtV~t} zgc)vnu1KD}zBU}c_<3#|ioZf0KXVX0g7kIp%eQ-#_+mI!*z5T0rX$qTq{)lQQs(85 zpmvYmd?b z6`tMH<1e?czTh3q-mrkqN=b77#UC6h_cS$|6d*Cm^GWQkU^tP*A-9TwvOUhNdVQg5 zYb%}-ugajy?}}A*pl)KNg74{R@X}+u;k|MKkGu5T?$+)L=zITjgK@veb2*~#Es(x~ z>|E@OeGso}cb=xKwJnC{+nO$ESUEx8v)3+jTglfaD#afp5J}my;2c^;J_dwu4c!AGK>rJn@So z4#)rA{=pN?Y4E~6@K=faJ?FBgdHEYre1C;L*LI#B;^mk~^M;kT2%1^-wN-o_VfA3= z;GP%%h`)Y+zk7wJX>2yyPyF>~a(`bpQX}qp&PZ*%P8}D@-FV4cnfu2_i>qiwZ!;7vd;P?y8*iX&ggsp(;?%2?sZkIy>f_`_31wG zXidb+`)lbvMb#n@8mbK`33dc7vr`H-NB+U<45z`pj#zx;0*{PwTI7$gexDkDO^6X3 zcAxTsEev&^dNv??Ty`gQi^nlJpxZviq8@${_UJi0W-Ou1tJ%y>^uiD7ywTSZnn1kfdG?g8d?5>y-nMux?UV3@iN|*DG0ME^ zRq4O(XzRxRY9O&)6#N(Kr)OqtmuAn45HP=g>?WFTM=0+~5?4UHxGMUMcry#(=1Nzu z?0JWv=Umo6yv{#({q7N=uW z;P#ES@O?i{LcY`P+VhN*$H%f-(}ye4zf$DI^;-Bv3GsaW)7SsIy+UKOCFYU~a-{h*Mudn37$07Wd{kytf~rEa@mdhV|Ld^zU( zi~Yu^GHt+@DmXFg-9ftjkD-kF1;1&P&d^8mZ8|R{m>Y)lH5Py7 zI&(@fL?2&OY;)8RvY%&zxgNQ`2rsV%)=n1VpKRx}WH2Vqw*^&b;+IffpHNd^f0i_po`iY%4dDkn#lKZATFP4PEEuU2c&i}H0&p7#oRrw=Y_nyiN``*FG zi@8gIIxp<{ze{9~Q+bVX$=I8uB3^2(_xhH%7DHgls(VT?jNvHURw4;{38aVx0V+AM*IiwbONeb-6EyFX$|u8;|T!i*I$d-48kF9A3B6 zvndRe!t-Ku+bHXc?dnbc10vLUouF6W_lSS?yf|#GuWK(w{@zLwtjgvhUOEfLL>FHz z2Ib9P+ma#qu_<$0xcdHfNadG%U&Dy1s_bt%)@aU9qPAc~X zaPP_W*B*%gA@72W%Q(vC@twPM`A%mmEy+(#)t7{NJm*K^`uPo><+Q8{)Olsun~Ax; z-2-MZ*=c#!az_pb0 z#SfbCeuT0&39AiZ*Hx>>ivre>+A-DN53-udJQ5?o(6Nsj%b%e2LXk6UtlTv!aEfu6 zr=4pE*tJ$0a3oRY<)ZUJcz)&_(tW1#GHb77+P1(BUgT=$#4n)k@6%2%Q1|;_1zAHc zL^`?un3w*xlZ>DdcG=D7{OH01ueafZQ2zaKl`TGH?iCOl;L&)U-~zSrC3hX-$a!(? z&2HqEF2=j*SpY|nArO=EJln2{k=ikpS6w%Q;d=oBjPCDkT3>~p-?j_gZAPe4hG+=0 zHM0tbie!<{y{HkLe%^C#qSEqz(B2Y%NZ`!w=jQ+ciQ zPg7jeitLeX@8+OegXP5c7s;DlY;lIOPnPl6H2lMV{r>8cJr);6zz{lJO)j765v9oM zr~k66>UaI1=*MYe9YLh8##_sn0={nmi@MDc=i|dcOa0!OPn#+8%F1}tPiM(Uy3bU7 zRZiyin5Aogq~_P}?;mE8uA>y+SH5-BQ}R;;oUtvxVPY!&SF;eWP|jB_Ppi=LR>t9+ zQCp0PN#~oGwN)ydQ7?ze1YAIGrwfR_s^+Sd{0A@Y^^Ni+>G+~d+DEImBma%vZ>qke zE|0Xx3;ID!NtM9fcEqdWe9=PZUKOb7P;B~kBpkv8Snsnepv>!cH~lV%?V1wb6Qu2N zgmyJh7?xnRkEYIxFy=#}VJ85%WeW|uvhU)t`eWr(ULj|Dt@#uXuaoOoB6gaTgS+g! zhzb=KC}iIFl*acTynfes$e@miYZ~Vvhw^yboZutCh9BA`DfbcI9?vI2CO#@KE@SYe z&M6$mGQe@-zns5nZ{~8WdEIB!t|`1a@lW<1i`ors%QTc-i>JXWcz3+^3pp^Wlw%iZ zD@EV?pZ1OW8PG&zUQiMtJ#Y0*{OS;jr}90WpUN%KB1XpR7MBMwyoZGxx`3&y7LQEQ(oOjJm(aB`T<%S-c#$_?35a zh2J98|3A2h=jd=nx63EKLC@EkewCrhtL6i7Y=^;Dt+d>D=|6b= z`a;}|XJZ+){zU)P$lIXMh`ncpJpK|4(CbYUA;7ckePL-I5HI_>;+Bg`@Nn`?Til$= zaLCXp6jLv!%qyU*x3x+2zpUT?uGdi2XT~3R>bwYA8p2z0RRBlpsn%zCf{d4wg~0H} zXX{A%ViFMJ3i3g`iWXS3OIcRHQrfHmkCT*nnaQ4;%NLQ2FW~qQw`73Gi_T@lsEnTE zU#IG;i>rbyipL*TU)sa6$q?}>`k1YyoVEcPM&q1c=Y<396_k0k*{)fU{(!Q+ ze&fQKyr%&j0p!2MDl{5D+hu$j%85o(eczN=R9(d>AAz zKPRXWDC_HLsk>Bu4)RC8?n|z(gf;B~79##&CHk@5Wgp@-awkY+ z?s3%%k5cBfN%hhqW_jwo6yI*VeQWXTdC@*NTBl`>{MWimPpow6ib&(*(lZ8ZW!&ZP zw(gaVzp@K_OJ6r%iPnLn8!Z0|>kk{AyGEVUc>eK6`s>?%BLp}QY2LX1kNIP)@zz3~ zBxU&FnQ4Caa0n#d@%|J=e19;d^=P9j;sf6Q#ouG~O1&|1Cgc@$e~(@NcZu{hmDd|T z|H$iYh?fx4o%4HEltY=>%G(<2U4UI{-l%uVxr#Iw^k;3M}F&)a@*hV|j& zWhAas?Xjw6b${Y(0>l~#XT1Dly%6KIkF$d%_e2P!vT^A=aHq`c*pm>gm0oBceQG>` zwR_sUMB;Ny28rhhNhqbVo=L|4`aI%g5#?Fatc&!OE1dbP^^f-t9ctVg#Z2Ty&(RYx zzTE}BE?87D{*l}s3F+>!IEzd?i`mWWsy)>0ajkdb_KWKNkh;&R=b{c;pK#!DO=yaf zh4JXvgA(ujAvO_L8olrUMST_c$!C6<7(?xv5>Mb)tTK|wRf3YWE3SMwbCJ}4l;Ty~ z%MmsjZR{Lpj5gxM{E5l*sXuz(iMsq1mI>l}|FHA;QpE)OC%no)$CaH%#@Yn}t~AS! zY#^U+6L>a!mNZVpyB4SKENrz0g4E=}0m2VbJErPuo7E-W7aROwg^r18lnU}+X@P{Y zIM(&>)8}4vWxGGT9I^gx?@L)5+L(mO+t;GK8(G^b++8fi~k+S-Vb9=Cxyb%_6 zZ_Q3tp1r@{Xxu)xb}u@wV;KYY7@-EuA90UfC3Vx3Luygl`Ddce;CW*r^!PvK+gNv` zXiFbX4oB_(^>^~|G5q2=+s}Oj&}(7cLMw;t@%f_%*O!l}z#gIQE>*{qaL_rt!}&R7 zUdvhD8Zt?Z5;Ib2T8iY#zz89%y}j5DxN8#DbVzVAyf)fZx_}=>v&W3)nTQ;RHE$<3{v)* zo+HQ}yZ3jRdUxurp)0|g8#u=hkjWf_R{$qSh%YDvh zla9|l)hlm0LbS*6tPe4^L!^G6%InAz#8H%=*`!0&y@%B8v4>UQ{r3aZd0~Y9E)lPqM~}9gxQpUdedVC#vU(_9 ztr%&%t{+?utE{=|qlTTpyE&&)*yA6(e&3+1|DBL}9d&zb=6`;+@3kK>-?zk{zX$0{ zdTx53MXM5sT*DpO&2%y3OJ+7}#r z);%@4hSsAO=Q|A^kx&GS<0UvL_LETf_*P$wH)UQwgI8^_W}>dIppyw+L;r*GF{kQF zL~+=;eGcMv<>RP%*M&0JdaJNw!|sFNsrRHM;yiggf$>uLF1cq+ojR|2{DHCDH$IR& z7Jm6)-$s(ZNMdH^J7S^)gE!X=Y4M(f%cbumHpWxtwaVI?w(r<~;U%8`{E`+u@Lc0x z`)0{B?9t5j*1Ogfr27177iE17j28tT(nag{87(7uhil8ggqt~BKm8!YDQ0*C_mlI& z%jkp*C8gm7V6k2D_Gd@IRP8+Q_BrdlAri9gUVx9ZUofFG(D;$Z%L0h?EZIztcSk zsd*l81-9h8Fn#$S61kkIMqOXy(Yrlt>3o4r{_|3kPblBQ?#h{q!v*qiQhPihfe;E4 zS68NO?4^AES9tmR0ZV`C>pAlbORhhon})uyhWu3`eN~>@Zq8eU#z)J`&RZY0B6~b? z?QxDS@w`XE>y^^Z}bkRk@jENMDiOVFjwG z@=#RRH~iiy6ykJM*Z9is4Dm877S3X0NA}pR zdMV+SX(`BG+I4VP#{u-yuO}4G|A)SQ^GXOd(b{?%Ti?g-O0F-)t`wmWkFNP(NUA?sb2VUPEeseoYe!d*zVO1At zb~Jxf8NTY*q>OkaBuFwPZzzSv}uS!AjL#CzA4hJ}G-K10&LYdcD zI?cx8|34o%J$!X`XE*+r!pQYyd0E*1TeKIjU0&T$CW*!eC5wWz;|>bI>agq0<-Sn( z=+&4ao;Y(}n8LTdQhmX*WBRb4epi>;3>y(HSIBFUNMEyHak_ZcowTIIj`NHTLz@Ky#d!4_?ST)@uK5x^=46$ zg9PD2%6&1RP`&*dlk1_`^AbLK%K4Gu^m$=5$NC#9y~8JM2c1wp6U+VV_k*>Fm(ALS zi1UG^V5qjfIp5hH@=tBH9KL1s+uw})Ve@U=Z+rB7zvqf*BXxU>nmhPp{HPajE|QS- z8rLEDuY|3vuI${h(AKzt&U!~E%vrkm`I2?B=cVRxbm2qU>GQ(-XKjtB%EM!vurgFG zHQ>fv6i;k4v@_#-jCid{Qg`iSD}@X2jM+lN253s@?rvbQq{z!S_2Jb|F4TE3-|~`k zHu8kKamTc_>uHgA-P@e7X6r{eNMaToe4ZK#bPENC=3SgQug((sB~J`YV55xlYyIui z^K+bR?jPwKBJOWJGJKkfFC^o2%VU%rXG7A6}kVCg2Pe80d`PXUGh@lAj)H#$}6+S z<{AHY4`AajZDai7yp9LBo#q5rId~}>pp$JF3Q2KhmCNtVo>$(UNIDsd+3SmW&-Y}O z$H?CsD?M$lS3$gn>beWI&nbo2uT7CPJhq_l;~5?6Gs?U=`=r%RIZ^jtW%j2%ZPYx# zQ}Iw?%LIzQN;gZnYZyQtJB8x#p?gT`M#YS%?z4}=W%26 zA**pm-=*72Nc$a)Z&Xhio+^PSi@FW=MA<-6gpyq^KV@D5_NH}%6WDwS^8)1Y1c9&9 zl5o!*a49csiLX*1*(38MyT-%!Wntaqs^sIjA;f$$lk0CeXU;46O+MR+b?P8+&0+oM z?&*(@$&+@cPUn=7{K>DDjQcr=@QJgGq4kL)#C(a;duV(-ZyEPJ+l`pNy13~Hzoji` zZ1oLv%*NhNLT-;{4Y!J_#i{cWq8IY#VDkk0gSVrSThV&qi8rw~Q-kDS@2Ilw(GKGI z(L(rE$+a`*Rr&PYc2V|y@X#x@k9&eTFT#kn+k$QtFx%2n_s%JojF%9eDOyPpYP5IDwB(G$Wrb6#AF;*RMnsX@UBZ5H9B2dB#z$1J}rS>4m0L%6lPR z_o`guu75)LU(%+VKL@jvfbxp`)jSKS^U^cmZF9|`&a1Taj_si#Phh)OEA(jr;$@h0 zg*!E7J%sNyELhGL3g@orEVI#`Ij>)Kgu_MO<=~VOoCdE1ZOQ7mwIu(5*&$X=<+Wwe z@lS_hk-m1m^EbSGxCBlruhZUaZ$s4{C+~Oglr1C1U&ptHoDTX(>Yu6OWAa^FCaWYb zz!hq&$ZSIKm!wfd`0553klbIT8r&TMZdaaMDw3Q%ui6@#KCemY_K5NQyF~Fw;&r9D~3P``geyhbF- zvp8swzkhIGiNZ4%#A~d%aKKTz1Vq2^?UvT-FYm@YUe7zWV$~75yubaxPO0>t=TWX$f<7dz7^qSqbUH%8>iT1W>PMj`5 zyuOtDV9_Z=yk6H%ng~slfJm#w(63KeqhCcio_G+kR9ZNZdi*7tU*l=jP8=Wa zypL!6KXIjB;QyenpZiz8 zKS#b^Sn%_?^@UOt-$&mNQr=yTcrofPxMIJu6neD!KdcnAhJEW~eN8V=<~6Ux=0g54 z>bycv(Pzy2>Is~UiMm`%(R>KwW%TYrO^-w<$Y%J>_c%OrUYL0`J2#pL_s`yc85HO6 z{wPE1(Ha*eHv5hvUK@7?$bEiU0(PtY%)f|OgG$3jo&|F$^9so~--(x{&a1FKGShuN zalNq3Ct5uW`7iI?m_08u-&3v}bC5l* zvkl|eyBYDa-M2NjlIZWve5LyA=2}Ca8I8Z08fE_#uh65j-;=ukTEM<6CH|8KxaBmx z+(LYx38n*Ug)*~|9!a2eQaSRqG8pbAI*E+fQ+_`mwf!ONdl#zTeYu+E|6q^U2mezd zf1hrq=N@E>o_DWbl9DmtkMd2jCLO)DttbNFx6AlEJT0N_dj6_-E$n$8^7wv>w+LUi zE4Ckk-IRPjWJxH@8$RL=F{Mq5?|wk%N6qq?z2{6hT8`Ez%0oJA-PPc|lXS|<= z)gbdt-2LkR!+Z$)+^-Vxvg0fFe)kML&*!zg*`)9lI{)jwqQ^!Sp+eaHWz3k4!5lue z9ne&;!SZRyd2MxiqthT}PSOi@zp3%pi=)-yOTQk4$g*XsSKg|Vu8(QD{qnhw6pRYQ zHy23x*lfgyxB{9nc_V-(NMF@%uiM-!F>y77TfLgBTC4d-3ViksLf$ zf2^!q>N1D^B8tDZaR&|t{BiylQ%{rb$EHHiTJ!kl@evESl^(lf;67!2)jm`UT(!&s z%v@tU`)MgZUmIswcWUDX4|u7;di0el@?T}wTDO$kl^~vzs1Z3Cdjet(b|!$!?Dh49 zaP7&~>P)J8@n)27qPWL7?^dED4F24n9zJ;jHpP?=@0Xgn zzVNqBjn$heLd&Ws%HEZF zn|OX<;B8Ji4og@#_|wjsfwI2xUa+}rZ8e9^%8b~-6b4d%PPND4n#ttzn>=8Jc+3IZ zGvx1`bjo`KL!_YR&7$5Gi9o1-6Bn*YKXYEmrAt-ACiTG5NHG1hD)sTPvY@Yb6$6T& zF-}-%7NMPKbQ0NPVi9e2;t=BXq(=O-Kt1j%>R^$qef=Vefxp&4<| z0NR4jSp@k}=Y`8+SrTE)NPI7n;AzI{EIh^uE2r{0Z)~B?yat`8aO3^nBysB^U>IJ@ zCFet(S4_vnmUGe;WO|;;YmxLe|5sz~z;kqRVv&L_>H5eg-KvhXQV^MSYGIFmARLP@ z8ks9Nb6%Chy7FiEG>QJcXO%+YG(@q&Td>LK%Lh^Pg_^5{q7)8pxPXB3iFzo^~{QfdcvY(XBk(GO;Fzmn5&U zbfz%krD|z2pEDWps$mT|*2qlRf0@;r%PQ=!fJ&E|67F%7$2L`8mh+cpydmlf|Mkwj zYHq~qY}RZsbgUDSD*t^d*L9);ewh1nri zPUU6&bgz!hcEszB$94w^*PK6+2aL2m=5wq=ycY9z+q=lhfFA41_T$Tf;QiJj|1bzYGJQ859L8o(v)LGam~kH^@8>ASvV}geIf{Sy#j15sm^@u>jVjjxGbWsw z98(DyFQK7V4{vWr&)4$vaU9yc3(be_-y`g>k1vI->kZG^ezSoN{=|88!<6TbHs2Pw zcpf9>S7b$8S7@i4FZbATsD2l%H+1c;++!h+@=abXJJ2~0Bm=K%UFqYW2SHaO-NQVY zne+Ok3mnewmB)nDx#`azD>L`cZJtlQKQta!&+aaU;)%3-hhsW85wF`#%e-%%Ed_~F z;=T{MsPpoj!$Z3#jmYcB-YzQz%JIY=w}jQ;=><`75oLEK4M^i7a6SBbfPOvf<8=A* zzAy+@)frsv;hH_KmCHQ8S8b!tYsm8Xh7unD!n&GbxgFJH`Z{gbX*u)+omb1%G-SA9 zJL0vzyD`3&n4h!nfNoJ6hb>6mIMAM{N_jnJ>$uSQ@Ng`D2kTw(`S$8B2X}4y=?Tr^ zoXUb$TT|6QVZ zB1^m_o1U31JYA1?6%p5@nuGXBUnfxUfs`KP47>>+pv%Q1|YB#WrFKC8t%nKiW{{Rc|14 zaZM$4UJRo0^DTzG;Jq=Ur14zjzoG^>oi2{cfWIW+#O?LLP`|7`dr8X7^+h;UoO3o) z6khO`u%3BCeSF0D5=y-qHf5Y6;}y2rWcRBiR#N_i{PUKKj^oH4i$>+@4gm3azIhyv z6CT;YhdcLmHuY2HRght+Q25mvDpiLAL{?LNzBb-Yc&u{15Af(6mhc`&`S)kov&(Y3 zq~YRT?=*NH1X)KmxJL%hoEO&q4w)0*Z-`D`U)bkb13`@#DcUex~=}hAOwn>`9TB|T3uM3Kykx9d({+SxD_8A`MPkQPF zNk4fPuDgxaCn824$)9(W212-W9|J=$c$G!Ib&QxjFZP>yRKAVo-wGB5pweIZ4r)a{YBVcEoe;{NvSlAi{X z&WKm=X$RR19w~t8qk-w>!SLcCcLV!|+4EAIQv+GArf-jw^rigCCSH#X%^$noTUGM) zAYQw;->(g|Dur8n3fD^!_hG5=vUGXwL`3}|>KexH5B7K#sS4fzCD1}ewstt}@vV|dfF1F|Z%Kl!q ze{*%rX)7>j9<#nF#6qgyRR5)Y@Fdsyqdu@PDXt}33&mdsuR_&tFP4Tg-_Nyms{{ef znDPeM{+aU{>c8ghNsL!YiF-{4ZKuKOcNY^l*6A}WI*tGJEAsa_XVld{v!eLEe_m|u zCQ-!eP*v&5SAC^mA--o`*GXIGjh7^xVKgRjm^$BndnIL!n}!us7F|BUG*2sUUI$mrU9!KjFt=;iV%*6O!*GAcMnW`;t?yCJTxtFrOj?<@Udd{-~ zuA7dfX~RoMT&MCn_(o&Fy$9Z~oyD^Auo8+V+?O=>eU+Dn4`LCAUj_t0CZES!6`k4p zd*@U07PpAcetg6@V)?7rQdqwBMI(K^(i^rqX;KEm$9Z`irES5iLO3&`o-(iDWwKUE zxzzprHo3kec_SZi6H^W~a7Fr>;BQr&@Q{Yd*Gu~Nw-D#sIA`ln{@L?Vx^eJ9&@|pZ zgz@;hMD`f$DkhQn1?>+tF5EDnDu#F+47tM}p;!tV+ebad8?8ZxS3y8ul`=1B`+1rx zb*S@l`;yO8Rp|{vS!(OfEJOObVlCd|kt_vA1_xrc$ppcxhXW^6W-%X1mV1Bg+}zEu zE^xT%efIR@3Bu#G&~Fou|HTdQ5?{G$fe1Z{zlK&wybT{i_PBl*yKdgXQjjuDmtqmK z2Fr$a&7yqD<71NFguK!T>f>W@a+dhF2rt+^@5SiqP83f#G8ezOa##w?1wL2X5!a&= zMLO87SHy+Z9sS&UXuJfv z_D?wHeX@jWO&NZ=O_bLQy){DB@)W4^n(O#sq}jp?n$4qEp2|h>#CJyPm_3COK<`6y z^5^wHSWtH`dcWB0{r%5GR|l*ALH>l9NmQB90P^>9RqfQx!Vs^!B{~JyXp7-MgKMhO zeJgNYd*Vc)9Od8uohj`2efG_6+Ci&R$>4zBbFxHX;9Y$A9R?&!; zN&Z$~EqlOHT!Zp{M^S$Gn#2+6`2gaJhaQCYxGUQYLPlRiLb|mCMfA3daz6<vUO-#49IZ zb5&EM5bPG!79)iCf>ZtHQwy?o&6pP^6aRN5Y?)>^X05MX-LL&7)RvR%F(~+CdMWW7 zHtamUIU8~>%^A%CO^u45tBH1|(7^3gR!Ui4cUB&xe?8Zf)P73(a$B8lbT!r$aGPA% zmnyx&W9=%4yng39rv&s~SuB2jJP17IFS0MSG@dapLfB&hbF2>Nh;GQ z>v~tL%|rQCR#i)M=ctH+8Q;F3t7<`z|1!?`u!hNu^@Z_T5!jMwIYylqc0FMO=gl@} za$Zi#h991GuO`_er?Hgj8J--HJ^mxL5ZzwQmShuSx5{N{?o6f_tUxGI#KP?D%rQXT*zlbe?1$QC}~5)U)FY{oJ3(179|RUZGZo;ohF$I+5mEmoUd_(A{(@)Z07PD*skrA!OVJ+pH!=M` z-rhW}rs(_sPc%@JNF;l z3K>!v^1Gimr;YnQPH%l5KYw)G`JCr@t+m%)d+ojFa7ws=dHXf98H3v<^_5b$_?uy@ z7$}wsw0C6$0Z;q8?erY~-hZUAU-Y3`Wee0?UMv=+F#UKnr0&o>r>)4}+m0TIyH$<) zL$Y0rx@!Gl4H3u6fI|s+andx~RDA^M{D*$tUfQ zdpWZf$Dv7og4rb_?`TErLR8O0E;|13VK(A*Vy4?Q(=CM%94mHDl2Uh*Tu^;lN)^99 zRBI=DuBDUM9@mR5wmC?m)PHSXPnYuz*(1x=YY`e5BH+BmMX6b?tM|^EMM>3(LZWGfk@Mmp za8d2PDD6KDUcRcCQic-H@Tr<(sowvx>Lj}KmC_h2| zM$0*0SPVoKpDBGL_Wyv_j|-;B$X@-Z-+KzYHUu8+U$p|t-DmNgK6jdmmq1jc89Ni2 zudw{yP5lII#Eb3C0);Op3ZUb}^7TT4{{ydJS3U{xy2;=BGoCmUGwV4|v@0BWRk`Ix zEsDR=IEy}ntBON)$rhG?eL=81vr6`D*EIC?C7I`J@mlDY4T@%MCg#;!-QdFo8-d1e zX2I@@7L(`y1N)fIpS-RPD~$RdMZ7qk&yBp$TmU(520wO0n!-l9Jn~31zW-7S={feT zkP?609CZFzg!X0plb7;U(v?lmTp?QX&8f@o$lo73$`s+ZL>%(Gl+zw;2m%$kdqW{@ z|IRBZ^c7>!{QYpKvLj-BCNZxcT>kuv?2+sI4Z&x0NM9!&KG_{&gLv()5dG$mSpahr z_a-}hFo753>e;(L;PX0L7Nae5hS(m1_FVGZbIKLwEU7=AQ;+&X#{^G$v~3jymBz*y zub6_shAY>8QN_RWqG`AEQ1vu{w6!s-y|+!B7s*^9w6H9eN?#8n(#0~^QNGYQZC4HB zCd4ai{aaggN`1M?+#FitE)!UI^q6F(Bz`>MtvX_$YHSE<ZBgP~}qwIlfP*Zi^Mxoea>f%eK(|M$bl9=#0$=^{DB!NZiE=^*AF9J-<6MQyPPZlvgz(4 zoRo1gN&9a}mum6#HTa~y+me$wA8mds^I6S+3p@(eS;irTcqu4-Yp`0j5;P>=756_6 z0w1dtg?AtRJFng|1J%nv>BElH>hYQPh~smUh)e0 zFa3CFR-19e%V2zC#?!Py7??2DZZ$IlwoPwyKMdiIr};(idYk-wWTnNMCE}6CGrR5ij|--{Xzl3t{bep4j;g%6RSl=+mkP@yBbMxp$IT zrZC^CG^FEkshb--?yo($am%hrzS`m|q!sg4!V(?PS10BCV2S+IwLUTbt}pERzZ~fP zUXb)Z@WMX$pCjVcSioHWH4*XZ-}C51+H5qwU104m<(g6mD?gMTWDqq6Xch3)nZVcA zxep7T?zSP;*Sm$orRR>if$ax}m-pu*f1e?tIRbSOaM3`Ey{Od}!p;syuUIgBUZ0qp zAI+ASKCj5dL1zXsQGe*_!)yI(&mn!yKbFhv{5|4NMCacK@NMAAnIiFgXQT|JNf4$kFL^PfzC6Tu+Yf(PDW|hl1ddLKhCMxe2 zy@>Dc&v`y6Q{g3!ze3_pbo(i|fI`WZ@k$!he_XgmuuC{p44xHE_`j3%1+Q1HXB;i~ z_wgZK?(Z>Fm7bFmb++e7$Sz+#!)k|A6TNI~H4o1hKD0`BlsFqVaim5U;PE zFKzF||tqbVqVe>Cd+u}T|lj1mhLiL)PM9}%EujW zQ4CBO2X8Wp_(Ge-g%SnVf9His`TsWuzI!_qzp(=Z)4OUTe-JTl!MgR|rruTIcoN1h|%LVs*KQ&#QM;NQCYl zVqUH%snyuY8LGasjojcy{wv$fF|5;26f%5ro_xvig)Q{& zUDs+)-yXG{SDiFSp1!^oX+Iuc{~r0Tc)p~{(o!@&WZQO5y8e3}uqno=y=gLmn(ykJ z6(oFK-D?i_@l*PHzsc+W_Q}1x`oOt|>agbFIaSa6)lj-LF>!676EwPczwJi z>mFN$cdEgw z{_BNIKxX7R%6Nwf{i#kJU#N+>P!`iOeg73H`N{U>#_98t+0hVME6O?PzjS6M)Qx)~ zUK>2FYc?6@!Oe(0Cr%a_!+N(5ds6-~-eI(Kqxum^VqS%Oa|3aF9*^*OW`IEPw8iR*LDzU_3}_tXK_9W`Ou ze+|uNP15Eu-*7_&R(+(~yJ;aATA$s>lBM*&{@ETQW8C9T%R&E=4LqT8)30aZx=iD_ zyW?Djrk+FFzhPLc}&wBp4 z(QI!;uU^}a{8zRq`}Po3#7jx_$=hP-JeYYp;(bpsF|VAg=4HK03<&-GL)T5xrIt=m zvDbm6_%iAbyT%D1W`un`SVU;_H<1fr_;qlsEF2C467}eju<(e}l-GF$tE!KD3^ehir zxo)$U>6w7#$cn1e1^B!I!oS9KXcF`4O>(kH+3pMrHQc{!<3_v|rk{V1)FuiRs~(n> z9VLTpdH;ur1^>>AhU{tfqN;}EF2l}Yy!RlNMSJMr@oh{_Z@~AKI;FWvzEQ4~|znX4yoEzuL`XYM6TP^%dz>T&lT5iwEt#?Yn|ey6NA-yC&;J~WS%9e?HcrmcwAVokG&z0CPg z5c#Nk2a_q>Q5E%ypvTYWEMU2!)km2>jQN2dN9y=Km9Cmj(!>Qeo;atiavs^ExZUeO zlPBUZG7{yXNy)D+uY0}t#LMaHOR~Burm$iPyna9NpO28g-!WJ3v|St5B(HN$ij6mx zB3@$h6YeW&^I*XG@(YVv6S#Kl0G*^QKCfz8%6@+C#JpraI7y!pbb(lbyMvWHsQ<|I z{>uy%buj=#heKVyWDwahKuTOaeO@B9QSJ8&EZQ{#wVIzN} z%}irsSaEQjy__IEuXehIyzQ*SyoO6|P594u1;-H2rmFQYX}`~1C@W9si-5GM`R$j{ z|FK8c`AepuuaeYtO?oo3~+g$`~}nOHEu(^F}B0q&>#Rii+Fk z{lbe=Pqy197e+fbRgLl!^GXt8I5j#2dlbGZW4zwq9rixYa$-(G@x9@_j*j;o0&qzy zLhX1Y8NO<4j^Y-W2Ctu&?4@cQ}wq%)BQ3y)qP{dkCY<=me2Nn-WVNnT9) zSMO{(gLr9ZjvuD8$pwQdzv!}g#t@UE^T_ikzW=InC(YrUORTT!ezBXXj=RIc-Ok^Q z+)?`|`!rkDJ5~fVE>9@vtn!73HQ_0CBLCh#s_{JkbhXS7&L+oQ@8BV>pCEIa#BtYd z06boa%GnxoWThv0>ALFWZInd*UNB0!j$>mUT<3`D`~Jy@ke7U0oYtL>#Jmhkb~pvl zdqUjh)pPi^luPVj{#8TWU- z<)8liM}vE;mAak8ylCF5TC^^o0W?dc*%QwDQQ4yc3(Y3dUNj%(@v_->o*YBGmhe~~ zX===azVDg&30fwEyp(rG^hHQKsI z1j6Yz23))73rgZ5Z@tFK49Bye59{*qJ$*~Xq=ZO3lE8XJEljoOA+GB}F-23r3 z#EZ7I$((K=5BLu&e{6n1tgnfsIqD^~#Jm=d4GPG(dV+Il$?M%Ys6P}R|HyK)k0_j6 z^hRcLnlHfc81v)SfA?Rc>sl<^*XY9o4YxKaHRAe-@}{!ZAQfIHSCpGMPm6jz_X-TS#=_wy#4D|)L3-$XKBznDI3~ss^RmC{!_9d`k5FGj^QwXu33-9#^THvk zuP9#iR$5!e;wug`^r7t4Wxf#8C%}13W%~NMc`f2i=M?IhdM^Y;uJB$5QftfR?6#oh zg~<+!SH*>&htf);@gc>4Z7Ejm`4D}d)-|Zg1VWg$kRpTe>rJ+b+J}2gVg942G}{9{ zdv6dgU6Y_tjrv2}>`w14s)|B%!jUWGTD~x}+U(%kZPVwa?Y!Juw_ytY3j^}o5!qvL zK7Xt>3z{#=ku>+)@mi!W6LRhv*~9tpQMvE1jUX|v3t5dUU~!O8UsY#@A0$)er_JO3 zq@^5z^ySbcoE|780w+&|nmV8Lg;V!BJFT;(&nuDPZo04vu|58VY4QQ$6={>MUCK3g z(tpLO-EQMuiS*^r5$4s%LV29zP%|h&%uA|fw)BC|#Jmp4uQhEba|1`GR)cfz5U=_n zfotbdMZopj!h2E!zVNamQ`v3nG}_0ovz{{Q8^K9l-COv`4m`n%(73iT&5s6yFIM4R??nF3fE6 zLh*zF8;fuHNf9u4+|tfg=L;!?Dg61C)8O?}BQJMW)+{ZXJ};>n=DRM3QG8#(%cXVn z7k%A)W)s?{R{+JLpBGz(m_iN7#=U+N-+$?PIcG;E5%b#nrn-&U!5tp5CVaixi1dZ= z+S5I^d(Kr~=q)=x%S7z|M_ya~tloR9pzp(R=SbkFf z>`PV#^-$`M4btx(6R@PzWA=0KbahNWKj9)dbbasY>GSG+n>f7Jc$kF6U!>cN(c-}< zKd~@wO+2$e9*DN3EWXJ{;ib`;e+%&QUm5NeJ52GL{v#AeH~#OuA}Om`EMqi)j?EDj?pKM|XX-VMG9W8#0rEl>?Jt5UB<%dLh@G{Jkn?<$Ynwc#+!nXs0YXW3WE(WN*?!e0#hMmB$0l5bNtu{G10@ zTHT;-T=aGu2WlT1JvT-T4hTZP>V5?ik}o9Po-d<4jrwRBT8DQ+;@q%vjNN~a{q*%E ztQmMc`vZz6CcI(A@`GrY8$%kKVzzN8Y;2sjJ+Gd7WJVT^S$|D7rnUX-6nNRa zJEVMU=%=oJsOs1G+g(>b`E0O2`ztUQoNu>lMZAX0<7^J?{)HEfWbapde0@FKzE4W; z6fv)_{AZh<^tu2ktD}0E0^)VMecZ|TlMt|V>C(t4`9ksiJr|a~{&#!Ct~gRPm@D3A z3cP-PAnER&wJdolq#qAa`$%@uJh!gz^Q6BYt(L4ysz$u_6_ct?QRXW=9eMIe@hC-K z1iXZUeVlID5%W@a{&3jts|(m3)+@7_jp8qHS^lJmbAqr%@^SqWPhYsv@j}#6{@?ZW z^YUq!YdY5Olun%&_PIZK4L79f>LjvH#uG|L4{bHO5U<;!mmXb|$p@xta}hxmVqP5k zj}>j$N6gE`^kCDM*UsR&F~|7RQpBr)7RIKijb-O z#_99IIAZa{O%vN&_f8^SyiJT7c2V-tuJkP@kM1@k)EASq&Rxr3;`Xu9U2jf0yEAlE zwJ=ulB7c9`UwdApogmCI72C0xGGAf2(}vZ&$EMFKa@T9!8u{t-V%BW*DB(cyy)Hw< z*skq}S9G_a?eiJ=le{$ij_~RR;rsgr7lsiv&#CjGwJNfi^$m?bI%~$JeEK2;&pdk7 z9X)&@mA!{oQtRLCF-crdX5Tt(*lV(xUHA?$uX2mnL$h88QN=@;T?+JPvfgr<^i> z1+#PPzWf=lk|Ij>U*&WK#?>rKMBgF*bureWMLmNT5^QftS;hOp!2PtHp~nBt3p#_9JX4Gre+bIs+P-h1A;Do&plQ$y!X_Y7iQG+foUl*Leg z_IFN{{FhTqcj9VP&op<*_Jdb{*;4;@hKikV0$P z6=MH&;PLsYwXv=s9e;v-Ya8Nq;p@lcv%3XA*F+=s0*@bTDu^(XOu*;GvBS@&KkS!z z{51z|c!{=?$Z^R|Odqj&HyRpE`SgT#R$_nugU_FTQM|e?xoD?NAe#Sp{Gq?hol{6( zKE0&6oDs_UCQc0Ai&BgrlAqCI=o>Y!a_e!mlGa$#_N#_Y2d5B!aa@=s7hT~BU(Ktp z8ZJimXuWT*;1d=>=v3v{+P%aN!V-1MEdua)MGJ0fR6lZuG^qm|Ub4kyIR`IHzn;lX zrTT%+5}L_)m7Znwi!bVk7f;L`f1vO>`*?VVw}~+ba>lv5NvGz8-H}wjRAwVdVts96 zndRG;?*gLfSu<-T(0mw-7cFPS!(%djAk^FTT&5PE7h_O$;ReRP=&Qf-(I-hY_+NRE zXcOC6t9+nQIVlVAx@{H1YFASL-sN*hZ_P}=rKx9z z+Dt57K*ID*|HF?Y&Y}mbvIT(!JUIE zt}FWix2#ab*lc`WgL?aX&wcxgJ^l~|4V2dJ)>b6u_4Dn291ySa=mv(1Jt$snZ!F4oX#WG*DG)Cj={$1&*yu&>&{Gv*(km* z3{(9o?u2;V(NI(&E9ZejfzynG3yh)ijN-)pdi?g0&t2x!%-NW|VydGaAM)DY%)f@l z6|#N4?R1pbGU;C>Wj8aucq#-@i`M6`+@j1Eol)|bo)=$V5}a=%u6!ZZ*V*=&?>^s} z!u&K>c0S*MPbBQje;koLCdt=u-MEMP+i!I|M>;6$Ct&Sk>}n90`6L&3xDM3_y)=Tk zhr}P{*FV6tW$xoC=GVJ(EK02O0Zr=D5{Jo79t$<+k^mzr5ZWN|mL-hb@%UUgCCs2QT zQNb&2wv=3$Vdbd8OIe2lj~D5s@8UamE|PkViQg{XN6f3nV3kWDWqyZkwV?XSGsqqn zGOcQ`yDSJP!@I|}X8FST`}SO_wD|V;q)Rs;zJS;swYTPhHxn^0@|`>>B^63M`cXSS z;Z`Q;#~nxddNBBAOy?%D$N8b{Cdui@9-}0MGuEESh2?Y9Loe^e=SAisbr;qokV-rn zf%!5qFV`9e{#iOMFkp5yCAI?fA2Wr--pI-mD1 z@t2IPvM>t=3|nN+uKPyJ3$qAPv8O?#33Yrg`Y0@inGVJGbe7`-t1cj3;ogdm!q4V{ z>TB{${}o0c&NM?*;4`&7k`D?BF&;cml3T>lGt+kp{;M)3KJ|_Z?Cm+ZdoULHuXaYI z&0!OQaPzaI(BrR^dbyeMwil1!^U}=k4!pANFT8#Nc(Y%cCuPgj;|Y@7rD)d^8C3c* zeO6KHG#BmPv2vHfx0k=HXQh2eV$|$gE)-vDsn0Gj0=a`9r1@`PJW#xvl>FGd@lg&* z@c5cdN)E*FL|t-Z)sk24U}-hGk?)uMYS`>H_qP>-lt#va#cdg-|ERuFhmQ_FUUk;> z;V2CLi#_IFkKi+vf!N2F%0*nKz^f!ppS+lLvc3M(9{=REjJChJm9oDH);@}r&yZQi zk_Vq(&3nS@jL!?x@r<7GBME1Td6{opk$U`w8^ktSjx1$E`MJzNwt^xWL0G;fTW^Ar zU){sqeNt5cUtfte#s@XN5Vw!MbZQeZu~WB4?0Wv>HTI3x+jK6fpV)J_vdG^D@ly5( zWz03qg`Hj3%E}WAL0A0N3)^VSKOuc#-_OMz>E6eO=?{CBx_vzS_F%i_O&6%jw~3G? zA%BnY;+vWqrygV3#tTRF|p7^{Z!m=3Osd&Nz-zV1Ia}Y08vbo;c zBmsDF;}l;$*%ty&%+YwGjjyljXZ-!Wr6sP(1l08N#)*z5SbH0VxzE12&I)n6ud*6l+5 zA)zlM7yb@@=nDF{Yr`I2xTbpJvff|DJ5GO&N~-3&P5SXmly?8m{KU^w82xxChJNkz zc@-%w@cw=qlqijNojKWPe1%o+JT>kR(Hmbju_lZAW#f^Bm+*v+1EhiTgU#CWE zIuP?p+-9qG*PEEv%OhL^rkmZNdS&5^k~@f(o9kdtZ=nG6@gzJS-Q^43lptt3BfdR8 zej*njmhl&Vf5F`|Kw}{=FkXD(o;&^i_Lz$D0UB-;uO@ZT)5S$2UM0dq^KUN5fswZ< zAa}qJ5@w5kyT26Uf$Why$7B=dj^iZJ(mb6V#n|ik`g)KSex`%6|D*LzqxVmkQ9hd4 znR7I-LKt4BG;Pfa@q-iXg*-t``0>O8J?}_`kZN3gRXjicxU9zyd`8+cRUZ=DV|4_x z+{Z{A$TZ)7fRn8qeeZuRWy~k?zVPKoy=eZUy2^*xlw9P$N?*UW_N45mmebWV%yrKg zxS4~67KdZ)AL2#QsW&0l*OI)7La%!qCypmJiwQ)Va`?cU-Z%WW+*Bvyonv2RGpn<=E=J;`ftpkuSRwJ^Mja97ftOBsKRZ(wnb$LHleOQ(VME-|k`+4~I4Tc=-7eqd?% zytJce{hWcueLl+!P(9|j^vEh#%6>#NO12mEkN<^NlixR0Q8!Y>^S38?9f<4iOBORN z{&3hA>N?(ctG!0^JFJZoR?U1S0?*~I?z@p63=IqA5?BQA_Y2scw1VAB2j3nejVrfb z@HK#pri=&n?Zo>@lPmfna#;8$@81tSA$=toXU%!xi01Ph^?GW%psi>!zAsr*F=s!e zy)4SUZF*7F94yYA(qSZH_V$ZCI+!qgNWVrZwxFL-b|YTTYR<8>$`2cSVIozhm+j=v z$@6?- zoP5xnMZbFGWW1Vwo1yRPCd8}QzrEOUXqNN{p{zH6$IIn3TWh#dEg>)a zU2)_5Da7_z&fKv($WL_gd?xS1fBrM(Gi)Qg(z*=IN8Y;bhC==!WRKgQEzB6+oCD$d zvK_jbbJvsJ% z(zTn%ti!3-OB>FTFO_aW{>y+{F6?G6;*~tHV1v!OE0E)8wAzu&0?N8Ocb=fk@5Q)J zu225+haz9n98%@+{P6;(a z;Njr{;lKsvaK&_9&Xq9CpCf%?-{17aA!yTPVqV!{p)rFSec?sbLz#j>G@e#e_35PB zmlcqE#&FPH)*k{D9_8+0!{^o7HG}-VkC+#&aQ0Tl)6?gLT_+Y#NQqP#EDuKe1w^r? zMVUuugZ}A68J>A&guEPrdHIv4FyH<7JY8Gi9x^=NY8V!jh}y?Y@p+qN7(}3?$c5$N zCVx0pS+SpU6hEH$JY!pJxa?p2{e|XbbGvtdVS8@F{Y%8WNV`w^y-r;<=^rp%VMpGb zt2Er)XHM$t%>H?uDti$x!_#;6v_@q@ZLNmc<5-G4#ali#xTxNDmJx$Pb(&d1MX@ zs)u0y4Z9oE{l{0i@jmOEeSvG<9Y5_+)IR0}H#I&i7Xz7#RhNZ4{GrdiGygaVUtiL` zOPL-OziQ@@_SxG_XCAUJ7fZ1X3`PWGpC+s*pGfu_im;CCh zc>^Af42V~Z<=C-|tC_H<|IG4hMwakxZoK`Z}t)M;<)zZ zS+gSdOy!mSxZRqiN6&DrSAL+k`t_vrgU^)9@ z*Y-=qynb-`^Dm0O>>iXCHor#eZEugb#1g#>@p>~?sb7dvr_mg49?l|d4nCX-xo;-t z|5N)f*`fn6X3L3r6)q7?*WKp}w-+rm7|~Fl)W3vz!o;>XacKLpgm3?GN-359^(Bw@ z;PX=S-g)TMv%m17%wMrwy9eYLSO!xVrVxKsmMAPK;)n9bdraSdr_SH^zTNZH>M*J= zoFH?=AGnQp*&Vdvj=G%sdu4;n_nM&zlnSO8nrR zr+?QQuC-~AKc=Jh5#yEGo%}wQ5^r?8;VxlYiO-9JZd0mgD84;nN{Oa#i(?y@e*dwW zQ|o3e1IiZ~#K^vP=|TN%8HRl&WS1-;dDK{KsIh>tF0W9I{nWg$^J)>hcGxom^KaO* z)c)(8^NKJnA2Q4;IMUfF{Y!saVBOOjlz4DX_N8P4Gk-`Zccf)f#OLL`fKk_>f|!>d z=R6k`T4H~Xo$>ESWRE$=PRWf{qI#2KHA@$NP(i$QuB5FIIGP3PGd_9*_*#I$&Zk|p zh8PcOd(2f@&?L73tAX_6mnhfuhyNlOpLQ>GrmPb_KHrb)2{EsbYOW^w|J%PyUhTz* zVj240|NOw1Pjzia@Ps<*4_T}`;h*vn@$zFnFWxS71@;)P3R}`+1q@vMy=IjDFLr)d zyqYGtzU<6~I@0aI(Ba!tSkLO#w?o&}4*SE6P4`Fgg^|9Nm{v47rHI2tCzFq>r31jU zeBXmDSMYgp7@QtHlt|nks%`ZvTlp(*dXzolh>n%8hv@y(bIXg*)UuFu*d zm5A5FfR|D7&vGC}$$lhYfdwuv>~|!49Y>;EF`n47)cQ)Ekr~b~F966(E4L4;qj-We zP*(XuQ3}M|Q@x)b4S?{Pr_p^?_`KHNTEZnVhW6$BQ(t5yHprW)0#REt?Y9XLw~tA6 z?>DVHw-#s^o7rMJc9VWw0kX%Jb6k#x-$eZ(PEFDh!5Wlb-5M-XxI;MyW@~RNgi&i? zxp7WMlN;lK?6G&&*gEC)g(R+pb52B?VXx!!TEE-+Y{Yh7n3E_)FZX)aWc#=$fc@@w zv6YaUb$Fe#8AV^FrrmCp_`J5)nUWiB66fJiTt1iOj>IR7%c)Z`{jbyM5_*f4~KMBNmAYSB8-#r4&;z)ZR z%#NKgh4@QIS86z!GM}o#tK(R_7~(ace3SK)pA=|lv8&9D41g`329@f6*^gDmk(KVI zF+MMYevbQR4%>k1R$Hf)62!cE`$p#l#H|LhV32Qb@-K19e{TJlPk_jW`295~KT&U| zTi*W&@lxcv@Wyj_4jehACz>T=0|Rc2%Be~Jg_rE5K)^B~v%tDK` ze@I{0_YW>RyXblV#uIy%+J9+y$5kh+^@TCJw%NOsQN6GBu_HS})=NXJ(V@$cdI7+; zaPwtaOMG4{9HJbqt;KI2$@=ZPheSPLT*d18*DhjSI!t`5Rx`2ris<}*{BO)>mc!f& zG8**|e`FgB=;#R(a23uba$jJ?E5{a+_-%R)L*hY*8;zepvr)5a0BgwJF z9FugO>cy`}zBi0`ov{B}h8YR^Bg|JiZzNnWG-B*Odbk-m6GRb>{P$bqM? z1Qhv}*+I$c+j{Qrsd-^vpSfbu&dx{J>)5l@`m*|J=bo_BkFpQ%x|g%$(Rhd6p#CNu zlU1M;9C69QBLMQ7Srq6G;Pbj}^^Ma{^FHavFHy%|*co`u{E#}&ikO#yP_H_J0!2T3 zfz3PDrl9Zr&lipPG_v*fv8bYWf=Pdc(bp)%t5t18W?xqh^n`b?UJ$ecZg2iIo0u^D zAbnxy6|r#eo>VyYI`%9zFN4}1`!Z=i$PuXTT0a~4d%pL)76u7YFiY8nzWtLo%$@LF zO5=sUK9fMtBZ1DZ`0)gG28Egn6wY}N^TOUwlE3}sLu5Yb$77@~M^ghyyY%&w`trG8 zm%4)zZ(;Vhd10(V5&acdFl-dJTfrW_cy2z$VSw?V=CyXvVQ+~d=HIYir{ zy)Td*S(j}-PBUqbG3Pzx(q65C{4e)UJdF>42g@p+I#}TAYsU^_&iGDZUV-&X@31rF zPsZ8zgmxuHGxn1zWBneak2+00Ph&R!S5lIE+ShGpeCXSG7Os$&NMHPQ zNwTCXxzN{8eav{4J?xfGO1pBNniqN2zCfG($4Dw8E4FQWOl*&A4MCn(_bBsS3awuz z&QYH{zXOjHH7%B}hSl5wMOVE8U=8iNb9R3CyfWK@^LVuJ+eh-MI_12bcNONJJ ztXqs-KQ1rq7XNlc`byi?c_qO^bW&d*IUa;f_#s}G2KJ9?AIya}7G|0madr^;bjQ8g zS=79+uNSdOn%HrOn3qb!(uK4$10n463^V#N?okJud-6l7;>7w=Q*kx>$Vj<=x@JPfVbt@7*459x>PK0x z2FpjU(w~@BcO3EB-6NN;t&hg_VmNu9u>K=W<@HO=uX9O-1`A3` zM2LCm-}}5@e;^Q;1L?c{T2T9Vvgqjg!1&d`FqAQ^vq2X-?NVkS1%;y^}X*)LJFl0Bxq=;b7>N`9|#Ri zZ}o!?B_F{f>EZD--ftBk7fU0c@+-X&MCnsQ`K+6A8fN>c3);=Q^Tk}jR@5%ZwtWtTBn z_n<;@(jHw@*rJrLUzuzlcRUN&;!fGGsZKhQ$;I6P_Y3kNu`(@G4958i`Sax7Fc&jL~_K%RkAu%^Q zNC017X)Yp3T2l2$2RQwqsq(_k`1d2ySI%~A8~#YrB(IQ4~R;xL%GdW~1}La@FMd{P_GYymWXoO61l} z@?v%Vyz;Je)+DdjLA#?vDRW64aDROHewG7BpE_Y=Q)@(}FYLS?W$vQ8H7{K_pk=HPv#r7dUTie#j4Hs_SL;#7 zne*2Eg_lfT)Q4@H&>FEbDLQ@nypjSe=j588{X4EU?>)^@n=#4jvwTp?qs|EvQlo{ zal{kWM9*T2JBH8e>Rfj1&z1Q8tJgX+$}-#yyuTNb>K%#gQH{IT@V%ZH@Ma3X*X}DJ zVfXv@BU;~N$i~S1Mh()}UA~JiWpCyJUBjDKcY5t$hvqBk2f3KNAznkH(j$9UHj=(2 znvHy>^k05_KlYbeUw3pbZ9gy(2=g`yn%R9t?PF4^-Il)K)gZe*#LFVZ6GS6B*h4Pj z^IA4^W8<<`{P>G8<>6YvIx;M5e15!Hg1CLe&bT*Y7YFb)l72i!{)_ZJV#cEzXgyI5 zG0_XY*{FTw8o$dp(US}F?bDt~%(e%yYZpQnYvS9Z*|OUuW}z5Q?AQMbuf(|{b4~<8 zY}XmZ54(~7n&Y~qyg*YL7W$?Ig=Ba_--?z>rziNlf)r1fWxpoQ=WNL2bsQKX))#ih zzaM|$WhrpeU5s(k-)lC?8oZ=O@mFrN+)A_Pe9)Tb71I384j!%BQu&1spOQ{8-!dU5D$%|Sbu^~yxxY!t== z*&}v8UTtTCD;C)6*t697qB+x1acV^nct#X9Y(1s?J)Jd~zrenonit=Zk*Ji*L68<2)W7sQ@?VYm-t-~e z(qNNq)|12K1p}}6m9)C?d3|k~u+}rh_h030kxO4FSOVQQMYZB;VqPT6cfMqEEMdqKtm=$Nx;!%-P~Twdi}^_M(Oq?5kM%vI)L z#q1TI7dzY5S0$#w;92S9J#iP+PYib&th1?H3GY9J%hDx!!r_ue#fX#myf&ni`Kxyk z>+9g+eM!N?#JsRG{{4t}4bN9E;ju;gKL#B!Jnh_p;`_Glyp;XHx!|~cSe4Pu8mzWy z>crkRbXjdSyEr?-0D+_}a%r0$wGZG}3DE zpsd5XgJqE%KCegd2@7t2Bd#x8rpq&!n>Br2N%jS6`^@G~`uoemn@SEQA$zP(OW01c zB^L_om?V4zY(dsJS9r-wEMB1IWmGH`&VCAe9iP|LO3B#dhG2Leq)Xdpuw~MJ>8=g# zyjU#?-;BdQuA}n+lQ5Qd#q#*^y;${@HA^aq=PS@J_%6E0dHV4xc14)~atYkD>tF=Z z*SO-qrGCo3Jx3DXYpvU72HUcSzvT;I@hCN~)$R7tjxVs+v2Ul=S3_>~n2;4Ep1ANq ze3u!T&v!PKm2OkK1lS$EUTnG81Gv?b!(yM}*JD0A#P_&??=SKF&r^Ve{xiy~({CSL z>0jh&A3*C3$$W^FDU(CIyt1TIZ**RPE%m#8pb6>p$?1(;zeG0fc_EpQPPOZ zqF0}%u%DrJt4;K!z#!1kJUDO5S*1yRMYHv?-f9(rTka8Wwz#?j%Z}TAw2$z|JLcqR z=N~aAp6@GLAnPTarOGBLS+^twQk zqqjH^FIu`2#mfCzp!tk8tE$u*#3M4gjVbNb4?e%}BAa?@u`ea%&Y$&gRvo54j3Tu? zw!cf)v1LCBFSi}m+dUK2e;xcNL~AZC4x*Kxnp_w>V6?B!?`{RYJ@OuXJ1=er{&>E-w0hjr6d#Kg)MDvA&+l*w11O4F+%67VMRa6rg*?E9=l%1Of2sG~@G5ghFDE`P?2a+T^-{LDC+0eO1Ckw-_JvE= zQRcm4pU3j6NmmRGN3FU*(yv_MK6-+fmqFhG+M>~5FngyuA`ph|XIjwU%3)10@ISvQ ze$E09cp|#n*kA^J`^aC^uTN?w_V<*$5M{fgw8`_v)7Ovd`tvV}zak%8D(h50{(fP7 z`1M3f#Ot7Ft;93F9AFpry05@z2d6!*-Q4h+niuv9^M)cgb^6z+u<6TKCaZ-KS_po|k>?TqBdmGN-lZ&{6LHxj-JiRXD zzwSkU-+7Ng3~D2b9qx3xgP~-)<^W|r9A5hvwsDzX_*Z;;Y|O6@$R0fccO^avJS--* zN3wXryFPYGeZ3Sd{kJM#DqiJ9ZvJG-dd*n8>fCD_;z6m4#qtxzc1p*T>q0=j!Opld0{JhRLaTidl>JU; z*oydGigE{$le%hW|1y6_CeGbG=sU4JI)}D((&Q5J!tU_zN2D*_=f%Dpv}nJRaFAgUGx+GLzct+a0 z;qHAA7*Us4J)G$d=VMu0zWrssg3&cb2Xa5LJz9Nw{qZ>c)bpz}PxkpFUs^2?iXCL2<`I(L4!X9MnZ%ME9?={m%&59$nmpieA^CMg}s*}URlw_i#CX7L*)JrqopC%;Qiue z+s7g5_zU~y^rX76fGLc37#h@D=2Zp4$2&y_H8&&w<@#N{zlTN`g2=uOO8M@fOP_EU z{!(8!d}y76?IiRts=-74-pqy>e2%aA7zb8r6$= zm!7hLk~d#^_iX(y`ucfdc`0!*bS}jD(uH?EXT*acN6G8bx8d!R@#@`ctDan?%zqp& z&--4?>;WwWp>3u4`1+cmSm#je`j>pP{C3rq3g*D2C)J^oM9hn1IOnTx3#A_J($Rw! zmrs&@To%%o)Hkj9Rl8CCYlAEYt3wjv)m*!IBJI`{pid+#bE+T4<<%=M-nLlcBFU+| zT1!ocIG#w8q2HZW8w@;0$@7YsQGPYx%kKJ^5>a5X?2(SR?g5n>%ClU0@OkM6uovvM z!Os^a9TB+X6J!tSes8u6Un2HjA;+~&Nc8N%<(1=^EiF2M+Q-u>?$5(SQGV4(G(bis zF9%ZQRIh&&We1-Gk1%k)q}CTJuYSV4o7YJpGn3zjOd(zk5O}(Yau4Cc?ZcK=_aJ{? zr7vQl!zBh+xLTVsIz50%FvcBP@p-xE7iQ_j{Y76tb6D*mA(wj4tyQEi~RlWTscOeqp1JL61HgkD5bs3id9nmWmgR_W5YLdmNMT*N05Uv@ahJ zFUh0jZdcQDp!Jm_?Z!LykW_N?{T^EU`U(3L`h5Jf#Jrwm%-pu)$WeH78JcdrK)jj` z<#;Xa6oK#g73LyuJs>~-W%+kg{Csp|;JsxF1o7<=yZ%9YoG#Vusd<&honcI|Odx62Y}+{h zEV2JGzhiRICixgFvRG%^n2+kCCAhnk8;=MAdD;2)(+}L?*xKryz_!p*yQ{V>?Tp$Bbx|w*-pU_F#mADwCWAYeZp9inPFeO@%L4WreY(Rz=1Dm$zK(^0-q#jLw3s6HF~ zzdZXeyW1MpbFUQibiwD<|HNkwzXEapalfQQ26JjKgy@*IGKr8T;|ZFBTD(_j1wl@} zqT?{h9hO}dDbm}3@9zyanu;z=BF;x^^YunuIywFP{kqhx?U6-jJ>(j?wQ(i85wB4P z{~JsTvq82nEXGyUijbH7x!5@!4aDnNU2DBDbK#0W@N9K8H@S}VHRdg(7DFopof^5P z`6>DPvl6^yEz0@~c>QhOB*O;2W_)`jzjOCo(XRsTPc6C81g@%mskaMyROEIeifOU+&cZV{($A9-)bzU-<@Bk7uD z=OhRd>&vvP=*(71e@Lo)r=w!K%H;X&puM)|$tfXVl+NR8CR5h4?UdTOiZXu?kC)Pk zifGa6f3ZiS(h}=WvXGUUBR;TT`u5lucyQ^VRcJn6(ynxd%fHlrskY4h{NU*oux-Ep zZlA6d2t9nsc94de7fsQ<^sQ`GBoi6Yg~ug`dC|r+osZxPrqsE6aw^!O_^WK0>YT^+ zf}qIOWbL!Z9pVRS7L8NppW^Ylf1&z`U^>3OFc0c=$I_y=g;-zM`+q;8_ObeuZr%|+ zv_3#Wr@*<_YmmP$w5M;nIqM3{ma-C7ptAxFuIDdy)Z_av8JSkcpvT1ac&jLIS!rev z@I7I+*x!%(k280*@>|IYfXhC9jfh@1h#Ot^hMonV*I09i?w;nqtcU#b!ddi9k5#Kr zpI6e89as3ap?pqWLhR7yA!Lt3#Wr_F2QtBxv?!xm!xD}iAm20ng3rsdPi$A8CAPjH zc9*Ez$15^%MVTc*z&g0OC}`P^Nqh9&7iUddECkE=t&;P1y2HaAD$ZXF@Oix-s!c0Y z#P4sDbW5ar8LeRze<_3XGvf8X)UrnJsa7finST50(rcNdA6J0%#cEb0JI@}?hw-b} zw)~4b(ig8D@00w;Ss-xPC3APU6>vpYpP8xsU+ht-QE>d#B~sPx@%U6VZ2dWWUblB< zkrdK`L8HE8&(MpYhEypr^J=Bn>T<3mZ~3873|kiML@cci{L zo(=qbO2-~s*x>Ra9mrKQ|9p)k*Ku-|@Fn&bzyFJ$5C%ReCH-25DTePMSj(o~g+S10GEj;Arcy~-x2Xg(V6cp@Ni z%TOHgs!5J9mh{gC+AD)zwkxc0d12?p(b>M*KZ}@Gv#!<+YmQ^^g1u8okr&1Hre=== z+PZ~7MAoQut^%cg!aDaI13$jLIBwdXI8%+Uug3H@nl<?%~|>VsQ)Og z|E;X~dnQ~yr|)fW$_fr#9%Ak0q2`5szeGrZYk3?ouUwu(+}R(3AkRWbO^*S!kKPiu zY3u6w;kI^9hT1Ml{yzHuqwUNCsoLKEe<-Cw<{@K6DMLv_N#{WcnI&UH5+d_F95l#0 zBxRoGib6t#N>Um%NM(qmQA9NPozK-iyL)!K<9qM@`Qz*|?6uZwzn^D4>sf2BwLY6a z3Ev(E8Fact#A%PW-_^otA9336TOsZE`xWISB5})f=Ls|)%`P)$yk|G6uU+f-Ul>f4 zfpIgVL5!gz9BpcSV*BPd{{CkHM>;J?{SCC^32(o+jpvLH!zr(WUGqxN_Y zx1^N&*FUiF#Guy>(Iz>vzSt`Ljeglrp(^m&tFUIWJr;|TPFyAD@0aidhsxvoFY9li z&ZOEu@cLO1zgJPKtTg@)`Wo|Cue{_VpS;Wv#p5NHxT$oc8;zm?GaMfP0pEj?UOg3*idUCuP=E-zuHJblo zSu?16)l3wsO_o=v+xSC$DE(I1dVGEDY36s^wu!d?dJ&O)Rrbcr^Is$%LU8@0KZcie z!qu22Lv(*Rdr5!RBr}v(LRolqhbWns*oJlV&5p41w3gYkIDC7|m)dLJq)MAtkYmT; z^0NoPsDUtBQ5E%HUx=UR=bMYd`6HE!s}lVoce&l;$hr9Xsxjp-HxZ<*ulVj^I<-~* z2Yvn2CAGeepEZ%Se~#{-BPJZ68cE)70-4_))41oppZp%ljtdE4LDfz)d6gXqN~JTR z?e8n3PQ^VJJP1#R<&`{kqw)RQ!(PnF4@Dv0++pcG%>XE55@?!Gz~}WPj>z=U6klJY zGyUm$(x$NWYn=Z58O#qMn23gK4hJ5jVf3|?i+<&lJ~}@n)D(1QP5}CThn;sf^9}NT zX&c{6fzAAVu)4h7-Q6EwUlX=jA7g@P>x=klt8ZMzL1;6-;NAKJ%PT-|^xiw=05~*o zyFGa`J}=#1j@43pwDlFr`0?4E$NxT_i0`+&lKl`}A02cnxoxfw%8Qv{!;sYRldz2K z$997(Cvc9)PJ_bV@cI-w%c9Ad*tX-@!@ZYi^NN<`|D68lAo#L1N{E-BybLR6({me( z!YP0LoHO$R05UDp^@Q+wC46){=ctVzuM%X9>YuNbgKO$Hw(TOX`$p`G*7`3}mrKs( zh~pT2ks?m`uaHCMYlDWqpBRxqc^%$=hCa#dBv>-SHOk*?O_KYk+ZLBcP{F`(p|5gth%)X>;y_}!jJhw*)1Q#m9F|Cn)Pz-zat#JvK-ESGUXfC}UPR z@Vp2!aA$UbeJl2PpCi97hCC-^ec{h)5%`Z@JAHc@ZGFk7mOW3jI{-D>UL58@X#BOC z&!PS_Ip518m^YXt;}0f6`NKSJ_`C$0=f?6pLZ1uu{&GmYQjZ6_kg{jqKO%pJv=h0K zZdbUf>=UHK6LYq`vhI*Td9jMCuJ;%(1MRQPYc_T|L6z>tTzT^QPsra>^XeG!IM*Ue zJ3kT1y5(|Zb~teEott~s3SDoqCRZS}xJnexM!a9O%ikY9i_EUABku==moJ>FUT&7G zgKv*|iME~QDXtJkaELrA`hU<@(+-7Wme~E}e3y1Ph}>9C;q^>Ru}?b`?T9_R8zas6XsJnbUWgAK!l!nXP?c1BJ4L^TXGTz77x=H_@4I%I z!fPVjLrcOP%gfsGsZrv}V&b{GYd21=-w7m$8C|^$J$H(lmzVBJ9r=?7L0ZJpa-I{K z|GK5Mfxzt|3LR^lmMHKAfYki@A<`fAH;HJyzI2H!{`zR^-afXGgL^?!{QKzVG1|PQ zxey4}SM9>po?+vOiPX9aR(xs{eSKt^pueMo`g<9VlU50f&VW>3e4a$VK& zCHNh_|LR_H;w4WSZU1#?)q)4Su0ZHltul0Z&s>Lo3apS>nsV^+vEoaqjGD& z;;}QZQsu!A9lITz@onHbK#qry`%wG)uzva=r~S0gPvi03f3?^@436t=D2Y(t@bi8I zLcUmcgoesuXi_LEBI%Id7hSxFXW@DL^O`n!R|L|Xpxr-)bbkN*#Nyw-pQd*tOR{Du zn$JnFo#i5JhMun=E9Vz58Fd!c#j<$w(d`5~>(utkH-E!x&wX7_oou4O#)N|}ZE2rx zW!UiW$=mOtP}Q8EYtgSl;Z@V2FV) z;@-qECu22?oDJnyi=4rFazsFXyZ)lDDV}*B^%*GstIu~ly(I$WwP>h0?3vXW(76@h ztRQFy1WD69iyk2BNwDiDoR%C3=h{S@7mFL`nbrmpB$`bqaU4QC|Qt_6x)J4&F&|FN37y&dd4D{f&(*}oU%MWR~}=H(wp z+-xF}U-X*1U-jRM=Ar1wyd`OM`zU zzP_}d8Z)su;p+>jhM;haj@+I9Kwq!!K6^BgLy7N`NBWHWi72n2lL<>Q=9Iw0K%*6- z=DXoadyT|4LwtRWrYxH)#6p{wT0)1C+m=wUCNQwdIgsbCetzAv_WZv8!Xj9}XOX?= zY5?5VRp?5P#OHPL@ol@{&$Ru0)&T1!-3^7b^@UXAZ&%bFP0H>BY1}~9%ayQeFClw% zNs=p(++7|dslWug!;PVpMc;cG*BHFydZ_jtv)eQm7paV;aX9Im$**b^^~Sfy9t}EHHcmsZe{|vO`6Al9h`kr* zCL682Gjfq*5uR-~fqG1D(47_N#ydGQ_=Nm1Ba0`fvH%+I~Ch4Z{- zCgWGIyoeUr!B4_ciH3Z6dA&=q`{Po_t9o3vodi}Av%30*6p?>B`N4pmCw-xF7kvWxV)H*<+_DX)Wz|m2)GB zcdvye_Zw9+4!WyfQVvwW@3>0{*jy?=E}f07l^p zU#_XZ=Vj*dV0>xK4iMSzQy-}Cj8Y!8zC`VX@AtkSf#}o8vpg&)uU^Az8zUF;1Lrp} zHND;d@E}{e3SZ{Yru`ZC=Q8|N9lSM`~VN*RAe;+KTe3PVy0I ztty1lQ4fLlVh)fZshlmfAD>s*@_Q{U_h{?uvQL=d+LPq|@s;WBUF9|n z9V4E+{pFUL4cXs6SYDuQJ3uLqT3>C&_pVhP4uv)1djgBO(0m~?gKopMaY5Kh%n=SX z355R4Q?D90@Ode!*dN%&g75DUotRq~O3liq%?tVczhCk7m0v8eZS)w*OZ|~s1jptg zNE{jwsNPCGuj#7Li5*o~Uda95ISxMw>7}hNhmwoeD@I5_e{75H^x%0fY;lPdxE4GV3@P{ zS>Ne&;&i+4dF?g%Br1FdWvzDFMX->&$)w3KQI=@O{_DlIDRRr4=`bFz7b)-N64{qi%< z_A>lt{P^?!Y&OJaN9WUyzvS*lZ2e!nx;BxxjWh@4HNdg&=|yt?AHM1Gn2$boheMi{j8j?J$ zh0iNp`bu}wYuff`;E=a(B{39?-tOKY--pKc2(OhanmUD70^!OnPUTCo`1Y8y;Chp8 z*dKV6H@Pr6nZp{d!IwENX245fVbZEQOMwve+F02h`~DEMzptFQkQzbWS09b`UWR|69^zWG>Qr0#Be`U^=!=^<&)FR`@s)u`hox;T=2en+v!gz;W9 zUhO>U(k`MV0O!V;7m3k*H^I^-z^nqe0>pF!U$S9I>7T$DmH&6?{B9y zo4nAJp!7e}?ZWq8FZ~+Xq{#U}L|} z=|rbtZBtdAM)z&)e zdf@-wx3XK_gwN}`)|~hl6?|T#JKx!!+HZsA$QKz)F3-Roe?E8;o4WOZ&7PF{BkiEJ z#~V4DT`M|KUM)=NnL>$0;QdA`qEg%mzBMX^-6UaoHGVVkNTW|6x;hxPgfW={>G;~9 z>@aj+RBB#+yS}aCC7&nFK4;(TTUXJ1^!SS(!7o4nlyBYbT^v4M_I=J19+@i+Ug zBKI$w-2MH+|KbUY=k(DIjl}8iWAk5AKQdlj6AcB%`t&I+F|>aSJKI>rq$vcJHpk|M zeF+5f3Zp>=GyMF0c;?xKi~8~PC6c3gVf{WcSYT687{@|8A5Bs_k|e^f1R~`Q0psSa zmN$hSne^rT$1esCl`#D15)L z9_3}8>lYdFqzFt$Y#Nsxb%d)YkEtYRVtEmS&%gBuIZR~##?GPjgmymKJ2{a-NF@}C zbKUAQ_oMMWLy3{#V%J3w{ z-y?wkcEz_xUuJjPAIzvduKe)+(;ce<@NCwa7a(N@H;WcY-M!KG7C@7w&UHowkKCkD}D*CZe&BW>Q$?d|~ z|NeXv@r9<&40yebj8nJx^z(kxcO{VDr$+8ituJ+>)oEz~=y+)UhigSK2Mb{KGdmNV z>OBxT(RoOo4%si^7hV(3b@~M(h=ddT+$rtGKnkZ5EFCHSSuWNd2aO!$PWpv_T6cP% zjN>K>FURMKK5K*dKs8`^sGybvvW8~s3g?^S=D(hWzh5@1H%L{&%^UvQ` zpym}s%zj;5@XLDNPlGyd3iF_pk&QkvZ4ao*cZsBXA?pW!;q~)T9HhIbFcOdM2aP-z z>i+SF#9B`KN5Sy)(7P%5gNNepP1qD%9|jr7gTDZ zJQqBD?d;NGz3Z0$pYw`luR7+Mg!1~d`dW3aX&#Krd4zA>y9fMVFBw^S6`vP-E8hpz zdF1_|XSa1cIgm$b2Q{yV!8Lskf`hAzf;OuJS2WY+rJt35XL!{N?D6Mw|N8~s9(NfiU0~}drNpZ*mbLLZ-pqyBj*Jnm zeRi+#Sww0QWMQXaKEhH_O;$&<%J>jR~VQj66o_LzEd&GoETyb$MkZI`EQ z2z=eIvhr^7j(4{o^`W=>;b7LGYuz$Uw1z zNcsISpLd$0i+F*3zH6!kI|-P}d~HRxiQ@8757SJq^Z0}R`l43M(c%JKT^s3j&e4un z|6Xt8fB3wgqnY9CJk0O~Y1-Yg1#J3WW zODe48DD9sv5zA|X!NvRGkAfgReYV6EDfIk~w04m$W~N1;J8PkQhAj#DDzmm{NsHm~ zl9&=cnED2v7s<$Qtf<5S9;>~S6gDdTS+76D6N+;qGHke^V(opVGR1r%Qh(%19Z$SJ z-WK&H0qq}!uBQvVnU@F04(WQWJZlGqqe6wUfmmJ@>-ImF3MnLR6A77Lno z8*#+vHQi1Eq4a)2!mFv7>x;y#csGv=yWb-$!4L4)oC0*EObjq$qI`+{|NDG$5hl?cz<@Hexz_ z5HML@aA`g{z7OGOTd#u8OL$W-;eEp&`p2KZ2E+*kEM5P7e&R&%k))-2QC{u?H#q#5 zb0OHV@1k(PZrHRuSD@MBH@tq<>%-iR^EYPE=JoW=&6euQAdoD7y>eqes;@vX*A;~; z_(1CiON57MDDYZ7FRV+|$F;|&jr)8i+5ey~-l~>v2tNsMG$z;IVzKTANBX0+0XVC^>ag)lE?CgMkG*Ux@5ZUjUFzqZ0UCKn%D-k zT{!)tyQ4a175jg$FVQ=!D;*Y~{@&-t@>1TuZ1~>WXd_T%1ClNL(v6wOehI(W<1`q8 znFh0?cOm-zIv%gx{RK_~ojDXZ#cRfXY{0k2 zI9t}Y0`D&2>+3wn0j4?fkPy@}WhY2`yggjOpy$3>76^wXK5Z^LfZq472amPK&%%7w zu{o$ct~44g=+erDS$h5>20LwF-`UpFA11a_`ekIjFY)Fq(;lT0#64d>5Tw59f=DWF z`2GlV9+lc2cW7^poly0MidT+R3*yoBzU!k|vQ8FrL+gaHy{oeyXgJ8<2;s)(<+5j; z*FlwwXuELi(fVcD{Vvr!O1*K{WB$d3`U5}yy0A#3$iE)dm!Mmx&^z92sAK)GGoHLp zNBOGrR;zAec_Af;4}Dx+(M4Ndu_enq+LZ#}By&>2))>@(MV1$R|FVV~im!7G3a;{l zR~OTBMuzb1an75wnqGDnX!7#3dhYfnXy)TbLihR3O85QfegA48R$mJ25y4CAP+mH^ zkH51fXMz9Mi*HKF?^(TEw<5!>7dfvHYmao_)k7?U$-K_rs@p%ANhyz7U(B2%FHb)A zgCDQGJV|aFDgF!LHAZ;*-Y3-$_)2Y5mYmSV_4jLRj=WdOz>gECI=!B51Xu3#8c{z{2g`twLy`?qB$1jwS>-_N83IdoLjT) z*4?%NVMfNd#Wq-dkyaZKk91`d)ygz;4hrglwX(}BgKKC%PtA*@;p*@_IRHf7u&;ZO zjq=Jq=A18K!~=UIu5|ET^Mk;x6Y1Mz@$J#p&c-6@&L8~ObVCW(ScjWp9@6GT(l!o# z^qd0-f!=j_oF+7Red%DUKi`G&+TGVtK<|_d6K3p*;*z%T!#RSR;DzNyxY1zq%{zj) z`gP5V`|f%`x+Gh9NdTXh>#jUA*?|Dqw;*5x(<0O!v+kzQ?X}~E%_6M|jY@uS_|mcm z#SiiAk#j5exyt>t;nggdV*c>S z(b5LEPqfip%D}frF9zq`?XS@NN~R0Ojvr6au}EL+Bk$L)^3cm-IqL6?!`rqA#_&MZ z&UxP|oBiO>j>I$GJox$w9^<@MAB?ZB#szMbRy{kRs+5a(^ZpEYJqxyoouVV}&y+GY z6c|cVU#593iQ<(guVeE98YXnJA;??&?NV)9V5)wg=%$L*SL1@iCHML>iMPxb75_M_ z2k}dxMWrm{c_UY9|0Q(w`XqY<%IoB}0QqxEvO$25C#2S4 z17V|SH8Q;T`dYue=lD7uvi~YjICRggAD7q7xZ=+QJ2J0BO^e*-D^qyoh*vDjQQ?K; zU3R2y@_z1v*Z6#+ui*0fbJr^oMk(=3K$Ed!dgt~@Pfd}&D zJNWo4CC9T@T{Myx;Pbj(&v{B!6Rn;;}=eB_Y#U2q}x@l8$FKqOO$m{%j zk6yy>A5*&|a?Abkd6D}0e1hk$g$4JN>PBV|e>LVuJ5=(DQ0jx|5xG+PuiO*s9@06X z_ULoyi+2`FHdOO=T{}bG$EsVgtYm^b|A72GG9DtR46hO{4j{JI8aWxB*M&xxjTKMi zD~ZVOQTzMt`{iD?S&~3NamVc7r>MU_`8lm?*?e9Q{62TlYc=xzasnz1dmHh25sH^9 z$34XN_oT-0MkNXlOPn?@7e@0thTX-JqjNMHFckLt6C2P_7ZG_jSjfb1+KKsMY zP6k;>#OI}~Ke%CS8os|Lm7Vpk=r9I8PxcY91GLAFB;M`Gm7)Sb>L1wa_xup$c_YuA zT3;MtR~KoRq4v1ZRL-bwdk$P0ztyl{^KLMWIejC#8_P@Nmb%x(hdg4HLVcQJiXMo( zPdTJ#jP7egt*^~S{$;f?B>4C!aX{S)%@<}@G9TjMTOukrP@LzSJb=Or3Xs9q2K_IP~`Sl@H5 zHJQH~e1=7o-rvXaLhh^^AjjUuMZ13tZ!G3`WgG$wlM3r4SEx|@MYHtwIlH&;LQ~f~ zvop5-p#SvaT7_l!yj<8HGY=`@`!A$kgM3;?eiZ)q{vP%mj#46@Q;o#;Vm3M#QwFl( z&Axpt1gX z@~8d&VA*~6LxMFvFAMoQ%n#!JpfAM-o0@j%fwTKWgKOP?&r2^lo%@m&>c5ibhuuv0 znhnj0n-oGcjp5-5;Tk>i`&84<4UNB$=eRE~XYK4DJs`e*x@xv5`rZY#Jx+?R@UY1Z zf(fMzP3iOKcnIP3UTn{LfeU`%p-E)o(ZAS$%BQN%hplsamyB!H3F--;DFJC z!iM+0a9_I6^pdI$Zayb-hqadamOt?N`5+$zeEHyBko}+Ya?d+HEV!+NVvn_p7s#J) z&xG|AE(=Ap3}Mz-VVMIzcKnFwD`s1#;l>_x|7=8u)cRtpu~05&4*+W}j`?*b)?j!| zj$K*#=%FvH{uD9RE`-lZ;9}v*L%#U&1X8arRg2V^bN+i?BG!WQjaQ<)ii`46RvycQ zOA0~<50dBkof(OSEk}OCODN;Le}x^9bB=&q*%8{jZob}r$3ZXv*4`T)t4~Aq6UhF6@Pd13C(6q~JAuS9lnIsMk7_!?48ftAfqjPumKP}^A~wDe0 zTt$q~bu-lZdY_}YQqn#Ejz)L0X7-@*m(fwFO>-wW;hlK(C9TVT#ctgHP8&&mQ-s5)wy<2q)x9&CAWL>bmAH z>rHl8=d!3&XTjcptIYK5Mo_$rxXDK4H@p@Y_k}MxP87@#bL&jp290}N6eH)O&z+iA zlg-tVroupIHD+1g>Vo?F{fX6+(d-Vr7z)t9f0AwhL#z z!?M*W^g_lA^fe8J>JvZydS_exs=yZ2*X@I67v31kfP2Op1)dlfKy-G<1CvL{d@Qzq z-0)#Zm*r|5jGn1^g%7`Zbfn22jz68_V(*RmuRF6!ibl;jz$IyiEh2 zm&fK^B1<#=;O~k^AoWb^+8dK zPHtH7H}RK}&%?@OcjDcVduqDzwEfqtuK21pVSgyq6=V*rLD%Q#=b2b=MsmQO;85c> zdOx_ISEB5`5T94;_^M2j7k)m6urD%1Y^@{|4ZX-M`akd@p6qiRmvKdP`<`u>F+kbX9#bk=QONc3Sj8Ap%L%QR&V)7Q7O=eKj;xF)ry#9A>WlJv^}Tq_fFjDPW$VogT9C~xjQnnXS^fh@fpxbif1Y~yY&?xz$3;OpsB)Ov$oGTOfFB$4?fL^g@Os$h zOqPo*rG6h8$6tOt!U}}jBaRD$y?|wO!1=KPKCfpkO5e4t&k>RKQ0MQbzfU04tzPr_ zdGaxuypZ4j`xRebUxl`tZTCWX75FVS=u1cejuvV01G{vOu z_ZfKI`yJokN4Pr;ja+d7o2*8wbsJ95e#%0 zXHY&xUn}k~2C{ml!Fwxb?UYu1c(77Rh@BJ32co!dx^T{bRslN1g*a@NU{< zc*ENpx<5>6cO1j#g|rVT2hSe$_j>M2ETiqeko(P)*EyHF^Y}DTd(;hiwc_(2nb)qX z+V;Ek!9e<6XXpq%FZ*^C>$S{?y&|P!{e6FX-Oe*i-XN5+2|9hz^>UrQ(k!dAIe;PK zu(6$>5BOQXH?RxFw?|$Ri5LNfU*-v^`bY833f<79j$n`>-t6f_+a6hainCnCWP#KZ z%+nF z+9^9O$Y19@dDwru2O0>8h9J~A^6(a0d8>1!k;mYLEn8Bm?`f}{C5p)XJO7Y**C1dqfOx#d+dZiw~I<&$j@yqFVPgW{_lCUI~O}= zMWMW|cBm-z8KgnbthSF0S^Cfw>7xHk39B!pUfxe)BnR?n^AcUXy#F$JZkxVpc%XPw&5fhez|x{>DJ_}D?Rt4eNpwKpW4&IukN{(D|p3p-@n zRg!7;kN;O*-v+g7yq==Gf-;nENEf9+RM6p0M`hZ)xHhStHQz?t-y^)(`G}tN#wf2i z;*{3*?`+`uZsft64sUp^t-EfH7(TC_#P5Vp{N6<5`D5&{KrN+r6d}EIdPxb?3I%S0eWdt-CneL#k-#wU2jX9*JZt>m&3)p$nVXqfBDD-<&_fZ zJHf-u4t#r9_UXL!h6O+9D_kw{c@ZzQF>Q~r!{tTX=PFWrVjJvM>{%#mF>_u-#;C=P zHTLLz|9T+U{*lnc-XLa!^4hEaF~71f4f0Ek59U_V=B3$l;+B7~CS>d1omD`@-`}Kx zr$r`F)C-WWPxzh@QAbV`vnuc&no;{>$HybhOsTF3W@QXaLx_gW*!fAG8~yl?9`kt>Pn>x`!U zkG@uRVE1{UxA&Ge)W7?vt4=--7;k^fznVe$z>g$j^-g>NLhY{9!8BAhsu%j%h zC%hJa{74kkvMvwGpp--1KUP&ea8j%BfY2zH@1x{A3&Is|ejDK>lER?arjU!?_pd+1 z`g@j}jB|E8FQD-937hZpW7%=|^ps22s?~@lueaNGJro_*fUVLu! z-F>Fr2x1d-ck7wVoL6>F21GT=IQ$8lY4?=m#oqi>!l#~KVg~S!c z*JdFWp5429pc&lcXrI9-yh@) z`bLSOc0Bm``{{NdpTDukTbs^=To-^Tu`drh7tTDsCk8I+K7BOgXT9+K*NWtW?>c!= zdptsZm#85#1N;MeJKw3$)>k>hW?{D*+Ay5xrK~;pmQo%yFV=yzB3Yk(pp+vx{s|*$ zkN#Q`eu6i-;4nv}ZcCOgoa%dIldk^IpkYx4f6OzF2mM#O7~Dc|Ayc)VTUx2K*2ZesfHHJ577M;%2AxV7?Bp`bRMzKJ%E; z4r*RpN!9yDF8ackNsaNZuh4v90c&N)8F4PSdZFadHh*7WN>5{IVE+Rz!mnSy(U;_- zTQ{$)1qRdHo0Z%%=k@Dd|FqxG>>;k&7b^JWX3)V<{us1}kaL z4~^D2jYMwu1%h|FPeQ07#eWICVVdo+fD4L@lU)scd?EkZu6y(UF#m{Y7nT1)eonV@ zzKImif6t3Uui^BLg($Dii+(+uin72e=oOD#5^Y`=t<2lMHPYsFbk&F_({f)pE}a&U zA%^mL+`r_S%^gnA6Oug>w96NwhSuz2a{d?k`U$N1eX-pdj{lz5AsOb9reETT20ga> zm+oeP=0)oq!$#WvKFnwRlN_?YK%VRL6`TLMYqf`=tKSEnJkK#I=*056GW9jJ!O9m- zulZsV>5jiY0ChRk{tNl}4M*+sI~o6bUS)c*PYQpr$Is*M`X{etz-E4S;iWU+l@a^& z;r2b``7yPzHI1X_d^)xNTE!HoVy;TI$E|E^@dT6?VKjD6f;tz-T$-rATuRp8GQVsY zBL4aul1F@$TJA6JbyC~o1}znt3}dJpUY9(}aOS*7x?S_$CIt|Y=Z##c>Yf2Y~(*5qQLaazMoXQL}p2<+`C~0&1rLsWW2Gw z7PzF(mO122iDQv?b*8)|dUOW9NTa;QIj_G;w#$TV*`-0Z18MVG>+q&ca6fH(R4Yvg zPOSEY%(kepS|xOTNI$k9;fFaVeEz4~=g z4|4qbZ^zr)+WiGPEU>&1nM=iLD&qe_CAc&@o?V*?B#%efGcL@m*-)wR<4g-%#r^F@sV%X6dv z$=`dnFI8q5q0MW>=_AX7=J*2Z@`qoecoZnSqNT1R=tgmXfL%hGO@J?~$Ps+(P>-LV zKol^2#m1||Wo{iECpF;a5jE*TCM>Vyj73FVg8!V?ZH2M)#u{||xbQRG^Lm*~kYv!v zu3@9i%j7sIJnj^2UUaue3d?5sfU<*2c3T*hmqMs+J4duH%w=;dG7|b1`uYj<-MP5T zG=3~Eg7Wri(UO18tLw+3m5N(XUj6qAT-C|_gcfCd9iBlCD%rK9(T8vMV)BW2`Wn8s zg^(5C4N8_9cn&IKdFd-LJRgzq#rUrXDL$K&e%kx-MZf-b=lTb>+Cd$ES>;`IsMq#~v7J&cyVTJAtg6->GH;>d zgGI*ADLY zE9ew`DBnloiJ9`cTz?>I5#RYzYFiE%v1}WuFvRk@rbX&rRV+vGUs0i#O5cX- z!M-6i#;3KTly*?_3NG>L+IQR^s>Hn?4t+z{ry#t-7_zL{n0!I#ytL7NDg5^pj!zEe z2yDl0=L_^Jyu_$h5L=(yZ*9{nDDLjYWM*4(62h;r2ynX2Ve7y<2{VP8$f`PJ%^urYQy{uKxA#gN_({?N_o`0UKt80FO&@i)&c!17tW#YcObkz zzx+zCbUPUKyfOZeZmEsy@AZy}3t2cEqm)CfFOd(Y)}C>8f<>DR(}T8VP;`XYqm0tQ z%`6wNyhP;gZH(3ZWu4Stz_7fuUe0>RC5!Ht_QBo3ij5-|=qkSQjaJgu*Y_Z^lf(9g zAhPS|pj>7@r5)6~?w8BO#SVu+gM=Q)=_asqN;s$nobd(=Mob1*B3P;2q2}-MH+wJq9)c-GB zzwuwZs@ykhLec%dJXt2`WMXq5X71jahi9?+(w1G+m%Wup@%M%%rrX70^kH>yS;hF! zeM))Myp|@`k4{|-hGixbgaHHeedIprf%A64IV|gLB{OgK89^z<-;T9Z6^)>W$q1d6c91!$fORsEgg0V*h zmr>#J_b9LaZUI|n3;mzvBZXq~U;9@?mp2avL5ICLDb!#C<@eJLpGe=hg%7SQuzuz^ zFBpQp%f0q-{r9}W*1^``V(fm%w@8fF`+bq~9VS93`a-HsgV*uf>KYa{sJ>pBkAL5I zG6xQ)MC?>uvV$hC&#K&8jmhga|5hAX$JF|I`R(PLZo?pOy}Rbzf^c-cgE+QLsGhtZ z^0Lvv*25k_Kzy83y`e`7*MCu$L+!tipFi&8U&+9S_4hiq=JX@W|2eO#{jZpUQc-<9 ze{UVVhA#&a*j5*0SYYEX8S4g{kfhBNdt?#aHgR>^%z2TP8r}8DK>dBp@}c@KqI{qp zcRq4RG6+iS*FVk){`dNl)zma>7)qts9pdkotjP3FyM*}9|GU0w>AD;j8=<_+j@djm z7SDl+eCxWqZ?W@3+6Vk?AJwQ)cmbU}T@)biGayeFtFL?AYBomIfgr+8|8{g4%ByI2 zD)p``ALs>4E_8iN=EcM&P%nnxKN3#HmtI-<%eq`De}8AzR8x5CUWi^So^_tQPI9_N z==(7I^ls6M@7qzn|9`c|7ZUBcZ?B@f7M(v(^|n3-@}DPd`^*UMCF}U$EEm0Q+)xn6Hz+^Q zyN7R&jaxQz^hy6=eY9WcwmG2+U|Z)J9p!-SA6Lc*#R`1-=e#)cI{XU0qP!eB&!p{K zmWBbR3J5vK=sh$*j?4D}7>ikw8dZb!bnoZ)b2RP%a8&c{U1oOe|?I(RK zukSu}7V&k#;NBnP{#FTp{Fv=IJ9)4_9#>xk!t)1m`^$)Q68qKK)UffY+D)0+<0toE z?9uvxpKZ-4+VSeb!V$gQT9uUbCYxpvUhWad2lmR8jW-pr@fS5O?{UT0;X(s&jp0w0 zcEw*Wck&!5p!-2Eyk?~z37m(nM}FJb?;}gzZ#y(7;t=PuU=VcZDQZ86uP@^Gt+>{d zU*bwC|J9>!VcIWk1dD!bFCqa}U;Z!G5cxNSV|Y#5H}Z)ee`VFqyZq!CdY;ZEqYy&k z?mSqsm1)$$oVLCk4oFHT^N{`h&h_%cs@at1PVK*>+#?)C8L_-7`xchU%;SSZdJUPm zV!?2Ks_OWN4n8lWMblSoyh_pz9gH|94?iY9Odh?0-9N`(M7+c^=AZL=@!0(V9w`zV$?KbO=ZBhEO4eu*DLSU^ABnpn{90%0 zK?SP=FYAVEirpdc*DU#UhHqkj!%N|-GT*UZ)~Wv`VC;B1{n&Rpiym}8n7zs>HbN13 zptgUneN7K-UJD4Xmw(t`07QceAJ%7AQQASxYsdJaxIA*a@a>Y|-D{UneKFU3&}J&< zgYpORir%~-aI=T&IK#0kf*fpwAxDCJP|%D!Z^HBNIoTw^)&tgR30zw$|&r1*Ib zP&gv-SEK(iTYtS@`VkynI|i0Y&)$yab6zKxJDw2FhY{Xz-QEh?_PFepti*zIh7ez7 zEMUxmKi(FetS;}+3W4**HI`pH(fk*}%Y4@nk|iMoUfh3H_0Aoi7a=R+dG&Jq_0fMd zll+l7TDq9Q6RWRgr-RSWjAQjRQ(m3CzQkd9l$W1km!qUwK8T8bxYo9tHZS2!yDj0N zh9Kg(>iOeOH!07ZT3;D!uFmyN4~E53BR7|^EfhuDatHnU~(8 z%`*D!w0Sk`@D}B@qdh+qc;)S*EX5%BBy;Xuvjg#`JrD@gynK8|4Hygl`TXN_JIUui z;_#ZifB6V(3RJiUNV-V)hA0I4l@{D|=K zeDUMQra+Vzacspl9ZhbKSRfg9Z*MT9RyXV(EBp7nbhWK|dlqB$r7>qPI^cJ_kgET6 z#gD(Jc}Ybi>wizk1KB0axtF!E^R?8xl%5Drs;;B0FK3zCDP? z<95yk+`#qy$YxugU|7XqapYFlA9&>i6kA;}Bu+OE8-F3qes-Y8Bh?Jct9XN7MT;s{ zUya{VI}a%wMDP38jj;AO&S-2!zZTV($!FD7?Ac2BV1GesM|m1; zUPm&fRJf|h`$ucOe_0iezdxn~lPJUbT3=w)EU-CFkJ=-mFUzA2DlKzDAZD9!;6?jC z@InOi^Z8T96NKFS{#GX^ND#TgIG&5;Rhy#DK}yE*BDe=>+|K>wy*C_vZ9mo;Y#)X4 zGL+Ba6rGg^*9$anSKq?UKT_M{sn^4=BQBHu*O!}x>nsXz^@Z?y`mukZa}1ilFT8$7 zYI`YpfAU(b2vL?0c;#McVeR?Pd5L^9eRaWA51a<;O9@&TT(m5k*BpH+gBVw|*a*Jv6N;;QqL3~YnJ(IODq>g{_huQ2xTPEGm^*QuB zm7fdeaf6R>B5UQzU{I_d%W`>y&#NbeP~I35Nu=n2%HPK?PaZp8VGmoaMOPFp#PV|6 z$Ssl1faRrEt0=p=|CjgP{we_5Kjwz5iSd1n@`6J;k9(fwz#-9zu$^1XG4n&z`f@Fm zZ{p_CfyP4BT{Z>y=atLd$@&s}JODP?ZB*IKj@l#RVx9DfCET!WQnfZRBN#a3HNNEK z;PWD0ud$M0@WJIpidqp+JfsHBhCUx8U9scGvAS=bQD3n3I8$EM_b%+Z_7LS22ge=^ z+?+YDj|U^>KWfva$;&0Ae(zS_05EoYvGF21kz$X{=jC4Uk+@;~!np;*G9gfL-QVGa zH~xC^?C5!=&FABAc@Z@y8B87bgK$TQb+Z&Uzp6XIGH1ym?09IVylRFdXUnCd{(k4? zym9Xj4m6Ok7Hax9Xz#oTqo~#NJKi!-6ZDx}t-y`Dkih?RQyf?Va@@V%3?u zbKeb6%A<}~hg)|ZTfQI^&W(hcef)}!w-J4Pl+7>V=OV$0Qzgo7b@;qU98nJ(S@84G z*#*8P_s-ixzT#I~?cc=rj;hRS#ouA|MWQPgelL8KQvd(a`W>%V%`S$AI#FH>Rl9dz zAfG$7X8oY)?gHB5?V@>wMLmVcepl0_{I48pUP;&d4+^D|Ae+~9vZEf&f7#E;+#@Z? z4Tl|He}3&n-oISq(39;8@OdF6OkeTGL&;h{0@x?BC^|y)wZ${mW}7XR7v{cyeFLkn zgi*$@Jv&f))D5)_C~?h&fo5Nks!-bbLPnWIAs5fm9zQDjY+G+18Ujl9-!he?q4C70 zy2N`UQe2SEy0bQdoX;tF#nm-bj_>a=?V{=*k)MaMyY4m;!SY(YY*9_(%YV*GXS+y9 zjSkA|SfOq9m3idzGMB2j%vp<_-=?<5sGh+n^~bb%`KgRl^DPbru`|!Xt4YtitZ5qg>as};>W8$LL!A% zN};?aq)c9mJ<5R_&O(9nOt8G3NCns!`npr*JA{KBG_`bT^J;oa@9Yaf;KI)kd;FL4 zGN&A$uAF_A6Vl(R&uuy$48buUFJ4{$2VS%1y-+d4pWjB>y?J}b&XLtveaUM|F39%4 z`g`Q>Xz_~ba#1x-LV49h7gz|7+OXo8{~3qW{24q-cSfCD$)`7b<&fJmzztpneNHbtpY`qyWon zIcG*m{{WWPoipiI*h6Xi`=iDz2^+o8`81|g{_zXf=EC&>M`A!XZC)B}2k&hZ)rIWP zo)6yh?hujZL+$TBE6fg%aVO80-8Y+k)gR@>wMMj)aWN;H_+Gc-w01Cv|2P!8tmY5- zufGWa-`_LveVE&ojpgMM(#+HT0n6)ew@SyHd}ZGO<(rMSeEVvjXh%}={!Y3s{Osj9@hFA%DB$8<-ZM8`u2uN8bGj$ZFz z==~JbH*4(Q^V;=2Az+yrme;Ah=XSQt`se=ZC^zGS!(1pYLj{Jbry8@tN6UyWgZuRXx0iRI<{BSQSZvfuFfYWaDyqd5`L6>_E4*XpI8LSZq=>sB6}x;=SbB~!#e zIQ#MxrMz{odUuUG-gm!o@!!M~GvpOV*BEFkjm~d>5xSt^_bdxq8mz6g$m<4a z+2ad`BAp{=Xy+$*zjUtPx)cCU+jG1MR-o~OpO%ckUGn~4K5JNv`YHp#VPE-%EA#$6 zuf5j~rk~u89X}SmV;N{+!saKYxgwwV{>xmnw{t28J+G-I;Er8-d^YG+>)tJG)uYMl zLj}EgtIZ5}sm?mnwaf=r-?rHABQ8&|NABFxo?QuS@Zfw@7GbY9sPU4|)%x=9_h*`S zm_=peId;7q!?hijOS4H7-ThzVuMZwPX(dW%ej-}o7{{ipY!De1qt}a~%`5e}ZST3H zZIt#S`ls$6bB1QwwQutV*7L^}jQgUz&V3AV_EzD5_C4pOKJM@VA%Pk$!ykf#Kj45BD+9?jJoGf~!W%wV^^v8zSn$3M71`=J1qSAu11o{G(H{Qd6kle~{;$E%C4MLQmHKzSV~4|fcm%z?9}uU7PJqOC7+ z>F)hZ=V|*dgcs5H?x|IkX#8b$duWGK7zglp4$L<(B)@-K#}lgdLiJzs;;V{Y<0_5i zMYsMuP_ z7gD2qoWP%PA^g6t55!9|yJ|nj=ap^H&h&B({(OhEw`>$AhdCsF@sM5k8q2Fqy1VX^ zCf46m*MIu|*!-8LXXqBe1!(`6VP88HL-t=LhSne7sbcN%1U=Vko?tXTQ8*;g%lYts z=QX)G)9uDTUxDqRyvf&$TrbZ**dw3)GQDr_$mj2^dZ$k6vO59O zRh!g-mFW4gFe|@)ehr$Jm8R$PyF|+OhpxZ$r49jv*Cx07nTh!IHvjq9jq4-O`TJcn zUfhTIk@{6XpB0|ob6Sw}UQk+KJ^B8aH#~VWI!ygRL09pOQ#4XP(fR!D{x((?NLL#= ztaJezZqNc}l;19fhg8Ju`{QMYO!}{S~ z7}_3(xmH>_pF_tJ|8v+sy)HagLF+UWImgPb5QmFsWvIkwTU3Hou!aetfut*??xayu+B$o&q^I-jh9 zCk0@~*GX6Jjkd>-=G2_049N2xh98T(t!MQ>r#wO5B8zT(Fi#3Qn z5_oam-Mhtin;*11z4m%xC3#+VtDO!_Wd9O}ljaLY&uos8b%TZat0${m(7c+Q8;;&c zL9d6J67PtK=ls&Alnk%LqlPrP6Uh8{+Dly57XLhqL*|}|HV4YQ>JH9j$hzr4UJh4M zbQt;mg>lr@=iSQv;pT=VAB@wHcy%3Z+RRK2Gx*|0y((;Wc-Rc6U$?!b^shbxI*0-+w~xM2K=D+JBMitF&WC zIRsyS|39buO+2ww&sokm#SaShWhk9~h1lbet7aurIWt^%{Iq+h{|E&6bsh`!`+Hsy zz2d%lN@!mBGbhB)uKVx2LM7Tb&SxXM@&b4HPaMmK3ti6_Hm;zouL;jpgP&ee_FwaT z?$f(f{6WBfxFmirnwKkg_vH9`KR7e?Wp;KId3)r0=Pz>GhkX5g)w+GHdAzQ$<4aR@ z7=B+yf=&qeucC&R>-haFOuy()SBEm~aX5J+ET|V4ct91KgDrH_L z`mer?NyF=FHT63SAEZAB>HMhM+44T+r9T)wxv<&iHR8Vr`r7VuD$Hu~2()m25BB*+ zUSDAb+s1kdZLkCmr11ns__1_hnGSr*xtP5x0L^Q|o^9Q8XMYpl|4Uv4OHvo#R3N-q zSp1Wji}K;2c*)%SXUg`tAr$D9*eKiM%?PQ+9UuJRKvHgwVWA??eg&)cUef#}*3 zI-M$h@bZ`l=SnT|`U*RXF~<)4ftRxAn_xv9C{Swq?B|SLKe|jMVpEp>Mqh=@zK?bt zMy|t;3xb{>O}{Ty>HNhWldrvR6S-9YCG?MaZ?>ZA0Z9GVTc71|v)+_>tzQ>6ajqf& zcKfo8^M6J2Ds7fhjKlcDzMg#}?vvzsc`&ir#HIecDPPG6MPqfpHGL2j-gg3KYnM>BA>*Yv!)o|1- zS?#3{R)KGhvTh_`qvrL#B6lmE47JL0+ra?GrO$&K_Vc&cdmk}br-u@vA{EMimLEKkEG%uEwPRC_F z7od3EZ97Ee%Yv*=ejW#Hk05NlJ-`y-74>*wjIXZ{@`g62hD}n|SIY;!Mx%KV{*HLuAOAq}YC5PMvGEsP(~8N{Nf!&jV7{j{ zxs5U}C#D;>y6X)<=fZU%e@61_?VXF^-29Tku;8d_{mBxER}UJR6iBUOhd8Fw=F@%t zuu-V%eU$Tm=f$^iS?>!KeF(`+`@YNz%`4LRK~Asg)A)oA8dfiiSH zr(t$FXrDKlSCZk)+KqF+th31IYek1lZ*>B~E56u!;~D&ZM5d7wGPKf^{g=`3x09QX z8WQb`aD7Pa5#zPp_ku$R)Koi+4Py|0Pqp1gjL(4`TATRz3lHG;n^w~}*%L|jKA(NZ^mF>fq_GivMnR|;BDhE&TkJPzkkPFQ{7pnm*nex zG4j*}DHdCScVT3-JHC*pBf@%UdQ)0E-wNdCgm|L#>WdJqxiaGS|A7m?4koRiI2&Z( z@Z>7euQvZKj2Chg!;^u*@rH5A`Wki|)}m(8fv~sXi|WhziRUBbB`z)Z)juu>M5X-o zgBOu_g20O_t0$^LFbJ-*@n2SABhRbpeoZ=6KKb=yn0prQ!4_%Q{3Xccqjx@vS2Df+ zEME@Vf31F6?x)96Nc{ffcvWh&mTJF5cwyhh$na zzpw9mRxmWoF%BeGp?ST2=6Oy)D+uVH1yft{kmrTegyWFv2*zarlX8+Lt121eq z4DL=k@%oVJDI8g!*XuD~=3HLH9#aO)Y)@R|f(IU+YV_}d;3lSbz?YYN{X|nA zt-jmtU*<6qUSXYIW?UJtu))1=!v`89jwkRM$`7SGx(02JR@UEg7xPPq-=CbmcF;9$ zTq1+;@`UMY392Q~)?Gdy#ce~;f5m+$v?#R2>x%*}kN5;lxyVo`(U%(9drg(-zpk7a z6W?*03j|nq4c(s)g3V{6J~*!<&uj9rOt|yGU-FX%1$m=+Y3x_q&@+hNNAI8gMO(q-d65I&-|rCM{$h$ePeN*sQQ8?n zmDFL-So&4vW8F@oz6#IVmV4v-=h#PipMAv@1e{a`wJRdY^Fob_L|=r@Z+u*Ik;+84 z5O~RjTC;}OB0v9s<(2t>zHE>O;g$J@!?nny7v4U`&p-Ba zA7goXF$8Fb{SLpohs5_O*`wBh8f&3f2@@!m8U)to0;!h&A)X-cAY6ZP>&I$0%$~Qt z#i;r5NR*V7Zw}(u{-1e04H{B(KzJ!zLvoo0hGlUv&`hb6 z@iCz1M^b(LTV7wgoLcKrBY@}nz}2-!k@@jhTc6X?%C#U&HEh#&IS95@73O$7BhL$~ zxmTagf&6+1o913MJY))-_ih{2PNDS`$hO0UYHc!6cLaSkg$*s`wx`^0XzB(3t?o30 z*M}tQSkvdl(42F>U**uh;ALr;$MuM^z7DJxj~=}p0YNfqVTyWaUiWa#9mb5oKZ+5m+*JZ;m$yAu454g~R2u~D8S(4G-BVTazVYPe$8%L%v@CfdVfl4) zjn#n6k0z@d?2Naph25);*_XEl0hNi))wyo+ynb>byrS=K6F%v@((|HZ4Vu^Omx=WH zrD$H5%}=cBj1{my|DOE)N7HSK$_`oxuiSXM7PE!EMc_dk7n%9-(u}OF zst=l1-Z)Pv?Kpb9jXLg+AE5o$<+-_yV&Vv|O7DhO-|_JTG>Bg)v8Bu_4#EcxGf}?( zIK#npab!gVj2ao8ED}fjedmU|e9lqq&{?t4YW{o>ID1e#KBW75Ugq{52{HJ6y9itf z_K1lZ!$>M2evRP2{w*)=743JA{Sx0dT`I8G?=J?M*UrbI9ntwOQh)z%c}1GPaT>c8 z4t81lIrL8j7t7Xq*B)~k;voc?<} zA>bnIplyf5_q;q-*$yX*K}vsS(K{8LUv+ykcw)cij@cppK_T&2>A&~c9SH)_$+;R~v1YV!S zKdfc{f?jX`&td;i&vchT=W<^fvVVs}wS#+|RuRm$ZQZ7jkIvtd^1AkXSfLqzf0Xom zr13rGz?H6M2B$!2ju$5>^Lj{4m43>N^7Gq= z-8Ifj$A`l3&4WF249X~8T`n^^DmMe+{EVFwE7u?Nb&2_vTMYSlH7uQ>U!r3d;G!R4 zBxKR{c)`);YON=lS6HperiWdW)^5W@Ta-bvTn2X?v@Ml_Vb9{z=%IqYG`O@A55d8rsC)^Da>QG@pPq`WMbt{Pkw zqX(0@X>9qSZxXp#NDL&S}V$aZFp-K9OsR-lsJj#>wOD` zstdp09QHw}_zlM(uzB^hJNpiKUYMo_y*gWMka3ZTzi=N9^g2wTd0pgwfKlN>^CEEm z`A$AwO7(j{{H{v z3;*XobpHw7j4K>RqrzaFC4E784>CUzc%`hIdNsZ$7)&}#ztuk^&r4jng8f@Hd3!9K zd1SKro)1K6e3!ozgWiwgstvdGjxn_V@`#AI7otSDenQhlvcq+*1f1qJu8V)_2_vSr z6{b#>Kuv5VFUyn#D!wP>g<)P}a@|-TbiQWhv`Re260Q%az9J{~d22lnhx&U-1B$VT zJw7lwv(zz=6;g!fOFAzFLxobwbAcxEyqXvW?feUW=_5{}uhQbTI}d;Ng(&s{F`Xr7 zeenr!Et|xzUkKlmV2^w!qK-b1sUm)V!sn#^i*s9G^}7`%5NN;s=%fOEf4rl4yUx<$ z;30K%d--1U{UK6b9^;8Bb+2@wX?=}xSs{7?nHl|f#esK@32UsAq=9QVf`qT`9$Wj0z``28OPQl&yTvX6jO|CZx9RHab8Rf6HY zIC?##a`tvwcie)VQFRWz$hG?N*o+BgS)#NLtud&Jn()Uj`Vt=#V z_6m}dyIzRqRmUMO&dzX=`2Bx=KQu3En^T^->u^AAJ0Qm+;13mgN)^e0r7#miyDs`9 zWnPC*iVImd=>m4Nx_GLPe80yfsXOL--iCvbcyz&;ScF$@i}OmE3|1JcD`W4%hQNL8 zi|XQ%zbgEw?)ZZ!w7zavzAsud(&bN4GF(f;DEMAQhoXOzUJCkPubsBY_}2J8W{;34Vrr%?LqukP)&)# z!)SIeVtkNAcPJEcH!j77t|QMYkM3Bw(Khn-*u*EZas+n}j_EGX4!?rtl|AB*X_H0s zYFd}PH!l_GS4p@Yr2Z=rXKPsDQ36{=-$+03@dU+X&s`2ymO$M8O8Tq6x&KJY>)DIE zwAIY?ukM4l^3PHf&D%Q7$%&ySUDF80S{$@ zvpvZ3nhUxnB>}(Ow4^xV+Dr6b>Acp8$(PyS>TrKg12z;^O~7*=0rI?lj)U+@ ze!X2l{SwYSM)Pu?a8Rp?K;}&XFTxRjzLSr?TzOI|mA9V-iCXys@jE<#_sWr5J%S|= zA>%2ZqlM0|lJa_%qxv$rigNzzyoz;2k#Hoe%apI);kb*)D_JhyIqE7qET`%zXU6xl z@_ZyPdiJIv+4&JQE)snaJ`WPwH`FC*qYg>PESqsVXAXfKRB|*#Vutgk(b4mrl^!7 z_o z2b;^JF&9_4!)3pE+vPWl;iS>(Eap-4{3zJc+T!CCC@7?Q_knvJ`dt~ zQ1j#QI=jUu_~#Ycp5~hkpm{NVIbyN;6B73m^cBX}-f+k5m%f`n27-%0HXrk?rl$XJoj49sUVAz_M9y|c z;Ge^lRChN<{JnHcSfy?fD+tQWI`@`^fb4;c3pbs~^ZGFi;tP5I<;)^gnIVbzDT2OQ zzf{>UK1Vnbco7fz;~(Vxm-nHF@ze->zRRJ;zCg_lHs1Nz;Od0G-?2YU=JZa=ywce< zuk-AnY>&SBUdw9qB7lSI{_6?b^A zHuG5&Uq3-DaCHyoKFa#i+um+zI82#Wdh?DVvx!Iu5o9skkcs%OFM3_^xDBk3nxP&3 z$v6Z8uW`}sIzpb;$T|1bZD)QN7s>p%dy#6uWvdrV#Tz&^hLoXrvBoIVPQLt&zV?VO z3wwSPOVA_XoitumUE`ZzhOfU=Y`G!MchDWy>?jEK?JEL}3m2_iaw+@!)6e83kDaE> ztBywYv#T~fzQ6p~W{()ci|J;S@m)uDaPoO{C6FczmWjL?d^$m%mkGNy8^?ouWLzY? zI3A?i^~8jNtl=e^qyjWA?j8&4-R@|8jSOwL>S=l%IqnZ&Xn!A{`c~`$UY{59Z7WmK zJwZA!b()(F2aiv`V%>8S9j}twBbBq}@r%Yrz^A$JHNo%(@qDEI-tobzvAi2mFn4)K zsk%|)pX-jnbeBn)mRGRD3cq)m1p}dA{QB#@@-yUl9gEDm*P%{6zOUG3=V+Q34KXyz zadtImUdtT$Iwon+yp|ZXeA(mEO#J>oaQ&v>=_pyHRmX zcquzv~<{c}U|4tl6anpK}MG!R+aG$8%_2;~Unm zF~HY@6L=8(*QEK`!rVKQ*N@}kud4csalpK#{F6?ZCu}#`>vaa7uZTKtyN4@)GOzfK z1P{$*BjWf8--DFbYZWE`gd$&{!Y$8TM80C?SCVdiVS{HYbxtm>3WXkT ziBB7g{yQ)1H)&OKD}6{|`TS(hZ~T`@Oey9XKA%9!tLcdtCN?SpOZYkAopgTOW##%Y zK^_MmV|3m+T6u!=!cpsaygxI&!M?X4g0jB2RXQv4_zfYf;^E14->wqRM{18ULA~*6 zLy=%|t=~|BQ60r=F3rOE*|!kroZE8?+wuoqPmOGomYg7-htz-l947{Ibj{#o@)fkc zVwMaXeuew*>qqKK20?38aPX-hVM4ykd;!#S%JVS-XZZFFmb1iMj=RJR$L- zg(JR!I1W-?HgpLoTm7Qo&5;V>eKLsu5^?TYlbgv3{mUz4{QN_JE>y(!V9+0U5rF)7 zC-1+gc2-(^YDe?Z&h_Ko%ZcWNHH_KOr~XTw`aj1f)z@hOtjQj0d_7ZY`)Mu8L;%0+3Em@E5 zfDjlnN{{zF_V>Ja?;MkR(p5pUJHq<0)R;49mIci#%&38NXdC7F!v5Nl##8upR_pg~ z%bd=8z%E+z=KP9MIDJ6M8&mR|_@2byZ=+wXgp;St%k0d1_YDvUocYRaQEdpXyRx8_ zF31K~3{wUc@%vjXh+w5pSCi+3HBpPvr}ZJ~7-f&g-&4nE?}T>R)1F&sDp9=HDmAEM z@$=hHnIid|Fq(wxgM&yt1L1R0ef9gVF`x4)h16IDi+4{vfHwav{g;(zA^3}A<%5xb z!ApJ8=S4VWUablij1EbWptd(cq$mejZ>MmpztZKyuZR3DnlmeffZfHXyVY#T^ZGds z!Yg@u)JxlchJy~RFSpsp2Rl;G{+@6|0w=;dDX+I{d(7gKaKP}~F=N`u1CGpk)4Lrk z1qtD88+=pHyq=33pyqAxB;HSJ+iLVa3g1VQ^n9fLOJnt@3wKx~ylXhMo*@NUZ|hFG zXN1?Xf}w~1F?Wj)cx1dIYMmQ-UQNadPx{t+VF}|Q)mNUO{AAQ77uah_&FkEO=5=T> z@T_kr+8*=7+WIn{QI1#R*5vPUFUJAvO)afQcwUx=eFgQ~Nc_J=}dWX3$IA> zyz-9rB+P`8zdsbVch%?DSO?gC^@_WRO%>7Z2>#10)MvL)&{^Wo3Gr7Rmog-M`{lXH z9~T6@e)Qc~_+iPiQc%!u>k6CpfKR<#rqZdUAj*)fv*R4v-;?Sqj5oBdwABFe81)(( z){@T`s)lWxz?MaVmE6o#cROT$^xww9qieznk{^x=ae0KmgqzIg+={=imkZ=sBNTKW z&FeDX^}`yQ|GPa>SJw}le!;>1V!kl?HV?4r{jlLYo}Znb=N3O@%DkR-mM`0NScf97 zAqM3~wO$d>C^)z!UL4U^=_^a{h+_qv#8|TxdLbZ`tuK_(ZT#2y`>~Rvrmjmb5OqY* zSB_iT$LVUczJA&_;T7)}$>^)z%U-7D0S@fKumPtIdVtP>*gKg6IA|0+JWv^++f+fB*V+7$>|00iyzJLSVB|~5@?tDb=5T}_IR|cCFRJ;-|&)rtG9Q*9dg_s z2SWR=Tw!i&=Hauj_hnfItCk0FY2V3O9L9m3^yH_UYBVqTe&_sVK<*C_c}2d##Eq8| z&qu1S^U{|#(Cv;A0L!P^RoiS+#npYi_%$w4M0yM9aJOjzJl=JsaU#`};)0e@vS=S?svF?y_>#LW# zZYlT=Z_t;Mq^vK$uMFL_mnp}q=KM1Tbz>25E5P`!qXTTv%^tgc|LqWP zZ8Sc2a5ee*3BJ9>eJ%@ssQ=1{6&)DX0~5a475aVWiChTtqwVnvk@%RC_;Z531~(km zX3a$EWqxu;`!Ay!$-CihWiVus*mFYK1000ir{x}(Law$#ME!F#FVk`Fo088E|79yZ zSG0MDE?C+7?x3q_!4l3x>c4u@)4i@4q6%kIbH(i3HKXQsF#z^?$~J!C2L4uB{eh2ivrYZAu=YkmF^eRK`r+9z9-( zsc)e>O_o>K(AQo5)Vtw2^Uxaq9yBkF^isPC{Ql2`>q_8-6$pxKnnIpGCwxx2evG6Z zJye)n29a6)W4JbVIODeeH7?uI^4bd$oTJBK)^wqGGaw$BP&^-Lgjxw)A zI`gl2nn-_k!gV3#Rc&s}WEBtzOcNYZk17y*3=}!prr*vAUn*DL44A_2f7Km(aPklP zvHmzehGdJ%m8!ju@d z)$bm7!48#;0YUXRSl}t%%X0|rzex4eIr*C2uJH@o(@9tU+qtM znp4)`>!asx%Sp`Q`!|g}`y{%Qygd>~5ynI6zX+dqaLLCnc%$=QbrUo<{2Yo=_DJ|U z;&Ffc1MTm7deWG+kDi4Ro6i>8d%fVj#Em0w58&&4V-3Fbyhq0qq`Y(`?8FKRDf6na zGvWKNAqwi@+0|`~h`vnwF4ZlDvVr{K_>??HC{SC!k#YXR{_a)n2L)A{$j7UgXyMn6 zZ#5xJpyTVQPPD!>Zm0Q;;`{p(xDfnT7=KB#w*4>n&B^d;TA#lA`|-1|HMUkzL(m(7 zyd`+7CU8)-y;C-R9Bq$Vop)c%U*$^7e;F}~J}G!%0NAxLU8*n1*GH$n>-}cq5e1ih zliFzHk^IEV!%yRBUa)~wGrMBQwoo|o%n8j*DJ`J90dL6?12#{k3nD^EG!BfQ*a z2VYH1v4V7|7rqBr2xK2zobA0pp4ZQD5MIgquRSY;t}a`T=2bI#s%M+tf49fN*G7+} ze9J&-?eWdNz|ULsqCqox-2usm zi2ow+>iXI}Ifk#t{QrxWQG3c>n+!CsT}g9?(`)%3;Uu$;GMya z(ftUoV`>J%4v$!&Lx^J}~hOykf0(#cZ&} z5;zgwNqI@e9bWVCco}d=(9iO_c!6nb_#5-eQdr2@du-oDw7x_;YBqIz@gv&fW51~2 zLzV^*X4k5x(}cvwGj^NeXA^aayp~D2-YaK`hIOTS=4&mG^$?$%ndk(5zs!N7Nvv+; z!H^r>Aa!w6=&$3iJX_tU4*b4@gmDw%`>U#-+nTu1{_E%O5njo!w_Dn;Nq+iV3ey*C zwJ_g3fXZX@-hTZ3$ZBsx-{}@~eIakPLP$A&Ct5;(rop$m%QCJY{n-iMPi>+}EJ9bC z$SZA8Nj+U53OYN)^77P>enYnJvoF%0!|z|ax}i^a1mE9ewoBz)`5;Eigq(?3NZffyrOun)$M*+3Ii)8to6A(K;8NCD+dD{bg|pzF|@wo&-#}a@@Wz6QT{TE zL1TIZII{@q#_dP!@rJI|mc!j#;GuG3@dAGTWUQ~~oUPm6^P-VWRMWOW>+5Bs+FqS< zw7v+ODe>wO$!F%(#ewFBOP|j#ID(7T%iAaAiy_cklYZ|(%KExWTi~+tGyeIs`*QLQ zB}hJcb5iBQ1K#_HymlWCDY)ho3D=gUJPE&#*y9my8#f(VZeTvoBA_Q942}Ko56ttD z_g{RQuO%M)Lw}}`(~r9*59&f~|4>rA4w~2MbeY>{{n7r5k7-}gz0JS$`6HvRoy@xJ zLp3;98FBWQ@%OUaVw5|rH)?+*`=1H0LNA6(> z*GJ9h)3()72rmYa_l)lN=R2f07qcFGLF~~uiq=UYh#xqt)Wz1F4hA{fed}L;B(JZL z0WN*I>|gTiB=#7l<^D#El*9Sg)k~OcbiO8!nEW6;vW*wrw z*iDzWzwwQN$t@N^rcLHz7hHY2}SI%C_8<+`((*b3%2@B+TVIJmB_sX)F%E7&gm<|KPM{T7 zxRs8o1X2d$FE^W^c_~k)=dwLR`aM?PUB#E+pbufb**-~TXNkPHTVp?*DMsS2ovk;k z`NehPBqsJsg-u;2sjA!)J(_ij;k?{JdZw$sV zv@UkM2+d18F@F7X2{bRl-~D_iAAdcfvbXqlssui0&t#QLIf0z?i^1%SVi4(!&tx1y z`+LifxX*J9NItsD?p08zH`?DzZlo8&IUx3^J)O;DXBq|9*G>-(hN=+f)05gOeAiy* zfxJ7B>t@75U|IUwnTHwV{TC*N-C6hIFZU%#c|Aza;L+I&oNh9b18dRoz5lo9#YkDS zJ%&AvV)$?_jCj2X+(_g5)wQCuf>&^GqFQ$QG>ojiCMzCz?>cAFRvQ8sZ{_#9 zQ_1t%zH%abfU^V{7YQ%EB|DoU?wCQ8#a5k!QS|+xF&-ED75zEHxStSDRNa0j(SWT- zj{5@`I-cODyYgj&I=LB*k&}ZxOq)CKUJiD zBEa-yg|ZmF-gm*j&G~98@qCk)>aNz|_w6I_+MPeFq}Ut{SA(f^LfVmhp;?mpjwf%p z;Zl`QAzf<-=$?JEB;~sBU#}n6YM-r&ib2;;6k^ow*7Blx{p3SGy=<8{0 zDnmd?6s%*299Lh0*khJrqjwAyH@@z0T8aH`2n@>Ui`?@6dtPU>cHOys3eBth+szBT z6KGyZn=T}^q*A`$LG4sc6Rd;7_hoAG*nQ0zqC;XQCQEQovfb>{(@$uBA0M72IH!*A zisAU0cF7ja>j`j+$vi-K-Ss0+&j*_ zTlFP*d&Dx8e1B(5KAuP_ow%c7Xa&-ThCX;Zp?Pr*xoE8&N6(MTQ(vsEl&K-w2jRMt z`ukP3#_l5c=VACNUWs(EyTHDMCmXGUN@3|v_CptK(Ei?5c+_Vw5}6-o8J~()^x*TW zXV^^{2V02e+b+i=T9uC2qt*Dm>#q*SKu9UJe8&OAe;H>-wyAtv2USje#*Ez|ATe&& z`(2$pub<-}ypqqaY8&lv&wh>eU-xJBO+4^L>nltlrp6}AkNEw6o&oLes~kmanD^o1 zuer0M%U?S}{yjf~dVF7_JkR#%N)vQ^&+<6=(TWToVm^oNLU^BxH2!(y6^e?1yKWHA zSEknV{#G!euZh~3MCpUkP$h7^i`5LVM^^2`0=Wn4;5fV20@Fwc)Mmb1(^B;Jyo?Vp z#jnDk^`-SN*|K{ZT3>`KK;T4pC)L;bK{+jTZ5(Xhb9hT?v=fM$cHas&rGe;K=ie6`H~o|B>H)n$m#>RTKNdCVkP|#0$I`E`%!b$*%je+C$@uj-AQi9 zk4*k@UOxnGORg$r?hJx}qfBjp1 zEidle)m>c-Yd((Fp89ML+On^D%FBwt&Fg;Ep7;Av^%Gxx;=Y{T>Or(eu`v-o=_z#o zoJiHen`R7%zJgC#jw;oJI3LRf?Ns3qgjW%Z znc=bAP^ilI=K3X9mdMLrlW(8+XMT|FrHi`I9t_*9!x{bGk=GYr-DIKuPV)2PUB-x> z1L`io{AujO@Gf*dS{%Qtuc#7QUs(0tHH#N75x+lS9Hjp13f0MHDP#Bt5?@3ek0`VU zjzcHw^G_7v_vd@}dA%Td|Hl<^mzac~yAgTO_tV5^(HTHXCim9Lu?6D!24y%#hgT!^ z$l(3XFu@=k6kj<`wOJ$fNWX^7L;UVWXqxVFxwsu4-`lmFPSzs7-Zm*J87q)FM?4Q{ zd_NfAX*d<)2CZA}G;_?Nc~#|Hxv#8`>=Q1 zoyh^%54<=k%vub$__XYkJka(?FB~mq$%d?lbaYRD8m2dZ(lyB<+Dm7M+) z>dZ1VDuP>|M=s>vLhDQ6^MeLsCggdTy9Q!bD-NLZ6EDQgr*4WM_a7CMgbr3Hhry&* zi&=a&lAn-jeSEj=2tTl2sg8*f2!^@>BZr=}Kk)i_;s1FhAAhMnb-lf<7oE@fI92$4 zcPBc3|38QQ!}G`!LYc+l$I1X36&}rc*bXX#kL>fQDS_D~Cpg4BjZyx-=gZf$tQcLQ zJ(e#<;Gc~}^V%m6lfGRQiC3*Vy7P~VhC|hLd7Xttq`&VM=KzT;b6&_k;Ck)LU=X;p zJ+3sA`+Hu}Du*CWAI-~s@$n$l99mzP*r%@pzJ?>Z{Q(diueQ&t#U1~@dSB(N&%N}v za78wiZlewU`MzDe*=!x?{gG9M*Tn3uM*Nq*nsZy$3H1CZ(s@#SzV#P>|Hkl;YJM1a zED^e92e)WOs_rSGc^zWlFf;jr%#U^6Mn}EowiD;O zpXZ$m{Yo3LWg!dU!UA61z>mqHZMJF{3Hk*mx&1cUp*b! zc$FG=w$#P|(E*uwf@+M3BXI`JYfr*tm1Grqe*Cw*DpQns@{g9m%jD7Iz>Bsp?j%f8 z;#dR=5121M4?^=g_lADeeLsZP+s%eIdjj+z%qW8@Y{@m^I5bT{Pp38`yfVa#){cG& zg58EW_`(6if2pm{mQ)nuhwN(;3`QpS{YM|x7c=vdw@0g6>}99kk+1)nW9eDG+|&kc zmu!?xevY=sRYA?|Pfj7c%mZevt>*CdxZ=@?y^Tn{Y|!_J^ecxD|0VqNxv9Kz zB%FAc@Zo*P9-o8oOwdaRERc&8R{QWf10_K;F(&Twz$I9uoGJf$hB>M8WsdXtT z)eK@cd-L3Ui{>@BuJ&WoB{Z+ROS3HdxRHL&Js?$Rxn5;p z38<%U>(H%4^Rh0N8+>>d;q_4lb0>-p?e7~nSZ*yIKzQB1?2^(pim#8hxPSiMYb0OD z5_WDj5;c4dX|}Ua6q-(f@PU zKg^FU!AxoPV`XqR`oO8cckZymCP=dm-(SW$e^_872F(j2%Alx-&tnkkO;`@IPaivv z_V>f(8AA$Ph&^gH%0e4mBxoLIDttzxMAR2g(j15SF>dG*RMllN3xbQ8Y6wgADp4UsxkuEhVbbo*=>?~$hx@dj<&td<-%QU&{@@M}t=$w4Ve&m2H7~Q% zsj{-uH{vI1)FS^saCzU^x!Kv`-cWfGVz;)qJvw&1F9BX>LQ54^|OgTn`j)j|W|7R{Vt3B0x zzzfkAIbNRo_b<=dkLYVVL%>S6H~4<;7Sfkrivp3?;e_<$DrR0-8^7)ehBg>xPNvPv zY$fl%_|C04{H{C|`%@3U(U(I1VEZFun2IwuY2!uDk8E#R*{ayl_tUl)9nF5*LAl?y zD{iHqmSQ;+)%5l6`f3NO2bGoGarpY248QD{t?2o&>D-<;+wTakWhDnRhMwcs+xZK= z-RVfbjy9DQ(pfbKFWpV1{4WZ^;SEiTFeVL|AI(>U`n}NQfiKfFsZoNId0~0K*L=wO zO~4Em zRF}KidZ5>j)-Tt(yo$^HN8d>QiQW3`)|B z+ieL3s}`3ijiPy(eN`Cm<3z`+F4Ql=4)apxr6%@JbAL-YSf^``(mk;U2V7jzL!x;(DU+VgQ90wh*;`X(k7L%%glpy04u(t;`v2n@rWhEfr+O%p|63y#k z;L!Z%gNXkk@*)|BzR|4hjYh;C7ll@zT5t&m=hv#{LU~BOKgB`D${x5OYW2O9+n9o3 zor)ThY|r2GY93ybyEuuCzw~?0eal#aw#OvZg}TWO%Jn(gRUL|Y_;uD%-Grs8+V-Gc zEPYA@zyISmf7L`YJ2WqkO*Zew(`qv_0yq@io%lgCp{zI40yx1c@<;`r_PULlKg=@DkT?ia6 z_Ah4bBE)kTOsMca=@l|uaCfcD1D_7OzV=VuUtdL@SCiuQ%$e)I^m!(kA6GN6Qawu6 z12{_J3}%FCausbLNHL3^%Ci{s zZ{{i63ZeD2+P6R_oxzyMt4(@rn_n&d`Pzo(jobO&VhK9lveKaOek;POg@35g>vOV;)gZX6fD*@rV_jX`khD2BoJ9C=wyCO9-|K8I7oTn?yayl_#6feN6+}q?nHQPJ;J?d zg&GH7EOjF_lmfvjq(M01^dET5^h})$G9#Xclouv$9jEV!-5}s^nMYrO)>ph%Yaab7 zbUi>-y3LovN|fur(pI!hRC(g?_0SSW=apTdI!c4;#fn0xy>wUI<8y?25s|F^Q;N6@@J@5aR43qjkX)!P!S8|lCF zwZQz}f9QCXO?lITm@^Jk1=|d3XI;SR({_&gwK{)?2?s@cn~o$Yvi z8D+$-H@Z(82Pv=Am)`ajrV%h8ZhH8jEfP=Mxs!uaJH`RcpZBbg>J5Ziv#&iJq2&Gj z$TFYk^s--`!yt*jFxq_H4MJuxr5PlgI)#q!`>O6?dOgv+I8+1dGfyGU5B$LO8(wB* zH*AaqOJQ8RE%6q(!>8CC`I-3zKs|keQ-=W^f06RC((w5Dx(I*&*4MJy;w|!gJt;4- zOhYZpm`M0~R%>}nD^fqv;pdpU?l=dyByJ1eR~ra>=kHAB{9*scs^n1f4NTH@J>Y)1(%0w>+8*VY`eY;h(fY!? zI^XO8^sqB1UJN~wb zV*nk0-F;ML*U^W>t2R_$TKFGPj=!wMwoa*zM}Umq$LTG1kougw_Vka+!mO}GHc_?N zB>>Ex>m-@E{5>yw;pf-Nozd~c220)ds{Q{xo*2?*Il}Y~f4?!z|0s6R6%@MZ-ncjA zL90s(_q9?Tl)oqCb=jr;+xt<B0ph&qat=SGASTYnG%F<5)ty z9N$9Nu8*z1^uZ;=%g~{@6E|1{l0IQ{*Z5q(Bw8rg&L|JIPC0K-52LKFf6J?giSt%K zaR`iHy8Sn(A-t47i3oCBox|=GI-=|sdI&smg92Iwf6t3kkmvKgvq&8@fft8RDW5g> zZ+Puox0rnHm*#A3 z9*iZ7htz*zr_`#SKTrg^0Z#GkEk#7TBj`(~^I1KMFq&5w_v_^mWd+D_f1Ch4Kbq~k z^Ld|Z5q=#!aQ*sQ4|o*6DfI(W9?V}qEqqUcvOQjiG@)01vJEiO$Ew{;kobi({$dYT zO5Ts(pJVWpDYx8OB>oy>*?Zy1NoM#MmGe~Sus`g2H1~qhhdeI~wV!+3per&iGXDM| z?RWiG=7lI;Omx0$nPt)b{;p?(RD>sGdt_MUx?n+D1i_)LH=7MS;GlaYRm+1M*kYfpb0T@u>gUlX1V)zeBLawPbB!V!PIlefp>h0cW7 zv_kNbOI@)(-5q#?St8BfW<$Pw=j!_oX#W*%C*bWoWl78z>c!R@OCFG-=)V-6n3T(x z1_S%&9Vdotka%^a(lfVZl4~IGn1u7j6Mo?RrMx=C{O@_4@s{}Gl7E8&?xK;)Gv>hZyQ zMv@{gmJqFgxc!09CNilP{T8V=F{K|obKQCkn4Nbu$R6>7fSayXp&vGq^yULKf!B(%hJKU>8>T*Tkk?n(NNau50rL5Yuxk4+57_i!;!KVEq9Ky!BJj!xG`w*PcM`?xb=Am~ zw5%d5C3}2RA6pXgAs^<`K7ANw_kb%E+p3S#obzp+V_5nThL;gS=Q?c*9 z(@wJbYI2O1V~?uTJ1b z8sAr_&wXTBfzNl%wI#`(^8iiB=I@N#bAey%l=uaFKLpbJ*E?P_9`^xrA}`*(in@L* zT7c=hQ(e2Vk$8PbdD$6Abo_tpy?IVl zSK9tJzw;WymNIBLV-ua%0*%-mmwf+h?{BZJQ!=7`9{}r;tBXT7BVLv-6n5HM@d0Iw z?UN-1F5t`>%bUMxbY2Z}wF)9Csc2rm*)L-Gv^{zkv|y`j{0ZaJjP~Ab)yy&3)gcV> zMU7((-amUQ8ute{C9lj!e%_!5k~ zx=o)|bJfJd0ghK6Pqa9|^@w~MaouVHB3|!$#~XP_(CpW`V_AabE?`{#roF!Yf9G|>LV11q z6XeH}_Wr%m{L*_-=sey3>Qu8cN!0!$ubZBFjf0QlAuS;$DsK92AcCCic}rs9yn9BN zP&uPMKKy33Nc}w0$4!^+#8F?VHVj8D9c+sbW!*#l(Fj=4#EL%depG|OhX%nQaT z{jZ527x=Pc`u2^dc>ZVk>QK7%*gX=(1xa3G4$bzy|DI7Fe|h^uaiSj&M)~@<@=MFB z{CJ2>{a#~R>;P|H&bn;8EEd|BC#*Z8!N_Z=`^hWom=Ld9rsFKtbLRe$*UK$0-xkPu zfS_}uR^oExuV!DKCY4dk3*wJujlC|oz>Eu_-@3C$=S7`A@ACr>D}tQQ@9X36c?zXn zwKGHPF(WU&rSoDEx1jTM`QkmCq%wa88u!OQjJzsOC`UhRjE8ky0U18cwEg%(O3xOj z$3nk->E3#8MqX2eT|ar2qV;hD2Tyhr&A%ky=kPP5eAS!E91kh;0OK3U8!Oq+{ZMFZ zFNaDn4;aO45VmjG4^v(xOP#kIo!9Tz8I9MbO77@2Y={dfU;7=LW#%%Y^K@Rn3_QAg z?Rei|TfZY7JiVXan-RMUvRyZGUNfTAW0Y{LyydupVSU^)>s{iD^&9BxqwkT#aNUPdS~U;trkj4n?axM*7H^W4Ep00WX+w&zhxi+Xa-?@Sb6z5ToiN@lNx|>st{7 z`MLbgEBNj6HHjL=5OywLA&V2EzbdjTDCALm_|Nx?>V3P`vPe?}M&zA-T!2x&!g>@u z1wO_Dx6=AL?Qy$7&`VWD#4#4K76n8qu4k05YfoAPE!~lS?-ADgrhE1>FkhJ7DMPKM z&*OJ{|8dnjMWrfF*m2u3$fE=4qw*~wOQ|3}V9)lEd$PzCWCVz#vWlbgD(2dkWnvOU zzs~Qx%v1EHtJ`QpdZ~PLPXwd?b*=cduSqgve1bY_#yP=PBmCe|crj~xMP_>@fc%oQ zM+r^4pF}N&QKu2aNKyUbv9c5M}3?@JY3kC>jkEYCnxUPiFoapF-fn8nHMgE)H&QsbAcOh$zF%}-+6^R z(JtvVWR$P%>%=z9DE{7 z+H!i=8Rd(*WK4wC%@N<5_2UAJ_Ny&Tu1{}S0_bwEU!LK=1;(57Nu8p_SFIZF2(4Sr zXzy28&$mk0g?K$UHecLBcNqk~1P|HQPwDgco!9=8T1ii7_RFNZdTmA^;&tr^lm5n# zUZN>uLd5f^9S}07*uh?G1g{#?n+Y|0>Elw;Yd$~q9TATn2FLKTbgR7BS5d?zXlG-M z#K9J#urpbN)#$&=7e!DP%q(*V?#9*(RlQiar8sx)8W$UQv(Z~K;!!j>-Jg7F(JIFH zy|7*v9|xfIu}-^we|OvH{;J~VracYo7C}AR^rqtJo^W3_XRe(sBd@h>v8A#+b`arY zwfSBQ;#KS_nNxKo0r47;>T9`(sRI9ofc=yyEVNHD+jzo`0l1a4Sq}Dkf~o^}p2>#X_^u)2UN~w*y}j z=R3~47)W_rMP;dD)JLcI_WYnq#4BK~-j*DnWzb*Mu(@@hkv=a{zPwFVygCW;ke}Br z@_mCB#Lc|q7Zi+m1%G%QnkvHwt>&|M9uK;L`}42zpQ&hlG*@b{TO5JnM8es94IWQX zKMXZMz~Jkm=S1mVZ2z;XEsRHyn$s!-Xax}A{WiP;%)m62E1 z>8JzLTcgJdg=0?eO}P4iIA3V6yZn9vOnQ}2b<4sInC<%pdn4mv!rkXP-i%?4-{T>%td<-{q>s2+&qHKyQWq^a0qI3_Z2e&c#?SQ)s+?A{1_uQ6}B z*qpn(uwlYSCxHoW(A*QMU(F53|2oZ4)#dejpD^X;gS5-E!)D9cDr2%9oI+X3ZhT_# z6~wDoXji#DYdfKndG4%%$$$4(DaO@lk)JaNwcQuWOztLtki1f|HrT<(RZ;!>GU6fc z+?EaMri|<3Rc7ZoQ)&ItWPj~}?st#e_m93ls`h?f_FQ%`=xo^4+BQIi0Nb)L_nFZ8 zXmB)R#|>Kip5yVD%a&2D(0U=&*7!M+FOtmg^9VjaPW*mMJ9#DET-Rt79hR8@eLXf44LR*Vi<)*!PbwbFLUfiH zH7;lH?`zqnb)Tz4`sf)`L&)gScm<93owDU0HMgGZ4* zlDs}MT_2N3afNy3J7jaIBkqUDH~QoGcRs%#qdyM&E?Qocf_PP&yd=BmeKQfBt`PRn z`R{q%y;fSFwa9# zzkhtkskWe6g)W zRau5xl_$DSBJUrmims)b>TJ<&k~9y*>x5tQLU*;%{V$%=V_xOnl7~x+0umPoG4e`K>f2md z)rd6K^3O$O}O_JB}=Ry7? z<#q(GpZ|3xNP#LoH5SyFPNXNX?tu3fr?3^*M8ox_fDM-aus#;ge7i+pi!oik6nAXA zDYJQW`SQ7@VR5Qh9$h0_ z@4Ob>dG#b{43aPPvybY0RyGkUJZbUOS%1$F##^6cr2l zO)EE@YGU+P_1!HW>A9o+p_c01)ps06k56bWU22e)w+JGhh^XwjZ_fx`pJZ}dFIf`9A2025x8jl*V>P0loD_*G3pB^w zBm1x@6cz#htPNafpNT+g+t$prEyRV=yb1ow|6sq?-={8QF-6z;G42TY>T+r+tMw-M zX0BzAJ`oSE{nf)Snr(*-tTiIMw0Pmb1bNmoRgCs4JtSkpD=)Je&Oz>PUz-Z^>F z;cUkGXao0Ihh9#ThqY}|Ti?F(f`uv?_M1Zxuco)FRQIP$fcxr(O-X_-AYWP9R;x9F z7s=!2`~T&TexLX2*`cFx+{?~%e?#(hCYA4T@P|fX^-Wm9REG5Qzv|=t7hi3e_~Ic^ zVykyv&^C}!tuRzN8v_A5`iykPE@N09Ct7B7%&(hH*GE`z@!YO2v+43O{Laz+V`a@A z|Mg6ZKzmBXy-Fuf@D$x??!k-p8;Td4P;H|<0chX5A|2hdA1;TTzr?#9@oI=Q4qq;u zL6GAPKa)R6J3}6$=NmTO+GZN4Z*Ol@R%m)f3vtAQ61-}~f3F8Hx3EZ!+IfrcRpQpSLg!(i{MFP_I;ppkc?_Ay$#mFyofD2TbLb@d!E zOrqZ_Zt1JLS2>XXDl=oMX29KcLO6d>LDs~pMAr$eZ!tUuP`lb{?lF4S4<k);I&`^+&Ge-Y`o4E-ZFjs zoXs2ufu`cZj%|$gi^v$DtUHDDu_>f>U`FR0h&vhT!_j$ zr|-@eb`ufN&&ol1W zJ_VSUYCFT-&l@Zn)_ncMhj=N^N%4PT&JB;Hwnx=vyMUm}R-^kgUe_kFFmXB2cs<=c z=YrOVzI{#pmD^z{{d1ujZ(Yd4ah0I@TWGJM_Zv_s&g=!723I0pd%h-gpH68dN-osd zJ^Y9JZOTz*htkxY^l^tj5wt#1iswY6}%#@TpQr2*xEWpE5i00oL zu~E0?GukiLF`2V?B@nNL>nZ~$*USN`Nz0>-w<+}Nk@8iu-rZ=^as_A=C>6d;^S^c^ zob>uU1M!M5d|b}q%?(~LOo6v|y8v5lnQE>p58SNSo}`$g48*R$T&3H|^l^vhLH=Zv zFKVsM*>kg`LCKu`z&k6%i)jPr47V$7MAD8n{Zsq?`}$by{a|8M>2>0Dc9mJPSOVOB zbvCWiw+uqBGV+o?9;8!!6WM#o?G?iOL8Hr;-(kB_&j||fIyI<> zC&LZ+=d?(i4@c{xj(L!lfG0P+%JZ7i>FfelF(-C@-NgfUzN_{GNdH+MiQ6o!!tvz< z(Yo6hlyZJ@rBLQS>OGY0hIrMgTovinZ6$W^Et5Jp_P_HYKcVhyg)*6WwD|pA$2i07 zHlQSQQu{(}EOZt|eD*eC)W^rO?G}diAYPgKVpCdUX!12@5+`*%ss|$Ft9av_oeL%@ zz#-dSVcl>yunU(s+02IIOS{1>{`5p{So->|P0=zJh|Ecf@wejv8;5z_K|C~GK7+fO zJxM?GcX>j`Brjs7v-LGjC5SB)Ipbo8c-d9Y+PvarD{*1^ifE$&Mte^U6*o<4asSDi z92e;$CEv~NJ#!*4Ly)M;cQO|d?`rE+4w3#lJ*0C`8_kGr%)^3yOO+>@0}v@dxm$-(tAb67kyL zllo=Vw^qV_49kZuO-A{mI*4`fycr9l#TX#K%h~6^D{Bdm@)X4OBJlMSr zc%?h#6vrmQ@$9C<9ETbG`)#K(nRFi^UelIL6H;`c@p^9Y>RT{ly@_bpb(htL<)KH( zXt5ie`gGl~Nx+>re!(T8`X3&e{{bbq4px{_cTr@rCO`WUw%c&6ro z`vhgVUA|}LA%eoKTyl4dB>46Oe4lBHcwL%y`Jkp+E3w%A9mfrOMqZSOb-Qh5{U7m( zAN(kk`^6sJ`Rn80BHy%#y^-6%W$-Sa4IV5Kbuj$(iU0IUd)8-y@B)v!4e-D zuack&>y3Hz&z;oA?t;w;TTaWvy*;1Wgx)anI^gHMUwRui?3>J0)^6zn#j*k}^|bp( zNv&C{o-nHb^+vTsm&Azr%QJJY=k8ahUxylh)iJc0{9c-$lcGH}if(@wRfc#OZq}0N zxz$S4JXzz~VsV7gUrk#ivuO89`nba*A^D=-IGTRsXaaHW%{R(i!#GeJ$93j``!?_w zy5YnuoCq^1TejLQVB~e6&msX{NtvpcR6VFa_WxdVM2Fk`uNL(v@7Jv-;c@j6pFqHXURk_ z#LL%f=VjYPt%Oxf?rFh0|NVZ59C^izPrdxK@3UB2e}m&P(>A!4)ABHKFagr%6VY$J zGV0@Izx(^@cOiY;%jcrXIdk;-!dM?;w!BJtm_;>?xL$AoSj|mE((@3ngs3O6<9fJY zmZZS5T>`EkZ8M9vy=oz3ZM$<@<(4u~wcu`N9?EY}2DV+0u|jo#R3}QlvVAoS$(KE*KldGdaU8n`#bLFM;2B$BUUUW55h zRn8Gn9n@BtJ!!D>U7M@Rr)|J9NhSIA{$$u0`PTHL0%LwCQM;P|92ephDtvjm`K`r3 zoSOP{$G)PU?;FY2wR6)oz4GND@rbKHi188Fdt%}fpLoP8XYJ*EeyLnwQG7*%W11_l zobf#p_GBTHgjUH;$odcR73Cmo>Pz;kQz%i(QgX6H5w9!6;t4jkt;9*q*eJ?sl*c7` zk-UeWNx4P5$p1fg+|O>NXBy<}%;K`O*#_k^m(9_(Nrnfe@{c6$GxCxV=gi1Sr+q&Y zHBzJEeJasvEO0jFP-SHIzsLiYeKJL`-*%u6)0Y_dCyplO#0_QIdkyM###&dvAJVpaL0ibLimQL zX@&)DKLqxLwv!a*zy!;+Q?f>U&-PcgsDzug?Zl&<#kSUC8T|>OX4CrSOfUMllphbo ztN5Z(XKLeZBBd(zwZN@3VA5D8!t`z%j4_o@?Fgam|5Z>BH7$!V9-ZL0{iFu%dqYWH zNrx+=!iw~OGF`4@+$+>SNb-_Oc-bj4MFDEKH(F~t1i@X3rd#whyGg zdeUV=#w1!x;juAXigCSd6prNUl7SVo2x|vX7FFz#-^>^<6lNXw`f&W{`P#0Ejot2o z_X*+YxnE9{roohwbDcio+hMax8jrxr6nHtS#6X~lF`t%N7W86;5Q;~u`h3i7oJH#& z%Hh9u+LE!~>gX+*)=9pL;ne<5oA~UH!kZ6{w&SlLUh@L_`l^C>pk#dBOY>q^Fn=vl zXV)qUw~G`dDQ2{|kEF_j%4nLO@bmK?ekOmCdP6(@$&36S*R0?+;pd1KXU1+ZC0g94 zE@j%etjGTl&mmv<6$rn*D|0VCZJfhfSdXPn28qBLw6_>2Bg>IeO7U6a&aCR22 zwFC!azINlpDJh3fB3?Zj-algBuBN{6(mHp?&kE0f|s{)S#JW3SCe|eN54p7czD`8^t$*Xp6a3Frk?N| z1={~V&~|ol4kW1S=ig4aNsISt-Jz}zq4n|qc=UTI7)I5|g?13HgKzn`?_%Ud-6A#n zWwHYr_vh>A@tn}dE5xV;gvP8|izTKRz$5DS;RNe;&=+J~y*DfcSlWf{@~s$oz3iR5 zvpfayQd@Sm+*S3@yaseQ!s2R=!fVUdpYA?EyzDhHd!hq*f&GlmE`Qp7FyEF5yk0*^ z3zkZq&UDgK1L465Qv!~oJ{95K<98z8BY(eGQllW3vyLdfB<(AIk*2p0aA&2|axfok zVbP63yfT;rvuB>~AVe2UvWv*~r*kCjJ@ugFx%{laQF)oG@l5A;y3dem!N!Fiot>S+|RUrQKyrd)3&G6uSeRi zrP1XAr}`9NVMEkb-Z8(D8*C}4yqAsf4L{RMn*+b^U5XG8#gQi7`Kv~k|L#qkm z#om#=!7BOR5h=7tN0(bBwGwXQQGWXX&vEhG1{$x2tJP0;BfCuUx^L-audB5f zE|m^&oSGL59n~)S8Y6hcPj6lM8F-p$>16#TsW?gnD`a-?1U<>EgMN*tx8 z>k1)W5pM5KPQBGeWHB!`?VlgbppUOW z?rJ;}VamwMB!^>_cBCsT6`lOW*jW>pF6$_{D*XemUnWM*>z9*14$@v*wO-*vyq;)l zf2qIMMyt!yf4^JuANl+;CKGr;xGX=?9lThLx&%5z5|1x|L} zH%e7tyAahx%T^+iqfNCka}jt!kLJdsZp%O z6bBxtDAY5l4R?j!VEf7Y&HzZLGrtp-{d-=-N~RX(p?z+CJo>%zCWMSPlWQYNA`7bz z`lG%g(%%2X@$-rN92n*6x{3BW7QGA*I9VU-P-F|9V_$I=EKGq8s))a{A|tN}m-lRt z`GDdRLf&)jm8LBlmDic|?{DV4Sp-kg4yx@0)V2MlNZnljyUg8cLrWOPf!xN#+LtV9T_Evjz{Bwrn6V-S2faV!dkdvPE z{6jtBRp@iu@q%eP(SBi3!B9m~H4$;gB4!rY{! zr$-@bmA?V66yo*i$k^fzZ65e0ZITq2O52aGzm7<26_4?U{8+)5boTXi%K#M0&NU&~kGN~7r~V~M_$X|>Hd8SqrN zWM8kYEht;^%uClvffgpuj%6Pi{d>(_hh&wikbKQ3Qm9DQ6CKsw+f}-Kyg`wN68}ey zd;E^VdhX@TUSkn2_4ctFC%@)}^{-YfznSO;TmcbxqxIF{3h!o)#w}{lCSm1Ic~pu% z_p0LSZjZ=3W51U{?KQDwYrB8`KZU|$HKPd58o=x|OHI3@(&${cC`rouj=d++!^g5t z*Fbd@bY91k2K6IH#6f#>87)!|ACMfoVrPw`j@H)u6ZPq&^U&3Rlo>swciGJH_k|E>&yL>r37^GTU zen7CZ-;%PPp9$aQOvrVfxdUQU%Auqu6*kqRBu|KB=SEq_&bYaRHLiB07W$&jr zx}MVYksRR1Gx<3neI)-o)M=Kdx^yNa@YnU~w41}pm*To2cB!C0d#U*I1&qAx^rw1C zbRb@D-zqDe&8P9Q+0C9{kM>g`{rj0K7gk?SmxmVx>u$BO2SZBI{C5WZh!@90U5c;? zFHFpNccSf(8~B@ku9T$hKPNgj^H9YlRVa=WeYMd(h8TXow0UriIgvjF(D9{leD@T{ z_ysvD3dJWn<6dkl;&n`*P>5GClQ5eat)Aud zsxxg5Wiz2|@+I~1cN>B0{Q8{D`%;15maE%z38O#ZdyRdzU>D*=T)(zKJd3tI8uJx= zzJ~G=BrlfU82Ru;3b6I_MS<0v!ElF7sj=z{;x%yw^Gfmcyl{Btv8$&F-QfDnBry&h z4XAoDz_Ic2x^~NftQ6)X+g#?ygwsTFvNBP-V)p7PxIesj=mt`a!M195)WQCk z*O?Wm|2`f~d|tJP_?AGBrZ(lk z5{7Zf_f1~s*JE@vIL^*9<5LPx0MX zP`6Lt>eF%~xR&+wv&Fh3crYoe(Q_JOKCOE1g?G`KXn&Ju4v8&36di_nk@BT5|6BV@ zLpg}LQ1!6rP$2A6_I&f^8O;y<`B*l|fBge(ZrDHYwak;Yf8p7%Yue>PbHJ5bEwAk6 zQYd~fzGuTpw680{)qd=Ayf-oY@_sMs{ifP!e2#7@Bk z#wogIcGV<-y8-V}Z3o8qef8IsW)tL)eED2Fy=9fXB*=(Le%5`&=zm2Vn8Bm=Ne*^S zSt+~6_%Q4*ey<}sjuzMY`S2BrF^V+c0>Sb8zQ*HRA?GqjTC3D-$W&*|Kg;p=@?|e_ z=JU;+h?l}PmvxO_T8LNawf37CNchCU=lF*TpyRj z!I&RX4NluQ7D2=KnJ4AL1L5PHQ!m_KAiG5J z%FdKfy*1Gl?yX5YT%M%~N7t=AIG*({c*(H8T&~bZ2%kDO>mKcU>dE&rC*z~L?TuC< zgUyfs-0OevzvAo+9H*=x$iF9_fA_z5uk~$M#+Cv8jhfy0-;E)qvS(6aa1tzaoH3*8 z9b>*${?)u!Pi`Sz6=x+kT}T++eto)Hx1OJNzpXBFF+bHT5S}-fJuN$c{HlgJXX8^_ zxZvH(Q%_6zUE!LDO3#c(nvf^Z{%l;eJWw>G%qE_7{cRq@vl+58#CwA~k+eQL>I<{F z^Vg6c#vJ8Us!)V@UGbi4B~;r=1g@Y?x>WiPyvQMcJpY6J+F3VFp(lipb?rO(T_gjl zj$eLWvCIS>`1!as9ZQ17(<>?;Ok|X=_L(d96Bn*?)x46KRbM{Hn>t=)OERL>69~txJzf6m6 zmw!DwbY#jOoZR;m*?Wb<${%7M(6$6QpXE~2#F&5lb%cN5_45ySUiV$yXPge6C-1Tg z$QgkqlkJ-C>9p^oh@ZV()`wBPSSFOWT-rhV{?(uP+NOTn)hAmdM%Bkpe1(@E^~l0r zeWv&K&If?>(Vdd7;AEUceK9AhPw4; zH6dgj9IZY&f)YT)9VCX|e!rLbz98+_sf|QKV1?=G4;Kguk6kCPg*NT|I0bhOBKfik z`&?T-uAS&}RotNy_YeNW{xfPDbw|&4ba7G^bLi9%$=`hVKNY0IeNEN2m$dKO<_Y!> zK1Q=&8EekmJiLW*ecYHVW7}+oC$|eyInV09>j9R}Uxv_4xAm?h?{} zr|Dl!$I3Fl1N7?*b3;F=`xyT;$)AS4f@j_u6z~2JB^9kg&*OBbd)Ua!-48BG`X#JN|JEBd=2;-`^FtB3>Ks z)Sj7SMzdd%UtVZlLGvJaO=j`9HdkF9%p(Gr;{*bMS|z&qWeKuh_r4vU-JQn;A`Y63 zMTV|m)|njV8>$K8CUfgG4$rEFP4uMsb{SFy(MqV=$-B;Q;x`0XWq-Sg2 zX+ZP*glnh1{=NNLq}e}xKo{xbvsoe$S5q4aqk$9+LG{1q1w0zD1xwPQW7R6wf*0$+ z>BNBvMzsAX=9||g^rh)A^utueI8R@`5ZSM#(^+1cl@HmAkq_kGJ9M9C_FN+e`@EP; zX7C5V*Kb9R+Y*ue@^C$N|BWRlSX+kfNa@@U>CB%Gt1Z$1-VV0P?b(0NOXR`2*ARs4 zz2f{!a~rO|CNh&J8^6i>hkVD#w@+V381+$KyJr5t`84o3)5fEAXDtZcN{Cc^7!Q(7 z^<4^+bQyRJh~1p4b&U3X3P0^Vds0(n;}82f@`8w0^!&7AeVVe+kxWtIuRI9sEfTCR zBoMDX1Frz)TYfz|bhNd+pY2s% z!q3kAtNzEJQVF z`>y}KA79`AXM35JKAqPF=KekxPECe+k@lIua;qjNNQ!4{z9~chyI~^aI#2TX{g^yYp=?~xQQ{+u zcp0s84A|T7is;%EQa9<=-}6d-c9*63elplCGK|%_vIg!fTX;;1DGn|jkY17)&&aD_ z`Im5^I#kaTcc-^7Y2)bnsKE3%x0p``JU{Un@|F8S`HqioQ!5nd9Igzk;GWXW4v_=Z_r>(B|+%f)w^n@ z5)E

o)Yr1B+bk>*q2;AFugY z9`(9QY*GBWKx0}e6jCCRG%MDDWu?oFK!rHank_P~bO)pT;$IT{&C?pm*A}%j$9qva zKrMGt?z$C6zaA-HU9YzuR2z^12c}~2H9P&lK`2Q=&Kk+rwn6s~h0SBZ-SuGqzHDc> zbm{P+)kW&ibC)&PQ{(UT@simrdFc`)Uo2M@zu&#xOa$oXeO{HuxSti0%%R^%xgEjl z=O4ISD|*@&k^%+(aozEWYa#K@_j%z4aj@aj%=>Uai$NbZPn~ppyB(6R=sEQcLPn#@ zSJ@3Gqt%ZVK08)!Pbr5U*qU?oXGoHxb#6VzsHMC+KqbU*pjRiM9*v+LIx)W#>1|rK>^d z=&sg#60x8u)D~BQ|aX{Ya>t{4rRJT!(;1uVhsPywu6_${+ibg?Q2t;ViQ0l!_unus4no@ zeZSLSd=e5pCg+yK6AZj&Z%m#pN$V#d*GGw1XQ|Dn)fwhR>Z4vr{reOZ33z906gPfjG4f6lNct_ z@6{GsH92(+s{5oy^MiR}Cvi&asVPu}{aFU$X4$QcmP36Ils>z@R|O_?ew5<4 z!uBn8KaXit30+r(ugoBPn8oH&DB{(y{*L$r`E0^!&GX%q_l&$KO=8_^{WqX-X+sU! zF*2SL=V9})ExwxQncQPW)lY&u7E^TIHyeVe(Il=+sR$5F5t+E&}n= z9eZlrRz8hU3lju-5(Bg`(-?Tfq11GsDzF=%>vP>$=~maIKz=i1Dn@s z02H@mEbo3jA1K8-23p-Sew)WnNOU~Q$lkhN+S@J~6ed-!)GbtF(8n8l@*?t0bm{VS zev0b3V;Q6S_e>6#%!}BhA?`>GkAtZnFe&RuK9xYcl*Dr~_sg@u$z}V#_1tuVimj#9 z)7=2V7g70B)#pRplkgC|`>1ah5gGc8)SF+=4jp~qdsN96jP!A8{a$K~;Ugm6ZG*Bz z!GG7s2JPaVzW2(AP2y2cG^&z7gL9F+R>o@RJX!cYaC|hhaItK=tID`O-uw1Oe~<<7 z;%5$~Q|E$>ZD|HInLYOK&N0oQP{^U{L4@#*; z*AbIyr)Ae({_niV@%8V=L_4G=fsAI_w!qpou*c|J--PNYFx_i2Z@B>@FRv=*b-s#- zmw=p*i-W@Gyc`aFo7^KV1*|W5%vx^v!oZY=@#C8jukqC#n}vc|p+h^)&|;%AJh?Nl zcVZ|_AMG4gnC+E;bLA($B_F26O@BQ4y}I6Ses|Xf+5h6o4t~iU4+vt`!4Ix;X!W43 z>Jkqx@x%6vMg@;^NFUR-+eZOc(7`myjs(E5J6o)FfaTY6GEl_*X+t8mI# z7wS2KLpF}sKj%r--kG)ePl*LVh1H&u7~}WU#M7qQFZR>N{he1s`QDWK%byWP2PdDJ zb0`_~U#(+GDBS~UETWP`R}8TGFTeMhmC@b{xGeYVvqt`e^{ELjr(dJ-3UI$26GQd| zBmX|@L=4}ac$!|lOrBh@i?%*Kjq6&o1NjpEG+)7+2r$37ttWZ!G_+ z2L0X8Q@p}JMl%`=<6?6@>;?vVpFgoYU$_|Qqx9kVf^1ht*GEFk+ivxfd9dHz#C@!a zA29`*91wk{%Wat!m`;8B6n51z+AZb^^#aRiz@3OXGFKWX|GEBlZss z*tGkjtsyb|dYV16;f=j-Lh37pLTy&qV!fCLn2&w5o|=#JQA8s0Ly}=VF`;+L(K)e< z{sg6X(QfZuq~A;0N#svZO2av9S4R@21#b?r#U}%EUHDiG?LJV8l>>6ujn2zef9Cd- zan7*v&3-;+7l2cjs6HvP7C`ZxYu#h*!x0}6^p9t9UL#HlANpMNDxbJolIK`FVt=N# z*+s2~^_~%_JJqxeC{57u*3#T{6Q z(Z^6w*7HC*^W3a#62_&aJe7SQ4KwRdUY(<(M^3U@i&yw>ZkI}QOwCCbmj!3>d zPf0$SW>rgcNJN!#cre;8^va>%$Z-)b^8dHjDoX5HmJEI3TeZv|IKVi{(mngA(a^gt zB`1ouPsH!??{_4rO|mADLZ*KHqV3a+f+AgbMb{k`zUF9$01(lcEn$5E@Ou&W7e(l z=x~CKOTD&_^90ZoTwR`iUmApkkI1{(qjiLMZ=*4ucMn0%iz>K^{o{-Rf-+UztM543 z*G-|gwtW4vG!F5a(ri{u)qh4%sP&i4nbCPVFRoi=)$F6&FY`6pU0-wV68wDx;r9^n za?sJ$v7@cSZ!4}P%h)mUiVT|>K2r?oqw*@sKIeO*^V(A}-)iu-By7!TZXWmE7wQ(R znmUe_Un4)yE6L8&`B#kttElWFPTe&5(pkfLZmk-;lSwaI;X#XQ1RQ#uBtIffJS}Eb zlg@}Zf|P!Xp9`6nrcjEVHgTCwUIK;^GNR2{h)dH;olQ>H9udtIA9uagM(63ggx!v> z?U_RV9LUdu^uMU72c%ah#1VTB-?d)Xjd(e>RVMKT#DI$Q_cFg6#`~ecdA3pEbx6LP zIXO7oN@?-0I9maQb`)nJ{V$67p;^@nq(E}PMYA?vKX^&yFf60(FG}*7E$x4$=Mg&` zQfqB1d*TF2mt9w;NvpwZ=?0wxC!~Nn*EWNTisXxY9XX$0kI|8k^1e?jXAzgJXG-^4 z)juXgR(MM_-9zW;ywFQ&zcK1#-2Sl#5^E|5ZviL2udFF_`_*7#(d|LYm*<_9T0dbA zqkQGbHK$)?MZ8{o+80vmJ$k(G@ky1E{=JegzUIR2`OEx(y>L1)kci@iUlMu@#kt2q z_??it8Z9T#ZL8F;_a~s`k)Cyz24kJnFVmvW>({eGM>BS{w;2hbb@1AUvVscf$HdKd zM{n>M{C8gD6`Y@xJGB@g`C`vcOFDZl3Z_qSI2UH5!eGB5MD}*R3(=r z$GNQ-Z@e0(8i7=l_K*hZ> z=J)8Pxh&HFUpN=WH79Nt;`P+Cby1=>J6vX|2u=%j2Ci6pp}lqh z%<|L|HjWG694pV4509?W=V4xF^nPyx^2Z#Q=C6A!(nb_7&MsT(QcQG>S#WttxHfo9 z?>}Tbh~!9k+hggCiyH{)V|%SHBksdVUgEhcw2Na$-@ouy{>?)@lNyK|jd^l0@kk#v zbNzTOe~y7w^5@zY-(~bCE^MHHNK4ANhSdtr}WTt@5|7%`9N0!XLW6L)Y@!I%#{j@DC(Qu)w zHMQ&tqdxNPmS+CKiS&{GH7?z_w zx%eF`oVjh*_U$9hez{*SI{#Q1PU#+y&VL{UMDt3v+{vgPjcPcPt`WY;m!w+H4ncbfn5y1wb*|D6IytG_ z8-$R2J-=7v9k-Sh%qJXppLpI0+LL0PI9QfI?Jcu4Q9ge!U;a@uI$Vtqms`^=iS*xn zNZ9+Y77Lh+&eM7QBUAGezH5-e}Y1lx+mnil%VeEuUy`oMcjWPa!2^Q z4&-E%oq04O9$g=_vSjJz7sRI*T4k#;(KIf>XVm`j5#i*R z89Q}MD&2nVd$Bqx@O}&!b?RuY-oq$gmaE5KbRdxZI&?GTV*d#(D0Y^#@PCB-O48ob z?i7?{Qs#lXRDQn!=Rw+je7h9cZX@|}Wty~jP9!TxNIjS{hZfICa9L@wS&mjur^04? zl=?639~%<+ZH-KS@*Ji*^fNLb@a4BsBuBb_Nq#Gj)DsKSdm}15P~M2-gG65LfDv*wgF`M6yIobjT8@W)dCq@yvXn_U$`-yy`kJ zG>f5mm~*%2hHN75fBL<6s#3k(%$tZG|2_0W&$&)8Ko#mvu-nD8BRRU$RnFGEyMbu@ z=rSj(62&b@UgvDOUY$ut;|`C9AJrl0~8oF=%aj$^yA8~yluE)gntGIrZ-k*lo zN4!4b^|9Xz*WVAv4{*N%?(e|;9k?Hc?61ZBL&J6v-#_C1v>`tUua9_r#OotoAMyH# z`;l?~Htu)F{d~jzG2S0wm;>(T!~0Ln!u@=Bf0G{p1H3-s^%1X+xZiyu;C_W+J;42Z zxStRA^Pzq|gB8py6My&$6v`}0@5T4}Jg)4Np?~8j6Dd13TW{Su-`>Gx429>%$)Ud- z`p=<9sQ$W@1}0-D&J;H>D{Du~Jz`6Q#gw-yib)BJZQZqJ@18C8n|JN8vK~6Gy~WPa zns(mNW{ZP0?O1%FqU>TRVL3$!DPa%czj}-}KRf2#K1n&y^tN#4`{)ZxU3S}_?L7{S znkmuQDFBoY?LOaq(AJ*J2cCK{S=SIh-uzHW-fh!s9`ou6>Zy1qj^6Xc_)X6ZcRa8F z`{c#mlAl89Tcq3()sH#O-b$2*`BbO(If04L_@45+1`xNDawF7j1Mqaactlpz5{4|* zO>(ac=&xfxJ61o#MjD8kkR-OlIkbJ%6H|`oWkS7C^3Jv!yCJyHd7K38tE+s}r~8l5 zbOt$rqP0a_cJ$ZRWlJZB?4*6gh2T@0?-hN7yy`^zht0u2Fuf>vx#1Cg9>N)Nrm?lF zL7_HQ&plTTuEa_W771*I#FU4b6OXN>kB4~;ee(TYnAb1|EMHh3vHimKe#jaO*+slQ zVt)d!kJ$gh{wntG@%<3Kf5i8X_SNW8!*D zT#t$CF^A(NxE>SNW8!*DT#t$CF>yU6uE)gnn7H0#*e}BQG@MVv`wij!necvAc>gcF z{}<{4m`0)OPc)v%yKPKKE6YnRA_g}^P)edHv zJ`Ly7a6S#^({Mfw=hKG$BAibf=79UtaDN)^Ps9CbL)HNIZ{z-LQV#I?IIItNeZ=b{ zULW!LNZLi*zm5C1asM{%-^TshxPKe>Z{z-L+`on&l zt~bH;Cb-@N*PGya6I^dH>=)sD8qTNT{%zd9jr+H8|2FR5#{Jv4e|y*;!2R2}e;fC2 z`j~(aJa6S#^({Mfw=hJXL4d>HvJ`Ly7hW$gF zPaEcd`_piL8tzZS{b}U=BJNMa{b{&A4fm(v{xsa5hWpcSe;V#j!~JQvKMnV%;r_JY zcnQv@;d~m-r{R1W&Zpsg8qTNTd>YQD;e1*V7mgR=cp;7#;&>sB7vgv!ju+y1A&wW~ zcp;7#@_qD$r7k#Lh~tGgUWns`I9`b3g*aY_O}mXf+(#w+`=Li1#nV`xoMVWL#f} zYQD4Os)6PwV%>>*KH-;Pr8s1708T`iR#@ygm+D1H3-s^^ueVygm-= z1708T`iR#@ygrh45!YkldQ4o8iR&?OJtnTl#PyiC9uwDN;(AP6kBRFsaXsd6yad-{ z;(AP6kBRFsaXluk$HeuRxE>SNn+*F!IG={|X>5mc2*MBN({Mfw=hJXLZP*^-^%1X+ zct2#kA2Qw#Io=89({Mfw=hJXL4d>HvJ`Ly7hW%okPs8~%oKO4zKA*;v*{aj^Bm_Jp zuYc`)d>R@xQ=+p|{=5H0v5(L4SEqt~@?!7*lJAfyJ^DmA#~KV-s+;6q{r_k0dB%zG z*Ghx}!Stfw<%UPZkX89F^{a~4k{ZmNI5<7N={WQ%f0mv+KH{(C%f4}z zf3qMJR`M&@#LoTqyxM~vL_LW6vwRsd`%dsZj(D~8Uf~@Z`Db1Y$74jU>=TB_uV0#a zCI8GTD~9(~e8d(8UP3VQnWwcHNEdW32_7E`eb#OquNVA1FRrcTdG|P|5PIa`{!+1j z&+Go0n)5tWf0nPlr4x1Z&mvwdg`vkDTgLoRzDQoxKB9@j<|qHCkA+@F(zhxZd0m?M z!t!H;I;@D>(JNFL3O?1>tv^)#Jumk+Pkmh-slc~gdwKqef6uGE<@83*9b5mXkG8ip zHqPCfMb}5!FNM!UyQBZeEAGzg;yYJFp=hD~VUfU7f8J z^sm@_J*E@_T({q6zMk;+ypqNJbF}w+K#cgI(5rL)Jukx^uXFKZE&pi0<~DPE&`?D3 zwP&?7;UskOU(46@PaK~zXZ)Ghjm1K1R_`=p;N@FfyMCMf66l(^X0rUJ5Qthj-BEtq z-}4GSc0JUe#%rD+QNDKWzvp$jvr;2a>d(9)XTH>ap@Mj|fyrybgMXGUV*J~Rq9P_B z4BI0n?)x(@r(_7}4P~@n0U>&y#8j0*q%k$Ha_Vsis+IrZIR5W>%@c8R6W8znJ2qop z>z?3$ZNDPAgVN;7{_IaY9=9vexew{%o0$IIG~3gElrNH(6;q1horht6)JIGEtJZlN z7c-^tcc`!#5O6OiMUFG{{`MTP)4?6M-;brK<)?5ymOk-hhfP(~C&6p>_=y;4chpvY(m zDI*D~e%JSVyj{=F(er(MpZ%`S>rcmV9dUcz&+~CU9?x+VB||*RK|?q03ncG{DE;1# zM4k+UZ>+JGy!-z>U#9!-1n-o>?)T^4H6D`~Dkk)Mq2sB2ry(6XU;ptsbjmvpj$-eJ z_WWq-r=vAQ^)DIQGxKsea^Umv>XnzSsgU6`DcjmkdcNe|P@G7=5eyX8&sv{`{Coe( zOXbiC@Wif1v1b%NI2Vw5eB3D+^)@FP)2o$sLBijK1wL>MeR3?s&X@k74tbW^(-QwVXA?**#a0=WuXk+r7CEbhPW1qk2i(tqO+97u+;NQI-y|3G>sDeGO-g)+- zNR1h(M;8kX{f~Xv{hmx}YPx`73irJ#d)D6>+v`iTjYg(3dcL3AtlY$wBn5IQO0}x1 zDG>MOl+Do3|Ap5^#=s%95csuBw_Ppv-@SCk-6>+DP+)5_e%YqF18nL%W~4S|KcU2LUI0?F+gQX_2E>CG&u2@k|p67>H8tSe5X1e z=@6*Ae^-R7>fgQSnehh=?GBjp*QSFy21 zp6~B$dlce+(FRqIar%J{s#EeHA;IDqVx9`0&Ev?Wa!L34FeO~6b~glO^s6{}zxy09z03#C?>OIpcoi)l)mrQQhyL|m@#^6n2jKGq*;}2<*j^*oylX=f(EV$i zH=dbXK^i0qigbRqrovCZGZ)Rsb4Xqv`x+a?c%Oy~ z67d-1L-yOc{Z*-azyD#s{aoI2Q}jmI?5eC3-;aI%HC*2H!v7RHU#;ibuiPvahYxes zovwVTAZm6PbJNI||d%jK-#kH46IbnJ^hqD;eJVNqyzdEB7 z=YpNDu>W`|+icr-U@Nv)zG_D#CC7is7xU9|N9@F)noaVWg>4E8V$t&sU9JkMw)9NWjI@9$~+4ra+taFXsykq~}XhnWC*oJqWDdB>Ar;{QJDx@hCYi zxf{D4JBOgM*lDc`UB&>Z&fUMB=1x27m!d+oBhFJUJ5U*?Haha0LEA`s0>)1^g~ z0@F`A7$T=g&sVV~oh4auApGo&I&$mezk4~=KC}#ZgPkuk*`0Cslo2n9Vr|Ek;S9`q zbv<9y!BVq_USsbczpz}k)@3t`772L}IBttOc|8U0jcd%dx{&TwB%mbnt0xd>zm0PD3;w&8riPC`PUgCiBn``=z3&p^}8Ofp#(2y zBwXXNQo-Z-nCgX5()*YHmUP!Y3qcT@xNqjwt$+6lZrPI@)9r$}KHmFID--ey@nX$$ zB)2Zao>x=81=Qq)9)!cZmWvlVu;;HGoDaWod_u2}t%oN^Jzq({V}`_)3Y+r~;4=Pa zo{V%a)q^>n6pevk7;xw`7wf-!ZJ_j=J}!sd@98!_>?l8mcukl(ZW`sp?qBO(*FL;+ zuo%UzN5>8>-aWg~=l5|ooI%tAM`10_p4G%G1%$sESL;!e?qzu|Y(VfvAnfJ$;4S3- zcdt($E?y~5!LG-uA0pqLl^|XV^j{TzNMNszWZO8?k8{1l&7Zt$XUKuQKF(HC{@C~p z-M`qL6iacAN&wGM*@I**X~3M-*poC!djGQP=Q#hdHV_g^Sbt3I`ggD6=fBeJWX8_d zYxdnwy4?`3$ziUI-HO;=VXT$(m)YiVuBz9L;I?6V*;2OI51OLADqq`Z8Lx=Iqu%Tf zks@iZIq%xS%vI98W_(=+WG)1PY>ggdH>i_ zno4o{bs*em-m$5U?%(q@=;*;oS%#f2lh?_?4<1vLviN6=Szuv z{O}+-wwIgG#fWb462f^j=gh7fvE|tPe%9QZ(aRYTpT{zQ zTP$$uc$EStHJd!%b&&3rD-iLE+dBx7FDN{CD*5mEI{Qq5HdXKp<~*@SgY|USH^ggv zs5bIQ0d~IFis+U!$PYlL?V{lIa_oHlRx27@QAN*RWVz%@^v-hN;wQ3y@MQ|H%)Doy zzDc?lFXjGIGb_O$5UIwi;`i@f!sZ1o*H5`)dX2a6j?Q}$^*Hy?CI3J^4r*BSIdrqmfn3{BwGDSk_i|*^Iq+;Q z1b|0yHX-@ny^OiPXc|tsVR|u@CO*iML%c#bFTDD30y|&pUP2=@xJ*85FESN%25C)n zzErr&_uM5v2Chf>$Z$3e!H4?lq`&=w86PP}r*Id~x5@zk5N5@nn{=2d3Al z_S^RZm5}Rc$-Vj1M^drRzi=?k;x)AsVw&#-GOSpeq#5Quoi-L<{L4pB0 zU+Z2EqO{a^X<&O9i@Y7&#fA1dMq%;eYKQ_b7TnG(s!xIs=YE7UwUh2uqQCK$Y)B|n zPf!(2&i=dC#8stDO)1!3nwJ{t&k7=5>LmvRX%_z>U+-7#>&G?#t8Y}j87sDzw;$)> z%FF2e<6`ZD@pBK3L1?hvxzxAGkaRoI-u7Sy$@7Hrxvvg>RAKP#d~?<#x&I&MtGxQ< zC(e6f)}!?x^WKX=h?mL7sFJO8*!N?G-Q=WXH{A+qWtU#aNtJ_vq z?JL8mLyFOFBAE@}v!#q@fjm-MaJ z81V`zF@4?q1-l;Cy(DQ%anoklUQuO52a2oE^=PSY=wy|q2;cdn-rgloh9?}t@6U{r z?j>beCY8Q93=ZDj$b3Kg-@T-0uAcKdhwVjavsHSj81ZuP-mGM>bP;o&aCazSNsD5K zF&FwV_IYeCJ@V65>CO(Q>yI0mdkmk(DnP#F5pK>+DWGqZ7gn@`^m;V0b8Uc36VRrumI{N!?mr#3f;$7m}r42vvJDk5d_Gr0wR#zMdv!+jf%uzw94X+UoPDn+>6U zK1cucXfq`DLnHsK%Ov-EuK+zok3UhMkYFJmz}WrI=MzyXpJ@wzd13Z1E?O?VH;S(a z`^PR@$#QDH`wrKo}mKR%8Bl(qxXTm zQ57GPpKB97M}3P*v(y23|Agpm!IdU`#4Dcs*YLqF*j^62XRqJ%6M=?D+};`Qu)WyA z-0I90&|Y?|QE#_+7{N+0WA1*>{46W8rmkCY!=Iaj)=+L&=@_ zh?k%`ryDND2a0-1QiswN2kT-EX82kBnL^^v>O;$z@+cvJ+_3URNPEyCL( zuWCbBffV279S(5lQ<9P0gb(;QZGEx77VWikHfeHk#0<{cl;#J{4-)FpF+s0j_8PXA z(FV_;FBzf`wCCYnU1w~smNWZ|J_e)nb>n)giEp(jYRDpp5j*Q>XQ+# z#k2L1WMf^J{cGLJ)yn7It&5j1``6+YffGgJ=zJY`%+1d{=LnZ(m@3*GTj1UM^F=jB zu8`dCHQR4G>tBzBSN5CQ=Xi~5BVIp548E{O`2$;U%Fshj zw3l-5)-r~d25@-AFXSyP(!XL4@%H=-!uHCie5rd?aT|C}D$g-_V0*>IYD<59i}u=^ zaqHWgSUZRrvh!X|Y5}_Lm031z(!K64)P`Pt7YjEM`jyjm5zm*O@SP>vqj;CEQq;MV zGC=O4lT6MZ2xo=DcIB&~z1}%!zp7rsuj6YOKU1|3FC&}VwUvF?UTmU$+{ZokK*hl_ z@w7>7udA)(x<4*Nq58dDO%M}#o-ITQQZT;>ZGliumHMd-qU{CKK`kH?&(nDIY(MxuX()b_9|%<1zAenZ@jw5+|1F2E&lBI+q~pZz7yx%d zQkt9u;uW*@ZSO8;Y_G7<*M99PTR>TiF^;AP+behCE}i@cI$yMhO2e`Pt%0@k`E>Ad z3#|5Bidgzz%&XTOt}D6UkAZso+VYWd;$AZ1lBX*y@h(r(1JtrLp=IlxQI>Mf8!RS{SHh4ky&^jRZJ2JKN$nh<%WP=>fykKVDEpo1Df z_G@iQTEQL?-t5ixdEo<9Nh5psRq#LmD_@0=)t(Cqh0-Zq~Fi1#kQGFbz-iM zVXNGxZz}df(WT!CsrdV!u)L0+-_mtZJ{(n#w^r`#zZGK*FX!SLa;sZF$omv7ZkF_U zf?Zi=2Zus5C_eZgoH@}&_`CmpKfubR(H}SbG@v#2`$f}69sGT`FRcD3`@x_`bf7^F z+Kchexd*laC&6X!A^iu1NWR>qZ(FdJV9yizzAguhoS4C`&;PMTG4}q^`CO*~&U9J%Cn7u12Q=-Lqv zH(R(fo`v(j><2>!_8)s%h5z|q`7#gG4uaK_aJ=qVvy zM?#mEE{I^)hE%WQ2EgU56nyo5BZSp z^*M!;`L#n71onOSLZL+5EBs#A6O%iL7hOw{uY9i`6juFdDS3|e3JlWYmK{6^5p_x{ zMhu9T_2;hHuJI1c{M=Va zo3$-P7e2BZhb5aK{a&Hsl2Pi*HiB1TwgN@VE=R)W*T27Z-D|V7%MrD1?CaZ!+D$RI zV01l758X_i#aTiXUAyo@{CUpr@-Ht-D@phAixIy$zaZPux!RLwnigXk{+7>Vom$f~@>!#7n7? z{8&K}Qjf&)l^+|nk>SxUsO!Eiq4g8H9=992^O2oKua9NdPQ8D0!~oc)R1NQOw}NH3 zFV(ak>0a-|EQVWiBjMF``9#C*#J#lptAu;<)d*f9zdSa!*Wu46ypO@RW_<4cszM_q~c^l=ZlS3 z*7C{ilc1uh#8Y^>6>dD4t|}=Z-RsTkr@61IBH_7P#}@?^;$A_ScNqN!@GigS<|jiW zU}U2-(~UeID4VqZ9`p_EbtYf9b;F_#oU)o0W%(cdE2kssEiElFPY|oeFwX~C*>tP8 zV>jAZH1PKaW6k%n4}_h@SJ3&oe!HDDhvp>cy=&8AH*bZKKj*iy-zWY4LUZ$pu^nBJ zu+b~*nzICPul@X%EZ+Ty*NN2bE9a}c;jG9~e&koQmtpS4LWwaQsClM$`$;L1FKtSH zE`?d_e64#4j}BgV>6L=%WoK2$(8_}L8hF86x8s>UY&ogTFR$4OW^-5Hgb$MLWw(E8 zeAekm;EP+E?vx_#6;M*e;}WU{`R*Di=N0hphl&{&v)beXDcP#Np*d);fxFbw0?E2y zvAZ=>=6{^8T6VmP;lTZSzH(F4D({zKdx`kozh%yZu18Sc(i?>L=VvK)lanxaZwX!kvuY~LY>F%-A4@=ly)=JJ5J+)}B z(ZdU@I-^d2E2S#iE7}Uxq3=0;Hk0n<{j9j|y>}Eurc@bmxD)pZ^u9*sTc`#&+VO_X z`{Y5ad~9CpnK$sLr@wFhFYDv=t6JN8Gx7C^<2`xcI^uO)ik_m32H8In>tAHsS|&Ch zxP$XI5KIpl!1g+)%n{%0iuU3Qj;V9mYYMwdp3ZN2)B;~Oe9-WeBE24yD}(>cMMpv< z?Y2ET1&DjGu+r?@J+4ad`fj%&sz%QnUVeQhUAwA^nkPDq3*DEcb%0@+cPo<@;-$GZ z6?#G%J74QwLBdz=*!Z5u>|f?S47$B#=>F9n$9XKlzyRc&$gKRYw?ON(VCc+N((~2k zW30BhI1((WV=mueCGI8JhqD)LL%b?)ei$@M^#=0R;7iYk(O%M*AHO#3(1x4L($@r? z5HD`>Q=3L~|J`fcEvJ|96t-6~@6zJVHnf-a+aO(EW+U)y)m@4hZh<4W;^eFjlI~^D zy30Z~KN6}EugWQN5ck@-GR3SOiFk=>$MR;g`hZC7*PmyvpuKK|^wH<6X+wX0$yX~| z#EVDa2~CC5zkA69QNByU-{+1sPh2qEt-AdS+G{eBVomgrDWtb)Ouc;30#6=Qg@8Kg zUeCy08IG|;f>Ga83CrtFLS6lRo0hBIepesE=j+qhWOO|{dVS>D@XD0; zy*AK?lT9+~AYQvmsTT6&|K001=liRT%-CKfetb`wr_o;X8-CWFEi?xq7YW-Tu2!Jl zdj7Ik1nFK^vfTEB*hhl8u}9I-4aD>H{nsh_%YRg1*ZU7N6s0zhLGi^Qq0k3TZ{6lR zu?y{GS5?omB&q{B^eNM2FA=XbNpgYTD@ebmq5Q^si{6Rw|7rbvV!eM|zblvfMhCm! zvkGXcZPi5gd&jqhCxegb1JssZ`pn$|UCl1h?_ZPN?{j^}&hF2O0PD#`^`=DP`Qjg@ z7KuBlLhv%>-*#oT-wUX<+r!6QR0x0X`t>2dmcG4W=Gt(M?lAN1Hv@$G0XC+l_{4-F z`NHxF(M~(Hrk{*ikNF`&w^UTn=Q%%hj+Yuont|8S>h<0D^NHmbd&3?Jke;tn_7%I{ z&5@vHD}F@l8gVa^_N5FPf5a=EJ6bEX#~T8Z!vrsAGKEb0FmN?|M{ue7-=*6uTZf6v?SO_iOJf)`dDLP>o5HI zDs9YibUNu?QGWD+Ud9pNd1#ydB8eIWgp4shqX-3bvvyzFh+cTnF&>apA{pWGVSK58{>R z@lpTDOD`Z(|GeGy9NMdQw^jdGoDSUE{x$n`65=%`*s@Lg8{##u>?G6s&5o$oQ6ahs zIX>)seRfju6FU}yI$s?dcMMl`HiQ=u{l5zE_fHfY*`JvIzgQnrJ4&sLQ;Ztv6`F^o59;%yz_U z|M}-Mr|$lP*Rk!tb}G%|?5KENUZF_Etj8i8|3a)~Fe+cx7tMCb_ML>WNhg{`>lTm^ z+SjGE-xJN(f4pwoV$T!z!OmB&!`kMPWkD#f zvjUrM`f8j4a`L815!zO$%`usn$R+)JHP4rp{=@AEn4pd@RkkPYb&!Ai*tZxJ5GrrB zJ?&%lr zoyp*m@pyy4x=LT->;?}|=iiFRtd$BiB`0gE0g;uAM z`KvN72>Wg2%yj+k!^r&;1<|=tk^AuX8~x{c@A`bd?j<0TwQ@ikdwm=M*^jy65=W8;!dg9tQ?ERw;zio}rm$N9ZFK6D0rR#`*h3$#lf)n@P zT*fa-t>ud(Ur)0&oVXy+=?809l~X1B>j}U2-(EE=+v?g^5HDw!VuS7^d*IA4YtW?u z!sph#X2mx>{(J%;g{yj1x~q@iWo^0HQ=FxR;AO1FW`5wV8R7GPdp(}n9ebxB5_6t- zH)>s(>FbH|Qc*A9)_AJ|A)lX?PBY$vcdAw_f4WKc+B(ub5T_jgyKYjMTuW^x{N8_i zm4B~%mQ{m)uW;|DGQ~sYkTI{<#-r>A3j%RmVqIvj>v=z!)@Bu;HtNC9Z$>0v0=dVs zU0XYmt-5Dq1He+IuKrRuaW5~cN$@WRV7t4iFmFf|*tlQO-iUU9 zECTF0s_cY>ln3rPT^-{#kTu`pDEWsS*6(-)i<=y=SoY(f| zFOM(BVR|*IxhbR-q5J)w2$g~}b9%tz`>frKyaC{`sq|s|`{YUezB$oAM)wfqKnOZ8 zI#poXLioM^&R5!6wzFP=0<_LmJycCqhpNlxSK9diiGB26*=hdUsZnp0Ajbt{-m51werFqW{!+;`zFvY(B}1e{Ou;OFU4V z`}83PNYXM%63s$;eRipt{>Y>NYCTTR-(E$$Bp1capFBkJl|z~Smfi%tKCXN1Ij+38 zWf(hO_m#}{4{M>lf|6a;SB*}>cZ%GzO6d(?9k+qn?*!@j`ndm~kT_)^G_&&<^+9dx?!06j5R&1`KDeO) z?$HI2%c+u{FAqQ0!y4WGzOA6bvfe`?^9S!!?viN$;Gu|5@jP{!9sHCDF0Ilcs>AVkL?U@eN8-Ha*f8n9>2u_t}O8ED@ti_DEgkJ zcG@0l(sJr6X3_KN=VF6LH(ddGd33H!-A25;>Mr&K+(hc}e#h2sw>)(JTK9^LotdIE z!uGnJE6I}TgU;7Fmi$(y;*`I?B7dhT}{fG*w9#CZ+zviBc3`bGrt zI(}JHH>eTazrsFG7jS00!cnIBKeJ21_IkIe{k&ri+Uuw-I9B%QL-6tTHqFQ(Kh#K((^U3v85`+FAz4d$tH#VAYPAk*Unm# z9Y(y2Ecc2z;J;7jXI1ppMOn0$W0&8FmFEC8Wb>_&|6^XQ&zR$|Dnt7H`Pm-LOY!LQ ziFGg27bbmlIP83B^0Ypj>PGK}O?hkCO~+CYWR*FC-JXfLT9I@(Gx0Ir=IBV9a@e7WVAd5S9` z`P%%7BktR2;`x$0=~w%e6MKDhSQyigK8D_JTPaeie0!)5POd*iBcT=UWwl6t-bs2r z(mWO8I86}*Qv(xL*K&w^(U*)=mXhQ9mqev#b0<4Ae7tjKX8tr3GGG4VxE<{!&wBDT znZ7){S+Mx|#uxD#>5ONjqeHw-1$c=D1pXJV-^Wh3mwdzxnw+}(k`CL;_j{vWR~I^8 zoh?n2Aem%`wH1Y2K zNGrm>yFP#6{^PahNLk;FUs0IzL?qpE!v-%8)P9@xnnzbmr#@sfoDqF^ss--1(ClZc zP9u5$SU!YN+mAHUB!OMR`q)PY?Yp8j_shy!FPx##W zyqXaDDN>+M8cuqCxqLVd@k*uN8swSMOz?_iF)BB5MeiTiz4Q+xWwZ&ZMd=(B16qd=#K*FV=Rq>yQ*W1F02V$&9zWjR5bESj*k0l6XO(Al(fP{0dM!Qtkv??9<~;g^f4<{L^?(-bSJJ)i z(#fz?1qHyR;HldUu_YXtiOrKkzp=dxu7uw1&qI4% z_N>dy{;UtTe)dOrO|-(rR~vfzR!Q$)zuo6Ws%nBE;b_I$ZD-{& zer*qDa|*M|KiEN(4~JFS7WDbV=%K`2=bp<$RL3g)qaTQuhuGy)reDjGI_B?T*=|}4mYJr74X@dczJj^^RyuA%Np#6tT9_e1v zIv;-r?hA$&-r7s?vc$bo4eDDN@aL;!k~#ZU#reU;Ld}-_mJN)i`AOS<<$(OzFHWaG`c zbm0(Jb<0v#D@esOB!B!tdOb!;UTsmj7YG-(TnNfMdY90*|L*q-H4QHUL-BR@A1_tu zNlLN))4;&}EJHm7U5_%?r2>+3WMTUF0|vp3h}XgopHne+kb1Ptatn+2g}&ag?nQB_ zpIzo`H0JtvU8{_1!(((kn(TbeBlJWU7&NNy@I7yZ9Nj}|EUTn@)ye96$;br3z^KiO z0UP38+rls1wt0nkJ!WoS8XItgteEjwfhn|CCCz4EhIv`oY%%=si#6g^=|%tWGyb~# z`aD5<){Wv}3%cLqZs~kFFua0eoowt18OOff!E=~fDd7ux{yNyT@2p^pE_@o8${Z?c z1*eWi>xYJ|KOa*FL1=X zaL-eVC=;u2!HFsGer;qlMrK$VS z-JlEfXGfU6H?@ITsiPTLB+w;HfiCY?{e$sjxam$ z`YmN7b;SWXvooIGu|Vg`P4t2Jk$zc-R=6k3w-wnBRS0ZW9sM8kJ;jHz-N{eU`ATsT zaCs`l0IC;^$PXrB=PTy|&!q-B^#0LCz^8i*UylV0@7;St+u>%(>8AYyq1LVSIUXWP9@SA(Cw4F0Hl*2{wW-jABBg*HI1zG_F+h4v~qb@u7OjChGunEO17K>Am^JJor!o%V!W6ZS&~0y#(YH7HzTlZ?}EyOD=_veDa|H#+<4jMVF>z0Ji|LwK8 zcJKl-6Sh}j!U++{0JK;5nKzy;jk>VGdic=NwhqvYZR*_9O1f9=oQ<08!vJ7qr2KSw zu95J2|2UqgOY3;OT>u!W`kCV@PXjZx6IUS{+Urt_;jx)?X|R0y^qXcG z;uYHFldH3e?1!jd*kud{q3;K1-N^DFpMN7b9d<5WZO5+1`PY4VGC^ps?LIj&qfNT- z(&Sx+!Hy32PVO?@e1!CQmF#FyYW=wY=xHZA^J|2-mrC92R82a5U;Q7ibN18rFYcX& zA59k(4ey}$+xs(mR1NUw{Flbs@{X4vURLzNS=Xi!FNxQwalOjO`~24PweIzU=KAG_ zRM`0%n>0>RxQX_n`0b})R-+3_9=@f;OYIO)nmf~NOM1RkEChU9z#o=-rW>|;68GAb zY~?MdEl2PobAOo`J7x`EHClfJ;lF?7?|S6kCHqNYM-l9NF)9tNS_q)$uM&p{38xBOFxXa8__(tTz7%%q-u6u; zd7j{jJYOQv?h9XY{_suv6Q95A3{P17tdu5rc^>suNUX91v&t9MdmiC`{(t{N-d|WM z?YA>jK^*c;3I&%Fk@JbFD4+Y?2@M15FPi&;+yVmB9 zKF?wEkdu9PQx{~q?{10v(+XGBnp5{RrjYFQbnwL3qlZ2qb$?kC#EE-_DdilU{wYcD zlC4+$+Mj6wDGTh?i94iFUeN5kyZE#ir1J1y;W_t|Fn?{Pe7+b^aF^h9Ux%6#kQ0HHqX~80$mvmur!0yoI*k%YjuhUl~oJ6u$(9#E~-aDS~ z`x`@GhFcxs@BX{r%ZA>HXQ&q^cujNbvZi|*!QqL$#k^1OKmRLVH>)KV#uJ3$@yYbT z3(GwOuf3PJ_Zg*E6TDoW&iegYM4or}J73)P-6jB z5Py-U9&|?(I+sSa?KhJ??r zdy)B!sntH2#GSa8qjq|CFs9d8R^__yoh~RZlQ;Em=8q`Cc$SvAk5n^6{^IpJZWlwc z7yV4;D5s<+3~MPYGICTCe(%4%&b5#~-6#TKL5AZrtlg5q|d&XURSjbMc6c;>oJt` z#$H}!B{*@k!Da{k{a;~~kIG-BknZ(t=k!p4jwejFd!@aTBkr|pCw2HJmpH*Ie`J!8 zN5c^EGTDx9mO(jM@{ zpp%_)-yOpL=fCqMP(QHu>MD}2!e5&2sj~DSQ`~vz*mktnr2qb!X&Zi^e{r(%3>#99 za%nDN8!C}{6n3HB*HfTR_;=UyweEFRFR`&#&JWYeZ_w}4-FACaJ=)YVy&V0c1VRDP zpQ8C&;cGy+y2&5X^A)3ff#xBfC&c~oUmRp2o-f7{!-jWr;-LTb{{GDSmY{5++})>d z1Y>-2`+GJXCH%eX^+?HJV4+nl2tmdj@%jBozTADjkEcu368cwRJl7s!4)pmdwJ@D` zp4%3nxfnbA9kAC&&Gf1*hO}W9c3y1n>2Hkx9q?5{|i#>u}84HJgZCk0!z*->$bhqrB)- z$V2uqsDXqn3(JQy`1AWZ+DlT(q_2;H6!8b#o$=p`v$_|5^g=veDbL3)oLxK$mQ5As z3cQ6NIP;0~#X|-#bi*>vxjGOpHJ0UhHAeKj zN_LG+kI{V|N8wQx6lRQ_FUR`*SNOQm>tlY7c-pqT%JALf!N+0z^H$#DqCP(Vi++z= z6_x(;f9`7;HFpj9LOfrZJY?#Z+{FoA8@JquY|=Gwf^5zkl8%@XnFH;;nM*-adZ(vI+mvde8L z$pCbH`?r|1q3e<7XXw30ONYQg@`)xD^<%=kI(*|yDfI%9uY)2xov9skQ1ux0OJ>W} zHcGHGIoLmZ1$$mSK&eR8_6wb_u#nE-12|PEC&M)krL}>Z*%1-)H>7(>U3_!JgWd~V zsJ`d*3KO4Kzb2g3X*eMX)T}$ozt?(#-rdU3J0Fa}n(~ICzcqee`PY7kr_q|fnpFtG z0(_!pEPDxFZCSL2yQYwSFBM=dX{>8Z`22eRTKBp-dUc!O9_)I|16|F&U+Dc~jd|Mv z*s21jjkdQoTxf%2!?X};$#|0UAs!7~S_WhM8rdBC}S4cen*=`dP zaAFfVv-}FZ-{v!S`TbE%2pBlzilhFAmyW=QZnYV5y@TuM){2{xXfNEm)m=M3wd42( z1$GJIu)XZID_v9UMtf68^bQyK9_@ zEX2Jil)wMnxfAhv)+&Cmq}UkxGtO*(d>HMuk$Iq&3dBe32hal&p-H-VKM6(W3y%(EK)@N9O@$7 z%Tpk^-zfMjh!~HAS&KNot+nrz)$^Z1Xnz|9{t!DKaBPZLaOs) zq1cy5{}TU`+v6RGcp2LMkSU)tBuEra1fd?CDDHO1Q5u5!{x`h$ zveEN|bw$Q%QsQB-ZWm0N5JkMI4}=EXzJv6yXLDmls}1P=<9fcl)IT^{y}+LDrSI{@ zY~PKZzf7zXc1njS0d2yk1lNidut}?5C|M$X{;FV$*rS0z|Du%^`tA}%+^bSUwZ4%3 zD8cJ#*~6DE@&+J0&$0X0|CwX|>i56dcBC+i90FEu7XGK%$UKoueVx0Z1z8`n86)>T zNg$ptwSc+s6?ts0_jS3woR#Q#!nN<)=^Gx3AitD0ce}e8gt((}U+0m&e@vqq3VXEd z0S~zDyzKf&eEyQ^?C;sPO`PDxCzz05{#g(5V)~{{IMMT0se5UU4mCfBM1(OwJhFdW z`4}d)N`}l|-z*P(c#(99;Jv;+uIEd-=JEmKF>lOzOd5)d+Y^TVeo@*d?MLieZ0~OuxTX>h03Oe0i=7d=ni<_j}8Sx7_Xc$8e5| zNpBB5{Rgj&jDpuF(dRiie7s$!+LXaIdw%tVY74wjuUz=qPI|u9gg4Yqn|VU}*WwrY z5yZVBhiQxjkKkP-tW%2*3Pa)dm~_4w{Cl+mO5N3tqrGOHeoM1h7Jyd&cm36hh!+F3 zq2BO*qs=@`Ez@ z*LO+x`s|tAux#iFtd`4!!&&bWUPbjh7Wr{&RJ2F;F=wSNTQ znEv4sZ$8B9_bz)M`359k<(t!NED{V*_1IbxRiVSyi=+4&Qqpn~+lw*ag!RxVv{y$# zjM+9%_PLpqw2nTQ{i|!Y(E+w{biZ%x;!CKu(t>0MsV6DW`ktBZxlFX?QjuAFbE#{Md{7tP~^9jSNFUO_#3 zs*M)3;O>r?@g4kaaQMCWw^uhv_nPsma-R6LM>1jxdsu3Pgt?rX^WMW#GT{+0%rS6#!eH?qZ| z`+ds(zMEu&leo@4dmE;`*ys0$1GFl_0@3pqPfp#?4+U)~-4GUZJ?hVnXdpkM6q zBO%B*p_INw1=$ZR7%+myGUBy5#O^tF4SoKV??$%tOmzt-9@D7XaUI+12IX{g@_F=p zpT1>lvbMK2@F=K~zp-nB_UFaamExq&tHna{LhBKq5xEBUXoWulD!j z+)4{xuzP3e=nY2V`^O-&hi6W{J4*2KACwr+WHW_~7aAcqmeBLP)RCW$tQv&i$->sv z(KCow>2l@{r8h{v##*e&Z_uLq*LY(c@4x^6uVSh*Z^sI> zpwj3P?GF6?(VSEByoNob*W0ix$*En5gkbZxLw)=2=8G8Qe6PY)98%+kcH+EHC zWXA6Iu^V`X*=N!HeprMtLcv56^8SoTJloj{Cfw>=LbIgLUynMKy)^32!dh2ymAyQ1 zujbg1ALV>VJwECB;_Yl>1Z>qV&>&`^@68Dli_QcdV5!=gk{;W+<0NP7qZb|7|iw-nbMKv*Q zY=sv_?>0mXknWXn#8#7C%L`WPwC>;7MLb{r&*xd&s*ghJ!-(VAE8;+R*^XgWzyeO! z&#k!kpzHC7&aGPAn{9a$Uekp3mT zXv`K+g!HfCCyH6L+UWaNnd2CbMilIYUv9#%{1V&C*;V&9eII)MI+>dE>yeTM$X|OX zs-V>d@?)m;i`z+GA0wI{OePB9zlWCnSwNpB@%6DUjsDIQ{`@}QYHVq>LkluEjWb^1 zzZdpRQN=d0Vf1(^gRkoo5q~ypODp_Z!Z;o=QO0$JZx!4;M0_^R@2vDk#l` znjX8~-*z$Jm_3JHAIbLPdb>)spy`iwmElMmOkCc-Vdf?2UTUXnB~|g?lW<1*fcWqP zaj(x3tB235N)x<>*5a63=F{A6W*OAHYEgTKO^rtsloa;rvORx7xIXlPJfh0( z05X5ESV*-BG7!&~zu2q=wp1~d#+ob)EaTGhf`$f3}5i~hx=%@kDXG4<$hVF;X()4@cvY7Jq_AxGF)Mj zhx-WJ*~Iz!kSKCLrc0Thtd$Ppb>Cs+i|`=vd^r}D^3C1H_WD64_BdSucxVNJH|%``@-jsM@27%*AjB__x(Ayw}g_3C&)tO`MS9F zVI44KTzJO2U=PusXfIk_lO=p^{eF|)1h4KV?BX!FvV%%_5%D@rbNkJPapZpA2R~#3 zBTUfyp*~vkJ*Q$XX9T;xBfhzKw@Mne>D2}p>?lGmee~2*4WpFZuRKw z-eZCG61x~b{V_oUmWJm-P02c7@%!&fL61rIIx`ZsH!9T^+UYlK=a3=pHT>-hW4gZ# z;&pwQ$tBaqXokKK{!Wmdtt9 z;2QD%ik8{^M|YS7P%u6r#U{-m6!fqlJ}^MV=crY^MC_w9ck(?)B6c(&@f148(x zx1Guhig2Vq|LCs4zbVqx7m^)wjGm1EBgD(|NMOHA z7~(Y^KyCNbfq1?cvBQa`3zkIY|Hf7FwCUZUS`;xL+0?W0Tvhi~+pOEJQ}|0}xt zgdN>(bUjMBFK|2S=|Lg+oqNrXJ0bdrceZ)+|H5nWI>QQ!KSad`Sys0Y?_b+wwi)>c z;@^+i=@&4oY6X%bgK_MeJV3{F@?!QmAmnnr9$!1h?`ouzf{OD$pVS#3&+{>=|L&Gn zLY@~*{$u(?eJA?+z;H^PET);yaTXINPbB7I-;X()HLaJTjeh=-Q+|zpkVO|}<)e;o z`_Ku(4!f?HERyae!}{R1-ZMY&|1~(r(o5Xy>c^eRC0Fq-5vFAhS=4aivG3ar4?N&_ zy>@j-724~C_qiO;ACeHTgIqS-3GvEE{POcHVsPWU zJ|!-nvAu#MYU_sF(a+ZksZ?Yh3)04)3v?&M^>zZAM(k6`A=15~DE!zbgZ<#^{gHy} zUBtaYV?|ky+)Oqoil-zIuMGm1tLAEnd##C9f2X*B?IrBoNBN8%?d8O!C(MEW z9^J-fBV3hsC*0LP+eFJjdcNKr5y=WL^@IP9v^$T+stf+o=hG zoxS(j=h|u}*H ze{CMJR_}jX-6or|7l+DQZ1j9IV%HuX^C|dE}nvptIYO7;B+SDpKX`BjUY`M@#$SRD2*y{f*t|NBmx=u5A3! z_&@!>{*1>oo{OuxVQwIMx^VBwfAs(A$QF8#faI&bIbk*bKmFy7IJz(~ zdE)oVt-O=)Jcn?;ipK;z4C=97W7PKscaLeoSY}gWL4F?? zZEjh;RfgZoL|ouh_eIpJE^Q~HBhqh6GnZynIRTrmu4Y${^#mbT&ES!UELM(4Fg zi!7S203RmT&dGn&FUGkyHk#SEZOH9I3qaV*UhGl*$`!1a!vldP@*^h!*H9Js9l7^g zR?y=PWgPMKo+HZTOo{S2$P(Mm7JTgK`ndXgeouCmT_M(37V>(#_UqE>f=HUD>S0kY zVB5>XvfmZkkFSXTuvKfBI2c(=Q`w56UJH8?j~1$D?87OXGs=*!hIT$oru;UMB(E4m(K&*Mb?vRpHeCgO{RDsQrO)A1J1) ze$~Q*-^)7$7ZNyvd*pTD!$>^8dy7OB7iXa!aXV@y`F0t)J0HWy+|0AY$KBH;J&-F(=FQ) z_WC)URn(A+&DX%tX;WupJ*4*Qv9L0IEpYyjSc$VG-m7j?Qa_vKIf%Xf%)y=#zgLI( zwYlE^xyMTGsX(WcEbLXdNpbI{J3Mu1`L^db*6UgGK-rsE3D|t$6;qZB>ZR_0b5CbQ z`&mVV-~8s0`wy>2H3wJIpW-Yk1n7b^2>0tqDX{L@^A_9hG2RWih;f59sAS$QuKG~} zb=oYHb2G%(dsWM!V{FAfP^no>8o3w0*N)h8dQ`@!m&iN&O_PgmP+-M$C)Ew>b@0aV zxuiK{{x!UZQDZOar7R<#8uFj!Ckh`r=8yqJjwL%X@L>v3fbI7u8DT zO%T!r>#v9NC9l+i;??k}wk6{8wO@d~;-jlCbVwEna^&Op`WebvqiHA)`VTyH+mtN8 z{>E-w-W#64-2IGK-c=q`zh2DAYj=|#gYyo_qnD>9=XfAzTr&2x5(0j6V6v(8}A*%L2UnfX~_qfi`R63&Ad(O>!(_9w{lK@G)KIb zd+)6WEd0JuaEI1h@g{z+O|2u08aR2VWY~B(tN9qP$%ZP`$9cik(Gz;&dRVWT&WN~c z*~j2)ue{+33)=6|>1}6b=TU(K&ohP5cK4C_ zdtvqqTN}c-^{91WFupLr&%PGYa*S@xs}tW3hEMDylS(yme@90fY>j63M}=6gPgYNh)AmckMQieieo@qG$P{NcXoq?k6?B}I z{DiGvTV$oOM;u6z^B44Y)h`qFI`-`yCvOqLu^%B;jzGtk9dKvgt>-@Th&DWY&&$S~xs&s_=Sw&UE)hp@vVDeQc@bNSGP92Ce zjWg1uss)7=^JLmD#E(ZMO|F|(7kr^HmrwFp34X6Hm3lWdNKmgmCqI%J(E9)>{o?cT zomj8Yn-&Lxf+ZpHR1eo(Zq$qKX+a4;AL`Yt=Ik!K2Y?cc%$`HJR8Ilr8P!4oa}>h~bO|JU~l?ccKb(n2IalW)h>k4>mob(mMP z2MOwBXTe)=3+Z>h_UG50k0kL;nzO%Sa4`eiR!MDypN~U(=D(L8#(D*D-d9f9qyrl7 z49^S))PTpfr(5{Mi1!K=diw0FtT$NhVam~p#qZ_6Ej9XwACjX#UUKCQyFZbigZ;{a z-AQrSe9a3pyksa6hgjPs|F;w9Vk8-6Yu3VEB;EF#~WO=E*VWs;`dVc!Q+$m z4soGf3A%In5FA#y^-vIpJRcp^q^7=My~L(3sW^B^zcz5! zZIAcLKfDGj9An24a1y)OXq8_R_DUFfuOG{f^;$I4&pQ{a1x8V<7F#b=Lo?3p^F$Ky z=K**m$;vV$Jwe&0RO(O?{{5T--t5u(p~ybOA1`jR@i5Y}9x!%hbK9d^*!_v~>Iaro zu8IOD`}nh;JJ9!OYDElV^_$W6+Zq|gPtKpho;M-+US8tg*J_D2m+`E6z^2q)2TMHgdoAyEEme3g3qh1) ze$PpTp+SLU_xBlh*cZTWfBh@gtB{G!)b4@^+{ihxfiDL=KheACYhLST^t?%2r?yBx zxdpa<{qgGPCi@<7gYbEPphwky<@{K$_^lo-t1%~Gb+7Frw|h1Cef=yuXheLz%vv68 zn-29v=3k;U`zr8z*~+r-$dXh5m!P2#Up;de+1A~))7%3N1~2Ey>tpLZ^~QztAmsb_ zG078ZM*r!LNpjQqM+^AMwL zYqmQ$g-`8&nT@Sq`ZjZK3wDUWr7dyfarS7w9&h8}`$2(vRn(}ur^cIOyhtoU&IaGl z#ktm*Z){2-oUiuFcF7zC*!?e!4~G=|DmyNIuTcOKqu0+h?d?EnQso3mkO%mv#5r{JVe_SM-_lbWInQK6qTe;+ zj@GYhMV58Pk;{+Q_E!^xRij@zV#i~juTx(c{~(UX`eT)NJmGxl_fZ|=&BEp@<(S%) zBk?C;@y_V^huqch;nw&)szu_xJk*17hq^rBn%^qdrHA;vmRiXII=T@rcO#{j@!DWe zH62dE<^h71Qy-TWv0mF*zWFszi@@CW-8L(wsMqml@#ULW(R?jdyxLB}g`Iz`d8w<; z_PRL|_BxT*AJcjr>*cpvbCmwN7DVq_wiCWz4dfE?mfTdt?|(T^9?A?d^n!gdmR^B; z_~#QZrUIq?@(>s4sGjqi^Z+MHdDU0V0|b8dJ^u9?>-Fjv&hN0YC={I4?|v_d&c9~O zcZIn4q4|nEESLQPxp(95`RbY%4^vB+`aQy4;#qm=Q%|v8jo&O|>r*ws^Yc~(W%_E! zGodJ;{y&`mnw)dxHWu>)v*m}?z@C5PwhHhL}U8KL~ z0V9X%PUi*5FP?(zu0iSn{N9cdfM~|PU zw{v)odM)kg8h-E!o3Ef^iBdtgC%7@=S)n&}gva9py7IUymRPT$i&|<~zfOQ9L-HHa z@@g2^!PqttPJF(0C!JcJ6ZV9-Frk`zGx)tE1Z1}eY>)^3)7iaS^sK?r(x!XZ*$qn1 z$_?z_haHc{>t)~Q;DlkzruknJylDLrzeVe`!iUy-hOnHJUESE{BY(h2%^2(NxS83X zhuqW&zptHsZg~DE19tx_=%GcDlfo(J53ScQSVg=hg1&{v5%1;IYkb{8#S^qXzq_A9 zhQHo(kqBNKcq0du_RPr>amH|(e`^FM(nmR1=l$iTuh@LmSGvwUq!NbU9BZ?g%cvJ+ zsefezJDM-c(;s7nPGP-B0yJ=DKJB>sdP-l4$O(JxwO4Vwv=Qs&o2B(=%jhYf3t-&k zpH>Zn3!UAp{KU^E8dq)_>K%82Nwxw_Q*ZqFs_*{zhHgR*VxAT`87;d2Q(onUlr-c! z=9va~u^F4M9=(=6e?eh5zg-5W&W3u~T6TVAdyIPBeE##>g!_I8F8fuu9NGiRF_tHw?D?2Lg1j=Dou6Lis?`6pQF?%!# z^&&Tj@;rmAheoVSepM7>y*S(2-e*n=!Nm@~wlWRWOXF+PSep&%rI^sK^f3+ld|dPD z+IdKQR-EwjF~9n3yWkeASNaY$N-{QmfW2hQ-AMiNxXhX0a+7#3y9lz4RA-!k;!Rii zpc#HIvWRaB)D-gT<8f>8ldcU>9`HL|uwv;Dc05YeXEA=K5QZ<*&a($~P_Hred{YKD z)a!kh{!Q)#?E9fLuTm;E2mQl@y}X}IFq53YdeN+g?!+yehJhPrg~H;hp*2=*mf`mv@{2PXYW>*y71T1xUzy&D8xg0zLk@r3e^yq>dJS7o%np+5H{YqW5N+ydh0x`rSN;De4Jtsr$@aG z@yQgVZu9`&l7R4IN3mWH1V4-#Bm0UkFUTEc-Hm!33!x|Re2#vP%tXn^SY?VGk855$ z<;<&@_X&UBK{qP3`O;VHdgx2zNy&%^UHJYkgM)2jHOyvYDfE6LzMq18M&tfxR<__Y zbNa^X$N0bR(7ihG>EM02^?ZSBdfW15cTgff9FtOstzV(hirQ%_LSTC5R%duIx_*4b zOg7z7fc6KNdD~pW|BrfK$)daSv)D)6RyMEjW_H3}B`Oc{8ac7+ZN{POqfF9lz+`;J z`}R2#FnQ>4Z^_-4==0>wZZDfS^Q<7uIPk{{dSoB$&wH`I&lj$F#XM}dsAX;g6TZBy zs$UK9dad36=q91kKks^Vy+6P;``;rA7hKT&ua7P5g3}u7fA8A+v>@t&%R0m5xW=&G z)xvxZ1kX=QwWw#bWXWOX6SozFT#mD!hL{~qV?xf=V0O;1T_%V4^`qK`LfvU@JJ9>F z|HThd{PX*~h3f~B?<)dJV$yeiFI}L2P*{4i%@bNH)BHcbR9ydg*UlI2xRw5P`nd?u zFEv-YUPH&DQ6!^v*Eh6&2`SNjxbOpeJ}1bgrTp?tH?I2We9)9H;e1sVjm2!w$DTJC zKHKO%mVFvjvs*dJ@{#?*!w0T9RuJ!HOK0XLW?~2Lx1R{3SHbV)7g?lqkx>aM7ZPvy zFQ0_6hnLjfMtFfmh{fS6I3wa&?zONl-F1^u~q7OWlg)FXy$bPgpH;u9r@m?mj zqest)*@4Z&w8Me_%Xv&mg#>|ndZ^cdDN&Eu1y5L^6xh;x0qb>TJod{`5m9JNdDl?u zfO_>W>O8SW&P}cDM{Aw)SDuQ+&R5sGT2^+nlW!78f|`5t+@T{0;<>ZQM*eEdTp>h+A?N8_L-c0YQ}OR@V2)wWi`{W>^;h6f=C z>!rm1=)-Kg9-KZ&Kd{PK1y}fki`hv+iLUoXy>Y?1v9=KI@LX!gJN(Z_^0A26-Z+Kz z@kpl~?{U$@9m;G^j!c$e_oEdPd2dm@6oS-w$9b+c^n20bZD%G|BhY?U^qz`RQ?l58 z9cx|@zejT(^%Ji5av%ArPxNBvUqBxzm!Yc-qItSZxFG5kB;Pq1aLIh#du={}`{PyqRD{!8 zmT+%SYHRm4@JvZE>^-#QuT=sJ{O4g^1gi3U23;21hbi`9e9r4 zOCUgZqs0caeihE$HQj^ke_hG4sqEy(dP!?9@=%Kl!k*jC^7nS4<1zhCoa>Wws29(P z9T&I%!p`s4ymZYYmrvCb&eyi8(HeVBtQY z`$j9Xge@fPe`j!e3BT7KLL@3Ze_mIq80zB&0*Q)N9LYxA#f z+J)3)eu9uPo!;2jk9tiT((h(lK)pDgt7>(uV%LwAzv>?CNT|jQQ}#8!Q*kCZ9@$yC z%Qkgly`19vT3XFcf`HV=Bh_Cjq1Ed{*0w(4^W{(SG@kvj4fJZ!%29gY_bRj|6(&_c z^7Y4y{$y#xE#$r7p7^<6wCAw-x+iAS?jj)oqd)wGmgmv&7%iJTXV!w=&*$cK?_G_I z>ALsYcwF<+I?Rwd^^0)6)C)?D zeKv6K@$R3|{P^b+CE6^}U4hC_$##q`I!76{K1yT1xXA-Jk`nHdKUH4;dDrsw_676% zsuTgpKlfl)^#=5Pnp&ZG3~LLTuc?&qfL}fT7>{+cn-W*v;a-k99+vpzL@-||y+Zv< z^;oZz^h)M(1w~kvh@NH4cm}$Uie0Xq3nY3xK2>FB*jj8226WHK-nBno|GEEOKe|+v z`VK2gtb1wq7mY+A>p;yicV0^uY(Ku1lZP(o95@1{Z|yj2x7Mt$AMJI#`gqFn*5|9d z?W|s&8rXc{{&=l?3^S?{C0y_IBU!ZXRAKw2rD-vys<0?PQmCz&tLIa2k{(5Fx;#&` zmwdX8=>;uoIL2NY^06-$uNQ6RkEzFLs27Ew(&2-te1ZH)=~0FE)dzD zXWGh8z3#=lt>DVe&8U|}Pv88_0Bk?LHLpdf>W&&+JA(OoW^(g*RVCId#Pj3K4+$km z(ERq8b>SJPzX&rEaV0)qJKpcRP4>nbe)_tNOdvl8zWs?|by8iv_s3xBD`lwFM>CMv z+;{NJS!ZDK?z`ao3)>Hd#p0S*q8|@1ooMbRQ$q7~{eg}B;NJZ8d_DJDsndzY_9v%N zJbyvfunX64ahI~EGU0sr#SZhxievi?MOOXV&+=UnB>I9~noUaJSnlc=xgzoP%gZZX zlmj`RT<0(P)#?;}ua%%5#TBNK&}VO`=$df?f}L3`ZR4H5=ymp%uqbT5v`~7LtFO&@ zAT4=jTYVtvm6-hAurv(ynl&Ncx1fa`k0c`6!!LYdaCa5E#08iLdpR&i{jPh7-TxY( zQht?qQ4ze#Tu&S2mjIRe_wsxv;=Sf}lD+AAZVj939~=4G;rE*BHm{A+lv(%Mo=W|h zSrB<{QQmH54#W0?(c05`&~rBr=ne%440WSkN}qH^LsC(%Q!+Yoqzn`lm~WyHBF-Nw%@V>B+PxZxrW^P{c_LTo5Nbfd$G{`9^d@X8hG_;IpS0C z*L(Z^1Cs&Q(0oN|^Vg3GH~>XOb6{WvHeXvDXU#XA;D%hwBD;(c)GK-_?m~nG>Ln0B zzejZhzgG*t&(@XwguRY?Dg|;O_qP1~{?Uu!NP9S?BDjs@gyQy;0tLJ2#f=)ouOGKe zzxonZXAM`xtp;pX@p~P5yS;4(ixdR4_s^6i3W0N*c=d;C4lui8YcrV!wqM#k7wS8? zNWR|D{vQ2_d^g>pm+ttr&J{MXHoXg>MDMRnsGOA^E&s=Obl%!aL0^Llmgb(?nLv2{ zHL_TB_N)MQK0(bBu;lPY0oqbo_qbdxh1I3A^X}(~_cE4vyIt+=|J1KxBTX6n;}P~Q zQg1=N;|sD${Bq-(Ah1SU9-AL`0Q;f4+t}u@^*(3Q`>Dr2xS{UdK6p2bdL1xL-8oy1 zdRezO4KOnNBVT{Kcv4L>_Z=cU|5Eu>oOGid>-BZw=#jTHN>Hdj-+uaLDX{JSSTp&G zcrO;QQ?gtg)*x%k>SLge->cMH_zj~XGVY?pB$QYVz`U)~>s$>-pxe$->fwYPkF13r z7UrVdu+O@&Mrs-L^5MywyC8+m?}fHJG7(uhwfv#MJ7?d4_40Xq&)DlVHyDI(+qdt;%k}ZN(Q7mh@n z;$yr;+z=9O8T;xm>g5t(P_C?j=4-Vxj_VgAwtfY1e(JIuypMbNlDcU-qbtGr>X`{j zrnekeFT1y@MlES-VAP%)t8=^*w7<2fyv-usYgE3)@Zfc8SWK`pE7ZX6^}t5Up91N> zvF3FnO@%dD%^p%YpOLlqV7(%pL`5{qxFBP2w%xoL9gmYq@wj#s)XTLuNxw-OTff%4 zCO^^4n!R@+==E#6kD#0ZwtiiB(07nkQVV=%1&0{7mx5OGm$y4E67R(pE8e*;5?K%3 ze|(-+4!@TLrIpiN=Kpy4TI{UIw*!T{(Y+pWSg&?fO|4M{E?`THA>VQx^%B03{LUyF zolk6`e*aYS5`M2O@s7 z2_NxRhmlFk)P<|1P!n}9(sr8o`zV%X8F{Y$vA|!PkgS{Wb^a6%Q@?iYlH~97l zm;Y)}>{J7;nNQv2Mi*hPEwAILFaO4V&&O}{k)w)M9V!-LqwV*V!Yc{O4?+)#_lgxz zCtsbm218AYMDslSUSi+8y=a(FFDrwiL+PdVu<(M9?x#4`%iyT!i!BUXpmK#$>dX?F zFC|%;v!jEkmq0?Dtm{AWwdVE2p&ed6A?%euc*iRF7S=1K^y{So7d05tmn-YqRSE)b zpAFle6Yn*ZX)=3g(i&c+x6+{Hdp+(f|UE=BcLyu9fkU*&n?-cBKT=SZ2zcwLTOW5l%-_t-360BG64Zo<% zergcRFuG91Qwo`JjTH8_#LvI#L|coBKUjlwwEYQ2y8QKV_4j^ZX3cxub_LWc$uJFP zo@x)5mes@aeX(9Qj%e+8aE}v0D&+IN(xT&$Oduq;Gz0aLb>`TYx*h99(ta-|c{v<+ zr#z=kYL2j%#uz*Gs$aC~*Blp=n;-MW#g`Qi&}i0&%N zIysI%Uq1uuVu~11uNa=K3d^eY;8K~z(58*`8Wgv~ZI0xCu`Hej#Q`*57mrG?|Mo<^ z9C}zf7mBdYN0PheBTaM^aLI|YvzEIFdl`#xE{s!Qy_(YLBx@YiK;h^3>Zn*Lj0Nj& z{9Hi%`wrRWE8jDktf6?xxQ~n+f4|MB{Ar|7MCj}1gWZrS`B>67#K{5~%uIars60}?Hs)*oy~^YsO1tQvI$^`hC% zMnA%d&DWZjT{4}#%K^e(K{6drJ^qm|arSqeIouXy=Tbx(SOgse5`2XZ?z~x zW^!?ppic?xs9t`g$V$9dq|31z?dPmPn}PjhMFD=Vp0l=81N#8^KJBVp&!jjMk@Aba znXm$j0{u@0$B^Iuf4>$3*Vd0Mn=XtBT5v$JpkiiI8S3S1JjEa_gnD&R)Oxo+$L6b& zbA^&GxdgW@rHt>*0Aa5p=do3hr&zBUs!WG-sw&V}qNeYTe1DfKQZPwdMf~}R+-94Q zt**%Yo~6Wt^MBdTd82zU%jY`kvaS&Vai7y}h<73H zY_*(c*CqIO4mq`u;!rt1Ui5sy0*T{zUhA(VO$BXV=H`TK!rZkH3|FsLV+b z_9_$|7@tnT=8J!y9ft<3Dr__pU^1b23UyDel%EMA-b-7jq9Z878oEeL3v!z9d;OB1 z53=J#`gJIeHLHD;fQyb&MXboa+1L9b6g!XqBVR(|1&Z%DKu6~4^V>RTz8nVlT8^fo z`Er{`6-mlA#d--o^g28I0rzw{S(f)6VXu_KA`C(}Z2h9Z#rx=Use+4laOm~lNc|eM z6{R30-V3A(&W|=&!v(U2Pd}RRdwIlNi1Us?-fK{&M3}6|K%`=}{#$2T&ovu6v_t1YPMnThbgt>(%+hT7`E~{6gKLjUb4Qtx>drcSLQpvuHJ%w zcm>}jJH=g#OG*B_|NiJ2YPTe_5uAbJM9VXuXY#- z9nd(9tzTc7x=w47m^jW zl*s_@g}YnloF?d%eHVs1HekkfZMlCIyMC-Lis_dYI1Icj#ro14(0qxI3hdrLiq?C< zp|{7#k6^t>NVXr?VegCEwy>%3@@>LiBMFJmdwa3(Yc+;d9~}Q85BVpXvv1@UK}g=+ z`+>g1dvzDaGTOYi1~d0yOZZ>b+svU^XQQQI-HRdq=(o>#_8@g?XVj^)*ym%sxfsrR z>LAQJS`3_ZK)uwDIl8qdqvO$Kp{0l>@*iFuMJl@!0&oWX$(!!+6ZVSy&9t;j6q_$6 z)2&SXj7Xm(sYgG0nvwH>%Jo_H#>9J-eAw-5(Ps_)tP;`sOcJ~xKtYSzL*h>~lvslASR^)#@QtZJiP3M^z#1LCmH zM-o!eBqr?&+=xi4mU1iM^GwqF^GCLw#d^6VJb$p-sQ?^%G#Vr*iowooVm`Z-c&|qy z*{Mf(ZD5y5vx?>^{QDD%8H49^k$V(Oo(tq{v^N0t9k1rwVr*gQ_z#*h%~-F()()>b zsfWPmlu0-DH0rgh|Na#Q8uWhTeTU47#aFQTsuXoPIaSt*>(!W{3^+qLU+>=OY^V5) z_1Y(&E9s5&{drJ7_2vFp5rp0SV5gWuyjSkz^RZeo8*t$Y%q&&K?`1||$RB+Z_4-C* zb|Oy67MS0tUObQVcmF$I&3AutO++7pw_TTT12w1@xgE_n+8#7tWQWpz>iFUJvI;Jn z;{HWAUq*5T{KA*8Ui9)*Q{#t~po!)fXOe3%=+7E^9^@k4>qn%+kCJURuy}Dq!<`ks z7x|!RxI;Do%idyD&o*<&usxe8!EO&Rc07GHudrS$xf`~nc(6fk#^vzs0d?!|x2K{R z9AAi{_etC2+?`^5j~$OK8{*$d6c6DdS*x7QtqJEV{`D?CD@kmB6Quk4%}xtt@D2T* zr7%?6_eigML30Wjnl(um$^m$RxngSXS%< zjx6!T9ajp%^Dj~EBg?UAT2GM|!LAUt|toXUXznW*&zbkBN@2ICo^TLp8T*I*$hGb=qB( zf+`Em*OiEsN<+ZrYt1Wsu^6l$5C$Si((O z=lO{D>TE4#c%xlH-B=ycvcANRUo|79jV z!sox*8_Wz`E#2QpslxZ&nu+g_ei2(JEZHZ2690UZmB3kLBVL1a!oyn*;Llf!OUSJ^WvJKF z&TwlB3p)r5y=~gxj?LGMS!63~5IflH+0~s=iasCT_8t3JKY)6v3WnBc@5kngB<)}} zPemq9HDA@?&=$g8%~3iD2{PDx>AvPX8mOoWQvTI_5iP|~YwHntuAlgPm42vqbVu&( zt!@6UbfN@*y;o+XXZbcMyFMPx-{@_IC0lrSVenfZEw-P+)YH3LdKlSZr*Ugjq$D~X zXZFR;E-s^9e&0@OuA1TZl6WMW`Q|WTufih}dNQim{fYcn35`c;l)*_Vq9jtg7@psk z*9f5@-pj_pL&q8EKmU91Jip&1{9f(lB~*pHsMr1HOgG4oJ}#~+?bm`dvGqP?dz|Pi z8a7BZd^ODZpZRKj=FsVCMRdMectv40Oc^_0UCURi(bWf{jfCf`*LFG5ScYKdtHbu< z*_(S*VBpZ{j4O7?^O56OOXp|ey_9$Ke_clU&uc^&+GSnE@AW%Qfluoc>SdxL`q3R( z2P^rq=&Si)y&5J$X8O6=VdTlstiv2SzgM9Cu@ZF=_2RE}R&2B8_5A~hBhVXoeh*tI$O1PJ3?yduZg$-?0EEAj!A7D;Q+b^*0$d&(fXBM zeH*0c(eaqL@S`oU2|HhH*K_^_tn&;Fz<3->auSO@5<1VJ~*_PP)09Sg!_) zxcz+Xs&I~z-G|xqDJUPPm@~dhyw^;}yVtj`*+8BJ*V2L#ey@FMdZ{t3$o|)#d+HpNYGUC04BYu2;h1|#GRsZ!} z^BDg6wKLE+-1WTdx>wWp`;48__8`xsch3DbwtfYV-r&!iV*`uj37x#lsFylZV{@b! z>h*GxM=i|~>qRm*dqBP*0rxs;G1~;0=ix|uOFkQX#I4QqSh59;Qrb>{NLz}!{}VwN zu)BU?Y8>me!QTB;^eqLrLdJh~~Eq{pLsXesspsCVfBTdr`NFk=mfA`1hkJ z!cw9yBKtOL`I58qG3Fje-jgV4=5&6<<||yGuZ5cD5coPixbO7{ov$){jCVO`ik=6^ zB7aV%cH3ZmzPtARan0)m%_3tgi#0(n+`ab=0e7(TFVYeNtz0r0(0xs2WbF0?l0jdv zWXOwXFEjm7-{27|Sa@EpV8Z=y{onof`Z0L>l^K(9sdcY!r${S98%@Fg@)pvH1K9UN zCu25;JmX-3Rqh{sHCrp!_a{CdF{F@@%w7NfQ7qwW)BG8|^}qLTuUygo77u?jf?kLG zl5B79bjF-FQI@FseTYR4MxO*4-;^tY-vv5`)@4&uf0!bP&DWY2N1s#-TRdT}_d}`% ziife|F*@VY)!>`bFuEyypI1X6)J*RXx&E1WuRfIzeu?_lAl7}$MolXlZ@v=i1-E!o z3$J?>d|~vpPjP@V;4`g&>lE~dRguf$XI@kt@mqQi_ent49*hvQgSkvA$7odHIP@< z?Lzj2>yq!_TwWA{v>=}d=`G^DM2_sFdlY94=>zX8=JoJrZEmq&5nJ!iC~Q`5K+Z)* zlI13w4ithsy%qDl0OGv{Hw@9n?X-ab^Eso};`qI^Khbz>`y{sR^^0uV=dlWBI87o{ z8IQ@Tt9`$P*kI&qVDhn>sMm@rA8RNRnlHJ47%DMCY`)gKg1&!ae;aE~uzrme z`>DkiV(a}jj?bf3g*e!C-QwiWkwS1$etYF}8u4CNxp-%P9<~9$fg5L*{PBC`yIkVY zdvy$ADlFALpE85_=ie#$&pLtHS+7BZ8SMG!WTU*pucSC(YZiN?X*=pQ&Y-Lx*@~Vw zdC~}#%CA5?WS+GxI-&Le51jrcnVjG32-DuW z@6G@5y#hm~tHJ%J+`#Uor(uWOle9Mf;x)U$_~{Aib;5FU)2-Bh_ZgfeMeaTL`#d?kw)<2tBLiX@ z{U=`|=gz$Hrg|PA_ZbbyXWV&|fvn@EzXksej$dDwZVc%TB(-c^|8J64Miu9|I&eV= zsh*T``M4%o^2j^LduoMH@0v97N9*I_@A*9gg{N!y4yko7<=(57q%oBBLZ~l?5KVGN9!i4gW{!N6|+f;oi zj2i>6_1^IL(`{T4((u5TPKg#dMk&^WCxVHFOVcCW4~8mRbF^_qn{JfY96NUEkf6itIb*)^b*kZBkK*H zwu}J$UjFvnO5=wKd&L~pTY?1ac&zMx|pwaU{$9+{(ObGELC$-BQ6^p^5?Fff(c!RT2)R5&g&x;*2bbulK$&&+G}alto^pdnsUbUbc(=Wj^{tPM#d`oL9Kv z$}mMF`A5`i7j1G&+ZgI4$oDMky}%hvzADSDmd`h2<618N64<~;`1$xyB{QgPANIZz zqbGHf@#Fv++_nZml?AXWa_m>PD)C-*t)?Vu$hk_%>$?h#BJg|Vm0Z6<>yLV|-13mV z(&G#pL>;t$=ODlT|9+w04;`uwNtr*-4W?Tp4z{hJUdP=mGYk%+_1<6MJ73Xd?EGuZ zOOUrbiuQ&n!FsR8JK_GC8|zi)6gBj51qX-xV>>!V3ZSV?U+7&Q@m>KUCqCZ#g4BC! zu~v^1{9ennUt}|OAUXPzuM;k>jvdFjKvUMeQ`Z%-`HH38o48od4Gd1AVlG^0f4P!_ z%(oqS(0Xs4&A@zC3OgQ2UZ1*PL30yVcA0V7t&p%+oiDX~ktEiOp^L)(<|bMANHW0u z#@D{~^Ofl;I~R-eg{Ts>c%z?* z&6h~d&jmpR9-yN0-6j8@ethG3Z~SNd(0uKPXWA$0jr9tWi8^BzpMz@&Na)#jim=y? zMe2up&9Gh@S!`(x9RN6}5_NV3=EJtY1Cce-cNj}bB*A*^eA9n3uMQw{C##C@^L$8H+${adlz6X;{1^01 z862RHWI?y|0)DTXGIra`K1zbg^oIVhH#*=oZu_c4#szrCDmX%aW4(qiseg4><^hA; z=#+RzbUZeeIy+?}=NQ-ab0qhZv+lh%p9l)K+?ym=gFB_Fo4of5VXy1&g}%JK zhP@xgzI-xvya{RuSy%|iby6<4z$~wGE4LJOJ*3LHTuh_D3vM6jUQ!)F-w(ZNyDu$3jlLf; zO3^<(VU6{wY`(TqI`9^EF|_LXZD+#s37+jLV;c7UnDvm)Q6}T-H)P>9!#jcF{`s)P zRXTdhj(9Ig9@U;^Z3p=3W%-Hge>wkUaV{&G>(6U;>p?UxCA$qTl#GB)~RyyiH* z+;!L^2mU9o)CT+I!(2BAK3Mc5I$u%J?5$@U9pG?c`+<);@b}{rySm5jwlEHYdd`ug z%g90Ds2TePVOP+75EU4rfmYx~jj>C##gio8H3Y)Su+8TArzZLK@bjCy%`1P#*^ zV(UGL%uE!!P%`f2MhdU9k%YZ8_y=y)m-%75+%EOGEi=dhDN~qK-|2kFKKHxC(}Z}h zWUaF-L6i=#??7M;mmdD{*u(YmWknX!Z|INLeg6DNYBd)y6gc#Cu>+egJ-4CJ%DqP* zN%zC#g;unFIp34G9$k;tuOo|h9R#-G_v-ujRhmVZu$P+i!;hYh*nAa!-Vk0UB@5|} z!(lZ@{ThE!n11LX@m{PoY{)I94sfT_@5Y9E_`TvN6PaW<(0mE{f7|1B*9Ep_w3?9s zHeZq440knckHBE$2g!#=(R{IMUM=hWPrewR>B=&G#d-yG4XS=ER&ig>TIk&XkC2OOZlOr-L~ z2>$iA+>X%>2@H272;kU>6WdO}rM5N`5Tnn^* z9cp)QJah)XS6Y(ekJ(tlUUkjmb)*HvQRG^`@tqXA1=O2wvxO}yw}u< z_exK{JyZ_oQ)SxY_qz7GFy_Wl)N7~Z9>E+3XV5E_|1@cgtzX8}P8&^zkbVjYJIN+> zP_L!VnXzGB)GINiq`=z@`+Qs*kJ_1AceUve_Tv3rv3g@K)=S^_I?t6US#bN&c-Ysq z0DPh&9tE=#KVQAyJ~YyZ^k=4cn949cj^8V`IK_lr3wzaJA9;hm3wT8y z4%niF%~yM0WI*^hCy;G#;{5po^$J?xHr)50{!A2GZ^`@TVDr^SF2!0ZQ-fpG)_ClS z^v@);zj~fdxF)g;yPs26W_>Z*TNdUV%uWk97C-|}uYz)t8`1NL5tp#3^otH48B*P( zGmqbETzvCss71Z@bcSs%eCz`A=T>b4da+)1NpG8!?s5X#Bl9nZdr_|+D-~u&m(hI1 zl<3@H{>OTI&1=)=y^(K8343krz_E=cVCVOg-@aQOLC&o;jK=z=8y7%}(Lu$7zQlV8 z`QC2dALD@B$8qS|&;P~i0V`v-rXT8cBh}Qs$j${^e(DZK$78)BPyUKq{K^5*^e)~P zU!Y#53VlnN_fapGeM2S*{MdYvTw%rO4@BcWTZ(zOZy@Z|aBolhi$<*1jc=PGx7W&o ziPx)pgNWCc&f&a8HR8SK>Byo5jU1qcE}z_T8GpWZ%D8Km>_EM;#j@ypEL`BABg{5% zVZDq#WPc-{;sBLxwbnUxsFy|PSuf!@)JxKqY}2JqZ2emE%Go%f#o0`_eqB^z-D_iu z^^)DTsA%CK3-1rACoVe`z+yj5!~Oq5zWhAs)1urQ;DBk+>y5Mcy`E?D#&Y_gUS}4= zKhYgUCWW7uIWWJY~E(P8OcN@2fBHDS)7vjTYYJ#OG^Q)q}lS z)eev`{;fHp0RQ~!=TV;3E=lD3+GCf(^5sqgh?0}GzITOy9~Hl2PGQeuz9)avWvA>K`tuME`C{s0}RPeV zfz|#{IPRqiCW|JrMG0T4(!W%6of7pr`j+LxXK;fl&WR;oHmq0M#}krIjJUw-=l9K& zIp}!2w@|r!{V6&g_ogeoKZATfj~S1Dytw_V+}_C%_WFH7CfJk;JD*tKmA=)r83&%0 zsdVbd`Z2}p_5_=#E7AFKDERGjr4i{PA^6KV`z!wUwWp%Z+1%7nuk-Vnci0SEAwu9& zj;;;XE7!5CEufAAVy+K$#{EXu+guxQ^7j4cdOI?NJS4%%6tCCl@%GxM`GofiCj@%B zy{WP10a%;IFIe;A!0`LcnOypOFd?0DVpb#GE9t?;^dA-WuqStLTNo{VuXNI5OJTQ> z@A>|W$4=q4J$5rLz&RLy?o2VZ-mBhx{F=Ui1731QDUEGKy-H0h$P`|nUOS6kUwQoy z`+f+QVjp=?;0ew$+~dcCLBd`mCo=BrNWgkg$Yd$?$m2jvzSz@NFdquNw=TTZCElxD zoYUb!3(|MCZ_(c_8o!ruV8NwfVWeNjU9K-4kEP(|b|q8WK^Ks>+{k!M8=J4#nEY^x zY7RKI@r_OUF4RjcV6xBs6Y8~@viPl*%Rk1WXT5;&)dbvfsM}H4LwGzMIkJhRArtG> za%_-d$WsP<-<@#PLh|)N`$WqD{&*FR z_q>ri=K>qfcGIMpV)NB!VVaO$gM9Da_C6=33w_^_#yQZ-=!1G~{cdb@Uuu# zc-{`Pzk2cD^DLU}IM83&*Ul7`57jEBQqJ+jd&Qo)ekky%E!6&wt-jEJf4=&$hDvO! zEAn2}g-&Y!wGo`I+`CJUemR5DTi?B2cd+kkDK6(27(5O*ejTMKCt-(_Iysl&2gqkNgP-O`-cwL=L3hO zRn9~@@m?SF3TPkX*+OIxe{u#RelJgxLDgz=WZeDnx)IPQ~${ z8k(=DFKx`n+Ob}h2k6dU3QNUlY^JXtc}CdF`PPC=$X=}1SM|KeWTby!b8ONTA%lF_ z-_x-{;4AT71J%@wUWv9KZy2rTz=hvS!2maCE{Z${9|>_?k3I&?gF6*1n2>YT57ioz zPhy{s3?^ zUO;m6$BU88zE7dv1~xJ&M)o4}vcK~+ueLK==OxmwqxG`_n=0z{*!i*i5FP6E%jH(( z*1Onzk^J#G6vy1B{E)DhQG1!Dy#dzion2iqrMe7E`_s3FndgH8{lN4<3GrTRd#jR~ ze%nG~K4q$`27WKa>SgokCd5T7vTa&M9B8i$%@}suLd!tX_TFKvm(2#>YqyYdVx^AX zXD9SgFM;pIa>lz+uhoR)0vq#xc-?0=*?gMIA3UYVU=y_ zz6|7Szb6oA`~Q)4-{DyI?c=~jNLiJP>=0#Sk20>)9@!LRiZ)&30Yz8qCgEN>_dWXV zg(LTraM=fb@WbtOXlqo4Q#q1%87GVHtuui~S13`&RYz#L-Tx)64?7>tvbRxdzuOPD zjjmJ>yP;l6cRsYMOrrhQ+l%s4XF0IFLXI7Z($cztd3DrqjO!Nu_@Wq#HB^3&-LH~N z4QSI?E5Kw)^lsr}B@ki1r>s|$aIcvwk7;r^onXjP%j2B{Zm+Ws9eQ?RP_Mw#ANp=1 z_o9@39AjrXiR~3u^M?9)79aeQKUP|*f_m{<-TmBBj(Qb#v=m+w!9LHo;l)Di;=N#w z->dCm1NZ26?0!Oh4t$0)<&k^T%jPnWe9Zj86sK~EaIe>Oc^-VcPGFJp3xnB%+pEZ) z`C*YW>ZN}Bhs&^;9ULl8bfS*N_Bwd9YSiEk9}J|$&a9k3z2vBgBezYVUV}T%h7$*3 z$5*5B&>I1gY)lT3W%|%L{9eV5nT?8ndG9xeg>~F!SRTr0nJ2dTmcXOWKMN`h3HK^G z{%(YI%@Le5OKzUv!tFJhblgYA2ld*Oc3+73jsxhQocMh?0o&`d9C5Q}CqKM5=3$Kc zNB^}-Jg>X|1M0=C8CSs`joZtk?;HKK|Dj*ymZG?D>lAi;9mAxYoZBY@4d!bff~iZu zETt$u1PJ#^_`K#uHsuKN%`Y=t6mX9pS0dik_ijbKwj`G?Z1Hpe$tkkV=N#Bx4s?1y zvz&S16Z;(!*HU!+_?G%bx7`!eOP?ycXSE1BA2;GlCTl6`0TurEx_-N`hu+!^YdrKt zDf9$Sv<%#H(R~_9TLKv*P2o5H5A{opY%a%P#u0+EuM9Kt;f}9E(YK6$@t|IdyPy9K z&~b#evC4WQMQkq)R?_>-;k>ZVKkO!TH|iC4Bk;w|f6RZa-&`-x^T3X;5Dv=GeUi5@ zDV2r>Mc?s{x4qgtmbdF+pYLd)iMe_xS{%&idW{swOJLWXHP=N+!sDxKJd1znq9bHP zl($a&mwuu-*ZPcQ4C3N4;G90l4__PaCtpVHv-11I+*p{0?bUC=wtehAFYL7*ulrSl zdSx1ED17?I`=Zq{r0b)G*!?PpE?uFOWjUsqt2HHgC;s@lFfjh6>M!rH>QiyI5KV{! z!%a$=Z_>zqOdigjPz}Ppw2%4~xBEN7DSk_-)nmBh>u4K);}dgaepTS+2O@TLXz;(U z`eM)#z9edXM~?Sm^{WT@-+r}C;)RqeLM;hgXuS^}G9=rvj^?B5!hHQ5D(v~M4KIEf z#vj^w_~R>mmhY^dD|Y?*I+YYc!zBTIqKZ1q3P^m>Ff*8V5bibn+ceaAmlNFaao2Y3 z!tG^rg*IJrD{_yC_S5!~JT1tl=A7&waskK7a!XlU*!P=E>)ateUBd@K{T^-x;;7eq zYWO})g!Zd-+C3USPGQHF)k9`7h534nD(h~u)OGw`=2tJ>w2#B?zk(=*C9+k;LG5?J zqZro`@OVZ2`kWo%UI8$aRHTRZ&dv=BoTojEfNkdVuhk1^<;`>mqjzb^9=|iyV*M?VS$1A>SF8uizkoQ_F z4zRuU*XnsDC`-Wf_G^yKi^Y)Fvz;%hgm5n*{a7~+GAC%YiIc@B;J)9)yDfj`nNx_1 zRdm31|Ie5mSH53;^TZWY4$?}nhhcmDC}wM*=;4EX;auPLV^FV7@yOCDDl{LpTt2W` z+5ct!>qXmJspu?B=g)5!Mp^KCy=kN#nDoQ;k{P9WAyzB_K38fr7(7ehq;r$b#ao1X z{g#@Y>sxdL+VE@pn5J>ZS95^nz>aFvYe6?=?aitSB;SwMWqO0{b@l4LltWQ`aPouE zaO6$YE7IM0$W{i8FMaEbdYX0Y_#*n_B~iOQq=p*5SF!p_g#&ijUS-odEi{7?Ao=^n zbQ_Y7m*gWZxJMH1^|8V@!ClS?GF};;y0VJft0Q+hI{ZK5#E<*BF7O|MC3cq&8D&mz zc8%7r3Rw^R_xwcm_t2%znFAml%p`Lr8THzAraLth+4r%r|NaC0s=%@fcKsrXrKJpd zc^4DnKz5OAJASX}Vh0%x6Kt;+<>Gx8b0xv7xNrR}Zwd5r_k?8qAL8rbNJCO=m=n;J zniLfX;a;!h5ZNl!eizyIubEEO_D&zdYkwHW?{S2!o=L&aI_(r5G0)+lM`$lk}Mwocz8vS`*#tdWE+O>P% zA^ctz#@0z^j$wP<_9@(Xu3i$5Nu;i9!4e1=UaA<#Bisu#_#f97IDyXZEh5$yxV?H7 zo@5+)^Pf06U|RC_3C3vY*j~T0j$mSGQOENOJHBXt#SQX(JOJWnelXpqL%q!EZoX}m zK;tWiZ|44l9(KJCQCXosA(n~p^LAE@JdEEf^1_Yhd3C&j0d?H0>__07NtNakR*MX@_vw&J<2+$38>O zF|LeNze=;+Zcq5~Lk+Kt_svAqtHGx1e)L&%zscFS@`l&N*!!90o;_%w^l8SFdeWFn zXyD(UV|TVEMh7|P^6&cr;{3)sSCM;0L|Uf>f-RAHKm0A?`2^wdwO^0+6Gxgeq)kkk zSDnG__133zeC-S3RiNU>o2Lr$2VNCr&AY%zo4n&IdKD}$^A+nst}cF9i;3GyW{P^n zTh{6uo1yoKJ|H^!i^dsye@n$R?fU16DLRRHrkbW*(4 z1#G@OQoqUN3HiIMzM0+qi&w|l{X;JV!H9;btW5y*x?tii#pHnAUs(L{Za;&7#pa*8 zv0jTYqfE;(?ZsF|Ni~rF!tb?*SvZf49eX@ckEP{nF{7cQ#o@2;d(pMN_S;^H z?WIUOro6Zy3tEpXjLsqRZj?oiAH9%sn}n`EWe2 zdu84osm7lF(i^wqUv!fPr7?b~U-QWM$9&>$NgH>9<7+zM(zF4Y3uMPf_RhHD_F^B} z8gu8h%BI)6eoMFb2M-{x4d~}2M&j=O|5`o2q?%xICz=-`nGD7)0|i0U<&!UO1?r{9 z(*yeTYl%qu9N047$9(%~0L*gpsb) zZcZK4D_i0u@0ZAZplteqNT)&&7UV7_NB5)qF@qN^66eX$`0|cv%@Q8PjxWpu_iuIL z%^1tZ+I!in_`T*c0#XaLvAr@_p0qc$0i>utV|GB^k2n_Ci|qd;-0R~T+Xt^7x`36P z1ZO%AZZBo3=`clAb?|s#V=4RUG~DwJ7;!WA0n0$+qC0--oADI|)<@GQ#b9Q6O!Ba+ zFbqzd%hH-hy&~Ivb)Q~G?~iW&9?EFE)oSzSH@=UMZTkxuIstEJN{2D4u0JNl(?m*!5jQw8B(Oywm) zHD$>8$6K^po|59e?{U}huXI^2kayeu#MkiqJF!y7z2KXoyW{c?w4cbM*z=OKb{D`j zSc|R+gQWe~;djPpd_`6IzTe@FdYR0#@@4SijxW+6YW+nU{9f(5l}6)UV%PgKMa+8a zb^z?d6V=HqrNAj@;5hJ}a4&(Y>NDXEE}&7hZy{e0x7U`ar?nNkkniAJt-|LFIdEwp zSLs9US1coCvSyS=zmM_nJkAsp?FLa2iES|$5g3!cayWPjt@m7C>UTMmqF(C6Z)06M zvEvI98TDKAcRA+pnO7G}&f@oqr8TS3V!@8DdiV6(=Rz>R9yDJZupPM%=3{NwViMtA z7C8_349rTkk@&Ea4!3i0XgYuA1y1T++^;|wbg4Tmj(MS8Wm<9OKa5bX7ZQAXPZs@UJS6#! z=BrCFCS`G--^VfhUO%I5DGFL(=i@0WmE+~eISViPJYqeuQrMTOncDoBa4)}m+~tp= zUBFnU_EoAW?tG+W#bo_fL&lGPye_v4j3pjF3!@K)z0WwK^(+4MtDziq4sbJ93CQ*k zhE~2_MRqo{eyukLKV`j&dc98K+@C6iosUF8)t?kT-^R!dhA%j;;rFs$u*e$=!1nrV zrI+(K1VCw#)*zXu6f%`~jkG!m_o}E&>!wETpO}p$X5lly?d2w;l8{k=dMVo^PTV|s z7O3@gkKgP=y@>XI;;cN>ilORwF_k1C3|EbkN7uHZUXeE_SVx!8`W2N=&0Cv=9bX$> z@?Z30QZ4Y;uN`+jtg`rFdr|L`tQH+Z`ib2F3c*`TVf5B70njIWKIf#=hfha^UBKJ7 zHA3nlZZFlRJARb5qF!xF6e-eQg5cO}1)bas>XkirBi;HP2UIjGZY${#2EB}<(_-|f z*XdioZLNDyufsA2UPKGx_WG=@oqsP2zZa*HsA<;;Y_Iv!q2GD#0QADwNg9y&x=Qxp zD&sT4y?$Lbd8w9xyt9{c?B{JV-0Nvy%OxTf4ypo&Lwwx0izOUcdu%vpeh%c1tKUAY zkM>_SO=m#Zga?Rg>9TZ6L}7;Ybib7W>Q(;n<&&B!G#@`b7xLEgIgQnSH4e=_>w5kK zGyW=)*|7`%`wBUZ_qO%8Vtd8B48I?|tN^Tw3G|V(B|x^5zD81+a4+Hc1~-Z#7q~gD zc+4RMcYO7D9nohyue|9+{Nwztr*DHGFMh9Q+8uO$wfNJqu2!urApP=s)b{P7kpIfi z?)U;)?}t|TCzX@Y_)47jf3xMIK2EP*774%MHvI7=^Spbi<|+2~(T3F5;_W#MTvj~x zwc}$6>@hJ+q|hPUi|kFioy@2UG_I%m%s#^HMGs#0-V&*7dYvyK3#t-12RUM9G88;$ zK1v5kZOQB62C<)lNdl)up@8}b_x9~*e03y!S0R6cdWC(x=9{8~J-ahE-4A_zUdHXXATZ}UE^=JpY<3~Y_mpnmJ*j|_P z%9ewI0P6IaR|Aih!WH_hB{|%LuXi-Mkc3y?cLlSUPv(xixV@YN7DPWH^Env(bn1d* zD!|5aLC(4_44f%eqqnW0@wNJ_FQ%HB2hN-D9ys|?48p>?=R41%Uc6@(ht`A8`sHJs znx$%j+so;~p^I(I_}|m1=oo!q@EF_6DKMSwk*p%Tt%|$t6GVO`7x?Sy-U4&Cio z>2?81>%i`BRk*#5oT-fYHiCMM$bS_0pcW2N2g&z8BS+(Fk~(*QT!$I1*N+TU-4cTe zy@uQ7+tKm1$C@7N-D1>htI7wO%NPHWj{;-#93P7?mZk~SPlNG$-L2Odid)3?B0gf% z+IkkrM;jA&^=qZTH?gmMJF_>z^MzlSizBRGxWG|fPL}QOaC;?sX1MeO@M>_BeW)z{o222*pn&Qe^cmyS){_||6Bi|n!jr^2hh z^b>!)?Cp9WA_l+LlgGW^cQRm)AIrI4QLc;v+!DW(7VJ_ApX;Rvdd)*=lH6;>9sO`Dp@Ht6cQgUdU@|cy=eL4{ru%=pdcgXj5@U#u$_9m z`_Wr8A77gnq@_xrUcWBg{w)%Sy_fd|#7;$3%NwnvuWHn@9|bwmZxu< zz~Rvm|4L-OcVoSy_+X0d-fK*dT|7Aco>LS8`vfw3Sy8WZN>x=EA*h!_rL#1PF?M`y zcm>r*khMR@@AXAZx2;3KhM_|N%HZU@<6g*RYP`)i8JRc9I2)7?&cY#NW z=Vj){aMv$^=r;<-hfyy|sZaaTHNxSH>lOE8+cw?8~>m4&QbLx3wLlIxV>8?~kaqlhT$dY{nOk2a-|O2OUX95q7xlUfk?}>P9mBdGPA|-hglw{yQp{nNFZ!x^S)1qGokx!Q2&V>Y z?k~*KiVJ;WEV}vUZmbVAzPjW-UQvjdP<)ouUXOfd@Vx(cv-)e{>TYeU{W&WIae>9Q z@=zG*-&Jc;3c6Cq&rT!ru!QahSmj;pY0F053n;4FU54zx+o(hT_DV_S38rR1?(6vD z6}-w?!IKdIw_<5#g$2=k#Dr0P-s&{xqOpwFN`xo zxfUC;pD9#KsdN_ieYIa!4NtmjqF!TJb8WkABVg-`J;DLn_-;L;s~z2A4kOO7dv|79Nj@pjw&gFo`OV|$s$OJ^Rd zQ-VV@%}0We`7c&hrqimcgnKz$9g3zibp}BNH?LvjUI<+Awd&X1&#?STZy!N)w-~)Ae@?;V!5lEBLpkO?MdKqmSXFC#r*01vU z#ak=4uN&O-dDTkEfmezUAb4V!?no&l2z+O` z*GTyHF*|axb4#i-n3PONN}j{*B~|>*F54G?Lpeh)<((ghHVKzam_>qcm4mV46?A{J zo*?t_rMJjCO||Dw^wNj|#)wQr=`HHTT;W{y>j2tMSn5@$oi@Pkzv4@C`m0i3W6~1i zBSWt^;;mn1mpO@j{xTn(Q_cRN@1i284kX2~DU||EMp<5_65(EqmIk+Nf4D%$3yswz z6n6?;GChSxx%4FApj`1^@FUy}QL+_C#F1tE*G zQ`E@&3ZBmU580POc2e7xEr+!WJdB?jCF1EQnV(f8fEJ-QSw$yaWEAFtd#^|;py zyWS7`l%%)#wqdvvI3Ex4;D3L+F>ZkM3pMus`(&B5=f5HV(sgp`E+Fp=H>J{teQ6`y ztC#xGM|UDOcy{Vba^WCuue5-N1`Khi*HETX3+<<5?>OG@Peg z%|RS)u!u$7*pGTi9}ZQEBu2gV=IT;)`eM(25y@L>xZJ&mX%B0otwHt`JT52Qr%iZucmq*T3yrM(p~v;l=MuoT+&kzt@!<>L%*%*!Ml|)=cxFdIT`kq4FGKTMFN9T$rD4BfNgm z>m5!^K*o=sWVQ-8BIoRIdEIG#_h9xMa$X@!2s$~B!l{|JukPwZ!poEU1&mWrukxuy zX&x_euzo>tl4D2=Ui9x;wbw$u#>UF2x>?cq;#1>N*T}Ql{Bt+@Rm@{9G6~00OaQ6& z#Py%}z51(&j}LFdj;~U-d1td*u{$8M_#a+52L;f1>ri zvR?1g*E?u@d4-W=W^!T2*M?W}k*D>=()hhR1yeGWeX#eFuV!la%U@Ff-@@OQo{5)& z%`c@Gv3SD04r=_|{nXGIBBGaAUxnlL5@jOg9S78F;lua-4>u!0gZG`pSxeL_B=q%t zzYF7-Du(K&iwP1ST7G>ls~Pq3CaW@dd;<0QM7itj_JgNz#+Oj;-lj-@{9bkRL3^%U z$FBDV)6t!ge2P#>_5QfHZYd0xZr@X2O}Li?eVTQghBHh@I5fT^$DNN1y?iElluCfH z(p<_6I}R)lw-U(~UIb57*1r4$sMj0%#+8n#gD|>WSr_(D0$!Zk@9n-+!eZ zubDX1f_)w)X}QkKKSCC+FHBve;3@^)ad)nXV8Xo^`6AnF7M(zJFf8%rTip3*n8ayM zR*ZVZI#7#?>qo);*2dEdPygXH^pX2*3n!3oUrxP}jGW7#zalr-h?7#2W>7ssfgm5n- zIUi17aE7+|?f0h}aOdMlpXA`{3PmX18TBF9&ZlhMnnD4z5&^O)XT#$!!%1w z0!jvdPM(&LfG~ZH@BmIUABlg`zg%#_-p>@WO)evrrUE0GoJQ$yi9a7d zf12l5a>1@&A9tV6N37*PaFi|^ePV2zZ`Au2)wzKKkR^uI*C zLJEm@yuhGd`A6vAHy8e;Uu||g+r)nzL$u?qZVpl}@b#-NYnn>9WwE`4HGf|^pri;! z6$g?JAopk;cQoMUL+)K5bpDHkiGq`%&Iw?zcV77e++Hf?PYxgGz<_W88@ac%4eZpO zb$ulOK1!^|U*nqo zOFyBQblbW!6C+fiPUncc`-;!Yt~$8q=p=UiN=)Fd@i9||mp)%l|3>QlIlkub$9D+# zYV0n*Ez02x>$CMQJFnvQQtW6PDnR-^qFU|hz$eEbggm)QvgoFf`8|MUBe?z@E$hb!@nxgS_xz*5o{Q!2qN|Yhx-lFvg^DBap|Gqi? z_;Tg)f5|?L?UnUFqUlbQJb3QL$YYTGCU@Fq&-^AL{JoRMiLwDH@14LQn_|b(Y6@2OwF`RTd)wA^QhM)Miny^N*=^ zRC1wSZ`%z687KdekAJ+Lna$iKbH?u_anwCYbQycSV|{@9RAG@aJlYM*>Op1FH^=n zPp5W>1KZ^MgIPm#efy%VT&5u%>UA>rn3kv&_WZ;~d>N%WaQqa;pN}_En^&u8vAyWc zrx=2)q<~wu_R+JM6677CR@*COgnRjt@ulWHbcUtt4``KgaNnoEG!iXoa0K zmvRiG3$*y%H~NPcd6mrdpnWh&t)_B`N*o-$3fv~| zC%!`GgPU3=@Xr^@ux-71rWbquisq;SdW~a*vLms@(~@~WaPa=H~leT8gYnk zb`m#ILcNGZvc^TWqV?WT_p5r)QSAPU$gBUJfr2+Cc^BUdF%SOu8X6N-V@B5b)9Akq zr~07l0qHs}$MvY`!A`5n^)V7(s?_BP=l>7=#3j15OCqh#aI6jEF=ve1Yn8&tmFEQF za{0j)8)7E-vNz9IOX3n-znpFH>_7L;{r_L9=ZUf$J4&7{VThWyc-~|b2eAXL!MTd4 zS7a0?>8ahQ7rE9~Gxk30d>p=KV%IKRj@dr%ckaP`2d5;ycH`adjFbR zC$abiPg$^UI8^6TR05p3&ifx`6P}MIrTt`L$a<|y-^ZBHW!zpG37*;pQ%Jr4<8>xT zuKf~yH0(ThqJ3oxw%5n1&um=hB_M06|Lx!qx_@ANaj(2LtE>!nok zd^P@jj0tpkHO}n=zv66z9#o_8Wi7Hzu_;XsN)2ay-Y=KHwEfR)tA4`0Gz+(zwAeYr zcL|p@HX7Vsy!GEg*;r67xq+Kv#A8t~TD>&`?`F*W=eM_8E^jg zRFvnFoL@Q&?#S&XS|Z4KnAZV!gLf0Yo;JEi(p>bT6HF@}N``N^_fMENoLZ(OM%MlR z#8<*rF7cF87hyHlGQ;W+x*vcjhy2c1=LlvhR?t?e8Ci#qKd@Kr4SK(v$(71%)vxIN z6Qz4{J!U=8zu(4u;fB`_B3fliY5ePtVuuY!Z)AIda_Fa6WebB%FK(8@kv4^bptZuk zjcF%xuL{{sfx!R6e)3yab{)(`ywvS}tbJd{eSS#z9A!JR02*KFvaCkI1<^3BlXvwS z^4`Nne9dKfRPua32z`er6pQVV{a*$BYs#Le7e+AMVzv?Wnqo*f6-fP@YVEClWYo1KxePm13Pj4I}k^RQmV>h>=UYN?TRAJ0ckdnUF zI9MqL7)Q-xvJ5qw`zP+$IHz3WK+kWxN9D$9J;(M6!Bn(nKhDFHcIqjyAH~1_{*lc1 z6=UR{7uS0MU3?b}H~;;bK3;K4Q^Y_%i^^(JVj+0N3Nar?)=vl>KbDG@t`&1SL+hj9 z=ic48`_;}}wNGT(k@d$vUMo*VP7f|dK_q>0f!%-ZE&o%$HoVk)`2u3Lh{K1{g!y0N z)tg=u+sU@rRHF4OYQacQTpj(MZN%4xm+6b(Y+7pk>)Scg3vG7!-cT&Wn7WG>&F{lK zM%BjGRiXa(n{yPm3&BkOwxdu7at={B=XPNZ^4~o_6}BFB6|?c*PZIv_JhAPI&Hqi* zX}cnb*wzG#qO>W6zE#)xq!US z`>m}DOp*P~gnDHM*Z<7zafTzR#+uqfxX*XAgxXtGZ&Td#@+pU;OTw37y1j6HfCHV+ z5uUS_dsoH<3|{IyzSL4MV$8K`MjCy;qv@fqjXxK7@KVU`c zp@ON2{P$LM*wy2Dv%x{aSHK5XXZ4Be|8rZ+C)_R1o;@GzAd_1fv$ zzMY>h3W|pJS3W6cER{g>6q!)d-KO5o>c(Y5iCf)cl~(vpv8KDwEO*+Fjg=KhI!rk7N<53uubj^Sv` zdCjMo#Mjne8MeFN&BvFip6zn9{t(bjbHCRBjaS1i?Yg*gN+6dvfuuq{?6N1xB<&mZ}iZV^*8}$YwtwEkRw?UyQkQbzFD*PWR2@Pw3#@ zFUO+LxA-d(_0qQdUBBbgMM$|Zn&**&#@B||PrkiHw0k9C>B{$aqKT;2_!Sk&uSB)r zv4w@&%D@jkQ3>{K`+=R0A%DE&wwhyHSn=1dCqb7uDi;0VBbiO;tTkHiPwz-q><&|f zjF)GZ`110B)}2H=zl(6M$;@=3?vE}oN&iws_zG^XF@{vdXP*%lZ&8L{>gF&NRciiY zOB9@@P3#dML*r{sg+pFXTLy~SjQn zD>-maV<+wE0^$9H!;-~39dbXwEy{ak0-3nI=zEP=V>0A7y)2dQ73=K22wgY&?4)av z-~a!=R?lyEd3hfgRO*v}U$>ptcpV!y`w6;7MQ-YDb#T$!=}^R1KQPoB=kBY;u3sBo zN6(5k(INNETX&Hu$qB4vzf(Nh5e9?GrXM(!;PKQ|L|`BfyS4V|)o5sT&{qiUAg zB?W$vVB4K4@jw!kJ*1p(mY`m@AIKfx=&lF1nJn&vk~2W&Qm21N4?DhS4NL|7i=Sc~ zs|?EG+wjMi!o9D8A>apxHB5@b)zSE3+{&JssUr@2Zz?_WYI5LB*3&%E7Q(%nK9tR= zN4Y@1!Xyjt3Ec5@DZPUw2-y-YW{!`pJ z*HAB;P|3U5y>-yc?$Og5eFkE8+VS(Ypnt!OdcWbtGyL?+tup-a8@h*afy$+kdhI&g-$zI5y<4YB z5f{-x2So=(cpqzekJm2(J{i2Xl6i)Dd2AbV4(MP5D%bqBEu@n0JCK>FL<;pfr`y%j zI#CBKd(4YYsRzR7YMqv$r^RMIZ+L~gu=9$E&Be@>G}p!;?@8gyN5#C|E}x5ipjAID z{Q6ro9|u~qJs-*_!O|!fu~ShFjM9XT^P~~(b;95ow;XbxXa#K-?YB;%x?isX? zJ=E)=in1&-%_9&rYG$r#e8tYkVIjfk?xAW-+j8Z>B`y5K{@KWPjxg?Pb zPAS|9-Q9$Hsdela3cukDP47vE{odjBT5Q$h-MS{X>E-e27{ytu3*bfK^(k>LnvcSV z9z9*u_NFSSp$o7_3pDofe_{Ri&}XM`}-JD_WSnp zg-%THo)q@)v-rKpe>GJr|_N)c0fWj>HCX=BNBBcwXzx(P6J*|Xqo0r! zAAi5vd<5nmtjwe$^Ap#(7EH3@P%qLp{l{`Hwa~*|OTw`b0M`%JxksMC&c_WeseaXA zH)H(yc)j(ozowoq=tPMRE>@%8kLXTb(J&@OP~LW^exF<}WVeXcPSgG!F(e}^JqFy0)7_Y8chynZZCE8E5$b8PA>(}uT)Jyl_s;RJf zE$BNWCPX6-@l;Vz>W1z4i&yN6CT5))4BHKlmGTJu@x^F6s3<$_3v_E)UyQ3zug;tn zI(-p6(4%mC&zzkLjIpc+as`CP*G?X3A&DtxsN* zd3f|-v@`W*YDX^8KfW?Zvmo5-0e`WCX*2S^JB`Js3`*QyK6OK z?oe3e38Wb!LC239Uit5TOzCM$0@K|iWQIEE`nEr1Ol9Ok4VXCZWeJ!KfV5(BN&6D) zdcWaSKh(18sW^VGtkHqgvaPHMR{5M|gS;QSQMdnnKCSqW@^vBu)ay0dF@@pZ7r|$`-G><&CvE&3k+`2P zH75lq?-Zr-E0+YGr+4ShKSaH9oaSDh{#66aqoO(g>bJS#mOTFWL-c%etwJV zUEE&#FFq<)M*3AE0~YbSoJwGQH|mp)UMxJ?Eplb~Kl6zH|JUmI46#E7Ob_VbPzlL- z-91vkR{8m$-8$;EOmvzqHl-Fcj!5d!?+S#s%!f@H||)L!5?1^rqQp9 zZ}`C#W>J3X@2HnAk4Zt&X?-v~s`=2pDhD~Q->~IbKjB_BCE80(k;wdnKHpBuZrolh zaz)7P6tZ9?-aR6J!W8UdSQmQJ6X6x3C>y_-?50;SH?hgVbWYgsedmHqiWKw-I7qQE zpR9S=e4FYz@INY?$n9+T1=@qLdselJ$1bOW<9ejv

AF zuQj_MhTao7@X}o1)U-6=Ug2~S!UxW}fOc!X;c4W);ElZccm4YC@U{f^Gt`Ur*+|ss zyaezL{%F1T6Y7!g~ks_g{DY6%2Ar{GjWTvlSUF>h*G*^D)moWhlDJeTvO6 z2PTs}pE#@$o{!fW#m0(NTp&`|>8J4;?tC0fJ9uBH74>Qv(`65IO@a}jK%aess8?^d zk@tDZT_8n&%l_?CNeC4${8I4%^{QT1i^&P7g{b|__tdHaz@5hBG{2(d=D*WsKk>)w zn@An=4o>`Dj!s5dgR8zE@TEJ_oEWWN{$rem;W{c{?(x(1)~Os=ah`qG$3b{}z0uoc zP`=$6N{pTblT?MCZYRFt+}kyIjZf6zCy%@$a-*hnJ#S2N!I ziv2I=MGuL5%$Avz1ENsPIvZPeXm4QaF78N#E$k9$t>nmebL0DXcGYx(jS;NM%E%Tl zk`UfrbN%QT8ejS;6WSg3>mYGtPik;u0Nkz*E)!|N_M+|YHpmfa!A#diiS>WN?`1hS z5q?t94<0haM^Yr9^*&}k_@|Jb5}fA08SYJo)UWb%N1<1Qdv&RwrK3@IhLuJZHU{Lq z(v83Ozv~zAwCge2bEsE6eW~H|UDv>(vH<*&k>CI0TvUQBP%xrTZv zW0GRE)RdssZuGMdZw|a_V|esoobdS^+c!P5YO2m~YGhWhTM@TcM_G09%si5Jf8r}u zK7_PtJOS(sZk*%Tj(Q1y{H(mP!UDfL+`SxI#b8yqvqD!3^}4otk7D_o4j*DgWjBctN|Ql4?~BFk;jFMhA|<@DT=1BKvD_sN(eJ zdRjCeeQvyZbREZ+o$j$S_^I>J7#+l2@8uLtD);LmE@r<`zJ9p9C|+>CXX!xl@sHPsAMu|<-du&$C_S6Z59s%CxZZO$Yk(X4 z{l%K{SS4V&J#wkwKHA@>MySv}HEsaqGnm{q%K$iCPvoqYja|Qp8sDYlG zk$+Eo>m4+y;gW`z{9%;ntyDub>SZ$8TzU6{3RL;1Jo4U>1KZ@r?g;l1?)55wYq-0^ z89obIX7BC7?IpT9>aIZ`;==NN-T$jGkZQ|#9r3;f_CtE!+Y(T(5Fg+*?;M=t=haLQ5Vo9NzSLMtSqDb?#NL<+>BIMHu=0Ki|g?(%hKW&fDN&qp(5odvU1n zfB4GdANe?>9eyvpvl?zNolI-d4}gQj^liu8{t{n*yt1YD2*n`xA>oU!flEhzL|F#_ z+mG-e$$zZZe)2n?)!e5FS)Rj@XC-pryv)V7Gu4E99X@yKTqAP+-sfSPTQ6PN=HL6@ z{e+4k`ANmosFy`b*1?|ctI!d~TsIwzdhxAmnO#8cE&Op{FguqSSs&^v?NR>6{a?0i zT2Gw{s$or7kp8}W0NiL2zh{iRIODJM9S&Gf>r75kMQhIU7cxEH+osOOtJ)ZZ`UAwdjdL-$BD+)^=&_u>;^OXl>kGFi00Cra&lcg}DNP#-jPjq?_ReFcWM z<&U9W*B4LQ^28!ulV+)UA^|XA>trdWirwF1G>e<4f(kH{y4K1mP58ZDJ#+dLz~Bpm z3z=0Ag~k_UPqhFYtr9#})xMU3jEB;8g|#(3CVYR+z`@=N#mIS@V>w(SGdj4vDsFk4 zqGkZRci6C50)%qya{fGp=A)Ie%Z0-gG*H6a@FnG_7+ejT^*a-cdOh8hCcRQo z1@cu9t9?5IAYST(llI@NT5OI|~>CHx>z!t2hxRSXoa#VwNEN9+Abm#%EtTUE&ZuUk9kr~|;N zF-0;V_T*;%Z06%1ujSl#WmC@hy)vao`VRN_K&b%Z7d2_r%ckX3zpA1VeAAxggZPakUVZY1BHW|jeRUz*W z5mS6F6E_osUOwAezjf5>ghE5rw$y3}>R{S0!w>*pFYBbGsN;?=+iKT=ssCX=xs2FM z{Bh)9^pM+vZhR=}C2&5L{~Pk2dh1=+!28-c!0>8Pbaa|yh))`1<<^eJ3?`7mUbd1?*9IP#~GGcV!6%udPxybwbvs7 zqTW6cf69*b6ByE|ym|AzP`b@Bvo%T#GW~)**#8k15D1a8B$eNcuQ`w4 z#dt4TXty&dGW{V6vnmUw_5YZk$nol87FVx<2?r`KCM3QB=HAYI!Ql1^p)RX5LGD+? zmyc@v^=4bteSv=3vywU6^Tjz0>wP;rU3HEVe)Xh9lUR z$kPlu;r1F1Yn8ZGh~}dbQ5}DNX*@J(rkrl;Li2IMi?Nh9;s~24Py{6tF*%^~If22F z3xmI_fU`+S(|s#`uYKyGgVX=RegKuSQbrdQUtrX7aWf7@&m-#%rAs|3QHMvm0z&nS zIgt1KZpc|TCxXupDRAt4NLTC#zHYx?=m+BV;vqHhs%^w<<|8?&;nSP!2{5rFbcwMa z-Jeseb@H{!Ic8XCc982XN8(HSzDr6Q>eZebrFW;S8j6B-x=&C9KrGd^+nTaDSI8{`2nVrB4hPNBDx+z?PkUN749V61b8V&94EsR}#Mub0A)_Iul9n z2+zmRnPTpILuMO zE_b(p12E$P_p}tn;4PJL_3eJt%T2_+Ee5Ig4lkaoPi_kU&P9(|hO)nWAC;$11rt|e zm_Z|HK@ERCnx9`bymG-8rqpg={M=D5Z;O+8BC%S)p>F4NpE(CKv!@0ZZW8XLL6u}4 z|I`s`>>GNuvvGTMt2I=LBJTwdH3zf>_liUKWw*mF{qYb`NAijM0O}Qg`+>2LIvdz( zo07}diUPwKuY!wXsMqxTmG>_nR>REABRLmI1EBD2-UvzhU;3|Y0$bJXV=!HH_9^+? z_`Sku!%v$c@%5CaFY0|T>ZP4E`l(b@8@k_xF739>0a+nao+@_2*S9&(N`1e<;RH;p zq$c@QxWA8A^{xdl1!ADLz+rHN-v(9;Nthxd5+KXE@;Gk{>h)NZ>}sk6KL{E6`g9YE z184C{Hn}z0PcVf?D6Q7ifM&AI14pD^{Y5sp{9EtT=J#S_JY;qJ5ez?hgt3_PGuKA; zi{h*I70;;(tX}xR^h9Zy-apPO&^>lv^F{6pJRuPBI8P}D*j~g^CRP$2U%$zoPmk6+ zLVNnza(#Z>UYW-phq$jQ0MW_r6M4G@K#N|8VLT=tMr+lEKT=_PDX8}-g$s&-U%%&N z7g^NnPL`bA_XpLGIJ_%8jywQdu3sIapT~}`4KJ-eV!P%4;r_yj$r&RaVn5K}Q|YUD zhtB6L#MwS@jZuWaMq!pftsF4cRMTqsNw}9rk;dfi3P+%Je8^L?1Gm?=XBUPX1{5G9 z^HN4ozZ{%#O1cwT5D&$vFRDy_BaVOOSBrPvH#B_C1~pHX+0XKd!sBmu%igM@UT>l& zpUC%CL&k^d%9R}fa8_&J+!1N3&7a?huSW8GCb}D0n5(@vU%ANP&quG~;e`W;*KLW2 zSR+bwy>{AUo54LF3>=83D4Jr+fd&$H!AJNDvaTt-dVn4EIw$bvCe8aQXxaVsyBe~d zv)rOL-oMj)^XLEV^}2KXVdZiB^U+n@TYmPh_<+OUk0-O+(0;XZaJVk-HU<zB;(vG)x6cCfDbB}q06x7Yt8?Y`rw?8C=_TgeK^Or)rk?45NkD=Qg^ zkYt>chOC1Im1Jc_R`$rw-ae_V0eazqornI^XlW&bR-bUau#|`&`#` zU-xGwV}2T6c#QO`zvJs&GtY>KV+e30A1&cU`q+PaHC;65zIQ?t_#*@=6xdKNdNwy@ zb;RpBPmQ1Y2OLNr|78C{dcBv|k0GJ{=0r^U-B=YnO2Y9a9}*KXTjC7VBKCYjzG(jy zP>{p+yb%MDa}&H}kJF%K_@&*-4dT7Z0<~Yr9m9f)B^}2_CH!8NvGI~7bEsEr6VJ?! z^C6G`#b&V_Xn!B_d4Kni4Leu`%xq}iB?=KeqnZx1==0Hb@`Aj<;|j2T!S6GUNF|y-$e=JnP??53(|l=)^)$K z0r6f3`J(h!WU;Wc&r!mlAHSEI_r=QZVW?Nt31dv{ks!!56`7+iLgS01J;FI+tOWD) z;%CRR8%2S#maa;)30*%y6H1?;-cH6{e)%a4T@pTwgBkV<% zc3RBkyE7D@t!_KN2lc836PFvOG0@h2Z&zDY8nE}I3#8B!?KXy@>QXg!+lPrtM+e12cZU|FLO|dGb;98~G`{Ncj!y2H-wJ{wtuC$JB9L}9KjO&*y5GIWyH&w7 zy&PnpkiBf4#DSAavhvAMbbfWU-jhrn;oW|~6T|4Na$i}Cuvd)g3j@70N2n+F=?K4o z*82?|6&WOAlJLOl$b6zmGHetR*rsAcyw{EZj#zTUONLT?Ko2>;d$n%=dw#p{n8eii=Qxg!wb}|M`V2pRVpwZc>CgZWhYA=t4T@OhJ`s56-I^t#gL=u) zvGkt~Er$l4j_-Gn@rl{P4TcrTrfYRzt$wMbFu9#b!_4c^n7-de*sFwQ(!+6HZ3{XdqwBRi@q2yWpf4$E zj>gwTp3j*%uAv}nonh5`6^*aN$aI%7RUWA5QNN;6Dgv0UY18N1Q7=2yeifWXIqWBY z>o$YrW9Ps*DVqGh#Mf^x*N*%)wr7O%G2L4!?$T#Ru&}tI9Q0Oi&Fj0Svp~9@24}qwV51H0-@ViLLZK%0rQGadhZJH`B?5_WaUqaoY!g1 z5{~^}yn44^R?DJBy}ZZclN=;NflK^lQ13p}>x~fa=1ewjh@amehs95sPz%H?^Ob zpz%oQ<@pTM3{)Op3f9Tl@0nk#9IHo!x0)8H+OhkX7^-IQ9 z)0~>49H_7V%5C?;LE4R%Z6P{oG3SDgQW3=$>zSS2hujuZ{>It;nNJ5Uld3 zUq=)55*MGwF^5XSVT8;Wy*4)O|4d=w1Uz>bg3Ml_ z;}aDQXI`;0Ndsv@W#uM;WEg%}T4Fatd_HEMVq@62Vhg_E%>LLS{QatFyWy|WK%|fV z?UiIxf%`fi0-c%7Ia)<%f1kLp*Ex}t6ts>l+cdorfg^pG={xSIm+A}pc=n<)Xc%h^ zK91yL>j(8MDt>?ICw_aC>=hiOk06|n8HtJ4zL+_|F3}*xnk=+_6%Kxp?-G#$@)U3K zIPqi{Sfb+8q9;DSsDrPxzDC9;=oNO*?ij`IW&3u2R{9&%OO}yU`7;Ny9^l8` zyE`P?$1NJ|gbxZP{IsJYKo|1ycxWZ+#T-x{(A!Z44`b?lGMsU6L%e-Xeh|NxwBh?w z&HvD^hJ3F}G>CBoYgU&x?e=IsYVWBW?LqcMuUsy)5mjZ6{?punSu&$+_U5;~uZz(b)C%9y6XF_oXBUQ#YV#b`nE4AK9FGxq0q7 zg3^sLm(nuSE9&k=jhj29A;LXc^|NF$cq*klKAcIsmwi+}ZL1s>=K0A_S}EbLUzL=p zIu-xi8{M{e=4FxutkknKznuvI38QRZF;~)o0y8gT?Nn;b@w zbkwW(c^vzXz0$BVEU*8xY%CE$t6Ys@%F>r9n3JbkklJoZ`;rA+PHr_`8$a%G2 z)2>g{JmPZ_Fk%f#3FFa9=GGsyUTc~r{GZXx_$BS`|Pq~8JfL-#vg zLe3?7YsWr&*eDEU&a4_K@{U)TBx-9);@V-=Y9$2!bjZ6oX@o5rWihKx_dd!7{}twp_} z@4CKJvX%y_7s#M0V=}O`UYVQTLwtNSsvPo?(Zd46z9dSJ!SB_?&iL+`kn);WOnAv_ z38669nrhuUsfK#-P)S~by4swt3DZ)k9v`C)vTz;RKP7+8``EDIH0`E zpA+DS-;3?n<(al2!udGp`;hp)n+mMvsF#R0uKl?JFQiL% zcsZ_!zy+rp*xCJcYv*Iynr^36=B|J@$#rztfBFfwoiTY={_=eM9bf&tE4Np~345Kk zEeTgc;_KetZ&^cZsMqwKD$yWUDX<>-xLhcd3})9QZd!y9@0H-B`0dqhEXc-?ykr{3 z-%kYHXAe~4MqDT^b375|gzZ;$N(L8%0%h5-fUgwll~{k+zps!AtiI)K_@W{TtO7qT zG0~u2z46-uxz3lv$%m#l8L#3Xh@y0gn{R#nBIy?Oiu26GxLIdbMqeY`f2mC$uqJPF zg#3(GNsri3uc*R!g9&#jSgt+t#T(hbecOWG!a0t3FSQ8g$Im6Oz+v#pUbhSX`}UoU z@`ujoBQ8TR3RKmHVXNh3Bfj`BxRlWy-h-U~xZ2-O-_!R!CM^I#Q&iDG$o!+=_=Qo2 zZ|M6YHvGzEH)Kn}{?4V{tdbBJDq%XJu+-WY3%G z3`zX~tqY`R{i-VaaoQRu1*OUAi}e)gaQRk`Zr6F@?}w55p=*@OjfJR!&f%aY{9fj2 z*=|0g0CjC@>5@!`LFf5bA&;UkD4xVaC{-rEaQ-}-fLgT3*MI= zQgA}u$mbISvfkuT&{e|};=OR%v5jweuyC6F&`?(=eyKigQ~OIXs4btj zn7k7Pp_M~#J~5&3r7#?* zZYb`Zr(f?yBKMwDj3W!9YQ}sfDv_|4LGv^9u5xG4J#}EN+yad+jWfF^oX$&u<@dzA z&tKBOO=!o9kJZHIBTc8zru5FcNa zKTDn^ZpMQ08CjZBkMYOXt8*;Oxnaot9m`RJ%cFWQcTWcE*BJ)eg75njWTEk8qHb9G z)NChYkc@0MVitvc&X{nAE2!5|vpQ>Ep9%;KdF1Yej2B{^joW$uaz2w~3R@q2XBB3l z^=%s;a$hc?`C8MyjxqINC$OPwEf5GjwC2_N3>V+!g@J2b+vScR!399obY~_YZ0Pc zoxP6G{bctpX=FZm_4%_qn;qkN9s?h#Eu&3`)1dFB``t!K;^XV`3BzUYIa_c)?j3kW z2fr82O;*St0`-!lw%#G+7z*tRl#Bk^e|UwnAGW*84Z(M)U2`c#f%F=~_D&JhD_qvD z=V=G>zAabKvD+I5V$O_Adw#Bee@r|)ZAkU0660oiq-OX(oPXbXE+m1zz!B1O#VoDD zQ7;!i4ZGoU7>MCBUKae224B=4szyB{zJBR%EsPufWeap)cX=CJEm`|s{ypEZA~E(> ze*@~3A@z9g-IJl9ovo_9VJqsDU)wvRSjY)SE+#4Vb&J5X{Z@LlR@5t>hKe+{xg2Pv zV*U8Saj=x2c;=qTU&af)9@1^QCX1==c({d+gK$0`+mgTQYrZ2K_i3{0k3svd)it?Ox^$Vs&vph*1tW1?!Cz8YhtP91@3su}9%< zV3M$W(}UwLDFIx7d+k)=hmjN3M^VfdPzvC-!xF}1U0ri>_6#0HjC65w*pTBd*MrdWPY2XIrq*Gx_*N4Lc!au)@5Lv?paVb=mwt@dpMYnP*7_Ul#4FOaF7t6j~QGd zeT4L<9$ufK%+QNCXq{UN?&#&(>8S`Iua0(Ito9OG&EL zsoM$l>fhCcQ=s*Qh@&R@U9qT_@8v?Dy_D?m==q60pLfC_F&L0qIf8moV$P_PSeC+} z;XBGw6z(v*?SRD+C4R437>Kc0ZL=kLiekg+a`{ z&6Y{9-DpN*ry244hqAjGMvb*?!K=LnyGaxO_(a8tO+rGOkoCU5t?hx63!RyYR^Nv6IbeP}*v^|w#fxQKxq1^-VoPo%BRx{;qPRXI`zgOQMNox5AdNzb% zD*d-;?XM;5^5bj3 zJO^$e+q>=snJpz&QnH0qXzF>25G?A1C^=c11dpj#;Kd@%8)ACLVLk!d1vl3s? z@h{iMH+*u^%V65+mQ+;=VK1dmCsZ3w67Ijg*zq2HeZT?e@~f}VE28g1N1sahN-ZgX zvf*CwP}Vfi8y<^3qC&h^OPEuy$~9YPR8&nX{9n9&F1>n#Ig8Zw-(K+<{06Zi$oT$`0B1{2LTj`B zeP}fwnZ6yJ8u<_RKkiDWlHPUY0=V(@yP6oI@wKCD+k+?ciZIMg+VPG#4Hh*5w2f?u z_u4_nO>g063*Xz%{qhmOA727G`}RH2K)pP#>`uD<#upwGdqTuNj{o!LmnZx5+|+Fu z9$1PrI-DRN3QK%^mL?NueBIG+th+T(2Eof)^T`U3{g1{SOwDKTd+~f95HyP;9A8dp zGW!a=FMztwfUEOq^!d18lYf6Nmm)YaV}8tPBJV@5X71`_67O}>>zkNH{;1SZ%?l+j}04 z!j6Z!(P(_ti%nEK5>Z#opsjG)mV#_rLpl zi)37e%R$6tTrlKq>v;&cFc_`i;SUQFksIs&*e{y@c0}|ehXBYoq*>fn69d;>55!*j zpk9+h*Y&SfS0M98p-;NMyFr3R$&K(@1nH=I_BCR1U1v@$VF}NtRSNd#m&9L$ zhqfMK=JV+Lwt^eGEe=`t@flm&F>yZ)_WA$nO-m=fpXh4aJi0&I78Gohlnju2*H`oE z-(F8XJ%W;d>iusoabXK8qaA)w+iBM8!}Ujey{2k-_>>EDKkvc?>We|y`_?!vd(iqe4ZZLhk+tBvmdapMbRgqd8^_Y+rGLlzZguO28EI;gg?IMiHjh$2- zN9+9;eRqa?tQfG&TD%pVmIm=gV+Gut#K%|cUFUW-q`x03^%*W!#UEcTX=*viL#UT) z-$c_U6F=C@^3^kJ0`>AGe{i$Zi5K?uvg+1f5`$}RrK5SHQLk&-HECxJDv)#Z-*gzu zyF;C|v31Db&P!ud;eO1jf;qsvRmClfa6WE&vZcRf%n__IB%VwapS3NZUwA@$cn}0Ms5D;StVp#prR<-;yUW}PFeS^o_Pf(K^9)Oq*~PLsZqzKU2*_gXWsP|XQY9-tw;7b zQsU!F+^}NdEf*HFD(CIRRPcLQunm~rW<|!ij1DlHiyweTQti2((SC6F^v7)Jedzd? z%!g5&t<5f2Aa_V@PZt9gzKCN6uBg|rqde#CU**VoiQR)VwC*6bc>JEm>-F_M-;_4> zOYIX(>Fa%U{XjUrtP@*vD%hQ1|3TrRVt#b|OT#{NsvL^}p3^1^Il*b5D*J8d@h9Tr zOO@)^IoeOQuy0G)Fc~#|FUPw$GP#dP9R2osKJPWsE$jypttltk=g|1#*~*~H2b}O@ z%C~lOC$hhmw)qU`p^;pHh>x$PkT*jwf7n8U z$M69!Dg0jAb5f`Lzo1?(`Ck;vNBIGjXIOIW3hL!sM_zoTnh$2-O>4y$QP5|ge?Vc4 zdO5@wd>;8x4)^rZ)K!rCnx>adC6e!0U+*#b!)0Ve1sHWrTn+9&tdAx=Es~|6e^O+im%hmyZHb^W@Wh-du3|yG#^X)5yC(u18L-r0LFR{ACvMn<_7d;4bSEaS z8?cbBM{}c98-G6LEx+3;i36B;7GXQhdNU81iOu|FIlO3`JeH#64$OO{$TiZ}=OfS42DcDs!!%wzueT6Q zxZY2!B-`a~bpcn>E5fHa(C4H4OkNPPIKY_Gn^Rn5>CmWwJD5gA{P^lDx%VqO0W1vf zeDLs?CVsCz>UpRB1!R7R>gAEaL~$6D6uo~$ArR6Ev4+}ztOv-K*lRCc$^$hKXG+Oe zM3MXZbaxDophdEt+U7dbC>ta5gJeDOqjiGMr&2;*NelEP^~*lTb; zQu^y17vO*VuAb?S`C3w@BgZ{*0Ct|eTi>=R9WKP5_)>I|c(0ysWFPNqV1f0sw33++ zelHcBsJLA{h*xwUd+bG7=<|^Ak4O%LT92YGiqB|#VK&X`H^ef+`?J2?XT`*TPNv_i z`a9}Xx23PqAK9O_{j~afM?-g*TZlXU_{sYCO4R%HnbNKU^QIfqXPZIT%Pj2J&NdNe zaHFpa*@Q*oYtJ*|4k=QAFP$efG01rePgtp6CRq^gb%NV|kPKPB`mpbtpBV#wFQzE! z!vpUD5)D6XQ7G4gyJM`HT%@-lk*zW2`aJ5z%F|5X1-{bO2U*=M<4&5M%atS|J$BwV`mT}AI5f0 zqWwg&W#XEk>q@l~lKEV{UN zy%&iDT?@rbE9T5EHv^SZguS@)_m^B&Ma~DvXL{S*i>}WR@TO0z_f>*f6>X#2D`_A- zuJ)nm|FD1PWgr`=F0#NTMWL+6s2ji6kLNB&k!5ge@#UXkb>5;S0DQG7XMZZL_d2mR z!=|7@6ud&~IamHzKXJfzQNT>R91a}Z#4)4p4rhN-aGWvy%X-Y;Ue|Dk?@nzb?8Wt+ z#pi;uGt4sRD+U}!`}^Cvc`0v@dBAdY$tC}>G%!%{n+)J3KE6z7tO}};_0cC4lk_r+ z@q2x4BlS=mM$Z5GosTmkhDutK{?I2Ay_h)pM?W!0H}PYT0g|(JnQq%53X@qBg?E3U z`6wEEejuu^3~tW^<{h(jhn?=HH=Xib-+!&f7oA6q>-8AI{lxr2Q0OUDCy=_nJwWyv zI=}rqCR$?}8J}wke=2a~6LKEk=WVByj}z}Tn`PaRV}pgK377b87UK7+iXWADwSaoH z1ikVS%=Cw;^gHnm3}}3L1+w%MgfqbH*XQXJk@bbYDtS!%NYH$Ays9FlJX!|I+;+*A zJ>B6@6gG3)?e$(>GQmo}+M+Sad7@vq0||Q>jdMmF=Wu}*qXnnW@@W4>rlc?bo*x4_ z@u#Gdo}_`LTHH=|J>tDi1ovefI)Md|b5-Ae^yBxsX0Bk*^Bwi7+OG82vNiyA%+sXN zlc8RT3wHIn)SNIYZT~}2M-+@JvRO{Gpz#&`oLZAkryO1uc}kYuatHp480Irg>-&jS zuSXghk1YPf{d{lv1)^*bucPh)4>ca5??Wds2h>Fp6+j|daI(BJ4SpE&7j>Eu@1=Im zX|uI47V4-~`D)7WdnwUey({+u1DKqmpr)(JP`DwwgXM7m+-8k&njb~;(Kl?&D@AxG zlwUt9SRpJ5?qYrSglbT)##`z+;&SD1eOqlOZMZvpskl1pY5y0m!K;;bEXy&Jr7^wI zZG_`1>EbpcS;Wij(NxUd?`S_UK-RO>0XaX)p7+(V?W;7PYJ1TuC`Y_k-zzt^E=4T7 zs;4EPIg7ua*yEs)I*^OR7qee?sF*(VkgM?Sdlvu$`(jcbA?HA@j(=e?Vh+B#wG&2W z_ZT+Th{Aw;$$<^cs8@p7q-x6ZGQcHCTT({4Ll~JMg8<2TFOqn>mt)Z>n9@w2MuQB( zUZwY%zwhR9f$E=w^tl++>*sqf39AiC5UKT?{UG8c)DUA)b&7beT}e+whApuWWA1bh zdlA1^XvgNO=4;CfpDpZ}% zLe~2_Y?Tq*_kYMo)4~hiHX{4^65a_b~ zW)UjFUW6+=C@p;uYMqWc(H@B7>3jBgC{15{lW zn!6_kX-1>jar@Edqch*m;awyZP^lN&L|N$$?{9Or&(%e->VKm7xXrPTjXql$Ldv*o=aBlzc&Z`>QJwk!`CL}oXcVN zXi;Bqw>zZFRZTy)TwlK!$EP@LH99eG(l_m@&n4_NI>lD@F$lTOu=L5dwx4J|K6yz# zw}(|3Ty9xJetww-)vqnB?*$O=WhsceR&0(1p$9!F5kdI9+;4_opF`GVuf`XfPE5S; z(*QUtbW-n=C+g*Jrr{EM_cn0T%HJu=Dh8C7f}#W(QLm-I=H_f%Idn@qKQpxS0I3PC z8V*tXUONIRZah;Z{CvDRNX764xsR3g@HgSp6{uIgtLZ}aF8~S{k#&29^b;AGmTB6= zdy!IJeOY=R3qB67lrO~M_nNwVTEDC2AD43K*Oh~qYG%<>c29jl|Ci$Di(ROfNMWfl zOAH%ONZ3WvA>$Lh;Zc_qs!=bE!pPlUwaUR@G&8zA!2=dfZuIw6S?@)1p0gy##S6o- zm|aglO?Z7_Nd66*&l_>@#d3V)6rf(1$C*7prT{{YD%_}GN(bub1D7<;5$|PRo&M{5 zAQoz)2WMXVFY{@eHpV6L`=DM<)O_1>-uuFZr%|z;cTq3?Jsz_aoxC6+v0-!RfGAw1 z|2Q9&gL>WQOxEFDEQ47RzWC#dl>Fsa zmPpjgOuW9lHO&v~N159C?xS9q9?pR`2e@G=NkUJqQxqsR6&FNKpk7u6WBbD=%7D*_ z^|AGU2hf;^XghbWulK9*)xTW(>fL`>f6rFjHt)OS48JJSql$DKux$5eXhjgyV~j#B8H*j4R~DJXJe^L!XcC4Z7)b*HysA?!tL1`E*!T4>((vMZDJu z0nt{EBrG&M+bbehh2Lwa%EpoJf9C!FfBjlM+x;}>jUQ~t=WN^aM?cZc^(f}E@iq`| z!jz0piz55z^E@wZK)vW|`VJb+l)?2pYvxNYJb=5&M%q^UFJ1|KhVtLyF;pVOTdkG| z*DpVZcRmLbT;Zc&tc2Gi)az)yoQ#9jVUVcYE6fb(uvsTQX(okuubIfu&ZJ9NP$yvp8zdi5Mwve*Ao20NU{{9{u+;9S!lerKcgUWxhn_av@XU|jSExY@c0 zdo{dO_d9Fv3K)^ui?2-3d~9yz&bh*`4_k%FTH4i+`{rcYJEw0F@Abo7bNp)}(oe{J z6cS&+A73TP9jZ&RNFT2tHEFhmA6h2NY>Zy|L1TuPy{$Id-`ACdF7Zupzu+_r4`MV%GB-Ji_ewBo$#L<+0-40kzM2O7UYNip1JZxi zLI3vRKPzW3kF1|iJ#;5k1erfxjW4gEP4m2dn;{6J7i4e}IZr{+4O1YDdNsu9MtYqn zgV(ZqSL(x&=VM#;PPVf3f%>+z<2b`0CD_b}hcG z2ecuh6<$caKkquavs;vSFOiySQM-|P&v=>Dy{HVo7t6R~QpP;uQq6VvbT%dI%9-3@ zyvrY?L&wI8Y0>x!$$HMXQ<)tOm|JY%KO+j>c9YeMZ_wvs$5iZ*LXR@Q?jIP(#dyH< zNBc@9%fIv!rH$_g$%`;W1@`R^r3rg^sm#@RUUGpR*VZ@Z^U?9uN&#K0k+dP4yyc#+ zl#vdlrsnN4|A&6{ip+br>-UiJ_)fbh%-_fF^~0EVBMuo~UG);ORo#A!(H~~b9(*t$ z`NOO1vD;Vot+4pa&g@N;D15`PkhM{u@ilUJj_*oL8SH609xj;X0VO3JOS&fOy}V4b za8L4+Fl38U{Z}msdo5kViqvd%Mb4jbuv;EP>sR|sgJrk9AxPfuwy}vs?rY-V=_(f@ zKE8Bjb~SM(Bl&0`Q$5Bxy)Se2b`DVD>}!JzyDqD`!CqB3Uu!V z#ZDoP7iyv)ceRydZW4_zKkk(Cy+}TuWt~oSM(X_u$DIdREY{ciRWGWR{^=4?!d@P~ z#!mXXxk9HS2j`Ft`g~OGpdX`VGlWL(ux^q2>Bzp4+)0aL#Cx4j;*!~zgN2Xkdziy& z@Oy1>cMjn7N8;#rKJI98d3{667h32aSzo;OM|?RN?jDNc1}pPcUg~D#eVd0}XWt+F z*DYy>`pK{|NP@kS;}<-@ysxG2DJy<2U%RqdSyRGZ60g)~n3UIh@rYI&eadeLk6&== z*ry}+>2#k-u{9yytFFJJ)2$c_=X}|wXOr=JajiUu0up3j`)@Djh|`7(&bOf4SAgc% z7IeQmX7YH(pxRCd74%8yNfHIkt7T3ufAn9YB1xj(BFbPFRp!}v8xL4=zO+4l@GoA1 zQtqxqOH87Yd-Xvg;v{vZdOE0UUT8gEOZ<8!+t9G(H>Z&E9!XaW z4YcrkrN3H<(JDl}4x2N+xOK=GT*M5+&Hi!xUBBus^pMLqumVSuU&t+O5h#!4s1F-Q z>lgi|H%#{pRIz$*6(-s`t7A^5^~;CldzYNpBRZ&rVD5+?5MoY zp}zKc@KbG}%soTQlOp21iqF*RK1J38?2R9!5BiLMKFv&|A$)@t z>P2?P(YOZLw;=no_j;EU>XqN{@Yg7?!GUY%Ds_?l$RY=KUwf5-dQB&=av4#Uft|IM ze3qOCoO}>R*-Wv%|MGfXv){ie7L!Av7_R#t>X%7st8-wM3*>E^GSZVk_dj0klvDd` ztq;u23r^o2r$Mv8TTi73;=QJ(^p$Cl`5}o91)Gk|_mXU%%rEr+J6us}$jU)Va;kEl}wK_i80; zWuBt>c!I`pF!H(~N(>+fe%T)$1 zvbq|unjY}$n#@;m#r64!d8&#FDbB)3Nwr>#njpNtc93c610&+a^Ha|ScO1>f{UTIc z29bs^(s63zr$=cJc>czv!?%g|y0?2LBPkB)zo>(N{vCd=DvzvZ6#wjpFr-ZO0KsyBkW3{G#SRuonD z0OgRrrM~C*y$+1&hT)h9$CtYE`}*(ME|6SF!4~X=z7Gu>eOWnFW(Zp=uI}@Gmi2%m4j+ z><;?zuw)k#sQ0}y%nTI<^NlrId@rG1Q*1RyZtW<8JMXGrygcXum&{1(qR*@!pYRGS zQ6Kphi`o9*tEs^};rP0DWUgShFS0(zN-=Fo3ym+zg^foB>kYx+$d-&l$bCAFwij0f z42X}f?)#HQv+h`MIBbtQRD<8E{meGLn1ALqe#h4x*Ak0CEgUq}gq5fMec@^#zz=du1?FAUPirQ^EX|5f{YtQewW(lC+`7_R9{Q?B>bgc{q5x;^;$xA zim(@%ZU*hbunWY{RWzh=qh1Wb?55jUj=+L}inUfx8u+kW<{-zsvzd{)gTVN_tug4xG)!Pbj&`5b$T(cGR3OP8}w;9V0#dlQRJxmn_M%I>l`$JJL zCKd6aEq$dx?so8;98$kBgbaiq(BqGA>JO$6D4na)VUsi<`hJCp+R$9j2rQ7=my#XG#eZH0m4^1GL4VY1PCj=N zrN$-D4&q~f8;yF2jK{gD&h%-4}>9E+jC4NJZ|lL%%=m7HniO?2Ie>KRIZwN zz*kKf=LoL9JRg6@muw{FN}HGt!SlaV``Lzm?RSM7DkYE@WAKelHdGZ5qJGT<=l>)dgRsmSJpnvj2 z%p?3>Bz$xqRR4&tvFT8MZx<_@z8czJ-3 zvWf-sU+z1>Y>jR=eU^gJ%bF52W+3d9oyxu=!2h3oTpHkBM!j~^eoXCjK-P6X9dL`L z2?0K`(WvXx#CvTCZJD}*%-1Tmy{;a4ir?#q(X*U?&S_iqI$;p{q5s+?_!02<_Up^2 zS2sB;)?~|0;16UAWxOQ}?}rCo*vq0`z1N=aKOt2N8}{>_h(6;1BHJM7TJ8GyTJ`$e z^L9}fx&MOD`Gu=q2FLYNtD4dNYt`!uu3_ozCM3QdFFzh+A>PY;!-o+TB)+bD6=QG( z_`Srim|Mv>)Qh4dnCk_V2lVo9$ty$d^;qpE93B)be>%wluU<$pUMmm=Az!T%mwoT8 z)%$JSCu3+di(!IkiRH{y4{+|D+|?6xV(tH5J5OQp+R;*Een@+J*9r2kguOJnYB(R| zx&puEA=Q%Is28a-58XNBJ?^&nO`n$KVDO-}7XSD~pXleKPe{k1sSC*djt9G}Y%B45 zrF`P1H?u{(LyS22vO4|YHG0lqI+YFPExJQHjD_IF7QM@FQcy3|Jz~e( z_=+LQ;)U~TB)(pHF^EYWSzqr-ZV5dN9reZNdM_xvjwS5H)=ebk_e5B~yw0j&=2q6$ z?uR);*L1&q1Aeb-)x9Akq-cEYoVOLZWa|R)BaGgg8PWK%4Ek~IxcO$lK3Qp4+$;d% zQRDk970~!PRlZ3;a4^GyuYz~I|<&VZ?3yAk>?&p1Hy#Wj7ZQ~^w|CjT>oW!5* z3XN7>i?3dJW||WZT!3o+;`f*WozdNd`98lZqy$}~#=S$mFs<4f=T&w+_ zDI1fo0Axzi{phnpy|$dpE0g|O0E-`DWaF{!uzOQOO4!U_<~x4JSIhEd=@wtYUJG|3 zO{2)2;nDE5uRq&V*1q?+=2yobevktSOK+_2%OGfOeET&DOT1T9;46J9TabO!j5f`pdw}`e*`IZ6sMq;C z|BzF+3*g9xtZgJl?%=h@Y(!u)elONi>{Ut9|L(;`HJCr{hdv*(3=*zxu~vZ3zj7U| zb%KG5OXqoaDe+zr`bqv5kozC6U8SF^?s`*n_w@O=$?V9fp7i)p_^#uAskt zXjtQl6FPd6+)l7k{C9ipO1&{p@2{n zJN%l(;cmX%1K(LB6%F)IFD)z#+*&Arh0?Igp&IUxb;aAwWPJVkn5$l?^w`PhM#Awm zJt0p8i3#g<^u{lr`+5Kg?;6S`KhZAkys^i55^9f#j*{5flB zPB-ec-zpoqF~bGwXFWnS$5Agkzf*Z)xjf)}*Gj%L7s*EC$E96`Mr9&b=!K)lXY?{J zvR0p0-ygkxdu{d6Ykk{G*emn#%@42NJA&WEyK>E5==T?~m%kEd72uWb-kJ%%VBr6% zq&V+Jyccb?AFtEty>PKV>*A4cf}|DcP|RxTfIB=(C6c-*MTqt*47=t@Gf&GQ4vSHmpN0!MjOQI zo6sg|En56ueIH~BE+?Q~v(e0#Xr8&iP({0jdIjoL=OL-aAGH;>rJnJ5dwmb|cX93+ z4Me@Lo%^+Dk^cVpR=vz!+;ukIAnYakdACUGHzyFUJ9ywj zC+d~_(I#g4jw1ZDKP$JZAr+`D2yM-KXHWF|Q1Ex3Q)fhw_0dMP4?R^L;Ek^YzQn@{ zWNP5WaqFVLmOLDa<8Q3oiUX0{%OTO6XuU`T1EzZid7*Gt`XqA%KfHO*x|O-GdhPv@ zm+Teutyje`_Vu=F!;~9@+cerAkzfD59io}28n>+u!_^=^^KAbqg6~7tIZRRh=Ukxq z{D})p_tDSeU_yPMAwUs+y^D8^PfUg3Ik?S)?B5zuO^?e>N8XzzKV4tB9*$XkKOw*I z(v!^g`PzSzw3vacWG6_w;dJ6(Wzw3~q=28l0!-HqHh8~k2;G?g#Jn^f1lu6)W5 zdfM*_=iG~#KKxO?x|?2{oibnp-*W-eOiKJv8Tjd&n-=QTqT|n7rcw--wVxeH#kzvY z&wKj{S=Nu|ta{b5e6@KrhTKO?HeB)Av~TVA;K>pf;;18E~UxN|MEBKCXIgS{ZS@XWD~#D-)IealC3RzR16Sv+;K-g5uk5 z-^@=T^IuYXB?2|k_x06!zv|T%`jI^!Ilq9A*JpVFA*Ly(bzWz68XBJ=UM%lOt1+Ib zAoMJDgZ(bzy(X=s9v=H{3!SYCq9qJDc=NHEBTtPp9rYqVMz=Mv&+ zIwA)xR@Qr6_x+dyocd|sI6p{bMsg_ssGJr@?Us5P&a^HxuiISWAQV!Na5fADgncDpBn zQ6Ly!nRoP9ZpW|pa-}o-(bc*e9L~_m7V%vNivb%|o|nXXb!1bf`Tej4v*X(3=300E zv)8RXX~#3&jR|@Ux7&SEv_rkFc&qp{58xnkvhd+WBp(UI7xmoCgHjd3UUNqm8PbE5 z*X5%Y?dD%;$B=q&9ll-SOe%ax;0h|xJVSJRJ$LhBXdbqOWZk4gQW1%Gy&|8!6qpu8 z^D&`usGI>gFEHA|g5!!dnvWP69`XbG*}y|J{;j?QKeGS!xPxy6TJK|$0vc{R6oHf- z>uYYrt956_%WD7)!qvH{a0ply!SRg)aw^_N10rUBgE}! zT~td#y+VE|XcyFQgU*F-g6tRhLGS##$;Pv&S6@5UBPOZ{E>Ve}ia6s6n+kp6f1O<4 zf314C`|Ni$K-NDJs`pdeYKc2}ko&EcId&f}M?cSe@u$T~W+hlS^uj$39aTHk@Wm0#G8 z<4Fd_5PmQaFg~Tkjn@00EM`9|BZ@%$>?~ESohuxU*yo-tj^8UJ=H!F=3c_Bcpm*{0 z9S4{YPSWizLdT;wI4;EM+*N|JgRfI;qf#MLPu62ssWz`ou?g)A5{PTN|=jLj^N3}hxzEzMLSTmNI?K=6PCphqi z#(vc6K#X^Kh(rk}z0|Y}dF%!>Z?_D}YpnMo5lQ)yFQ1O_!le6zb`thd<#9Vh`P~8B zn6+!ZF`@levdg#O^aF~}GT2$}p^yqELI-zGpCCRTdtZ*rOikGWMeHu)hQs*#i44{a z4_Gx&uLw?kv4=QEz{;>i9x_M0`lIP+s75(~Y=_jvh)4VoM{(Y$_z~KFHJvkUrukkB zj?xoDf@N;N`lk4T;`{aWYt`#cj%&3k^7#nmW8vH9o4=?zf^z2Q$GbIX{R-;=s@Psd zc=AHK|By;5+#1|5p6|zOFGTC<)<|2B{QD=BMup#qZs8? zT>hipGm_EG&~hT{ney{JQ%d+@_N})={}}4^LgK*m-tl5s*hTGHS?UIo@||X(X6wDY zwjAhKL^h0XkUw-}Lmy$U2s;D51dJn8O&aqSJx9O49exJ$ni2{?@qFz3O}A7SS$LMf zu1maEe^X|>{D3X+-R#7UXyf-fTt}53G@Pq-xzW$J>J@VGj#;%h zKj>WP_T@`Ny&Q*5ijkEUL++Iv9BYgl7`Vkh>*2-kW#8p_Xr_d)S6nswQVO{fsQY8t zj>@9r(IvV=C7pNVAp+>)DVtNkpLSB>XZ&fRpO3$!&Ng;bBKJS`F6dm5#b3V)m&3!i zJe6CkUr$!b%hP!-!oK%SGcE7r*6PK5Eb_VZt;^>;eB{&AY$oSFZJ|jzf-> zJGK@Bg&U_at+pFjxUe%8Z2Zgo(C_$a4GuYJNJ-eMwB_gf0%b=~&2GAC@c@0lig8uP zvRNSaCLMHNAstJB@JVaTX-VSaEB=0{@{bPWegzTo#)Uro@x>Iohp$age$8vkmM4?V z{uklry9c!f82L4?{_7nB5hu1o>XKx3;!b{WniM}c`~>wnw4Gk*3bIa?lYILtj)QKX zcvOI1Z|;t~pP$e1eEi#U7_ZLryw7Xg z*F_a9K$-3c?`g5-3yWxcB^B|cjRya>i52Ic; zN~wOd>?r{j-Qr*!Eq4gzO5Rqrz;->McNUc3x^ zi&t);^+EqZBhe97Ik521(>PC>3Y?er+eF{T-%I}fk0)tJe8nDA%C6+cJs*8c&P3Ms zJnBU)ST=WS4DkvO=O%YYy_yy$W)5CM_M5ndD)IL7LfIFJkT)!7{hEv*b_g>n0lIek z?wzLY;NTL?^WJFl{1?V|sk{2CU zI7ccR-2TkfwGV%<)Rd;K*n#u#Yq;r612b-~s}{~@PFu z#Y_1P^$Mqznkgq>1dofcth?5DL2~ruPQ8n$R|?_D-FbIQK>7N^q6gOQKsm$wDewBf z#1|X!E~7{B7~x~r-YFpOXKoc=G2BICN8Z{)O*2W~ZX$I5SDeCucG6KfmY=|ax;!*@Ky(f#wd8q!Y)pX{Vyki_j(Z>DqZv62k<{GwAbyLk+9^4U0YD_r2t z2iWavu-QxM+Ri6KA9;Z%>+QQLHPowiqN3y1of0^5b$Oqvvpb}&)CPtA+x{EecIi@X$c0g)1SG zKa!6hqzJzZqV?$R4ZAR2klo~Wn{d*Hpl^$)3#qR#Hmo{bnJHsBmUzd`LkgY-2_&k`_WjlfDCth zvB|z3oH#!$v5+|T7;jL4EPb0r0mLR3oIT=4hu-bH#tZl@KveG661-hApmt?MUUK3O zugca>4DKahd30am6@Paq-3LU!3^&&=0zwBPJB3Kh>$Js_tu0%7Rh}ry`h=V#E5se(bAaF;GPA`LxuvEt~C60UGC)^m1+dy>hf0OtUzV^^W30yG`10=VQgG9u09J z)C+pqLwkHO0|_vElb>9tOvb}O)|IkQayb%n<&fy!mpW|^R}V=1cQGVwXvWqkf%R;z*Ug~ zf>8mFCYSK9Uzx0>It$49cIj>Ve(qr0UW?Xldd$y{Zp2r`K^pG|)GiQUT@%al8tw1H zBYVSX*oojpfkLqkmIrzWRS$m`N4+|;of);_iot#MzY9vhh z$V`LW7jT+YuJR!6e55>Fo_ci@84oGMr{;{w!r7r2!GHi4&^&Q}m%=>iMfrfMX@z(n zU;^(w3$)>Z&kR--BYvot_PaK+g?VKDOE~FM+9h{*VO%x5@9^gOg(0UFs4q^&94K-i z6-3T!-pcD9CNS>Q4F?Dn+}Wa_gRa*e?6^24{!9kWu|;W7DWrhUKJ(vE7WjK@_bl2o zA%xWXet~66b=+Q^6VaE~_>unWAFor_BrGhE^9y#IIa$6Yi{_(`h{AGRBps;9h5r~$ zmkv4jfHyLF3Dsd4>J~h6dW5tQYL;c)(Aw$ddUw>b36znQl;15p0umKXDuJy7Td6 z@TbYm_ZOjj!3MF-`AR&d`#)U0(!$SKXC?_SZkWBflj>_=gkRF46dVQ5U5(R1zgsuF{7+3!Mg9r^#8{9?i=gV8#2f?1Wz1|eAl#kt( z+wd|}``sf|;S6Dw5B+-6(f+XJP4}e=?;jZFga8LgCT_UBWP5dV^2x^h7xRy8vQdA; z*SO^PtSsN=_#&VWz8xFphB;QiJQR!kep`8^2M=8N&F2Wi)Q{Q%+tK$Q`;sF*FVG|N z_or*}c5Z0)$ve3Cj>&;X z>V>McZ7G2Hka})`+y-wiGELWQuE_IxjEd~ZvtPKq2zGdBa1kT-E68@Gk`12#I-}&l z*D-ET#~_rx^vC-h1oWrh?{uVtrV)1Kylp(tX+39P--5>1gWEgjnjDIOM6)s>R?Hn- zb*5H4iq33|57zfj5b*39a4_@3DE|;?qy8`YuN1NA>9l@Fi1+k+b?iU;lK<~((c8Uj zymv}d4sKfy))5k?z#M5uoHi@|Ub@J!KR>Lk;A|Do>1Wxv`w5Apoe?xlsMq6yV>8Cc z`r&lcw8j88I{(`_WfMmHiym&;ycs_ElN;jl4S&%MpQXrptm+U1!{JjLU_E4I|Am_QyvqX@r;r5aze^6Jjz+Qv zxZLY5;T0Uk33VPU=iVpqzz^5n5 zoZsg2?=h3QSw)4pnDWa4gEx}5_Nu@*P;Xywg!YqfO!9Nk`jvg2B4$sR9B4If+bc4k z4AaWOelN=LuU|Ss5_5~l`;S%^`4SEk;NCA+nRtsvef8*umofYGmR18~KRJVt*G0-d z`iY@c-;y3GhUwdx^TkTVcju-xxN>*SeQ` z!Tu@h{;lg5lbJ_;ElHYYV7oby68Y&(}`w8uO9<$f26i`FXNb*vd z8w$61_}O@(Ui1eKjfK$^gNAGMu3>(6s6VubiGR2G`MB;COg>1&hP&p)PL<2{$5#uPFmH-{=IJ17WUoMxV^sR z+)N}(N4-XpN3LiSxEs?~}VhXDCoFZ?{V^`6CJI(-#6-}m-) z>22hKxZ|rR+fH?4Co=!laGlqgS_)PjSGF;9xWILJqP0>cw0@a++_h9Qq6N9@eXl+= za6zr+?&$hI`ujq+3a5}A#jyL}0qsdHcW8Nj-!vz3bN%vZoYSYE&A~JZ4py%s_YrTE zkJ%KhgHgzQ$xKb{0oR@A^HKb85?N{;2E4c?54pcj2KL*fUvHV=@AZT8LJWLpdaquTE?Jw&pvJ}Hjig8pXlZZT#w zzC2wb)U$sULG8-L3{nAi=<@3M-A}%`|62FbnPK#}*uAya%ZjuNRYT;w$54UQsAp(C zHof%nmqqSrIj|=;SztOD9(|`=F|xqli*E3zvH)_Q4i9&y>UUk-UM=E;lR~1X*F)s0 zxVchfzEIlTe;ILH?++()q-MFX`#{hC(>+yxF31brna_0?^`cVm*b}k67_O!X>|jIc zy@!?3)Q>~B<4eYjL+5A5)?VFQm33@7P7oPomQ62?=A&a-=xw_TvamR`gR^}{3VdA= zGN<|<_J8$U`LqS+wo;iNV4b(YD+=;)$cW8Wb;>67d0=S_5Sv`zTM?cGVr)1WH$OjGPwVmUtvb>lf-v^qEk;%CqvZ=dEeEv zIFl2%m#xOvZe>-}Yu9JTA>xyc;KLR8o9z_pRkN4qlELx4KbAu$-3H4o^D!;V*2?%Yvi~kem0$cZ8efuq zDrp&nvQX0RFgw-O-tZ+|u$@)D zFX#Ibq?Eb3WO`$k9`bj!h-{sYkA2>?=L;eCE1Z3i*)4;Pw}WEsi*u0pT0E;xL+hFh z-y~0~H0I;)RT`eeY=+ziCK=(=Oh$;?>%%wYoWz$%9K9es(|_~f}HpG%igvSj&1J0 z*1hsH-EHPiZS7@vJbQQ43pZF5u7rU}G#_hox+QLND1f|vW5?qg$)Kk1=hQHOzt=oJ zy<5Aj6&$rV)Wau(+e^RNBI^VlK#d6f?S132K;g~%gTWkmKfr}zR~iqx-yKutUHr3` z4q~J_SKKAIKsUKmjRj2+s9Da(pX2rbBV*y>mnECMYSyYo;uvBv6TeZy<{q8CnB9Q+(Jxb`%LV!K zKJGUZQLo#TVQkUvg>bd<;mT(U4-ntmAe<<@xnEuPQYv$ak?7gFe$@rNbyfQ04tMkJ zKh6w5`_-EU<($u715|$CmU)9shV8#}^DboI@AcY+@_04krJ{YNLGAx@{)8fx{GEa) zsF!QG`U@6%N05Jgy_n;V^F$3}&3=vv(Siui$@5lnTp(uZFq~nHdg&)yKQz5o2uVid zjs6lI;Iz6I`&k>e*O|=3PRd_fd)?#Hx2__%0z{d!l4l#xe0=_pNbW^3K#Issj}vCe zpbWn*JS@Q9>!z2fC^>SzZ{&@yrO=2wzKoLF3x*R=FZy8f5|wujP*DA7x#5rf$l8NR zw08rM{Q&wA-eT@tP}r&dE87_LYX9c2T_ms&MA#`f3D`V9iP6R&Lvphhfe&?Hf}k*l zinU?yllNPD$*d`GsC2u7;n7PN#SqlXJ=(kB>Qex{Bs-h7h-4`HCY0Cu7=JJR2|dxG z6l8sf#?6t133vZh>Ll{?tZKL;Ojg9C?J7mbkA2M?JH($cK+o() z8M-Un@bN>c3e#s0mtyWa;>NTxPFAAM*8-^cXCVXo_s`M)mR+N&d8 zK(GN>4^?@bM|On^_4>H?_+SEZpOjDbQyU7-6sXNrPaUbne?Gc6|4Z$#1s3A@>S(f3 zaeGxnE*|7^MZCmsj6{i_hqX1Tvogs0+ifq5XGud*FR9R=C(~0rpto?IJ}{gYe41WI zDmkP3ebw%{g~&ZG1~2~)V?070&?sg)IM{2tY5d3@!r1pNvIaA3n>c48_g~EC1T~H( zh9Kwh_ZoSNo=5jbi(E9|XT1fmwBNB_l{*E9Z=0G>>f;|@W1$NRyEU-DeAspAU<__A zPsvEyTn>5AHa;cstIGru_Pj5D_st1XcHh4Bb}LMHGRgG0+oIgyJlnX_hK1sIT5Fe1)fv=FlUe3%aZEq2LV~khL@6H zeeH}YGM}?3HIw+qc?jX#WUmwxu)#SohB3uzUZ{PErMS_JdTFxmOHN-chD@#KW6ts( z5JjfE{NBoZQ+yFTc5Bon3&$u3wikyhZM{CkM|ibmVEzi6*4x$VE{A%FW>2IlAlJ}e z9!gJ^x}E}AE}G|JQt|g<{4~?~OdbooyO$e+@8kAT+Fy9yh6AYs*E%eS?#scXIkl{i zf-~^iUHe>m2z@@r1a?TDb!P>Y2ZSPX8N5(TGAJr=0(~FZwB`-HfmJau_8I8z*71NB zy&>;zS(|P4>JmNQe&+@zl9HBm=JeKH0WS+CC49X=Q$-6~(uj^9*$G5Fx{>=>E{vXY zZ@-oTrw$}}7DV9hb(CSR>#hA*$bQs)k~9kU`Kt#~y0o}Uk@>=ZygZo8&da}Yf)z>A z)S*VSzc+kOEd3742;2QXP#Grk!s@PN@{%;vEA_P?x9iP9&>1RfdgAW^9tI~DXooiE zBf&pjlzCR>mcm)OWO5L`3e4B zL4$!70f&$qm~UP`b5z|CJgjAii4#gu`i>R$HAQk(4%qF&Xuuq|&V> z+0G6k3|!_~SR}rJgMEcYQ7`dX{ktZ5g+SnHH~u-)17zr_{SL_g%Y6Zs3|+DrX&9~x zR{cFHTYCjRJIcGHyxHr~%+qZTcLT_=o_dlTo&uCNmY$Q<6O&7~bVya$}0o;5xsS+lWU z?#!yrOzc!4{5o$fHWG%+ufCRhKEd@bUeg+N7gfSADTM{S%6GQ*;xHtCEb+=4i2Mu3 z#NMI(M8iRiuNCSFVC-3|aM(TtG~@WEWS`^jbt0h9IaC-6&or0o)3tD~r?L8fj-?Sq zy-X^jtrL)Zyw*^2Ob~f)uFt=BdcQ9}CCLg7+I-ZJfyhK++63wAX7u?eId&^*lMIvjYmiu5#>F z`u!A03*YORz8C*~RWFl8-0CtG7#U@W<-Kuxbuql>OtnT_o>AzXq>+UAx98qlYdOP& z=$?~bub^JnYn=SsYT1E=vzjD3e-8xmBu>N%O@ww zaC=32G#TDS?x);peVd)YY{lHg7n+ilBW0^luNv$R+o@jw7dw(~kRbiU5CP$f3%2;j zR~OS0;f!Ngpp>VJKSqh$i-hK?F|$3=m;U2rU39*f)Y1uWulTqQk`gIjxTf7-uR zFzExul=Q43OK3mwo>RL+NKgR=_4!)uQd7XsHKX~dAO2o(%H;PiX=1^v^6l_2J#Mf5 zW!1}r6li>nFYHLmB6SAu58;pGSkd^(+}s%MPGZZy7{Zr*;v zl2!<>&9WENJv`vS&GRZ`r#7z-5&ToX6s`x-=pNbHi>s+w@7qaV5D{<9uPjBqqUGI7 z7UC5kshEJ{$b%Fxb_*ysDZ<|?@yLm!N@FaX7VG>LHH+Kp#%rGQ3ggK0v5)UaTdoSc zFei7}*5nLNTB$F?0_x>nF1#Gh&jw4Z3A2{ZDPd90u-()K~YFPmY-ZS71P-GesekG zH6&dMKudJI@oIVs=vt`JNfhGm<#g;bpN0_@Laxgl?4!W_d^D{lWR>$mT(}cisfmpM z!(K-*Jm(AoOpSa&VW?LqeNFQ>y8WE%UTS`y~H8OZ^j(RaH5)@r~~{pBwZ* z$xIQ5`bfidZ4s|%7bc7@{`*aS*YB7<6^DhcONted%(%S{e-$LMh(TPQx_|$^stT&I z;WxrDF0d*^Q`Vk^db#^Ojr}^o3I#=Smk2xe!l?b3S2hW#m-(6Dsfh=LFy!btSb*I3 zG%>|k@K$*9_|a!5T5Qj$1dN1gp@qks6+|U@PaJO$-EjE({KDnL`T`*_KVY2u%=fVu zjW6Y}+eJ%t3Xpi1kIeF93S1+p?~;hY-|Lx2d5Clf7Mj~XUH&MJ+v^fL!RKCj)T>vO zN7lI88G2PdWQy~nUW)~}oXx-4;keNKiruxe@Tk0bXP7VQ<;}W#ez#*G7+>3OL|}du z9Aa+jaIQ&QQov_R!)B>J?LI zrs_G%0%7(Y;qiL3P<-PgUCJN**9ryY$7P#BcxzL{7$xEfVT6Y{BZhIuS2J_)e6T)t z%lJ}{;kITu=?7Y=6Zw*YXnc*wz8SrVoL3`ZJ1i@okOHZyZwBOv+0^b(TYz@dd~;0+f#Kat8X;#2e|O} z@(nC(yU>jVH31IGyK)5^zt7+2Jz~YVh-2#k^1VD_e6!4<@BC2{(X;O0tYG6{+lzWV zihCP2aGMiu8TNde*P({^o#uvzhS2$F*(3RJbjbUUmtBdtDi*!rF5RkP5VAhCK0a9Y z@_A_=oO~t@)3G;Ku0+-vYAE=#unuDR7xP- zH*~7$a|(oathOJu!N0#REqr;9p$H2Xsd5>)_u=+hv^ua)CJOc1=lA$&*McjUN<^#E z{^y+B|NC0>9tjJwxGTU2n6dED&a3;tN-?}61yC=(XX#&NYKkE0V%*LuX&>!4mu0`G&g+1VhUWl!rI5??l^AP_F}R<8H2*Mey!~>kiioJ}`Ytf%j(N`3>*& z_`-yJz4dZO8s;Fm^f=ect-Y#Tv+_HYe1Kgdlki=;*2ecctS2?f_(%~shalqc2y&i+ zPxZG9!k_qiF-oL2jN2jgtCa0)$zI%Ey)}Mr%gBDF^?cOqG@Rdm&l#?KJihyC5?b#G z#^19v_p!j4^Wujmv}r&Sy;)=jNmh@qOzYpwaj(Bj+>)giQ_qUg{ z!l;sC5OVA1LCv;CKny54WCYe|vpS=k;e1+S)6CNbX70q8Fsv z9bg-|fUehe(#m|L{-6Z8jO1;t+f(7#bk~)q68QHMXTI3Y79;%Q*<&HhSfhz`zAIr*P^)&B2m(R$JIVo)-|WR7{q}d))Drk=}C8tXF;`AAOACUsskP z=gt4fiLd&O&QCP0)n-1Mp#y_+rn?heXrP&Il8*Tk8egQtcS00b3!#OvIhond2U0RH zx}7`zrGEY6r7LMWMn$!?SHu2GcdPWhp{e)P(^H4g{_Ehgk~|CK9*w@Ci}eASDX^9< z^qcoA{$5-k{06FyV}axrcEcOGys(^(_WGeO3x)X{&*$A7%2h~(Au z$hY=-TV)wE7vKq9vr$-fCA5CA6wvo58Yx50-f`JC$a)6@Q?ZZE3jY0>O5#r9r43{EqN)^P#y>r^NnGlquD%iXZIziIQkrz)5Q7?Dij*q^zOmKm-IJ$Nc zIalHLyIZA?P%jz&h36?vh0uET?gFus4=`&UmmfKX+iN_`>2C2!#A`83uc~VSv!3^V z_Y;FwWj&Kcp0J&L=}vhkI-m1l|M6K5Lsg)ALr+j2mIBVJYA4?K;~!t!MUL`Io<+t( zoO4DTD!9G6OM-=`5+$H!TqSelvLbv{d(idnmlNz(fBtcMXSkV-@O!NmQsSM&|ETu=gWX5`94SN}58-#AYu7dOrOd4%(Q$;kf=qe=HDe z-^JaS@Cv8b^T2(6<;B;)EQxV!Ar-A(Zan^qCRwT=%@nyRg`79U&ca0zy@iqmWRqu0FGCa=NMu$9r?*P0vOFM0ZxQ?0%V6g!68F}RJaZ_~b` z5c(hPS8(!(Hc>i=h0qBSCe>)%_3J>-X|BgtQ7;}HQtWFFCs0q_rcrnk^_qP6(xdex z4Rlhrd#J?GLFe&Yy3|-SzFdux>Q6pK?tg46>f?Ik16=jwF$1Ij(%=8%Ma0Pe-2%D4 zV=J#@b*Edu3_XE0ritR~E_8iJ>6aVZM4u|q&J|M6RHlN}TElea2mIrU*0|W!^(q$9 z%kN!LXTTj_cIE^xxR=Gz_)4#{U>I?R>cp-)qok-;qQ<_525a7f!hVQeg&&+j;}=e%882DI?V4o9ZDqB)?0c_&Ayc~ zv-1MK(JQ2}*U;zVH>R2LrE*o|UgAXCg7Fkcu(I5Kx%)ES?{^fceK8~V#)1mR_&GZ| z++IG79+UcfXnZ*tmAI~&IzuR>apGAC)XUvyXEwoOb|@I5(^8^kfHs<^D?R3DKDunr zRprYrf;AD9Wvib);70D3vJW|r_V0XL_fl-vsS9MZ27-M&((ffc;mpVURXM2&20!4q zd@`qo2OST!7O%d}&{l&N5>fA1kog?#rjhac*YNjJ@!%}k9)yK^+Ku)ZvbeoSSp(z6 zZlhim%ss1*G@M~l`n!HfEb8SmyQs|jiUID67uMf-#{gNC@^&@uXg-dAzUZp?tOy)$ z@+hVY`$E;U^?dMo-0>xH>9GaD5yY#CJ!k3kB2KT%kDt1gJ@AF7-^{L0nb7*>B$O9L z6s!gp$J-CYKTUxnV(+LIi}CmRHs!mI;4T&vCRliO{(tT(-__!n)bA>}k&itO;`>B= zonVfqwj?}Iaw9(44yVTu`*Hzu@bzm^*$fadFYJ^$jCxr`%d6?s6~U>Go>aG)ePO4D zq_BI|W-p%@W%DmrA~7$nn$Vw6*xIX8d)f2+PhXg!>1WdlLZ6TEy4Z8CKB!txFfg3|7ZM|=dYqG#)x{c&TqeRQQismP*DV!@}pid za&!)}44lAjc;_*%5CiPao%?F#gyy5}@iW8jDMg@XHTOYx%?Ae0#%5GI;Pwh#9(ye% zzO~oSxcNygEq{om6JZ{rN9$K!M1?n9xe9p6ztJKpOM%?S=c0bb;qT>6fAB=aIV_|W z%BD7_;m$`jj!QbKchP(-Ncjz0g)KU0)6jgxusgrijA4cY1Cgf)Sm;6DmHknq zDC+f5iItg4z6ex~V6PXE`GWFc;V- zOA##Fx$N+MK7Qg5i8b!D&!p2ZE9RpZGW5xbIzW!V9f;CPu{h^hDgO14ETert3x*DBf(5i zUsr||=4O_Xq}fL-m>uzM+GC8IoAQ5Oi{5{ahd9zNe|&q%57Nb)ML5r*UN3D<7kKVAR#M}u<_JufJLXW*9?#d)G}O^!aEQygb0aN(Ha{b2?*oG6HKvW%j{h)GIw_M$N{y7J)+e7q*PA>;3)9;?;g|P9c&-A^60Gzm>@TBE@D^7W_4aZ|8pyd_lDF4bA+PdN? zzDQlz`1{u5tJ64ua6q*N!%LOCXkLl@J7;eDWmnww-h3`Mqae`_*@u%-mYI$AUug^3 z{VF8t;Ou)`IRuF>!4~pQk+k@G>3H8uN30=7RvzPyFNMBT@V4 zb-5#)Q#G~RDT&5cX6{u3*&PB<-oQzmahDBb`b*W#y-_c=#PQA}{m6M~es@JlJA6SS zhe#6Jb#arI&)}Pv{2F&LhozdiO8>L|y@l7WVfBf^AwN(nnBw^Q23>y~(8vfBs#FE? zN~cqeW+~7RXm1!^Ws7$|aXT%!<-jE@@TDtPHt^&2(&YWhEgyvR_y2g+^HZ^he0Bh# z8nW!ZXtaK*^20^Ou}X+(x|~S_eKpbX5(+Et;6zucW}q!&GE+-|_cKc;q+w z2W2I%GafaN@T7&&m{dQAf67*M`4W|6B0~ zdl0XLuAAqx7IAwGJ_^)aR+j?IJx>MZ6N<3iKEYIZ&=J^Wu18iIN^SgpO(UND0m%DS zT;~~gJ}ck==Bads`!eW$lSLpQ3hpg|*5}~^+1LDF>IdfY_{6_>g*C=}_sqvgeS5Ht zesXIs69zv6{wvp^2_uyZ@LfQSVmWE;> z;`{MaPk!L`BGIbsKBI+tMKHO3Hag%4k}ZCW2hO5i^yv+06CLc}Ww9S?{E7okFG=~7 zyh7uPj*2QIEWZS3Q!!)Wc78BiOa81WX0sQ86=OMT^)<}c&W`ET_gi}rFqP0aDP0F* zeuwv7U(n|x!|XBcsZ#))M^>aFJmR6yP2b?=Mf|;XIp@>;x`G9A%5>!$HW?dn@^}AL zj7{G=*p7Os4IJ41#l{gHggOg~yhFY6!-rpYDA5AnFG2I@oc-Wja`gHzZ_&p5YWop_ zKt1ICwA82f*QoFK0rkW-qy18w`w4)Yu`?&TG(*74-@ddhm*T zUopt~b=+|jLH19cpMELyFXN$guUFpuI+C^*w)Dz}rIn}Od|@Ov>*Zn9V;g^m_IR`X z5grV%+n=~U!5$9^)zxc#I{16FnR67*GGf7Mx%`W58g8#W?vlUM5>c=C7aCvh{IrM9 z=O*sF$wR&Jr~AS)FwBtbDxXm+$PT8huWnScpDD<7#LKEpJ34Xm ze9pR;<@Jwgceicr)sXQ@Z@;52$i+OAzt4cq=M0S!H(f>E&kUvR^1c5)4(>jF@Su7C zf3IINzJgdeHgL-A|Urax1Qw%<(XA~lZ{NQEqLPK-g=JEEr*QJ@Huhnx~d!5Lr zPw*r510p@6a4Ao8{+`X@qFqHK259&Fq&52#2O>S*dB1z`_agHqJ7=H!1Qh(HR2ILikxl4+B+%Z2;T+&sf3dQZUB)t2<5GPXaQ=eStDCFG8v%mX^&+^%Q zMDc!Ldt>lilPp@lq%!&0(lg{CdF-O_&u?+yR{Uv|_8$ITUSDE0NK>%TZlj@Eu87-< z=0FC^@&VNArB&*FZWSB2MPQuK#D{u~tiIpBkiiTT)O@G!HnTxTxjwVC1nQM9zC-2x z&SH?Nw&h@^^#cXE`|2jXxbv~`Vs8mW3=&^kc=f`92amfia-L|(F?&9=em$5-@q9Zd z54*FT>fcz7gKKHVn8W`=y*GMnkl~K>U-jo&QopI<_HyGDi=n%UdWH3K6}q_CKr7{G z$x=7ei`|a+&PpgdWcI&TVBO6Qb?niP&3BSJv@Nh%g5zJ{iGP`WjcL&sx!a_R-ZhIEIx#KmH$*p6rkJ> zpL8Z2uQRcO4OiB$-Cqhf`U#=QyVT|VMG#J$i#>t7kCA=Q`3!3kZZAU4)Px1vt-Tz4 zV!w*@Hbio7`*dHb)AH|4zQMq<$G<_+8p|eM7PaZ+u_i5Kj*~5Xm z-s>@04+yRz^G#diWBY1?-nT|C*v@>0Z`*xzy;kPZskNy>IUs!$Zst%D2Qw#kS8GY( z@3kxAzW8nhj72;&OQJi;n>ScLb- zmSgPj_*KoQ-5iq5Y4J*IK0_t;;a@D`k`jz|5hcsKd7}79znjaz0|4^q9y?0Xo;+1h> z;nUHlm};VnxfpJ1_}Xt%;_{wn*5J+%nlut6vhj2N_KLbHs&hEP51D7-NTpvu=O+S8Ebc%k`3JWQutEb+6%->YUP^Sx+feVe}HXMvI(Zm$O~;z=~sP_JD}Lj$fCk^7PJ zAIei`pOs-13r@S*%o#N7x>5ce$T2?t(uHgU=oM3NJjr7Lv=dQ}0 z5|JbTuUCxX$48jKYKO8as{`s)$+-9Dwu8m+$jT{KM93exf5GzMC2HJW_uRywQgCap zWwNy$5i)*|RGwUGtBjuKy4b}RWcUaJ?Y^b+8C3Bwa-9A{w*dZL)HBIuWk^15CsB>x zhs7OV@(OULYaVgwT4^KNc@WA@+-`C8w1=fn0W`l!j&68`s~Tzc#IXWVT-3?Dhb&Mu zVdG?uoQJs{Uvs`he)K9x{W3kMI5KFZ~>2PX* zU#_)c-}pIy=i_kTplKD>4?bCZU)ZH+wDCF3JzwK0NHH+}>|W?nX#y;)Z2zQOWsUdz zz1?@kkkWb=A%``vAE@!}zO`@TbANl;<>hG~6!rs5&)y3`W$1W_zOPLu?lgdxUbFJy zwFKD7@%EyFH2(2*cELBZ6W2o0SCP_lCtr#9$3njYY*bRsE z`Fl!%AO=0UTFi0NhDbYP{jtVjMOTZSZsT)*dp-VjjwFG=A21ic<<-Ab+xXo5HDBXz zy%_lB+9kC2Qv%E?+BYvA#NW$pX#4FkDr@#t!x5oLWF-(h7%9CF;}3Sm zw~ZvKH|OKJS5ODBK)_ez{l_i5d^A)tD#m@`kxy{Tsv6pV#T<9#K07K8tB;M9r!o>i zht9M4Z3+JQ=xmZfG+KoP@kr_W9CNtq*W4@7s(@DM4X-E1TVD~R*@GL+1<|2FwBAz+ zo)9jn*#{M)#=H8MSi!0zB*tzA?Y}aWjrDf86a%HFo!)4zKfDhyGd|6SJHA%+?i;Epd3%t@=fvs-)Vu-&G0sqh8MXrt{d zROtQ*ogwChSFhy3#`mYvlk5br@5`rgNBTK@&%bBp$I^Tl#zNN96iMk`-0?LT+O&uB zEgD~m(!-=r>n?)<*?7a7X*9ktq;9+lAq)_mR^G0d#R8Ee38XvUqh6|}X4Nfv#n9Dm z>1e;>I)r>v`Jk`Dke!#diKd{f+oq-hU?ZD{?PH{)w#V z)$s&~y_KQn@C$!07Do4KUQ%nATh=MF2RGi?7Z*u@HkJT(M_6lnc=+%g-wGF1%LcjL2O>Rt!s_q`AM zoj2#B{fi5t=2G<-7Pm~Vs0?dlKW25>K-5pn`g8Q}djGPRoFv`b7hdEJyhv#?-0*57 zEI;w&1P0Rmb?h<-5`kFz%`3}c{Jm2AO=J_^W1+U}M--PYZZCB*R^dz8NF0$K34EKr zgwgG*pA=wofK{dCBi0pYe8rg5%#-?X!9$68ijQa6z}os0Cg?osHQ0RD)EHS$OEWIn zA?F?dm*?O2Tt19={NLB2_xkg3-D^cXb9*HfvR_~eulZ=x(8?M=c#zz2yxR|rFH(xn zYG!BUU?9y*E!H6cyuTi(ibLjy@Xbe&Yhq_y!?Ez)=PqOLLEK)!L2m^c(vWfWKVAj} z<0JbS9bm35Cx^KN9d8qC{3@)!uv4E#T68%gjnvcxnlbyE1MPNl67*)J> z0~l{a7k9>P?(f&VcA86VeG8jama`gCtmV*0PSS? z@WVWR%;_rnMa?6?%+8u!O~ns)oB^e6XXIPtAl)c|v^eByVJ_}e1B?%I(2WW3`G`gx6RD!u6Z#8u^iu52zD$lCpRcw2WI z=#?><(p<;iD@fqGPtpx6kg8FCsmRCeb>*Qj_8aq&4X-_&WOtmq9DwSi5Oe53bi7>? z{UQ0tsqYwtmnO0wcCjG)q*jT_iqZI@nrnL-omB+VfmfZ9i*AB)tS}wZ)aLosbuUpd z;xSJCt-UA;C-r-|ec{ZD8l`z}w4b1PJa5?dRvxmG)m&a6>uG1PS?U^j_rP1xbPl$Sj&sfsZ6LCQ9;j;F3511h; zqE!B9E$VfPjx^!~S22v8I2Y7!p3~Itau%JBzNCGNN}UM<9y3x? z2j$}7Y2%0Z(f^_UnszSE$U)v;UCQQj^x??fcuxI2zuIM_?l5JK#+Rr$W`aV;0jh?R ziQ2u;dM}j}6C7a60iQ+2Z0mcN;cGrw^3ze&i-$Afs-aB@ z%_F$IhAulA-N-ru;mukTIncF^gIDwNHpJ%)q+OLMIjVWL95{6$B z0<1}+Y{+=vFb`=C>NW4!AKiGj9LC>SkHy(Uz`oxn$^*(adlAf^Oe$N;LccJa_CtEyj943{qbMU&sq2CGgeGwmayJ3zJ9w_7M3IP zY1emsit2}vpyN&U|E`^5P?eD2Tl-0%If=jNXCI`8#9Yp=cc+RGA?p3th? z$rM|K`hBV=OPfj&JDm8+F8<~Y8|>)P?a%(ldSB*jwQTX>I&e`gf0s;q6;qO@l@Li zrsrZxV`P#3a@+8Vq#wBPGcg>#H#r1PlA^pG*v`4jT57_=j+W;aeDjh1A*3^>kopm> z@mM)u{h;H9BM3ihwu`32&dbxWX|!Mv<)vaqvfIPi6Zk@khA+KDc~M-k74R-(g(Pko z2|s@}WS`n3>C79HS8Ugt-V?la;Qi91hp+D{)V|HQT7CUr)&uxd?O*MTz@!qfP^cj5 zbGGBvsc$;e;~EBv-;Rsl{l|Q*Vay(yMr%!=_mE{Ud6o~9=|(x)9F92i>c48kLifNC z#099e{gtuvqPx<2Pv8t1j~unmVR=YD34(?0W^EuktA(D|) zx0nL$?<+H#l3}V=2hn5|gG3F{AT4zG>$4zY#<$tkhucqpt`b>WAVCVb&^J6x!d58v#yfP)BkUc#!!8)}gK zd~%sL?K=pt^CEXlELW&T_2t`jJ^`{hK(1n$0!K4q8L;WX`c z%AW=c`Yv#z7FE9V2;=@9@$o%^@`f_0MG0gcD23%}Q_ENFvdM5l^K0=Kv>Jaye zIbcaRALI)jek`xXofqEWZWuEopod%!R&!(=`! z8jl9YMVlT9(u1#5?6;!mWK1m>$Ik16R7H4#BErSb@!Zg(gCMAqGfOk&4yS+8Enm?_c?H|1 z=+(Ssh7)wcQ{#l}FfZ0ypGS=H((ZQ5+8taABY4K7$4FBkI5mM|0y#f)Jx^Y*ui|-3 zeSZhxm^}rMyRf!BFOIXFgm-sFz{IGLbN+ebjlX9<2@I_S8W4A(`-S1zJa~3q0*!`>A^Yr&O&ycwL%S)P0%E^7j9hASU#u!eZ@i-XxM1XD| z2mIn?^?0Dd4o=5e#%rQcUZX6LO-Ikwf*0?rjPqV8aCV~VeXG#F@G{7+kP$nL=`S1} zEBLiNFPSud;-y>RAS9zZSx1ffmuisAo_iwd&^lo_NC0_|t^PSe0jW>ononFDHWDyG z`ZIajH9V2LfL&kDKZFNpK0dg?>*vYJg-74q;L|ngDVmqazyGbj+d5nvN-QOT-)b3q zR?f4*iTV3Q-v2n?D(c#8kLYtXknJ^kZF)~Cavs947@^44c*Oh5%QV8TYY6GTvYmgK zT=DhpTnPjB^iQb=F=#zUhu~~K)tV+;n@K~Kn&pC-=`^#r-$|T#O;|6-4Y)Xh(VUW5 zlNxqjJs(QSicX@uqTfX-(a*br0{17oN_UhO-d&TL+yoM6Sfma!l4b>QLD!=(VQ76K z(eRMlSI27D#SxaBbTJhQ1y=)YW489QTIY4tQd@1;ABSyuX~y*(j0M2>21 zJimDsCw!yI!{Eld=TXp1E@aOa)=9VG&WoB(xE2B(q3yh~BXf% zN)*k-{S?^sWhrPGD2wFd+v)2x_3C_|UKr?Qo%?-69<8swAbtJ&q8#$Rd>=V|e=Zk% z0z=#BEOF=6TH|xk44MBc2nO|TGwi(fzj18kd5!XV(Ue^KLD3!F3YyQT&;5hfTuM7j z8y$r29a-^3#*a^G2u%EIP+kI)@uOy!tAWz(yY(BNbT~qH?x@f&?D|qJbjdpO3aMXc zPRcfoe!{HB{pNi2eT$W!aCj($llIE))kOF6{gE84DBQ0NdxqxkTmH@kvUfV6=P%;U ztMBFrUOby4?3s=EAfi&cq2tZtN7mBcG(P{?SM}HLcY6<=C#3KI@~4mWD;rV2$Gcp7 z&hiWmyjYVE7)0{?xl7i+x>itL_v~lYdySFjVr|z>5tRWI8bhp8$oaJE95&vMANKOo zXOv<1!)g*b0I5fB!>dQZ=J-rhC@|jK|4COL%_mx=moIYBYD3VG>5ug!dBE4-7h&VB zjkA9Vd5R|ub0g6@&7U4n9`(*G{7#UKtEF$tOTgi?!5xe`1h{%K6rF{~#yQ|BL&S z(kn&myvFp^{8$%|xcjRwE?0a?tDPP|#n~v9yo~(&e?QAXGy271@41*kyD^*TtP?ws z`;V*`jiCC9&5zv~cB}^CqxSw7SV;$k9wDEyHthPkH}CmzZ1?uOs!5J3B|C+|J%gBA z`>E0S$BP7eFaPOLgHx$5imSx)Kr#B;?YIcsdF|0DYgR($w<{j_zN-Cy>d}AnU$O2@ zI|Q~wu^bBJ7*LE!ojR=Q0h#k0-Z}-SzM3n=>pWbLog3YX15wEwkT7#JVn->;tMP=3 zwEl7poXEUztT`#M;1c;g?raPYI+ zLqAfB#^Z{B>Ki?zA1g>u9>hMhP0o6OJFf?=k51_#yyCiNn-oH@`~8o1d>;#V z5nj7)>E^7zHDFM<9Pj!$9<00`Jt*dRgC#@hVu zJw@}?G_IE0KVPZ?K@MBl4$?dbc^~%W@?qS0(L7qKwn6gyZVvj1Iz{aDRdS=*fE>z$ z8@yCIikC`M-653RXiiKqMhj*_ zzOR3I_>-;ixXx>+*!IF(M;GqKNIQ@X+LJ6PoD?Z zPU0U-^TnOluLm+}6v+DfYMzd-Mi$t4`9k11B3qQ#Te6|W-@@)Nl}(v;_7uts!#zx{ zNJImn?r&09c5#4=fCq6qAsUa}9+w)9&eT8^t*Z^4PA2GhPabbd!OqLU>-!*!h~~EX za;Y$v$lDPKbQC*x6@5i{`S?YM-)_(Vt(AkCzRYlMStonH>Q!x`u6hr3KrUV}DywLvRPP#U;8Xfw+WSKNduVJFJ# z%A7WD|H^cka5e5_Q9Q3_Cw+4NeHjcv_Tc$w6XI_3eYB$MEWqT z`&WgwUZe528=SpxsHa^J<<)L#?50i31cg?2d8A1>;9P)MF!KT$k1Ub_!4(rVkRUW` zBAK*3FT7{Iy{ux}^CC|vpl8V4%B$6h(E7

Ny0niy^n|2bUi6IEna55#ae!j!FQ zZf*qj_v4M`69w0mQGGSmoL3djbc20HGmoDjak`#=O#OXrJ4tif*eGeKH$#lsygNH05V^q?dI~)8JUMzzwgcCp_dYW zu1546go(L}Mx)67k_9Vz$4f)*kYh=Jc|C~oqP~3Myw?p*m_8uIL`2C6^HMdVhaRK( z#AM5l+^jdXAgkV=?;)Fom6z?+1cIZQRTx)4rYZch2(N2}fx|0g8`rlv|LTi+&?%*Y z^ix<^IUn=_9Y5j=2^X3RX@Detn_!+;E;PSUds)VG7-xMAHC9dzGC6`xFS}DP4R&77 z-b_-zCPebDlRegKSsIXd>f8NsM-Sks*H*krisq})=l!M3ZTaEZn65#u8W-?IA0%j~ zM|oZNq&!cMTn`EK(On0S{hQ5Q%Vc>+xBC4$uf-tY9Pat;c|}Rz5P7c`3Va+Z`ZE!z zz6zdsTb?jRH6Y)1f%-xzGLTwcUtZQr+v+7SbOx znQS?v67??u!?buKNey@;mo^EHvp~SsNqQ&;cV7G3g?RUvJAv0dRiUKP+Kubk9FI}s zUxs$fB5~J(A9cEn9afl*>g|wMaV^%-=p=3n9#AO^P(ECn~m~y9HOswS^d@eYTS91e{%BmIiw1~`K!P4uVLq< zodIt<{_`IE%S)U$aSDU<2auM#G9PAy@{)hv!l07R2CDpeV*9MQK_dE)(RH;Z2nufH zToX%zQ>DK&7L*&|3&FjNAI90x5V|0hn}(g&)3I>*iqPh1VQ8Al-BxqjyKQWsO7!*ka!7YouY=xW3K)_2D?b z7(ZVWur+oirnZ=Dd~bw>mbBGa3xf9?S9~y;0cY+jT|MoIJ1>bhm%Y5i)nVr?mA$*a zW7ikA*74^?$axfa`K1&b?@dAA(){4VPEWX^{jmEHKB}*&$2ZE1WjMj$odx524_+{r z%LzL2x&aOjq!kRmO@*M}^b0zaO)!(JXvodAJ+BKy`8FkahqvWLtFjn$(mWEzwHZBG zmC$&6Q08ZlFs22{wZ0YH0+}#C`5~o%Ru^Y|O|(ZJ?Z2!JnF+xzLtkn)p37!^F&%#Q zYc2xmzv`_UX@bU=@NY4-#7yD zoQ?2aRcD6AISoi$zExigX@)?zy&be;Iq)M}r$^2msW<=q|LgT=jFf?C#)k?_7N@t% ztM@vPOL~Zl?>F}KUqa`#j?g?>o6e8*;mUU6e<7mHepM8DfY# zukHgH@ija)u0deB#4YfG<(}Gy*c=GsCa0?W(S2p1H^cvh&3eCwK4x|2eL%JI0OBW05zRb|wvGEQ*yLyl;k| zg&d6LCv!kX{4!y|+Q0DfT8$HliNySTJm|x^fYfKU;pO9zOdn+w2`1s=4cZlG{Yj0acd+qK+5S%5lBkmo5 zzD7@yKlfSC^WWT+odLlu6s+Q$?5$a#g z#|Zf3c56fM_~3`TYMJn|q|A)%3humoce!*t`*jS8+!8zI;<5AM%kH!n$T|cQ*_YKQ znT%n-bze&{kr&uM>1)05kNQ2^S?;{DRSu9%x!O~Bl@E4jtN2=8ZG-?3{o~zH=`j4> zeDa}B3)Id9T!=`>0eiq(T&3pX0>&QsVLoKWG-aW(#ES z!+@Qmf61qMkjto2BT>o#@*QK;x!o-=@kKPd2kCEsjTgo+EBI3T9n4P`C+?J2x?tOS zdspTl^~UvW<`vTr^up3O68JdHCFWv}Y<%y41b%;ezBWwXvOgf8k_EH1{A%|E^>BVa zF5i(<^Z7*e3B<$s8~ zzMPH?T-zxHpf>hYBNFLrvwq#1_2pWzikFuqwZSWiKXyoJ)f0-lMMn|~(0n4{Ug8+* zFLrqG@d!T86FzuGw$Bf9q6HR%9ax!G)4-z6* z=GA#3N$+^f_WF`02nc679SQTtdTjQHqxt<$x+g;q;iW*#4qbq(v+y_we%VRUi5^g&=_Q%Mx~V}bOe$y2y6?g^3#gNn*ZsJ`kk zc*VvKxgg0%ov+W1AKW=ozkhIP0sPVVE*bB1kSuxP{57K)rf(2az^5E2-k)+`1lj+; z{=4hE+F7oD?8#}tkoWO2Qohs$gA3oT9>T-!Us5VQBon@o(3oIxY{xUSAB;4OPf0b6 zHrTs*;)hmef+wk#nAQW_^~Lb})(YEeQ)uN0D?R6motKqGvS&yI%B!^C@}7(VPte$N z;L>0%$}70ePy2`?JIvD?q0E#JgzKY$J8~l&oO{8&8It*Nnq)#V?Gr`wP zV@84ucmKjz+KOs35jq^zo#*ORS?Dx$&hwGUwqTn}yq1JgV z^!;cVb%Jl$=rCN;x*2Qcl?f-mxj6JC{SUlusUDVkr3OVSX@qa5u=5hA7WE?iis)$P z;a@V%qM)f@5f&xr1<{lh9OqV0ePNnTEQjlH!Rco)BmAEQq3`Ymk8Gh9;CPia@Ix;X zk|SnbG>Nx@O~JV}Pk-#ZFkesXV;ro-D38!SQ#aKIce(rc%SR4i<+azS;Oim(NHAMT z3odO&-;Xw}b@i!AI-npS?$vLY3A-L>R4|R>&Wko{pO1{S9vG$Bhvjf!k4IwumcGNt z`c+JeUZ+QjHdvJ^^YSZrL+jK0!aS#msgVOz0r*CUf?>`e#J)&&A-}J?YpCq@#7Nnar(wkAz(PG zP;sBE4(zGIL&=UJ{T6fX-J4Enh2qyWi*?JnaE;$!{xj0QY~2Ue{XW>UPn7LzK8Ahy zeafZc?Rlw`Ew7#6iUMKru7ObvvyH#!OHXggBBTS%y@Qf1)ENM$_=vl!apyIwldEQD zr4NH>g*u=5HCn7-1}kLW1BjV-Z85@hb&jIJf~hI9_szKZ{>oBrL!^k z1kZLhUbB%u0QfFy{*3cAFgGewlbf9d%FBMQ{2bfCfuw~^B_Iz-(vRp4(wyG-|LeT) z9;bN9NCsm@tQBh(eGIqNSJ;)VbMaJBpeOtUPpSyj*GNyS(5qe@I2P(>%Jwo7%ADWL zea*$4R|91`ryp|uA3JAM%dhdOjqBd*_ne2Ksa?MzT;TPWxI`ZKb6{fENpo+&kG}Fy ze--6b@kXGg{RB5GCwE}v-~fzWzC;riTLW+Po5FLhX2CBOPMXKg?I18T^!2Gs{A$tWB!K<+1G3cY!g(|cWP-izPCA_=pr{_fN6{gSfD(RSF=QX^%Kqozq z>}&taD`_&hE55`V7!)F2zy6Hwcc)9SYF~LT0QO_LCca+|K&;9UpP`Co5cw@P9>JZ1 zoC~jT+#TU{O>O4|pSS`bBb{icNBXO;``^00@D`(t_8qmti1>5m)yW#cggf4wL+#k# zk6(n+R9*~3LIT4uIn8&AjlTy&%l6S)`q2CH3z0=_He7btammpUcV3^l^2KQT4WK0b z9cSboNRJif0Ck*r)&aRr$oKmroX>b2Xk01nRFQom*m(JphluJ@83SIx z!F~K^mN4tz+pI4+_8)(CrAGpj>Z>s-sbd>|Z_4^1y?~ZFGQawcf5bBrej3??Sg_*G z%TVao*jFnRICED?V~q(ruUi@?yF#at^C3KgM9N)QM_PS)lb(PRs-U zO4epp|4z{MlgVwhFNWV@S2$lzZ{>xDp?EAmbP^NfZK>xH_f9K+t{BszW3bxWDDFJV*AE;<{drkHkp)S%d~$i#d5Xupc`wGR6c= zer}Xk!kv28kmufTd0HjSS^(uW!5QZG$Ce#*mW9=0jkw`SU*hX>WIq}kdw#l;dmhM> z3=jrIbb?Mz+DI3|>pIscX2$kkc)hk_el?YaVT-&(@seqh z-;Z~E7-Y|cn>A;DpV7fxUoF1}2w3TKfWE=_O8idje*e|ex61K9arc+k^<*x_p&#DB zWIQI~-HhsMomXj#3X#+#H-u_s&z@4M0gm_IbGiQ@ey>E4eO>o1G%TfNA6vuDYodN< z=;z1>7=BCnAg$Ni>#KK=U?NN~0)h$8eB-*0`WMeP8P=O3hmn2+H+A#zG9iRxIVp}1 zcV0INE>7q^(*RwIhz}hz*!9I(ZXDF4hV)1J%j=4x-JX_WFBthb78H3H?VDu4bL)1) z8z#8$;FZ0BBeFiHB!+zCGt$4^>$h;zY#xBIPMN|$Cmc`W_|eQ;0)!`hynI95H~eNj zpIGOW;y0gLsfzHrc*U#A4EuVy0ujPKA0!?lh3*NIxuE%%NtZ`mGnz4`;YzGTORPVD4Hd5t`*rw_vO0=vj!b4o#! z7hhiEdtFx^C}c6T?lj>7oq9T#J(1PGcPyf#K_VZ!(xM%^i5}h+288Wusfx+ zEIJERlJ+D8#^KIu=`mjuU!*$JyxAGd_Xaz!6A}BQ4L-|n=xdj8BA#fJCy05dS6}`9 zk9efBH`6TR0j|4uJ6q#9z@f0{H{Z@0@Z}qPx{nfx$J*YtJvN>2B-lwQujUqnO%+9c zqW+ih_FrD22~MH!Y>l_&HEuylCx07>M~rfnwLMz@N|4+Y!Dy`xjloCDKE`LlL_=q# z86WPvGJcfIpJi1;&XJ3l%?r)%sfy;?H>B zd-qm-t@BzG`n5N$#b{ezx0lO@FWo`TrzSY^Ou7cmzx-5tP0I&VfXD4^?>S`tQ@vg= z@Ju`I^FvE!Y|Kv-Rp6=#>4D!C*m(ubl?cz7+rc zEC)?Y6MF@zF#y>g!L~ncEI^&a5++)52L`MAtlE(M0v_lKdytHe}zU^Ks?Sa(I|{wx^+FD=z~l za(lmmreo3T$hwJ#!SJ(OM-7ae|{IhR&OKIW8?`*E;tn&))J$d9ty9v;` z)f&dASYhRbd3xpgj+iRUw8?v3=2JOPBDzZU5prQ5V4r5f`)IgHs}%1f)bs^3&66q0Oj z)ldnVZ+uV7rCR2xv=&hQVaV{L%LOCOrW1^Ixbq@flw!<4`fnW}^<~c{!p=*Kxj<_7 z7;;{~jT2FcIj4b$95Y^!=LwhMV-GOAM(eBYJKZHGBltj|bclZ?j2*;xS3pAj_g z?vBt+uHJYqoBjS++-xXkBf_g<&aTH4fH`L3O(l~jbnJ>Qzut=SlBalrPdX_G1_rEd zHj?ZxCiHsmHQh#dc(coGKeE4-sYmkQ>{16P_56@{vs3{iEUWH&GCenbcRe11TawQ4 z>@UYy>kD+;_1vCU@(;u3id~^F=Pg7fXo8N1lzJ}QACc2Q=AkUkmB-~mz@P^2Q!(6m z6@GV^5!1GSk;6ZpSXW`^ViWhppuEUt3){2=k^UwuS0>ev zx-{lm;_(HoMhN$(b8kZ8QTN;9u6x!U@G02&#od5PC|-G&JVN;|`m#HxG)JC?DLr%g zQ;zrcyatyn6ars`g3-awlMKsfeytH35RF^3T4MPTIjT@pDDjH>j>s`@UJ*Gm%aa^*9~zO>5xnS&(kz~A6_ z!0BZJyo$H03Vf0a>4HZ~D}J;CVZoE`OSzSBY~HTl9NDM4u7{0y{L8DKwx9eP@AkZ$ z@H^97D?=f>+VH`^do&)`f+aIJ)74?Xs3JZD;Z;;HR3t8pyS^Tpy;&>ysRsIP)i*yP zb@uh^-t1q^K8h#wt|H^@zwzk(mN+Kg(;Wz`%RaZIqVLD+T(@JL%=jQ-mEabyI~zF3 z9y#-kyb%b~5(Kp`=Yp^%!OrjJ+TeMo9rx&`O0fF?RkA0 z8I@u|c+r#)zBz@D@^T8#ZX0*k01J&EnIBZSAmwIilkJH+FM7M|`cw{0uz&QjV&W@y zUc@vK^hPx(FV(C!&YJYBU&2@$zgdXDA-n_{Mw(mZ*sxCIbw>XQp zJfCx+N%hmWs9^{j;gVNsX(ZL50xyBqw%WpKft!)gSWMHlM&_Nc{Z5EexJ?%>9Y! zH{H-{leUO^k%}sars9o@B;_FeYkmqE{Gn9``LnmhTb8lwYq4J=-B=T; z=lta*=I?r+n7|F`Gk$FNvo1OwYG+T2+Eu{_oT80Q^KC4!+i>@)_9w)@-fAlA8Rvl6 zQa<@iL<<~=JkeRYSPAM8Q4%uY?i=^F9*^t#A}7xoq1$V;Eibu7H>cPdq<`Vn5J7(d zosGZu$ILq`3fXU6{p0S_2grGudOLl2IXZCXb)7l2frL#9cHMn2K(hDrp3{vzd1n_-Qvd zY#tBQTlikO~}!_!cP|i8tlBN<%;h+pV^+*66XwF zx%5_Ew^fFtx_ebXePq&)gESZJ#=Lvi`ajHfj9r?~%OFyRx98*eO7>ybSCZ47XQLem zFFAahxyu^BuMo-BbIujUihNEtccc2MuN~0W8hSX3I=Yd7@1Iscj`yd17vk8rtA z@clM3A2fRM9x9%31Nm#&oT4o#ugvEsb-I5FL9&6c`0LNCP#YpDB16&)d@r7|bz*X1 z%4ca`-tie`Z7zf$ibwqiX#`|lgia(M0q24=+Qlj55>v`H{ z|FWh0rlDLI3ay8ZJ8;R^Zrle(u$tdNOdTvn#}(Nw=7LK5_kh?D+eJp~63alTz{@&IkSVVcE5Xg;Anl&UnJB@P_R{`sA) zY+&uLbb#tm3rwztUYZcggZ&py-Drqwhrv$;1P1yw!1H+E`%R;*`9y`x^P#SW2bjbp zzg;BcCV+`QPx&+ldw+oH`?g7fR$)M3(H_-iZN0%Omq$jR0a>5Bw_P=Kk7h1h|5|8s z*%^0U*`1|P)vER|{*>{&T{?DN-4_QFO3x$brzMP@@Hf_n!pNh}yzK5Ec!!|v5HfzWYOihfY6IV-@lm`&kb9$VGQ3?!F4_ z_@F1!fXpALI_ewJ#N*DZOk_@`ds-bd-5T13-(vT#5X-e+LzXD7;JupRf=X_XZLVsN zV~g^_)5^IM5{M6~qEe3D0&@zfFSi}-c_X!Pee3#C=61c6W5xm* zG%~|8vkkESUSE3Og-oD4GfJz=-U63p;u8C%YhX?C-O~*HH6gW&=#XvG$?n z(NsMUB%c^;`jQGE9Y2IJ_cy^Qzb-t&=xQKS@vbG@zjZ&~Ixl&_+sfR@$o`>i{Of@1 z^D%Ct02r)dUr?As_5RM1KlQE;28i~%nVB(WgFr+yZ)61S`U<*2{`EN_25tpP<{1kzhJ`a2BFNUmM()G}yymiJ3Yi%g`xi1dHMWp_WK} zd)>!3=M(O1Npl`rL6CZ5aL>R$`WN1}ahR-~P=c4M{zu1kvVrqb;iZvNxbx~GimznS zR)WKV1CtjUu=`hKSXUl(8lt0bw~k72h{9xs%XNiRCwLd+t^YI=)tAw#3YGQ>FHA^q zj>*k1L+M`l?P$~pj}rG<2EI-Mf@QAjQufU-pj&5ITu=k2|9oL8VaLwvYl^t@3N)gf-LI_+<1AeJ zFIHmb1(E@#1CA)KcP&5n#Va|%?I|G}d*`jZe(uYv-CNHLwiaWiLP$ND<3>o~giJcT zj+xI=yW9-s=YDd;2-SjiEZ4Kh;H|vy9Nya*(??wHU)!CUyM{w z9pzpHKqlX^b&am>2Cv4W#iYAs6+lAt;GR#bS-?J~vZE*scmHA{lpMz!R)p-q%jxw_ z*m>PBKkzPx5#<$`*;*G#hx9ea@cOZ_qP(Qr&vCzY;ef)+%{OnQFu`w6`Sus%4WRkv zOL=ib8aN&*F1Mj>2CB3jj@-yOR}2fz*OVV`^{;haE>a8q^vHVC?RY)KztZr4Cjb^^ zqJ*vy8*K2(pFa7yTvQ16$`QIc9L!kcQ2$!z)g?i+#6QFg+}Cv3f{=bzY;T7TDlMmh zGTQ;XgC5PG`7vC$=YB05JRwv0)NZT3+V`5i7B?)#ToU9?Vb3r`=812I?Bu21;IP@h zXm9w@cSHq(8QEC>q&6Cl&8|MDhFLN2MdjP?$M3R%COfoqoEdjs^sHpWZmEbLa(?bv zdWW4?bP-p-MUUhLuNG>q*YViW<~L#0mOgKu_06PaHG?T z=BQ!|4Eq*j%H-C8V$#(e8YIpex>>KUPS}6w{MuTL8LhM%FrhGlgoFz=)2tjD-`lJ& z!<)nkER;bYWU&@SO>VmJJ!9j^yBTH}WIdXH_e6OP*t`{{wz`ZvueP3(XHJA+ApVaR zui3%6jo;Z^p9o~mdo)FWaGA2PpF2wdPkS8sL~@eA^fWh1rXA%KdLrpXXcq>$IO{D=_2Ih#NTjT z|3$IKC@uQyad?5>sTleH^h;Y*(UK6)mV8hPz!SA|nk>f&QhtQ6J8uyj~bC z>oBPD!NDEIKAb1nfVRdZL7uM(V&>#_2qFCgTfV12J!vbbgb}TTbkxJ7@Iiv*(_8h` zUREAm+I|~znX@MFW90U{Y}eS$wRwVoMGJP5i?)OSzQpzp_#oy4TdX$qi{P(Uq?^ni6rs zLLQjUewac0ew%m}OuboGe!YPvRV%`GFkB{UmQL-GV!> zh(Dv0c=i}L+~wQ+DhE3+wda;Kin|VO=qvLrZ|bUq6G(oQ;Ey0f{{6q77cVl0y=5Zuy=}b2t30{)L(O zsiM_!9dk@S_=pMR_WBABJayr9e;_Pcd+p!xkMoE0SN`0X^;87eS88sBh~MMAJG-PGdo&P(;oGv@&- zB){jtKP?*22+s&wB&TcBK~F8~``xH!IAPFGnf#<4gqoZvtU|ZyYn_)W=#jQGBtsray44Jd61x{$b zy3Xr-`7z3rAU`Ndpi#9#^oQ&FQL~8q+glACsGD98l#JY-S1?2R_~e)ZupDI)jZ!(X zh1W!4(X(KWNz6L08-uhiZAf3*(3;BNw}Qy}_l1!L|L0MVXO;Ol8Lt65uixWY`y8?` z97N|G9FHTsu6e2B=izN|-JDNI#I{`z`VX&*N=}x1==-sJT-?Hl83V4>D^aaHx$xn^ zTF#RJ?D`TR`_#>s*N$1&C*FOl)&PMKOmJ$RH|bI(X3Qw%bR#d~_tpJ^yk^+r(WpY1 z`e3~ztn>1mJ9)1FnI|z1T)9czjOO>5tLFYY6uCjbgC#zYn+4dNeWiL*o1o_BSL439 zbbw@q!PSV@+N|V>V;{7QAsLyA8qbxz= z@q^XstuOJi@aQ4_R|~#eXmFBT``Lqiyq(=lRD4nxJFnlydHLZ_#w@@%B0|4S&ArA7Y4qZKJRE(sCaCOc#Lq z)Tz=GDs+Fx$fLz+|2SDNHLv?zQlAYIMQqVbC$RG(xEwa@lHY+Do~vI&xPao@yPyPQ+NFd6Cr*OUmo~|V_^7^i@75i?62<{}D`YEF+{|Df zcp*NQqyhHky^yj~O9LH+Ov?%NCUBWO_r>z7FT9c<7IgRB+HZTE7w;eboo^xyw&nF_ zkltvi*B`u0@NTP$p}dxwyp$Hb=eH#mOLM0xQMH@juC@q*qjnZ>eF zCLp3Gy_#p!08Qr0i%U~!a78p^O|!5Gp7l+-vK{aP^;ez3;wD@5g-5vb)uQS;=5bH= zHQ(n*d@VmO9ieL2xR1^LRq*LrCF!XPu*0rffVCa<`{Q;`_9s4-2kp_y5EV6KeU9Ie z1D15y{fqNY=DDyP=sJzf{&nx5UbWxaQLu7Ka&1$<&TA^rhy3DxDPWs?(7?o}3Ms^% zKC1r6xyp2XGBfmOJ~8ui#&6h&AL`bw?tDPO3>6f3J9%z4!t3@uv^N{l!9Hq+X2*Cl z$OJJKE%#vOHF0bCbf07`25vs!I-Q2x|EOvuJ3H-0+;8T!$gE@e$|3-kIz6nv5}@@7 zQNF80#bxqP|HPPr0?EJLh&qfV3u5OLSWb5O5g9sac!-k^Pr zHuK_&6!Xa;2KZTWnxIk@JFjn#djeybP+mG0Lz+T@9e`#;dC`mw<+b?cWERFdHCA)iaz9{`H+Qa}q6p9C71kGUNv`U!O~j&2Qxu?6fvqzUK-? zD3*EupFu=l6k>abr}DA#x?H!b?dEg{D$CJRm%TQ{y z2z@_t`7ul}b@9O1VT+#q$C%+)ddOfSQom1Zd%^VWVmcUSD%bK3BmI3Xb=}@0`9!&7 z&c|3c?7R*VuBF}dKy(AwEiDy_Z`pW_dX8XP7;#6_a=L<3t7o13E46e zlD(;{N+@N#%oIsTDqd!#NF|k>A_+-lMkRjt=lgi){yg~JZ@-6ce|ls+UFW*5bMAAl zv)|ykAJX58;D7!tMfsNos5s6mLk-Pq`h?Uxa6NobkzfQn-*2xFOJH@y_DW;z#mg}5 zg)E}!?(H1dUhILtjSz0^+^&Jkzw!`WwqaWA&BgDJab z+5l2deD$$>WR-=~UjlNahj7ULfC2^9Il~ZmNm!yNDv#~ONzu<6=()MqiUzxFjY9;i zXy#F$RYbieojs8wb4pNMAi0fhDHqZ|6Mv6+uyrpgGG#Bdw&%DB+x6$ieml18)9yaK43nY2#gNL}zpnGroGeM}JLw^5Y9hBBt zF^@Ecz$5$q&*g&HURBRU6*Q6lQ=7$?XznZ8(mqAp8?omsFm zC4v2ITRnu&+i<>o8v>tRkNXi1Y+RpcmUxD%CcL8={X{y~wPbUzXf^KRB&Epvp^Dx? z)-g1`)WQ<#@<)_Fbiv3!S3D24t@I^|oW+i>wzkDXlu~1uUSUyMuPUFa!2P(!!+RRB zgDQ9&1R!{6yAj-FF2c^UTnKxRo~U;g$= zpZkI{!F+UN#q&uWG$^0os-Fo4j_issRE}-g<^g@ z!1lTtr6RPbfO>sZd~-(q!cpMAYE*h&1@)S@sbx5!&I{fOxA_>kSV8l;+d`yD9WWdm zKWr?N3D4edo9BL72PKvy$GcfVK>CuTwN*W~*Q4ZDhn@Cs?v;`lk=@=N1}W`sgUqJr z``Qe)Tk-EsDuEYq*REpZeOgUDd8ODb?D%4dpgBBvdKA-3t<<(CE=2{ne>8mF|1a;i zXPR}o+ZK^^mw(1%!{FV=E~SWDe?GXFmO3sP;&R*=tYpuX8L3(Lo<+SVo&?g&Ar$TIo?p^ z3tje^I;c7Vz5XZwpWjXr)b_EY_Fa6b!@ljRDh;A zle=3FJH7@9_Vp+I*LWoS({?@dv@_HR-s%2ojKZjNqHZ7xG`~5~{jFEkQGCZ`_%!l&9kdJ5iDnV)Qz$SwwFc$;O})-I05SLEpz5<?}u)uHR>kFI^qSX~yJYjr3&ZeDi)CC)MCt9#>3I$HwBaHGJ&1_tlFVNb&$4rDsR~}7&MeF z+K3TutiS$x?KnSo{G8_IUT=?EC;kqPgviT>PRRb(`^PJLtdM={D!|n0teuJUd*yp< zIW^vi?bUNs@PM)>+Lvv;SF+WfPbDK3FlazZ{5THVYrnU{NI@>LUTAzwKeZm2{0+Sb zu*kHBo?FR8GX-dTC43%`DK`;<_wENOV?VQjyUXc*_SQOBi9b?A9+?GgvK4z0R_j4o zR>aonSuou0y$#Mr8{;c_@J8Iz)mqKRcL|S$BlE{-K{dkcU5smSvp&8OM$Wp;M?`^6 z+V$0nL8M;$pMOhHy@km+H^x=ql?|-k|C|r9-Z^t0`LNgTDOVR5RllL@-s`;nd5wXHnx{#zJVl1G3wQM#0$A4+8(}-wwb$@ZuDwkOi1Sa%?66w=BfrK>)_t4 zFP^)ivOw@Neepr52Jp|~!JX_32CEt?DN%uqUIbGo9jNw4XkM3dVVoaF`sYkj-L0=) zUdt=%y%@Wn(`rRUfdu(*EpawF9?9ww_6p@8_1BJVHBv~wNy@s<6Ix{0UQETeVw^;< zz3Q%u67LCDg(8;AdJn^}z2yFAq<#`ay^6LQpa1;I4(v2?AMO%Gy;$5!_q?m)0o!Fs z{O15RD0|C3*rHVjojtMd0|~O=3ExBZ+28fBW71CL)2?6$ls?n^Nnzu7{Oct`ZPQy{ zwYgWw-R=*<30f|xSADkc~23xSK+Q3)UCHLz!5cVhtI)==zu-3o)4897JRS>Ox{(7&{W@UX!J#VPIHgS;?>0|kK+^^rC z<5ke$eJfoWO47)+qrSSrR4@Cd?l;b$M_DzVn1SZ|^7wgX`TepWMl9`-Fu;YpcM@Fj z=&yt24xzre^c*-f)>>7N*bKQ}YdSye4~M7O_!h6sjb2qY%i?K@fPK^P9Oy>{WF=$gZNuUNMo!8_ceq2^DTjRDe6;P2nPZPA6t4cYHNT);kI zXjuTFPZw;s0=O{Ppp5R$%)>^>0Xiv|jy?JB01^mWQDvA1?*XlRic& zQAi)1Ia*v=u`@_y91_3TAhq`Oli1$-y?P)G{!hyIo*v@?o2Nc^87!N?vE`0^LVFH! zPr$^m$j>HFo>2a1QXdYY?;mqLmD=b*qqoPwdff3I9Og z$UD@FEH3Wd@jDuz$8h!jj1kfgupl3=Zn|}^*9D7bc>2u1`_u&%@~7Bd?>@PbXWf*7 z=*0*^9Zvy}$t|H0>pTipKV?;X%Qt!{vpQ-8edPhe2YH%l$oY2vs=Gv|TXLWE;^>^zpD?&jqyLI965DIQh(o-nZF4Wd$L9^TtD-^uj~>3D6@5O&55*-anQB10 ziIe4-{RPlDm-d482)5TT2fx^!J#%Z%;q~!lH$Ri-qwWZ%DP3k|gxFq^Nw~w!_YkiH zt8Be~1=wZi__;a*d7sv*O%+j#dYKX?SLCUPK8XHi^uW4H}D|Ja=G_l7+JBDLxoY-q8)G6P9&P9Xis zhr4*+D0Pvpjr;XpoV@IAb2ZWM+WgbGlCP-OF`R#v108^2X*ETmU;%{OXf~C@V|$&d zA06n{z+S&s{Y~t2_=Fa)kPRzb)W-HAuyCus79kDMy2#D#cho?>b&{BW!38{Fi1>0e zn(uv&3e4)i-UB4*msd02^T5;oCZp57^$=w~_^0qu9ysgl48n;uL#%o8qHR64mxKHj zV!q_tn%N(JtJxk$ytetxThwpI^!n)X%h&h| zpr>2+kUMhz1FN2R>%qxT_6$3|h&ZOUop;iLi0RU2a@N>h%&aAIw{1|b9o@eCRMaj& zO{a|9D2{s7*QK2?8WM#BlX~^<(L501p-|Y2cs)KYAF~(f$2Xv>cI68aUy5}6b5X7l zAa$)bS}S^EeEs#Bw-h7aCBL~>?1CIgPh$+Yihg_NXNP*d(`1({vep2?IOUQ~U8Mh5 z#qZU(=dr!^X4>;@Do+`P*Hj529Xo(jRm-4IBmH>poI8 z7N{4=v!HfeXAz*GZYkYO!vo`)UZhhJrT z$W%zY*5c;x{;$7Y!4ss-#mIdbo5k0diTP-5d<;0HICnYz*Zi^e1pCk<>;|B~-d*6M!HtN?H(*S>q~`<=C)dwso{niH&TsTentr1;KFtj*4cg;hE;d3!Mu_3h;9N){Xi-1E&5uS>8@&iLzl^f<3u^iivpuf-co3qWr87S7*uD1itd=cf2|j? zORCxkol%D%ec3QqihStTqpq~MfbA9AnW)PehQ3E%|9lkIt?ckY?)&W{KWJe4FJ6Ho zS^d8MaS_R`vi;5smhnlKOr%`lt(>y|wG`CrXliI%kD&<6?>_$a=pr|K=(?)Tm{$k0 z23O1rk^7Qf3(HMhm~Vt&JHy7;QsMBEd?$YC+s5%2T^5`{aI`{`v3)Lb>FwrThQ%p2 zI(lND$jL( z^yZxr6KHAul9u@W#@hE;A74C!4@=uGATERROoV!Da9_3cFy~DdV5zXZtb7smqM4z# ze-$GEzrLLj_eREJ#>lno!$|+==F0h9CgZu_pYSEbYkwn%#Y{B~ScSu>>phC|>Ko_V z(LCyNQZ_l7WCZsO&%E)4XzA7C6VZRxevkF>MOZTUB1}3O7z}JRQq|D>b;IP<%kCL@ zST)Pjc$t<3ADG({%2Ki8OT3rByhR(^>*d%lnRibBqE~GB4RuN}y{t=RJ}ES!UWL4Q zgKhVY0`JLX>0YFsC-|R#OHnjp&Xzj%BH+~;6qGp04PMtyciZf%hE+U!$;Y%@5R;8g z7Jk(L0R>_~x3a@9y$C!TXIIN(HTRc)3Je=Z;;XAUL2aIP?fb0vGOdb#ev3L9Vg)<< z_L3f0`+tXJzOZ?SszaGsk#{^9vVQM*Xs1aow$~w2K^>x8Y%kWkBOU5j0JJ}{8Q;*u z&iA63+V3d$%B;m#kzQ?89h);)bv_+BuaA8EA1{5Icf?PRh{1p#?}~>8FNpckXHA}~ z1NKpuhiB|^f!Imtk-*1Bn7l?2^nN-NLT@f7o&I0u+kd@s+wC3>BKPTR=EdBITqY0@ z14ey}#v}^pcq~`_HU!-oa3!eAbm$C{zqajhQ+ST;MMpn*fLeM<^Y8DoempwAc? z1-qIVO?|`Z+(2Vv|5v$>5Bmh>0^6C=h5+ar>Z2HWe;-RN>We#;Hi9QWZVB> z^%J(&T=sO7^bEEaRl%Dmb}KC)8r1mt@Cde-^i93Cfuc;#>+R$ky&`rf<{Q3qGVY#UR*2onA`!GR@q#Dn1#fk@#Aw{Cde_dhPGN-s|Pexj5mwF)+Q&)TlM>g!dn2Tz`3kH;V0p8XlzTJXr^DqY6E zcs1~xQzbZ#dWCw%QfKUQg7gWa1u|dMtL?=L<1;Oypc>&Nb5)oZPH)#PqU5ND)mMGr z7qfFApr|N6*s}?SBJF=3od|{So~mZy5Nxma=XUkzT5s-Eb3;_MuqPVX?~9M&N9vEi z&Ii4V@#nd7XSgH_`nBB z?D(oa&s0S~B()Y_2{s*vMm!zi8tbu~5)yR&_}7anPk))_A`f(q$;N3Q`&Bs!D-Jv_ z%7KBqI0KfDMktx*H|vfI1|qNi7F*$s^#s8`@ueSAM{d@Odh#%|Qu@#QJiS~%#h3F$Vnu{PA$ zUR5cI*}}n+YhD$@K$RWr095hv6{(?;Yk$wbUY5ewNxy3Gz^N?!lN0yqK=jt*vF02c zNY$mxkh?U3Mv7!i+NU50HR|k3&e~W{{PpT0Q~k9Q=|8)f*KzR=>9OadAb;`vwd0@k z*M6R_)SOW>)M{XJs%FIq>CXgLdtcNc`|q&EBfDET$){LsFW=Ognr{4BkQ^ygVEZre z{?MKF6c*!w6C&=*xLigdW z&tKQZ`SJsJqc?a__7B{4B18I{I7(gih{*@u_=TH0aM)gh+Kl-4ZP;GX&N{DmN@zj? z-2|<|cI^0CcoPz3e-ZU+wCx)nWN`ps*w({+8TE2kC^HF*=7Xt=lkwq7yvRK%zGZ#UXeB*qd#?Q?nR(_|H|S@G%zX7 z(b~?S``B&hlGKqy}XRbNUyjPlu|C?2pI0?6J&C2j4uM;$K$KtlQf^@H@X(Z z_=3ZgN^9R{x;2OOt0L)YSN+h8gSD?YI6J?w%2dJm)4TW$a?fYUivx}b3YmE!9(4jdo~4)FIwfCi>!;t z{R*t69O>EIu&wcxK4pJBRA@aZSGbu4UW4W@cTP8ePH;!mFWn$8DJ#X*l5LEyzg`5s zl+=onn|rCzADk}U5e+@wQIa7NCTp$-1SeWZhAheY*NlQC!JU{f;YixOlb8yLL z)A%y>?|jb98^$%fQ@^%Csfjj&pC+auvYAf&ll1KW$>iX;2A$j!a>?^NW- zZi@nfuDA{g3Dj%y=AY#6C5mvle6L;&lD}B>=pSS$Z{5q6sc>v3sRq=}9_uvc#P*U& zKX@`lf_E){iJ!u$)k!))R~jW=RTAyP7=C`$1pA(^gfz?_1AbG4q8p#;?^Nu$$jC<6rhqC`sMq zo=oLg^NKQ(%Qe$RMMp1<**q)>F~CNV#JlhdG{`Xd0ADJS;u`PaioiYHBM z3`oDU>wRxdW;6hk8J%Xnl0OjEZEI1dcUk-V^0|a$WWA#$GgkLM z_woMydDN}tNGUCmKB%e{eF3)lAnIOQ|IBIY@%7^=#jAodPHBJzBXVWNhXFX z9B8xmemR8fPyOq4SeudR$+RNq+#2T)GtURIJOS1T!L57kSXR4AzFQr1lA7!~M6tc{ z!b%eC^?BC3%zoqsU8Zva#X?$_c>^@Q&g7URR$r3@orI`{V;wx{5y z8BI8h;>{l&4@s$f%!Dt|`>58B$E7<0;eo8a(EJ!bsB|3N$F|-}ReQ{u zXC@Sw`0L(M{#U<_yNckhKBNe4>NkY`80AA{>gh?rlUw(y;(QzNolXso?we;JE5Y{K zY1nJXDeq#U2nOA>JLUcIVfV0(ZUpz%y=?c4eMv+51MF#HT3ijs_7ZuPst}rhdg+(56zsq<|jJIz2~&v6PEy=`}w;*Hck-ljav;4 zLA|_0_$25T`QVou`^Ca6eqcBsI}smM51)C@>YRI%1(!?q=+`yW!?zvG@uO4zpko#p z=X2X??e|+h9&uq(lZzqVn|ghT-giKLF%&ePP5zKFL)QzR8CPaKUsMEmdhqQB;e3co zkySUa-@4bKXI&3lemlVv+XutGNI${9`E-4~+CreI=vT`P(erwNlKE5ZqM zEYj;2@u-)OCiO*CNg<%ylk-TTlOJ}NTwqnJss|}Al6eOz91sN&z1qjv0B3LJCkk#0 z*wiaED672q{pMcGTAJ4OFTz0HLoz6C8Xb?LWj`Cek@vONEu9{j9L<~rR?$@Z7k!e-~*;OYHzJB!k**B<{QBpy(VweyxsipXe zDe*&u>@gAY2TdR|_vEu{3J!R_Gp}k?i z^yI0~3_0pm{y8@GD47Bf9a$QGh1>@=`r^{@!|7Z1`oa6{b@FcwIB`LbuDci8tF}$P zCf6Hr!Rbjo&J+XP-ZBOH9l*-;G}7Wceo}g#t%nmjPc&u{kp9sl z%T0Z^8bIG`IPa7C=3cE`FB_^td?5O?d16}pH_UqCJC!2|iWc-+{JGUM4I5$cWlS)0&FkPyz*4%H0t%TS-W$N z-5yewUrydSgL++335VU{!brpAu4@MZTFXL6}fjjb-W~%`XCWybJ@8(DLms7Qu zUF|vvhXn|=um8Z#UzX#RyfVYyn|kT{`a~TX424V9)!IiE(R|;p!72VuRt_ekTG`_{ z@&T`D(*7cM>t3U8vzR{%X@XgfJ;gj8+skpdNOUe7sk8oxFCOQXYa$QqK~BJvJvb7r zzvd+yDp?5mz`Iz_h4n5UBp5n2HmlbIt!XsL^E)_@XefMY%BTDI`& zU=FsIn4OS7`p(VcizVYv-lAw2Tw>m1U~jIsmWQVUM!d=C<-z5FY-ERgKG^8BhBQ!Y zy`K1@VmBhLkMzf^B$a#g5!nH*W-v-5+M9w)unI9{EW7JZ!Js67xCQ6q|eL>&gCc zn+*fDt{g)yT68`nN&CP#81dS%U1rDY8e~6}IB`_O|6x7)%(LfN{IeR!ea5bidLGzb zPOqjvgcqP*rc;NOEX(ZS6o*{)uB)h*X>AqnpLzkXxcxZs*dcy6Z*#rWE3Y1&-sHMT z^&=bXOk?PWX&OQE&BtxN^8TRqMxTeA7~88i=zfCxr_IOXCE~Yg^()Bp@omIK?Q>}U zs&8A3|K%eKM3PkOE#|pkPfIxHkht~y)e>uwnC_j9A5rXB&yo9BCpV|<)uL!l5@)jeqZ0$u-_zguU?%P9j9 zBNxehT9NY-qr^KN7P@cw`50)feuCzYDyTk@>zR$f_9At?ckt_f>a2gp!Rm<*4Ot{t+Y+IpB~>df0 zKWmSjKr#74F@ z?cg{8-`+ayRdTmj`~3A@83BB3EI#g=ddVzPmw2`YLrVQ_jq8)zYhO>qK-^(}wG0I2 z&s48k;E;8&Wiw*ut;ZLWuyrNlDO;e&F})C^!uFyfxn{t@!~>e)f{BASP2fps)o367 z2qgTzP=jYi$74IKAJO(gWWJsF`b^0L58SDIeV5D)5By~EWxHQwLS&2buL)coAlopF zZshtwQ(4BYiB|0Rdj4LQlCHo590D$dj*(!WujTto7}wfv@3!@W z9p6_x-6*i*%j)arU>h3`7;Y}l_+CJCzt2F@ zeU`iKI0!m-kE1Jl>t5GaSieeoSwa!(;|IcA*w4o!l2$&M-?)IFl73gR4J8n!d6ntg^~HB<8wkb zvO}ADv0WKBbf9%}FP|RyD(bEf;0an?-l=4`_Vdiei*VjilL4#u(Wk$9>*?t()*S$=kiejeO4we?lOl$4_faox=TagQwIjgJ|N4zh4eBM&_p?#<4nO$T zaP}!y@&b{dmwkyS9{hNY*ldfI$(0kRZtq)&IJm?MaaRqEk^NLi-f2 z_xee1`*Yi&P?&$)!ROPCdf9$%E!9hvgH#=I)l;1~@Se@p$~E2kd^@Z}%Gt2N5q`D% zemSd+9bbAx=X;xcc_3vX_0p*n9T<9NU0He77Ft<`XO~W(^;gQ{(wcA_FSLJuTRvsN z2j5yqc#RwJu=v=Ypj;pq!uB5JqReRk-qUKt6Wu2vU(N5-Sn8?~`3U#tWK@P@LLe0KGf2dQ;=<0H~zsN4(Jaw$SwZ z=MT-r;fJ~0_<0inz+KoK-T$!xY7XE3YS@(v_sAE*PNz3P;*;B#?|t)!A^sfF_F3%t zqiKg?xBM9|Bo8E)D!s$rFIRt2cqO?%96nezTOURGbN_w*w0J}pgigxA?N++LJiQ!v z^5W6*JLY3so3|2l{{?;L+O_Yq-s|Z;UeWMO)XT52qWpTE9e9Prv8Dy+oN#+YF>Xz!8D~SN!H)@7tM3 zFSa3G{zpEzDq+_X$0=O>6OFl3n6LA}yEhh3&S zk$rN|wJ7m{5Bj8@fbK(?79*E(Kz*Uh^Bi*i-B>)UwTCtUq;ERCtI)t+&k?q| zM`>NVxtDNG^t&_9!k|ZjP1WZTT7Si-uq#Z>$w9fT#oTj`9MCn%dAA$c@4wacdr92~ zy1UQng7F+1-;Pl1@#uDJUr#Lo>ZMS*Yan3G9>#^LmOc=oUX>z|;h zv-&N!2#6YiollQN>1qxzI%a+Tb-4*FRqg0sF8golWluQz;RfmPP2-C;@ek{f8(}~~ zKB#)91ofKOUqu|zB@gz<_22!t94MW*OS-##>t3JyXTIyiYJ*?eYg$V@w$}isLurHq zG9J6|{bn{A@IdwLn8rmLFmYM96y=C|@n!wyCI2R5)bT_8U z|7*XAoN=Sv#8(+8j4Z9(emw_r+%;cLC~n=0Z8h>-;Q?(3cJjlsG-G=;T-cL9p@@1( zC`3VM2!n?@i+?E1!~%{7E=W*-%Q^5Tp64(9P3M)m9#ow|8Yt zCgy-UUO8yUW9we|y~imZDi|XB7=(T34r6;6FV_kznxbCXQ_8x!UP%4*O~vE88R~V8 z!`AA3E+dRtB??A3@q?lp{)<;p1I+3+kO5aN{FZk-J($@HXShCCnjrfNzdmX;*f)ml zrNpR0V(5&#-`1EtCBlJyzLtAxeE!Es1n?N0lnh@&_urG0WLZwB$sw6mjxska2l{q| zXv-RI-Ro0hP=rC07Hs>{TjQOJ?M0_Gjw{0>>tFwjM+@JpK~yf*kUbOFpN91N`Ws*R zLpaau3}KLb$ayBMfDZyB>e}@C8X(!!=1tKu4t$oY6ax60p}#@4+WQi+AHcT6>)ILY zd|zrD$!AT3c;zW4rR2_P{{24d>($S*BZ6&(;gGw+yJOEobbUhiTVb~816f$U?wQsd zk^{e_;ux8|x9+t#7}oxJ24HOLi^Hc0Y%fpN40-u1)XV8sqA|$>D>!0lZ6}_MdUo>vAxuO zCT1Uee0)+f?OSpKQrcCbyI^R~EanCkl;)k-i93T8R zCD_%Od~qOXgwq-?4*7*;gX!DPpXRNaK(~xjrd2;+Q!kMmhtjAiuT8zUmgyhQNkl^D ztm5qOQ`D=xKYH)S7HRmkAm?Nqh=atN)>PwiTd(im*abW|1A`6KAGnjt|`S{;)LUZ5fhG!c)=ug%$C`Io8PS!$qv3+%* zn!4b)=#dTKCvPY;G&X`fcam@c8=0o?N{a*OhW)CBI)<)sSH`c2vaz!K4 zh}ZaHybJ^U`*<34rPCTuc{FVdr}mOID{yy5nHlv*>n9h<@$w zvOa&6*p4w@w~hq;+|0h)x@i6NWxJ1VILAC zzD_wwlXtq>zI;G8;!}|J*xvToc_GV-} z)?7OovJ*XrzTWHJVP~SZ$o;k7vWV%7t5L6Gv{^s-jb)+v_0!)<6FCqk^!5Bf_O0iy zm-)|slolZUU_>g??A$hbDNM+{#;pWECzI-dz9XoY>LKC4hx%qvUwnDHiw)`({sxzu z=f)3avJp-fLU%y;c-nhB;x+x$!B`?62NW-_ACNiT1U(NEm}q4KfX(vh*VQA~@x_?# zWX+uF1zfj$>PlvJt+}rEavOQ8?^GHA=JYPZCsojT;!+H?uVbn#L|;FJH2B7gZ=USL?woMj7MZd~wZO?&#V)uu$|A+iV9C#`5T$&b4 z*qsd*9ZWr^kDh+LZb<|ECMO0!5?IKl>|&Qu)X$rXq+ob z-#otFK~qNwvJfCtskJA4HyU44UxubWkIMj85$E~3c^vF(dvj9z*w*7q_p)(m@^(uY zwKQhA9<~Zu5z;(_Yf8G?XBx(G7HGz7S(8Lue zQ;32<+A~wrOWUEjEQ{_kvi@~A%}2+@H4ieJh2DJK(*!G8NnxKzPr{Av@+iY@tF`aD zKHn!PQVI*4+T6=zVYFB6NI3L3Uu3;-2dyV$IdNwQK1xB&;|X`0%Q%3SUy99@x4wR# zGj)+~N52wy(~}x5c5GaaCO7VTQ-XLoRGboe9>%^FUoHxHi?ql+1>Y(Rl)i=|AO9!5 z9BfnTEhhLucCmtR$afpk$G3amsay3RkYLX`Q<(!?$K$-i*c(6~BaDDS^(34SJ9Fl~ zg5}!hulKsbIhAqN&~wxHx*KGyv8Wn~oLd^JH$J&{?d!$NMfxNikb>Hhntl%p9JJeW zk=c%J{r#giuU}Ae{a*O0G-6ouzw9Sh>emhWZ8#i5wydSc7r+ErFZ#xm_R=D_A+!&~jdwj6#ch1QJz>JFghRZJpT3<>c9Lby>w|`l=g}fV zXc=_i;_^p6{*PCQT_1}S1r5YKJ(xZ7jTE^r_SgFhOm*-|{Gz}Qq(4AO_vrT=`Z`$c z4dsqc@`Es|{U2tJV!wY3>AzcO?dHCz*DpO{sU?A6xGp;Isv4P|{{4B_&V*;O>WBl| zVkvC~^1j$hHe0G$#%0U-%SUU9@ZL@xaH-b$x%9vEv-0*;X>~)qB1FuVyKR}*ycQd^ z-YNO(!ndvg89968<37FuG`#3azVdaF7E*$-eT=JlG389ARO+&%pz zyaon7hk4dy9tWG{B1VU*Lu;SEemo9T>-mY^ zv8+$n1I*TsYbB0k17%zj&GhfBd%2yJe7Pv&46$EVv(M&jJikq$ak1GJ@vm(x6Rrg^o(92&4sY--SZ=zkgdJbePp!s>>5%ivRqcNFma*S=&V~kh!>42P2C4#<~1)T?m?r&zI$Pgx7S>T42>_#k@2WyEjpmxH}g@KfEbQS zG`SPR)_^$9f?0|-3&_VhX2PZJ!O!!9RK3#Pp!9}2tiBFAzH*Yv*#tU{ZtC@fgU2k# zE)WRz`7zU~yx062@6V-w9CTolf+zLoQk$OQpzFbCY=)!jmc7)S?}5{Ib8w2*z3)1; zF}~Dd2d#|}ueOE}W_w#!Fs0Ur{`B+!G_#8ieb{XPw-WiCrR-VPoMm4hmfrP55MIyo z$j4-o!oA0?Ji(9f;1x$GaVRMpnzH;#$_(*vs>ADWA>&P&;LU%`*ZtGq| zV)VRJlKbE|j*_7G9$B#=}XIJA{`HZtlfz zqil3z#SLgYuiH3{60iMT)_a*p%d(#!3I(=3qYPrz=y?U6mz?vrM8$zhsDs2f1_xxR zUXGS^TlcctBRmqIrVq*wPtvWdY>cmdib*;( z#iZErWv9<`dzIH6n&TfDrgqJ&jkEP$pGU21$^z93$oy~xF=g4x#MD^_P%#QU8t zP{`B$lnh1dFWFsHV|iz|psZ)!j;M?j=%Qtt5QPVv z_QkFcUO%u54oOb^mwGk$j0WlXJMNHjm9X(iC<*3x{CrQbGHNvh-1}6Z8L-^Fz=R?IRze?IKAA(tW>I9>68{^CHWXOmAcnzKJ-(^pZ zdgZ;jw(Wd?9{4s4`tehsUJ1_~sRFbG!P0jBD-XVHF#W5Jk?&DG#5o)?VHU{-w)g4W zD@Z;Y`^a`aWV;`He8p0Gml%6IvbB2sDMs#(k^Fkx>CQdudSdwiND>8yLh_IT+e#cd z-|q7|c;LODI7lrntBWNg&qpz`4}jeNy4Cr1u52z&+`}1edk+~79NOqbQ4!iTfq2o$N*=4Bs=Faa=tdJ zJ~dYm2YJGpiSFm|;L5*yW@q&YD5!4Blr+Mg57Dt*Wffd<-_*;m&BW4@4%z>ur$k!$ z2OWDPvjTDs!e-}VtreBP%u)|IC+w)? zrE&0lHBJVfQw!Rixi>WTp8(SR{;XjZ*!e3*r+RV3+-*~@UO8UknZOVr;H#$B-Hz6) zon1{9o)5&J){uLTsUZ#;3vNEo`Mve|_NX|%BD`}i%nZG*F50m%zB<0#+4dRnI(#_! zq5TBv_4P^i(H{>@fj;QC4DlrDrO2uPV_&(U_u6wJpCMA1y!%`29)CTsCfHHv-^D?! z={@Pgt+lYv`UO!F^$ECr_rrnNU)Wy3yL(MKf;W#ZkR6(QcPs=Vg`9L^tR& zawc(HN)W=*^rEyO>v{&!3SLVUej?Mr0|W0{#mhhJ*X9NFZ$Qu z;H(CjphRCS_=obGlT6s$Yf^mVbfTv_5X1-bh>JbPtS9!ylCw?Ohk)Pb$6Afg(E5ud z)MbptQUW}l32Jm9?;jP5Gzi8HZ9RV#(YV}xt+Wrq;%&>9^*4Gwz7+fV2I6(ZTuFC< zn-zxJUPiHjD)6sp{H{4_29I11?J42i=r#Mggrc*A46b!uQ6`VAhr;s*eEcVI@Qr%A zXcGq>_6;+R?fZEGR7tC%7t*lvm#Icsn%aA${%ZOXIDZelmvMdl#Swe9`o^6QNIUrL zN@*yX@3p1Lb{Gmt0OPpSF^Z!&2vdkOx-GMHujwBi7oz1I;mSoX16nz3ulP1$bLS$| zs~CSsK1|RI9^_M}3tmUP*seD6eo2Qp`Qv2A!XQa?hDPQ!>eU!+XlPNm2Zn5g zx_%?)JG9Q5{f?a5x>xk#g&PORT%qvVr%RbC*k1GdiyBBjA^H9ku~K-r0{mR|<HS0%~EI+f*h< z*bj7*B^c9?sU%zixc$&fn(IfJX7QcDHHIz_|Zq5({ ze)8`sNkiw4!FuPC>pt^Al^z$VNf9a3ml^xf-l+%94BeYUxAVd04RLW{JsuRUywUG` z>Ibyy-*3-NS!2f6506}y^?Et>G4$qh-<5OONu-n^V)8j*8|DUUtR~8^5ODNinIEuc+fB$dE3B(obPy%mT6>r z1k>wvlWZ!*|A$wjfpK)u@lbd_{_L`<7J6PGa!x-seN_?$sp3VCMI!4-A;l~NL0k8V ze$6jHw1PYzE%#FARbzY26Y%u@LF$RW@%51)zx^6=uUlzC)IjNf)f1i7lpV{tyukEo zKwb`6&tXbcr)OHMgDfVieQf+le-VxEMV83=^tXT`7b+D_0`&03Sgv5lSItk=^p1?p z<11*}@8t90q2N7ta^l4!O``9$X~@{N0_9KIzr_R zy=MpgvAwRGv>CqW!Mx^WAU|oghwK1+nUZ?IaSUDeOks-4o-E^pK{Fy|pT{IH|B}u8 zRzw~6hAqBfpvnUQkJOVpgK8negksylXg|28(-Snj3){;+^~UbgNI%QX@_q5!aQ`S| z{VPGKhy2YE^nKcDI|*;lD-j?#c&*|t&UcdVq!cUwug-crggCcvG*2a@8U*(T(WdDhlz4V?_q{I-J&?6@{ zfk*n7Y0EPY-~h+X|Djo^2D*5A|1kQTfW%xwYmw{N`F^4#lA)O%Igh-_^D$`u8Tadi z!Jt#8_46?SI)5Zwh?9GkAP%_IiepHlBp{%DYddDQ_4Tix@^?9SypF*&;+BIBaoAp0 z{6Fo-F(O|0!KFM$VTkdd_`zak3OkrC2{JOHUNy%O;^-AP;F2+!Ua|!-D6mvTDKXSR zv}cd<`12f)=dX)xDye})li(b;|3}(+$5Z{k|KH3OGBUFFD4VR#6-gp{WTniE$lfc_ zkP&4hGa}j9k_wqo5n0(SBtD@w~3*^LhrK zhjs6RwzrO9=S9OKBjo(c0VgkdmiS7)$G)KZfXIBGIl6xQW;H9Sg52YJBI0xrUNdrE zx<^)Gv-RKevLMkhAaz3K*hhvJe+tIV%VUvEW=sO%C7IqTZO02nxBL0LY;~bd@0al@ zDU=t%;{8{ec|71m9Q1kD8A5pZv+R-qO9NnD54`>sbQ=y=nY1V8*MgFYkE@jOdFb!E z(PD*^yWB-1_P}Z+g9v-@NkHF~A8rjWI9oW%EuI4SgPTRH(X@ewntIX%TL5? zIpI7{(fPqq0;tWiU|P{Y<|nkyh`JxlfZnY3b=}5VU`xHo9CYVAa^C`%lvfM(^$>+e z$EcquZeEOTX;d#Vk^3zKP3WpR(fX=#HS^R!iztjHx>g@S^u>-*9O}CE_q--)W!CQw zSi{jX<`+$(vGdCNlF>N2hVoL`c^-lEhj95XV!4rL zwQ;`i@e`ik*c@6f+^nKkv_bkEp|`Zi)a?;o1JB!8H~+rB{ozE{)0ec?;B}vtvueF! zTPJ_|wSWA5*Gmo2vnnXB@v8olpY(NryPkVs`Vh)%$e^h3^BYDmxoPw0B;H;~x^v={ zTWmelEmCRO5Z;FK^wDRc2Ww!)TE2)};XLI1a5i{H^bfp-W(PSA;O50?Wf9r4$7ctx z6N^I{K^9^lnatn&_E;LcR?_>lckb`?wQ!qHN=4WQrlS12Nd8BE`wy@0)||y#lL#-@ zO9>g=+K|tNQJ*T%h1X5vqyEz%W~>`czS#|scj|}iW#UW- zn%(>KE>Rs!F?w?64W5Sw_>`|`Jh1cf2wr23Rl{wMKF^DiW@LS#ZHsQDi5>M{gv43> zKau)GM!do^c$EsDsgtOlIQ~7aP-AN0pPANB^5v`}F9y3kdPwIzkh;aV&5PC9x633- z2b>6+0z=aow{^*AGXCS>GAFz!{3$cmxEDfRtmVF8sE0rjbIuN-+i*vl`q!XbEg0{T zw-qrv4;@>2BP|&0yoT=`crhP?TVHxaYUTL0KG5fT`|amg)PKEL8L)Q_L;6Ekn1xG5 zQ^9IrYn@)_?|GRYbt|XpI}7xI%F%IV*mtNsspC`=H6d?hVt6f zQ#|JJgA+LI!aSYwko@-0@sm3INI!f0IPvwV4B(i2{*%tK7V6aBiV(}6ho!7uLlIr< zyyhg7vcz3*^J?v*)VXEigYf9LQr1BISF>sFw?H-#@Jy!>;klCvDwO0Os+a%Xe^o|( z5SC4`0{Ws^SMz_+SKvni-H=6wZG9P%Q8JU8=t4@7(dnAc4BPsW*66+3CCma1hYq$2 zM(zc}hx-0<<8^S8ab~*r=4~hu$|H3dsDXjSE{)6@cR1+()iCED`W@`gdf2@4?Q#11 z49d56<3qeb!~EQxTxT@DpJu&7Igld&N2~?lS#1h1>VNWBnX&!re%ioE!VK+J3uxr$ zcKUV(yT1=jW4{m+L%+={(|zUk`t1`CU3mL+*){qdyzI+P6=+H@!3VYKCXP3?5XgVo z?$G&kByX(Z5ca8twWx3-ZW}is>#7*(R>!_SC*pk1`AS+loV*T8t@nHk@dEkFcDv2$ z4sCy)z4~`1T5gGe_3cM vb&n3>2EADJKjxBbT{)&lj0)8>#gdxGBQHg;YVA2n5} z9Z`J=EwnE1-_nDsoAhIRXHk8z-NOrF-e7{$xA<87tC-;9q)wHKNgdp8&i&X}jnwZ~ z67(zNYCwxiu3U}R9S8~qUddEp=OsvDXZ=bIH!mBJcA3{+-oTmP)}fY(>T7sUtQ%d9 z2vo?Q5qX233S~_TgDIMS&r94*Ex3%|1RiiDntpkOo!5uc)C3B_XnahJpKa?G)`!A(#tBNCNg1w4K9P3Wd-t8obwJhd2t=iP1Fb zg*kmbusBi|*~Ec7r#rXK@v9CPZ!%tdJe~pPc-GkD9cv+{!Zu2H$Q`&E#+@9Qu={(b zi>n6e+_?Q$M@{U8T9psvBvE>vC-{$g&T7uLx8nPuAXk+{z$q1Ed++piocR0r==I#E zul$oU91=a{)%ycGFR7dxJ_c?muQqBw-YEwIxFD-d+vSe(GHX`edm&l`Oi36N3h%##IXzq~vp(`tb;^Mu9Od3SJH5opWXgPoUnMYw0M25w%Bd&bTP zuKEDIe!wqvc9a*Ur}3**st8;+4=`cTPlZc;k#o7o|NC42Mcm00WRa)|ir;@;s~p14 zE7e`9vUVC-SO3;m1H;3|MRW#WP|eUMJ&Uflb#HW!$07Gy$rSYTT`k-Thc0PI{?xCB zxYy4gR5WG66@8LO3`Z?&9Mv%{euvxxa6m1lY6*LP`wn-A%tspBynG`hJiCK^z_BaM zmYx93zlIj(c1_rd0z)NUVmY!NVr+bztbXC|d9{vym_0jr7-R$qp1k{q`}qc5)^x9> zBKee{&78qrQ82`FuTr%!0J+X@!ZEi{eaSbryrnP_gQys{&gA92uwS@9?p#DY5T9HL znNQ4wJCgeQD3EjYEqdPwE9uU|kYi#8#arz9(iyPx&*j6-i$$h!ddt!e&bTzG_YI)_ z%c}NxW6YKqtS6CF5ND>sRfj!B@1_2Jzln4yxqULf1+?A?NMW|eUSB1WQL7&bWC6V9 zi)rKyN+9ComYTND5IFI_-@XxyJpcD=E%bdNK~B|{95l4yE111xhN_m1tD>*!;jA=U z{f}pvuufjq-#b?eyy8C?44lrxQ!U}(xLw%mIlL>Ep6PwU&5OP5=iEnbKTx`nX>vRl zwMXehMh~e%F}T#tJ17yI3Kadu+TS?-p4W{3Egj5#BOol@YKT0Ho!8oy(zL=Zlvmyv ziF$8I(2k01M|mh)k_)mAW|W7h(#z9B8Z#o2wcK>Cnv(Yphh}hk4eI+Ls->{-E0QxPd zOS!?QJ^ta_^v)1)1A zgdF5Q_7C-mE8qQCJ4zA1{+pLP*>36$JAD|fcIhoGNBw;mVfESlMC?FeWqLj=a4+0_ z6P7&HUI(wglAD=EWWcJefqhbvG+ zKk&K~^rGz`Z2N~V^A#bMyuwJIp_tcPGZm!FU8U8F|DG2EKl^v9Mn@=EjydY|5xc%R znJ>%o-A1?=_TM_B{sl8jT&)uRULW|r6PoYJM0x3oDi}14P=d<0z}y*N2IIsF<%f9c zfv|i~acVRJDtv2|wU5_8fem-#1OaYdZ@NBwwW+|(%cOsQ>0RkSFrF#)rp!X~``VtZ z=Z7=+;M;)jK;uy)KDO1bE0+E}uc7q;-3=yt_<1^d)pd0zuZO2--7(1id%KSjhoAnd zPXC+N=5%aw;9diuS-)SLB!luQdqT7-CWV}97^l;{#?K71mtDDxDeIvGlku`VKLeH? z6C@QX)BvxpOJOCa8zh#Q4hLN}-PXf@@1MvTQvdQx7&kA`5BpxOV*-I*$7jnX0F93m zG&GBer-ec3OM!A=c`A6NI8Gf@`Fnpae}yE7S;QOIGXv{;9CrG9a`&p0!w9dXC4w%g zBS=2MZkNK|r3iJWKa~&Yh@sE|3kVD0#k4=D2eGTd z)8pb<;9DQJKto=O%mpiraG}G^i(&i$%{?aE^_+QzmuVi4gCOPJ>(g>8=>CbXiBCMO z#6*CUj6H>5CKVdLEUXP5`}_Xms(nb#Jux4U_BnrJym}`usakzSRfLz8MD}HcWoE!Q zB{l0^l>`1XQ>#S|BaqIyRR`ZtUZZKGgp9~sZpBBQGK^$wK=iDnT<<3`|JcIHc`^ps zpVK?~*vP9EJl;+S>K$`~koua&Fa0O6>dShQHgLIZ1V!9ZT^pP`(9&R;F* zZ7?=N>R-%x3K27_snEGce~T^n@AXC6bMjGhtRu`Y-9H-Gvy<0Cq1W-F2(Lq8ckeWQ zLbxnzoqf%P4;;@2I*ucC7#hMv<$DV#FSXrxDi~uS*ssQtP~gl41$0`=6|4NDc+ zS%qbLhY8bwpCM!}TjTF}#eLd4+U|_(hx+bk+wx{7FNfQHs{i6euBg-D(24Rg>JX5( zW-)?YWV0BGZj=`d@!tD4bD2S`%j}1z5E3~r?G=S|8S;CX9EbufvRbl+N7HTtW zq>I{c>r1fu*}V=<+`P=Q)7AVWgTbe0of0shW3QNAp#$oV?}QQ(~Y{bk~JlAPw5Z@0Pw>|9f7eNoT_a zr>r2LfL+^RKXzVYiKRErO(FX+??lB5gvdkFi6b-zCyimcy2fjB8qM$j@G{gGnH6nd z1L~-Wo0ncR06&H7;#uo#Sfu-Qw&YxWJq?0OB9DO@yJ~GKkKRgj*kti76l(i z{|~(6_s^Cd3tsbbUk#aRz$W|Km=yOhBx}hQsD^xyN@Q-{~x?E27F|bEy0{7 z)!XdZPJ8^*PrEk)c}Je0yaJWP_U9=Y1I5VC`bSSuUbU4xNhZ7Lp~Rv`THA#c?E47v z>5=;s-m|r1sT6V7CHpnP+B&Y}oask}j@o!-wpT z{;e8D|E^LqWhsIA%ux4z0HNM4)91b~OggA{!F^W^* zIXknr!5TZS(olx`iz%o*X5br7iL;r4ykS<%;WX49#XBgU`MR-!5Jr+Wxeeaa?8CaEA_zYrymG*!@?;r$^Sa_i^)jnA|NJaU>A> zY0}qg`cQqH%&B_j=P3+_K2Td6+?NJKC$la#ng2a6n|bG%x;u`LO>5Y)V1b>NueqF3 z$a#c|YJ~H0pa9%!xL~wJX$s3!`EgqwC@*W_)F&hv$iQ=+_5i6@Y~a%M_HsS4zp$mF zf|8ax8xC9X1WTmW0qrewMQdWDzm0=eWj`(FVhnCx!u*`q1eyYXz?J#SSv%C<%a*-3 zV*gDLZM!-!A~|!7E7qbvf~j%=H!| z&-%@as*b|ptFH+>l+N0qY(w*}wFl+TXL7jUa@umkq8l5iG`>zY`CJbzH1F9{ijehC zX>k4imRe}HvSuR`a0eS=p{-mQ?E0eBq;@aahnv^S!y89gs{ElgMIlc12bzDSwVO}% z#__?3rPN#Zh*UWGoa~hL=HIWkV<}W$lOXlD6T6)R#8r3lBK+9qJ%RAT|14SXF_dYW zmtM8Ej@AhyD9#S!Js6HW|6O0jB>nQpuh;{r{ZDx?5iIbee~82fIWIITdqX`sIurJC zo_N-Ir3U)?yP74K&%qqu$H%r5*w@>A_zeszs<`9hTH&3h?q;O^r9yVtd|Y|^^TbXJ z-6yf)g~zJrN83wN!Eg8Vw%IFx&#U4+zL)Z<9ddpaUsvht&iX{$Qk=;M!b^A1oVjKd z*(dQ^Uvj%N7r#~*0DI00jgjx@{)vgt8B4~tOBm_pi{wnmx^lF=w`38?Pl?)dTS_lv zKx|RTgrsE+&=`jG3|hFtoc>7VV?gU)d*%1o;&)2}o_p)NtuvbOLp8^BYix)pprRA1Ip^t5}Pu|S69wS7g2%y7K* zz|EJ)oWHg!GVYkV4AA(JB4T!{2GSi#O5{X#@-h~ENv7!{v(3w*Khns!9XGFIuZReP zJp&;o>1fh`E?WP()S=xXroji4w{sL__|t(}@4D)~uD|Ey{y|(=KgtuX5;obz2vu+E z_|JOi*Gbw1Nev3>CfbJi{PX?LL_!xS(FUh8aAGo?2;@Waj0go@4rls=lc~Rf5 zE8`n+fL?ChAsvpL{!1&>NP!XG2MS1g>lG1y|A$u@9T#g&zac!Tr9Gq8iu!v&gNyYg zUs-{)g41qpH!D=rpC8ae@>5x_OGcEHnb4{qz*BU#7GlN^oMzHH2NaWg2rfTG?Xex7 zs3ECK0o7M-=4RD?+`LLj=T%59BJ&h(i?X)Vq4h$AzH6b(<@}Jp$2wuDG8OJoPKLkx z|JYyX=lJBwyJ4neEBGvP_s{VYghrf#_?~_gjrK z!V7Qz#QJ$uUw3c09I+r!+~y@3>0A5o2yR{%6jcHl)BONb{HA@M8p`W=(71F{vjFUM zbv&4nkqUiu#N|N2IP9Z|M|dTA7nv!jpuCLO_LvPK zbEsNWC?#zUp}dxQ^UO^%7{Hc<@R|by8=N-6W4Vd!m#Zr@7E4jggw8G2DsMkTU%Y!b zs+0c*uNbZP^(}>MeZ2}PZtBs+&C91l_v5D=U+_~Zvt4CHdClfWTx8GXgAj+aQq6Hl zK4JLeVO7rG>q}9%moVrt5+5UvZXI3U86N|FQ1qA}yjqXM+%x)D|M9oJ{F^kV^^O_= z6|IVTd&{JumM(ofUT`Yu&m1LJD={B4RZJM=Xk zU7q^Y6V;bba{F-POWg61fR5tEP^%A&whfTBa-#lT6*ImxDa{XsE4iz;!%`vk#nOkK z@W1DEGgMOcp|(BzIufS)_~%YuH1)fFtt0;aTP%&O?M0N=F;fp0B1a3jZ}n!$Bnst4 zY0wwgz{CafH+Tr_3t55x*5iB%$~wSghFlc8lL3{TZ53ezwSYH8dgYmf3+P99aGFP< z{yt$a!cXru%4^S&mXn{#aPzu2D%Ky5@FH-ob|ue9Nq3dninvq`6k=J-QU!ew(T)7qDm(8C~kWkCQ*f2e?Qo44^=o@fUbvP7)p!e z=8*epc(elX-=xCmDZG@jgMZJf>1l?is<g(ICrB|Pk z`M%MTRd%nd8Mk?zDJi-~fU$(>JI4hs*P!bmzuC2D{Z(ez*C#R*h0Jp*Y2&(bq^J&5 zwg_)8Ll#sGtA5g3tc6KUj|(-(9PR(+Rr+hW?D!irU$r$b-@HDK+a4zl?$@Z=?FYP4 zcg2L3(0szOS4l=aR|pjM6sH;)ronJ*Zd}sh-}Cxmm7j9f)Eb1y;pRT(o%I~2u>|7I zb#EwXAG*@;iwVSSz2uRlC4!bf3TxpGseiK*^ z^8&4$?zyc(mjqeI8A*Z%)k8adFiw~V7M@74K{>UeSQC<|G08hTldq3 zH{{Z3aALwa22Fc1(3Os;}xulW)Ua zxb1P!gxWLgm_JCS-C$51M0uU@(IIpw-~+>RlB#&fdE}~lD*7wr*m*H&M1;PJX~F#Y zK6o7#O~JxLn6Q*=Z;Ef_7&UXx1qQ>$omfBf{9)a(z2?kb z)E>dwg>URNAF?k@%77TDM;j1gI%R)h&nK?)sr2c2JjKfEm)*BQ-CS#^G*sPu^lWE7 zK@?o~&;sFglCz4$cm}b%-}+LB8ERzWvVzCCw*JPms6DzLZ+~gZj0el>&Dt%^)v@V4*`RMUqHzq_a*|t4j z%3--6_CO9WdSq(Ulv$XsNn_6oKjXGX_L-OFdJBGVoocS*)_yc!6*+CRC-e;;FpbfU zk3kyT=raJINCf% zxEO6P+>-?K$0?T?d_QEn=HU zKTU(5H8SUWEfkd3%hJ)h?cjCUAa$JUE5hqS>g3zU@tEf-)HURjxOvGI=l{xN351gS zUPt%DqP)64^>OlC7KZb+tm$};Y0xB`RGUwXU0)_KLLpc7w_)X#l{BmCTz(poE=JG9 zitqGaq4ZrDE(ouH(z`jYkbQBwZuMpF3IOs=SFF15tYI>TPn16!I`rms9FIHrzIcnTOK85WuXeL(H_>DA z5T=E}`@#PJV=9vIy5RqCzN0GYH2c9#WZvVC-ZON?D6fKdH{ZF&3Bt|yB&!=o(vbP9 zg|cq{(BJ-Kbuv721G_!4aFZsGCRzjAvEr7TvYqeg)VTAB)f>@AxCi~gxK~9oUL4I=ovYJT zmR_q9V8n)Kf7WOeRHqrEDFfKG9|rdG4Jo!o`n)JN)A+#t<1p_UyPN#KgA; z7RsZ%vXnd6nmRdw=*}?Z>uFAybK$Zbp{j>|YWmCLhqFLtkECgdOEnaH_9@7|c@C&f ziw-}!iRvrc$Jk?*Odgtpr+yk4r(o!d!{V=TMM zgm7bq(-T=I{^oU8(P_>Yxi3>tm-O_47rK7@!)v|dOC4J&7q}E1KdVMl2i^n=ZBL^z zLGtq){`=-C==5#>`tE@XJXK6+RhUNgHO$uMy|sblU%z?TGUbSNR^Zl`*V0E7`+GZi zWm2$rq@?nL0gq)_$Lmx$>?|e7@DKf=v|TCGKmNhr-!*=0di;z9TuJb4rVwTl0B^olt=eXbVG6OQtV!ZcOo)N#WSf6wB91vdD3n2?V|vkv5s zABi7v%z!IjF34^jsD@{w`J5_0w z$oY^K+nsd0S_k+VSVo!jf%*oyLc`Dt;@ z^bsf6Xid!P++YLV#)vuN^*R_%pK%!2%z#hR?vu~ts$p;b@Hg4moxD7pL~Yhe{L+s(@}2pY^7Wa5M1OD_4M^Ctd zC6T=PxlPkz1dIouD7-`>>ul!W@P zgVroOFFW`_Ew}nu9XoR0$^5a;zP#A;34_Y9i>30|d0{BaRLZ9G{2u5p%-zZtE(ggpPVp0}TF>GG!H2O)Wu;p5_L&^Dpek?^Yy zNX*k_wN^7BxiQA20BXRL&^RoZ~Y7w;WCVh->8=8 zAZ~x}?Kwh!wImRnoXPsQ|D(QorSO2GOaecY_$BR%Lh{vePbEe17nrqrV-cXKDNl}}CLuMl1hnyqKYst_&>1*gLH(twuX8F}49R^S@vBhZ7? z5&xT)wyIG>)Du?l^#+bPt9sCX&1;&XnFWRdKhz9tYCxjqeaO9e*B!iQBty>8i%W0w z;?Cj^`M8XmSCez=;wrMgFzJ=k3M(CYUSW{o?Sy0tA5ia4i{xKQg`kR&q}+e-_Zc0w zmioW2$44cD@tFB&3$T2isnqsA`rDg+Hom6BKG4IaeT*}KVVf82N6!T^P7CO>c_9Ds zU(f&jS_@s=5pWQm|u z5oFcvO5(BW>=Q@d$Ifb_UU)N-GpV2ar3&~{I-X; zHwfB)ZEAO)K-WWC3u6-liM(JX)@7uMrPq}YBNNOO**M4QxroA8Gy^4%_wB3xLXY>CE4|6B6c z^BGZIH%J3kXiwngme9vIH=YDIW8+owx0hM>Gs+K2G21z3W- z#Kc|ye?9;AYb~@n@ZPVfC-e}xM@eC=hy^TCi&qPf`Hp15Z#Sh&(qU%VD9?4a8cuY6 z3RJ)6yo1-b(<@v7c*XlsIBSjq0pS1Hy6RmTust2G5< zuI(tV;CB~-FP*glhPIm6tPYgd+Hp+_Ja$g_N+csd*UthS=@LP3yB4@#vbSZvx((;T zDZhQeRDn9@p0>&x&X7(fy3xam>Z`wvs4bHS$-l_$R9G~wVM?+}^}P*n^J?dNBYSjT z5V&aDEKgFR_2|h+wtXJ-Px_W;4F+o|>)ua61O9ptJ@~2odT=(p{Ia&0kY5A*0_4%x?wkX5%4q(* zv1tCKw3|IfqDu-erqAudo_Aqn$u?K^ci_$^uKKf@u1tjj`NSg6V}EoWG64~rY7-A9 zd{+_h?~h03@flsWiU`JDUp2@as@>!N3@fi`+M!fskJDhd7V8oFKh8hC(74#bg2cy9 zk^7TcT$r|b>2}A^Zz>u>sLfC~&pEU{@rM_`Syi*#Di@IIx=m-z)Pu7U%|@+y4t#{3WQ||NCDoM9&c)THbza?IkCkPcfIg9?@!{)x-j6j9eEc$ zue1I8y&EsOk)|R9Q6G)8^cqt^GUaU`Wgd23Aw;IV1kbSZik#ojli{*K?yt>XioIO5 z{XT#6b>2aghy&4AK$ch4hB32w!odZW5oa&u<>fpk~tL{^wJ9*I`5>=|OLU_H1*w7O8#{}ZNlw0q^ z%`3t3jiCE@ICQ^R9@AYw&v!)K*g9~zK^PofIkbAcPX{lv`!71EvCnfi%MxG+HpOm_ zo0rPzcDs4N!|vo*x&E3RyyANYnY`p&lrM9rp!ZD zF|;0?Wm`->=OqFuqCpyZ6I}2uQh>Iav4>)~5wMwOIze|6y-&J) zx9FX!KoR(+p3_#aoDSTT^|xM9V&}DXRzdo0LkCuSys0iGB&Kc)jEAqfjUTSv{@j1` zh_DK$o)hA&1>USbMJ7p6!h)kdU*JH6(-RK z-|n0nZeGG}mPg1tBVgI+xXQ{)-R;k#txMGUYfu!L=E9mk&8H#v{szVA+GFSCn@aaO z_E;BIUQL+Es^dkr(63W`K}@rH`+ffK8nQi*D}(TwZKZH9EJyY;QNGcMiVy~h*NRN* zafa})Xpktbie>xnmRto7xI7dBYL^??yVtqE;%910HnJa}WaDR7M_(=|?He7OvU&jG zzEbx_j(PyIotoGYZlr(k@7G#r(FLmmtzl9Sf44GusG|yF7Q>pi^a=O*?R2+@SDyzi zz!#CD$%C);xBuT1$q+3SqW2&x9}B-1X+W4{Eq}@eJ1^k`O|fg&u-l{N)y>C)Uu__# zKGRv2?|<0iHB}#GB4q!nm3^(926DdkH?L0oYXz={k$F~S8dD#%ShqjlQn7NQbcpaEE62eL@Mw2nTh4A&UycFA)hj;)K`37E*o*6F5PBg{vzi7w zuRPhja}0;D^Lk(?erzMl2E@I!)StEP)R+A{Z`m56uO1;CHl>%y{;%J>DpQ*VKji6y zdWc&v-5_d@oZjgjx9|y|@8Q{Z?>}+EQ1;l+>PkJ(#9tL-iO2y;{x=!DT$Ny$d5KLr z#2rfQsmn@ZthRaGUVeIGpjL8QU&P}SX%Y8v>+58jPLxGqILx*6OF#I8_R}=A%c~D{ zBIm7+fX3J5+oCnuKt*H6 zhxfP=+%hB9tEqPKQc!Se8csxc&4hO!B0Geemk0fHOp0bW%&4A;iON9Z;|SGZ#p)p; zAdh=I9p#Vg|LVS!s_N<3D~`m+eoHmwkC_e>oH22rrZ*p6aOYjF08gW3}!GFT>q77UJaWU`>=2mAhXRo`zh05gu#+p+-vv z12pX0_OwKpa@-E7uQJAb4N#=zf{FVs-5VzxU|Fq%pg=AMG-Ae6U7IU_|AXprDoYR8 zpi0+Hc1HadX)Qn5w{}Tby7J`dz@>VO711$4g=XCGQE*7rFKImtz89VuCUQpoy%8D$C)W6QeE*<6zv>yVDL&+9+_2N%zhiJ{ z{fY3(>(Cp)>o2tR!N&D+G*1VrFZY+tRj(EpVQrv5k+6mnXpi~GbBQ&; zJ)sRz#`J8U{#NBXOuvvjUhdZczKXz25B% zZ0EQh=-l7QtCZ88f*gsD0ye}dUkMQ%wKk|RSMz{{TBLSStpPZe*Ca)eq57KmY`E@> zoD;p@X)eVT&INa5GrV?RXaENlJ6R}F3@Kx;gNaN(OqRlu^3B^fhv-`v`0@Kq)=l-E7EezEw?gWJ4>^zeJPUgKU5-SChu^q>p}q6Z!$ zuOw1+fwbQRNp^2*nP2eh#B3fV(y?%9H!mo}bG**$dw$d#duo*6jMX;pVM3m{rQ;|s)=g>sE>+z6 zDx{G$TZ3>=h)gctQb7BU&Z2kcG=uj8Zy?i6s)006)608zZwfmv?c^VYqHMib{Z~ZQ z7qx&!FX-<#?GEwYX^&G;L*082dqmFNWcaagz@>J!7rT)AyX^CAdt*NvfN^95=>UY)gGdSzvO7-N{+_xNXx!ji5I8X^A26wV5g*}0U?cIa} zFPh)4ktKPSFeyO6{(}#1DRyHj1m0a{K7%_xDhnUqC-CzEG^zBgR0^W{ddOJeY#Jj5 z16_Hfoe$C=$u^1Na4~jXOxCGOr!%mxhp11f4qfQAfghG_y;olCc zIvrVfw|Q~i8gU?KHvpobVx9NwyxaO%Dz1H$_*D=z4TzkGu5iIH%au6yn_~9bSI5euilP?FxJ+iOL}NN zVJ!1lceYI&(reT6N~+U9C_rPLv<7>8JouQIO=u0hm-D~ttIT^_-uiGj!|wYn&8nw% z<`W!rWb}cEzEsH1G!+!_L00#&%7IKd2pJqNYJOk@0m1L@MwRnz^LkJ2doA`6AL#0b z3%tj0gRFH9_X+trIQfG18>4D2kg>%jUtf3t(Sb6yEbE@IyD8JufF0GBOJVHM_eT_A zf@n=%^I#Q5J>%uEiCNtK%jS`g(2?G7AkPrrrLchB2WCt^>&nC=4k_UThE7O)Y{9#0 z_MrqjucO`eVZD#A>noSAcS~Qv7RD*)3(o5AOakRa>e2gbo}C~1?6&r=(sDxtNBimz`|KLSOMF?Bn3cTo$f?@2T7-@fun-|}{N3Js~;m{KN z=1Q6f%8Sk1By*8a0=h1Ke^=U&246)-h~j&&+oQA>4e=ve?7TkS?VPvoM&`Tw_RS$VBB6e`l?OGQkn&iR@gE-ZizEmiGYkmyiA&R`P<6{(YKxStu{Mu3V4cb%dA9 z-D}EY`IwOxO>#aDaQpk5ptwhX@M_$(?@XB=YL6an=SdQ{#etgEz(x0CDr`M7J=*oJ zd1q?*w~O{ENtw>JFZj z1gOZ63+&6|hBtL_JY9x$kiXCWeR*~sOgW61^SM<)gGmj`gU3D)em@{KsSnjxvbr(*-zzNLLeBf(%2yAwX0v{gxBvu<(gTb72HT&K7Z%M#%3DSPUmM;cgPcxZ7g9Xqdj=M9z!gPv{s{O|ZUYx^dd7rEbB z*T1}I&~Ya(=9#o6F{Ga3#mu%!XTT5E{1bX4Y7)TemR^7N*GZTnewHA364h59-Nfu} zEkW2XrLDJol^ZS?Q+$*%u7!2|Cy6PW@ei*`P9I&l z8)mTYSSITkWZwNBd-Srm-tS&`5J>q|t*?^uKq1FZMos@3*d$=YGxN^_{fUmX0FwtG zy#A#BS++0y3Oba`WR3D_qRTFnqCt5bIC+)uo;q%SA9N>G({f(~gn7~IJ9G_i`}6qC zGF=l2l7!{M*`LS||3ybJdgenkc717R*s`RJVCO}ktZHTR1p1Klb?C z`@{ev5+B)eh++>42m+o1?Rz&PX_%F_3z$D=2HXjWm)ymW=l_1Kg=S?b7p$v`0H>KC zbFmQ*tc^_Y9_^}yAEQ1p)RB1*aJro_#=aV|9nEfJUGN1R<*b6zT$I-yKAGn}j!1u- z=;TFLvn!ZQiqgfJY~1#Eiol~ikLdzLR=K>-AV=#H;(5+wJ`GYp7v!W_eJ%|&pR5(U zw8PHp{$O+khY9xh$TxN>!ycJ~Iwee%XNsIF`18I0@DfySID824_X#h9dg)yDZ}Yk` zaW(Ja0W&CH*GyXSMxOuXRd8aYG(AulZq^6isEgr-_A4<0FU@MRyDHRD6eB4^jUm z!S)H3dZdBCuyg5?dl~?7&uUFG_I$N@pA;wkg_brWys57k7*fZHMhBv=o4sCV+q?HeUsbP_#1B)T zeVcNf@VyOq3LI4aHh}7jrdZb|*k2M}@{l&fHS++8u7yKMM*~po<}0Any^GYPmB=3` z*1*HHV@rbIzk`?g){DGbh`)~y2@qwy^#~L0D`>4qhg)BTmI_;KvKJs~etv{C%Xpi= z0{P=1Eu`NxKlMIEC@2{y74{pou4A{yit;3%)0x=)m+PomKUaq{XjPZQWM%EF=di|C z-D^U4@w}s1G)ohMzS4oY+(13ZX}=`xGi3>vzshqs-4WaVd`pby-rK9}1%2BNOUG+G zz}Y-;@0)!s+^BLE<$sb7SM_*GSIDa&R*PBr`iv+e&-FI;Cx zEWaie2~Rlpx>j{zR47ZZ>nkzUwfr0r_I&lZWqYWl z??q^=yhm|vXlK1J`IXCiAH*J?(iVFu<{k@WfhtUy*SmQGTip5zJrqaWkbMc*J-H3cyNtH~-(`m@BtE3l zP;r?&B~3B~N_;l`_WwiutH4(%%Sik==8r%A@BFKcSC0Ewf-OAQqw((B*iK$`vwiNp z$o`zBtVbuk98g|)jURr!usQ_}mh)MDt|+gvqb9CA@fgSvDhVdO%nM;*j<3y-`pp*= zr}qI71#mROii=RX8Yl`Zs~jKr@8G3JY)xHAjPiUJ&3O{LJ^B{itKuoN1+J6nUrCg9 z@;a26W+aC2;+(gxOO_LdFcDYNuTByGMty>mHm9IQW)T0V@_+C$^KU6R}S z${aryQJIIE*D2Z~%@@x`!{|x+NZVdiU(p-Hfv%0>V3(LwPDY*#TD$hhkln@}A1z*x z-y^?-y&jGKqq1OK-WHZD?hiI^?%Xfut}`Nd8u4E=cP2TU=0vx7aq@7Avz)Mm?qxV4 z^;vZL`&lM4++E5KM3a$clmA=qYE-l1U ziN;4i%||@l^(e1`DLRQ*A>8rtXVG*8YicA6-N+kGAx8VPN3E+uGS5rEW4fAxajl#1 z^tP#cv=MfFy}s;4NG*q**N>x_EH<}2Ahg6))f%Zw{ju9W`B#a(w^qY1Z=lSoUZeIC z+2+Nlz{h;h-yGi1(J3{Bq4`8sq!W3$9VO(vVbydMM(SUq*_2;9E1@uxB<-EjUC94# zI9xi5oKKr|&31M3hFKo7;`|kK{pd}1Yew-uc=>)%KjMbmr-v(_IQYWMp1UCumKg3! zJ&u;${vK4%pV;-pO2L87$8oJmH-QqPFY0v`yT7lYVlesHirs(R3D&-vXB-9GuU5p` zr*`t9$R#ifLU^6)%&R(}FABQs6K9sK9H4K)AWKW#5-{~oqG!!eef{CpznEs=slo%b zKe&f&P9p1{goBsAe=h)!JG+e*u2;h_pX2TrOJC41_#pWbIY0jAcjDaSrdVB*g!%81 zE5--N*@$s_VaWXb3X&5~=@0E*}42+Qj%a`o2 z^V&Dxq_pof_WVoB^Hcxb=t$6^TTD5|xHF$P-1?<@0Xe_Dm?B21v5e%azj?{2dWxnr znghqzHJ8?(sJ<}EarnbcOBl+M45OqYydX|-)ZX)KHM}_4P}?b4gzQJ88stXiub!YZ zxt6c$3nF!QuI@EJ?Xf=V^P%^dC@(LefDw;5+`MjFA!NIyluHCv6=M$#9Fr@aBD9?7X6m+;Avaz+TVU-}iFL{;nmgo%>qgeq^V=PpFh5jzjeI z+*tfwF*~ZSr|$2l@jA`n(=BVuP(D;&7$@FG0V1RzZf)#*!jKn2jAbXEA@kA9mq^~` zWfsA`CEKw?#X4x7q4%^xcn!-RZvDlL@)|;Jw}oRUFAu$E7rJtA^QsT|NWmKt1>K?_ z54`e2^Hn#gLz0ZUB_Sx3cFrR|8HD>{@~%!}_g_?8sY=$5vGaP_?7MK2#|&-;**_#x z-^uIrDBt90L3S?f?MD?{6WnNV6YytZi#!Kwk-*KJfNJ% zT{Txz1ByALce`YZpc>ErXpBT1I7Rq=%f9LZL?7&UZzrJkNa3B7kfen2N@<;$zZQa< zm$j>mPJ?9>1jSYLs=A^3CpsSGjTB|FRM++KUEjU-+Iz2+AKQBr(j@Op9Q7vVCEoEWwSmGP&2!8$ zKD+_bSD$AK4Vj@O{CId_Ljo11ufMz~B0C3r|bFsm2>hsTi!%N+1cnRj#bRR zI9(n|b%n^HKCg@eKI3Vyjp_qU@EAU?-%nQDEFJLuYn!L@^tlWlSdw+DqWmxUedzs^ z*(fT1ptPl@TqqYq^K=%%72UOf_u)QHVbf#Ka)CzmTNS3S4(YVaSzQsZeRo@mT|y89 zO2v{oL+U}{$wl7hm6b4hDa*a$Rx?;ed$z_#1Q78mxcPHo!ha`pjP#bCPAw;d@{P;P z*R5YKY{=M$o}k1)&>g>!m3!FziO+q<``@3EhLJPl_m00$1F}IrIzdrJDnqL1P6L?7J@q2P7vA%}S1yW}rUXfa?rR4^qYrIaIC2V(aFo*uqk;va>qHF8h zJl)||ZVw5#nO;qLZ-pO9hupimXc|DL)74_|#1(ksPL{H#0rdyC^2Rzz!WUfk8t9#u z#{4TPQN4fbpv)StpvZaCkEmY0p1ydN^8{FJq9I3gB3Wt(JMVZ@Kwy0sDhb@Cem$Na z(jZ_})n7Xs-@k^lau_<7@$VOwE}A*8W!V6gG}WQHO~kzR>t{cbN4&P?la*_FimdT^ za8-p|(%ce!9^ZZ-81NtULa#c-caC&mA;+RpkFs$7@!ga`HV34?l7K zC3Tf-&m-jbg*(1BvI+~Y@d_3T&iAY}1)i2H5@AWlucyp1wQtK94bd#}we*9u|hX`fQOUdESWywY^9mgt>apI6O% zl;nv^QIJyfMO@EOW9{>lBqdeRmZJM6E@lm4h*$CXq^vwnd|neae4~%~@bh~|8kf6g zLhL|?@}p4S6JlN+2A_+oQ2)ZX+lDa=8+NSm8oMlbZ@R+-uBWsO&~M(c_W4dyOx7H= zK=&y!ThC=|;)lAq`*y;Ujqvq*g2~C3WxzN_`SW~tBe+N0$y7Mt1$D*L>_RUvUavNo z#F*TdfiNixieaue0@>6to}G&8^BR#ajX%;94JJp9(W+m<-XqJ>{K#H>APQn!Wvf*m zQ$ayrNy0V^|GZ;!&CyzKC;a#Bb;CWc&Pv*X>~Hx|PT|_M@BMc_R8HkWQ;K-K?%5BE zLzuqQRNwQkvlv53Us$pJ8%$q>ul?Wi79_#4FCd*xiw^`I2z+NGGytD+OGBYuDX_e_ zyWNMi5h@jZSt}Dg;q~yyqxNZReLT3;QSkI7jMpzNo0@*^^?9{a(sdZ8#elVL?X97k zn14x=pPaP5EdmzI(Nq(|sW8BtTAi^6pVvMKx8zs0`0+R-q%IQG9srX{En%0)h3&1Xu0}8 z_rd!6N7Caq{@U7c@P6+JOZ|=5{jYn$$Er6-Nx*r|!rX4*G?>5pJ+Da#KOTAFKGA0A z;GcI$x7}3B=T3yI?;{OTvBbO zYb_B_hs9%6i+%|yE9y_~a87(`j1QjfSW<8vYXCx9ysFp_jttYFd?IJMgn6}@AiAu_eK~Y}eSH;ox=eo~25cjF<_W1-eS%U^ z>vr*52{4^Kn{zNf4JvlGOEgsC^Vc0wwnqSj>FBOUmiF9OaIka!@eE= zXp2M$2O+oOdMB#WmmQRF(phUdlNxRAZ3on*zO3j0msDE^j#;UL>${YUh(h3huD6}wv7MFxW z-qu(?k@&Rq#!F2B*m;WfbLS{OIM^9&|2WbFjPi2Slr!b%9QP*u#dtIiOj*t%y3G$V zPdB&kp2c|aeC9YhFNyJrG;tg^I=w!xsTa}Lju*v1a_Rf}j<=ZK_dinHytGLa6c@Mc z{^F1ZzcxNp)KaDP+`Rn>KmTyJ&@gq`XfoGQlffv z)l%t?pQt_&Hf(3_?d1!W9xUfIm@!@kA2`x^Np`L2D>;kHO0;kNcs%)=Ebr&97~q@! z^u)IedygzYGn8=6SqxgF#B{$Fr@_O?o1|Mb@p+9fR^RvL$In;Ii`ZkQ%k4m*L%W>( z5;3pTtWk!KNMFn9bT< z^A&65;qyvR;fy%{8$TXZ<&DgG1A;+1dUs!&91 z15jp@y_X4{Lo!u{r>1gZ@z~LD>P*-Z5Klq7T8Qpt=zo)D4%RA%*ric} z3-V2(jUSc@Wck%OX8BOkZ>{?#<_mC~hBE5zp|mk}5ky^0@?v_2^|xoAFyO}v*y zkuou_pR_{iawr}@eVn5|=PSI%>wEc5w%Wi|_osbiLmC}^~#pX}@NYK4vBNpEpAqrn2UZ33_jHrH8UgITIe(*h)@cO*I zt}f}>sKr3r!ze5AJnVkqBuzpe$30PS(j2VSu}T9ix|XNIB>4MByJkwCuqpg_d`j1@ zbiKeHa?a55M{yDFAHNvRkt85qQw?WR8>KN`f$v5F(^ikdSbR5Sz9Po!&5M==$8a97 zc%rUSqRbEVTA!)wel&n=(ZspKpGv{{*y!+RZv!~dbMs1*`hpkvQ1*`qOkcf8nlm{& zb^}SxauwIzL_*aWf$K-**XK1CkieIt83TJMB|ndoYOVQ$_35+%7A;{A(2?BFb0`hI zD~*J0euA$rcfF7MFJHswwRh7uP2cUuA+b8i{_7&~ekieh{@6Jbk3u;Yir=$iy!L#R zT2(hV3SrlqEmn9jUe*JGi9au~0^bhV!l4J~yn`x0_&ik;tTctmkGPhC!M2Z+);*07 z`?w^3{{bHenQR~YI)mk3p4`+vB+SPFEh#KZ! zmm1~YbN&(mE{-iWHpkL{L1^o-6aT~S=WUdHT-fma>)z|^t8E?*(D1a%t673rUoj=k z8Noqo!w!9e0S>U+vf zaQB$IGjD$>h-;RdwnOj4fBtB*RQlipt#eCn4AFb9zw3qNdnxfCX-r@G=36Nj3)bhQ zDSR8X4~c=QeZi&6^V)0w-W!RYM@bt+A!BchmQP3;cy4*{Sk(YO9!a&jeVR_<*P}~z znPofE2SA1JtImQ|VtrBGI`>Osvp?KAci`BQJ*b{zyZLwcGi8wcmjLl}jl!+~eweSlC?k;E1QjoLG41v*gThA2(wMtwo=&021s4%NB3|ha z%vToIm7FKm*CcgK#c`D1i^vVCZ)?~A zvFD6ZC0-lCXu*${s_0!0$DV+@P3V35Uwu>sUE3+TA_9TjJCwPE1;Aw!eQU~cBczsp zj_b-Rhvl*cdzaj15O_K3&2NqR%m{jKVt>?a`$CazmH-08gZWraZ$Kw@clC-&}SQ;{#M}tKdsd` zHql|ctQ{?WFW%#U^Px(Ys0jjaZ6-tGb7&(x^1WkytFs)m>idr*y=sD|wG6*qZUn$H z^01>uGZ-(a%)L)uv!i(@qLr4uC4PkaKNtG<3$LHwQ;0vJ_um)~^hU!Mg;EZzeUI~} zeW-pc?f_@fyzCBC|Ki)|nF$~8{r=hWxIl(bd|nsIDo<{E>IV0&Mlsp%BGy-iP>S?c zq%X_UxhtD@V7vy5#GFZ%9pK%=CQ}<}j2Fq#!eq~fEO7gc8rA0=0#MjRE9~^75mr-J z`!`vYL)|XXlr$eSPx`^-YXzfd&Pr?CXTuHH{fTBv`fV=^FkYjU%pXcA*XLzkn+SUI zaWGr~Pi{?Oyk4hXlU*4T1)+||ip5cBpuotTr_7FDzfX*M$jKmxACHWw49AA}UExAg zU&_u%VqQGT=_$)dUjq(KZ~TXlPyNvs)1ILpUO^79`*BLq$=8^F>4daJ_L>PohK|e- z^*ld#Zz(c#<7ff`$S6~pmxD~)!#$Z#njuYB?&98U-}!*pIP`l^yhX% zM@BYfflCg7;XbAL9@Nirz52Ztr}9GV^LSu(R`Tep)m!`b63j$7bdL*x6Hj~K8ICmA z!(*oP^f$h~9{MKU6`n_A>)!Rr`$O{=nCF#Q;7CIx@Y8nId_$UW#ug@ zeY+-54YEI;WQX#v=CmI2k61ntxj}TU>gKLBUOUq-L|pl{{`y!(_>|IU5(`dWiX86F zVf8O-d6mJ}fC}<9QbRvH@qJG1F^+7Ugn|!#R9eAXdEvv}+fb4yd z+5GuNxH~QW{(VIW2wZ^t_sIXZT(F)al3?FU|ylczz=`!wwi zAADXsdA5)F`tjp2cPNH`>4Q5=vvqj%-X!Mrv_2}K7scbp=vybRaR@=jSi9-Odq??i|MQ5_fE0feyHEPZc(7ld0uD=%;vIvhT^eCoZ9GM34Akmik35P z0!jHsuLK?+;BcIbQWwY8$4H~%?(Me9z|)!Xe&kLWVQNDNMFH3PyjozJ{XSBN>Z&IlP&s$=`1zNljyHZrI`nUB5E>jYj9 zQQJVhW3~}?USsz-Wl{nvUgGhS+fjdH!6MFD8XpiEcu#+f1mo46uv3Krz27Ejcv!AM z?@HkDSsdsRUY}P*rQ?|+k79ttJcwP>1IxeIuZMjYH53AyAGzOmP^E$8H*t$c2k?3I z6x|;aEXUW^wS8B27A`nI?}29aTTR5gI2*s@b|QVT(?8%NGs5)MdOc1?=&>2ld^y55 zXol(Qe!6*$w=JuW~|Wa`u+He z8BYbqEUs0?yZGlri~X_YhW+^ckW{SU*VpHnM`iU|`1@`3PpMOH&zu5QBl*yq zcEtJDi<@R7HE4ZQOK`ea7$F3QuKsv<;EonVrhocC`N<0G9AYlc#$x)~=GY)l>m>@4 zu5L>=k0X7Rb;UkgX$0@txGfphC?2mR@9Hvbgo@cQ5m5>s*q(epd*e|o9<^n+jnE!O z`r6kg`13<4VPB~St>KaN^|f&9$Tzu>7%=E%z0XM7KpCevXi<*be-xUE*%G-|~ zvunU|OiSPJ18aCRpdG)Z7t2@kthi|o>!bIXRCU|=(EDxGuYUOgNKZ?OV-3Ho(EE;_ zH(Bk?jqt#P-DOb5mxveP!&O-$@m)Z$sMd1&)l5(dtb3J<=6SA{uQqLcPdI)l4mOz_ z@Z%oD>Z@Us!N2WGg@Ibje(Ls*R2aTTdAsL-)K||$7F?#Q!q?ZOxIR+J-;O|Yiis(w zmsnpC+u6M5Q9M>0nR?V(kMVk>J!C75<`38yaT^UQqZg z|K^QJBcv;niMVbrg{Fq&nFD9h`8EyDtGBMi`ihu6nCQ`=yv8dhF}X&*Ykj|u?c1;a z!z>;qnO^SG;=%I!*!}Lm^9FW+a^DjgGUYTF8}B)L^DMr;SlYc(v+v{UYiDV3r9wd{ zsASKvve**qD{uc%Mq#u*YRyHE)LcUKiLX{alygx(P1}RdM?(;=dD0(HjO`!qXO`J> z$w)z6xJ%D{UOo_=J7;3{x&b695={#Clz}1lOhmO$BQWH3DmA6~LPoRpENvyWAG)Ee zy)D3X7f9=0?@-C^Aey zF(N!xXR#ihM`blfcYA|H#mu%>+!(LBCp{)SJe1aWNldwPB)F{4tEYK4JCy?JyX!1; z$Z`|5KBjybaQRp)04s}6uDnF|O}15JerVOj=QTyH_mW2hf4{AK-gJ(1(g7l+5bezkL3AK(WvoFNyB0rinfl;Cj-&@OvfZ_dH^~rk0tkK-u>-xFL-P z&N$f3l;}5rfTi9Zsjnpvb$$MFA$L89Fdva7d+!A)x9$(yeZ>5JGHt1vTnOVOq^)#g z?*01l=n+Z&tSK`FuK7j1q$=OL_WK)TqkE>)EdV;f4-B16(jd+1w20vqd|tLp}mZpTz?nZz5iL4iENG z9#R6rTocXDk|ILV-Wvg81MBl*f6CS`b21ug=&Az^|D&J6k3|j=Dl~t4@Z!a-^Ec8! zHbyP{#S}g->)+w-VNLkF9)7cKW*cyV+%f&~*ki={qJLt$LkRJbm_MGVV2JT@P7Jf( zh32@DO?@HlGRJr&Zq2LmrB#HQ*z|z1c^-H=Nrmo%HN$wOjzH_vGSF{!;%K{E4<;`d zes)fJL(4`&csL6dj~?pwL3Dg5|Jno;b)jAa`PwJ%ZJw;p>ukE&-6Bo&?jpi>OSCjr z|7zo5F1M2t1ja)=%w6bxhup&*G{@rbc{y})cpXr~_pgcY;7jD?&XBAT*R9J;%&SdS zyWtk{FS4X>XE@ric=YIcr>U-B4doT)4}@Ant;t?91!(B1vzc z`8*K`hHWHd${Lt|T@P2dv=jBK5#8QGt&PrW3qSJ|?y|w>)nLkCZo-MLFT<~!c=ypc zgZ<1^h9M(jUW@e>E(cKlmGqeYd$$k9YuEM*e1~sXfjzU>SI0n%S4ORgtmCK<)VcJW z`jN~7n?1NP*^$1kd=E@yNh^ivS?=wG@p@1X)!=3H@gd?hes#}mMFqwyexFiDV&nR} z#91io&Ge#RWL`nV#~#b?Wk=gsXUhb^E5XF^6?&h>!%-J_I z99TRaEG)2#|FAwU?i$4o#<3`PXnw>0{xHUiJt${Gd4(XfP$ce)*^~|von9OiyYYGH zhcHM>u;J?~zIDd`lARmqioYuVpiJ!d7Yful`w_2D&$+I{ei*Oa-J3Nv4%(o;ldLu$ zPh-3i-yXZ2_nZ!PMtzo|bmaloEZXF-#TIn#CniCDu@p$jtrxvMqx*)AKMVX#e84?6 zFB?iQUKPd)Pb7CEUSGpETHJCc_}kThjcbbt<@SJHx4qsu}3e07LD_gV3S6JV=PO7e`6nAeHx!Qv?>9)-*m zH=LKkc*TU$`n~411=Hd#N>T-kS6FOj;0}3yNPBRBl1HBh3}~ayWE8dl?XtMXm}Mzk zv-`C5TYzSuigOdGqcac zWPQPHhns)n7*?OizTthh-v{IM(N&qnqJ4c{v1HbVZ)8P6!$Cg=^WeQ}`B(Yrak7I! z{P5L|)59Pj4T2u;A#K*e=f&B%={(CI_IvsF`)yVKHiJ<%S4eI<8S(kQ%uiz%*6m(I z@p$FgLAK?es2|^-{C-5(+#~I*6=-u)_Pt)k`tgN5uJTlO;{>M2-o1*2oS?a)Qk&$3 z-oM<@J)>q`3=y4i;Y^eDaIKos+W4_I)amajP#DK}xph#uamu% zskEp+Iiu62f?6xBHC~7PwntB#5rl2J7Zmt&(m;Q4Mj)*cpVubsO0!fw{CE`lxkYy0 zU@&}*s=s;MlUQH00n@o}5U-^q!>&FpEFKqZ_kEC@v<6?7jG_4bSUjHOf9cHgNEqZk z#u`c3qj}}4q0a<4TOg;|XEpaL>Yt+#HJ8lL2qEv}&y|z;pnC<5N8`h>cx+iXkx_R~ z5qQ)J@1~O$5h|&lZ`8SS9JZZWT#8SdApG5T{W~7Lh3eFH1w}%I>70`KZOp&8Dx7uy z=^t`(p(5wrlQh`(V87sj!}$JHY9=>f=!ft3&l?Y&5BePfT3&CyJsKe9RZJUsPZ7mq z9Q|&x+tPWH)=}CnX~GRnQ?ju? z23w#YU{68U=Ms3!c`v#7XCqusH2WDQ-~)?ieD;fZ{|7JG=;mc%C3Np;#4uU3jF8!O zjqx&?-xUVuzDNsUbEp2z>!S7fQ_N_6l;k;fjLH?Oe@PRXN9{EPL5!C7!^}(?Y~rz%J2qYUj|I-T_MmE(OeTIj(t5{6hJV1X-_Hb#{Z^&6gW!-Cp9-Uh^)+>cH}D@`LcA1W z^_wwXabeEGb6&?G&5frodJ8cxR$j*5nn_MDdAp()_NEn9NPcwGQI~-9OtyN7bsH$T zTl>?s`oOV-ZNEZhG5@O0A?Y8{P+ZejURr9tPvjz=-!OkNC;^qQ>2u|{GRY! znP0e+8*H{(NKg5t!HweJ6PK6p>-S1#Y;iwmF&+Qg@4r2ZjhEc)05Jt;b6=k$&R3^7 z$uqjq`)x5gzoK8J0&DT8q_*Y0W3e6dCuptgF~|J=ufCjw{H1lNIY9b#JT)m@8z}mJ zXIy+y2&Y;-j>cxT!W?OlYCDAo)SPT%isZt0$#w6lwfrK##*35IrIg>#eqCM#e4}Jv z%OW5{e|LGTHo8Cde}5K3lUplQ8kTsW`d8e%&f_$6Pe+@da|VCDm70H55NRg<`q+Hw zvWai5BSeionYy@__9$UssGVI`8=7UllRDN7>N+ahTgf><9Y!-_Lg_ z?y`6TKP%j@)AKxJ!U0=m_CKLaYKH|&xnB)xMX-%zgY^BGHfW&@A9go#g5%a%Jzd=x zuiI@yGXew!AaUr|_tncI2$m7f>)G0^%Zs8ucb9}j1k`mV>07*2U;8}VZo@^E_MGsO zvcGuC#Wc9G_&~{V7d|i6hF5j56Zq#t^ZUbJkmkAp*}(GQk9&#lGxgn2m?uZ)+s{2! zO^$EH^p#e+X>|*$4HS*Bi0|Hp>5F8Qg=*n2Il4dboL1!~nkQ3~7Zy;_4wm~Fhik$M zp<&TRNV~rkI<8R~96#m`B;uwH?9Z_GkNw8y=>vr2*LbZ8b+=s?T%VVrCSPyFbQmO? zEvhg?W4yNdGrJ7fa|841uJ9G~UZJ!z>Z?E!KChd0C#A>A@$Un0MOo|c__{#w{7VBH zAl4V3cRi^U;$?QRB9(a);o7-ko;0~6xA{Iv+FImT0>SqqBYg|<*Y(y%dWi$tlZ#;xo#b2|z>V?J;;ClX@nJi}zis&~ubl>OxT*Go z5I(PiBgR|kT=DfKncKGI;w@*$d(=-grb5gs)N@qhpM15>Ad6uzk#CJx*+i{XC$9~B zeo_&-n2LV>$1h0UKY2B{m`~VetHvpMoDE`5@~ca(wn2LJOB=S?0vMXKH_Sl&hy0@} zwply7!R?*~u>(giUgmEK9p{`dUWY#(m(KIGU6&VO!P;@JPdHeUkZ+TB)>!*Iw(*fI zLO(cQSbj-bW_uc(9?Sc=3%!59>sNbq=<)H8ANYPRvgdlo9O_Tk^6Bu=)eho#+#(wx z6N`9FsHg93F~E53-2Bn)>@^!8XD(^fKZ@}(xzZQB?-2_;cJ5tTF=vHhdGo{vN7{hP zeAD$v=>jm7n`nAUQx8fXcKeL4x&WK$#YeBtVZ16G=RXSUz&2r&-pT&uJz;R4yXlu0C$=Br(n~qJOo{eqoG0$> zQAF>D9B;W_8OG;zcR@aE?0?(`&|4DvF+u7I&O#@3caReEvWOD*l0xqvRXm?&=A(Ed zoVv)s`<@quDrx1~i)`V3LuC2A|A@!?mok)h9%Y0)rVHEK53@m)_MRufH`;*3h^^*) zU;zwVHQRM!dp%qY&k0i%uBST8FkV}B&qcQiVZ6THD%LOMIQmkD&pzbKib4}E3E_b*i+OMx&s{CqXW4}Q9*y1;GE zs}-i|#Jm#UKX3I%`Z8Mae9pa*AKnO0kr_oxL*11Ve8)WPAXe+BUm6wv+UNVr>;Bzm zI+wHApwX}Uq9^$cD1FY6k`|B;9|zBlFI=kw@|E6U4tg&zJhOFjXBEbafuTTDaKO`M(>S9VC4gw=TYguS9USy|L$eeYyL&d#7p=iN0 zxT;{qRq#Lj>-yB5nVI|ee&3W_o7QyF1>I}78rAob*uNr}lajR&FG|&6g@rD@HD2F0 zV)6Kwm!r=>m7EM4e3bfl+_9n!^;527`m{A44t;ngG^$(&@}(L| zU74PcF0PsU{nP97I@lv#YWf1LkMA`$ZehfDu`L~%_apE? z%VN+63H~%NGFScTm5Q&g`F3|lkqCT!owWC!7<=vm);~2r57ZFzGQ8=vO&{?Z$rjw8 z%Pt6Ev=et&V>H3OHiJ6oh8@(N8F@v^hvln(dCeB?eiXQs9S$v0W{_TP0gDi+ER*hh zD6lCiV0c~!XZ+3&b+>!L$&vJOhZT&M0ngEzw^7LND?H`ZKBN)ul_Xq`8nOkFXTrq? z?cU+)Yd&jOfVMUqc8WLzl`&!U!dI%amOX0RkYW9l?#EIpFl-dI)X>4_rL684d4Lw* z@1@?aqz1UTf^UfX$Giq&UI$oj1Tvue07r7uPcWc)6n}Z0mTaZl=xPU>)2EVDc(MHJ zFRy$OEt!J?Y~YpkucT<=q_h0 zKCipYET6XLWhgz#sL4?85IBbRLv!mGe6vc01n^IwD>VqTHP3!LeRY z6gKyR??3YUUma7g_gbNNjQke(Odyf4Q7fD@?3OJM_YwaiVjZp}^;cSIEIMxR4R1+{=m2%ccLDLO311zK*=GHl#rPSKsXSS9U%{ z%!`Vyd371Z<4Mt)ghLS+FM~$vvMuG-Aar4psR{ML{OiMi^+mFxckd)68@h~lfiiAi z8+dx1iwaH7hlAs!W&4lR!|iV}zqcU2pA)WmDhJzxS5Cnf;W@ z4I=nnSUs^M=B27Yw$BIY>j&kX1KejZURHE$WR~abLF%iDS#B~hFQ@1;vq$fvcYSIS zmC>SYAbEgBLo72Nwxpi=QY%#tGjGtPBV{j8^9)U$-;eQkcx`=N zoE~p~e{&57!x{VC3gcLQe};$u;8h_mkY|xm5wlN&{o>;9eqF}rl~4WPmVN|2ucu$` z_1N+GgFP?DSKb0*Ua~UD3@nHj&GakV@_*)ZC~dtc)yEIJS}zG5`(}^&Tn724SP=WY z(fjQYPj|AxWxn>wPNz256{>C1vy=}HIaOZwTA}*@dAeO1Y~J8$U_!^X8RI2>>5GU) zA@Z-+uN^nsxJdY}omh7p-QQU+9#05R_A@u4`5k5Z7LA>-eAPpDUrB8+CkXCi?fb5m zhUN*(>fQ;)=k+SU$TO)Ki~E1qzxK6LebV~s0_xk;Qf2Ll>tEkS?2G<6uQe0BW|}aK zbo7VUUSYopGG7}|tI_3T{Yh5Uegj<_uCL{dQVEX~F6+hJP32*;cCdk&S?vAgo#Lqq-!>hc2j4Jg6&2?g4mtsfv zS0i58yyqV1IwKurcbZI~zI?TDj=~qpY{07Yk>Dp!Vtv`qS&K3Fu%mf}^Zul#Z$Mj0 zQKX7fKG+^G3O(dh4}}%NJqEg7P|NY*;rsuX7qF9Y>-aMr#A{Wpo8og0!KvQoK%cxF zBrcKbZaDY^*Y6Ff#rF+}p!sQ@*BQsMuy}k?pW2!5j0=uEe(;kO@!D3ct-4(qzrOll zKSkm>b$ni$t$rj*rCu;IAYO8~oS2t(Uh_aM;`MZt{_gds7%z@fhLOb#w(#mnZ-4kO zF|S8Y?lC2ovcuc$y)_EL9v5V}MhkA6=H4q?%#Uih>@?d@0_T*+L@H@uYx zeMgM)@;9S%l^s5>ZX>_H;{1_5xfo2w9sfMkn2r3)Vst}-a|f=z5*-zu z+M;=^l&v|d`vo=E;<5itrCdM~H{{T&hBU>Z{yFkNp;|xjd7a6v-OBzJUtdo-Chrer z`-102-_m#Ah6uUk7QgS*Dqf|`MqY`MIH_nG~Y0*xCQmmYa!e4!?KoGUzBX4 zC)CkgTl(7~Sy`nwz(QKB@zAk+NEDE^KUY={`!;k&%^Rcr?K*yxRHMD7^$ z)4N*_z4~&Zu87y2t7+;3fm_!}8ozij{OSFedAN?h_9oT+IgQn!W1DZGXmsh90@|OwJ z|71OVy-Z&G#E=;T-R-UMAef-#_IP`sqD`xXi>~(6;zW$U3@hpe4o% z6gA1(&At%x@}s3vD2`@@le=@7x+q#9&e53NIw2Pd-UUkS3aNvyH_f;knLHqT@kO)( zDb|l~%8rfmdmrMpII8!SGnSC%U-G@h5!F}EkaAXhE?fH^|MvTRNg}@j+e1O~b(NzL2h>&r0M)v-v*8Je6570n}v z_uEw$&n|=_UP*72M5phgy3HS6&N3tSc2HOX&C1W_t$pa{|MzDx^zVM?isg;?U^y1B z7ngo<1ijae*ruhZ)_Vz(C%;wgJy;7H$Cw|-{&oiqg_k@ze%O4`whQI2Z*rk{{KHFK zacipI4eNDzJyP91D7ifpMxtj|6@;;TB671>M`j{Bd=Ga%} z0;l;t^WVXERh$u-aovFNns=6%*z`}mhwS0NaAe`y_xQKI&QaTC&36Waw_cvIdwpG6FHT!2&K;fjyxi;>Q$Nu=!LJzJ_{smGuTo2v zY(>N?QpKh0fgi>zwW#yf!h#8uiw`M7>qNHygwp< z<@eWFjtD&wWCugP8+IvLX|Tem$AA7PJ}(`g4#T1`{CMOv_F6Egaf6E2lUL+5udV%F z{_0EMx_Rnm#LKbQRYp$}%y{r9)34Gtzu+Xkc%!}uo#gE_*Opq4Jmc?z{2$8O{ zb`h~T@G?F3?j`kVpbXwWyy2`H1R35rF#HzdWn^(#?Vov6f8#MYdDpE2HtX{Wen-BQ z;2#Vd&MLT}G3smIBcOZonaER4*j}5^L>7+vZI3?ncf4bb8;?~{sS~LT`1cDpr6;nd z(?`JXmkA%l6^Z@+u}Gex30faNX-wA~w@3B+_$DdKDMR?U#O`N#&K%l@22y2QiT!>r z?VV`lJXYX&Rk5k1y9NC{Hfx!g%b?3L^07#~4rU`f%*;nUz{Q?rKmwhU|BYwa&3iY? zs>{KE>INyj51jkf1K5~#mX;vAQPkeKsdXd4- z4r3*I^Xbt2d)rSRJD7v4aCu3dQcfVZZe9C-{_S6)8dB!JYaP+M^xd7lI>fwwyI&U4 zC-aBX9)tWeyU~36KfH?M{A?msO<;Sa{@?_GSYI7IV#e|eOwjDp#h|Ly2uXL7T}0fo zp?4$83Bk4+IPj`5!Xe59&g=9fJX68e$A!I)yxh97YrH(^Uq0`DVzDl-o|#ea)V^R? zsVp^6t|F}Q{~pUv8jQ}V78XVc{#@x$Z;*F}@30ImFYO53oE9H^{~FrlU>U!bYJEF{w#*ly!9<@)ZhSDuOnBM%+et=Fag?! z@p*}@az|x-#OD>M^LW$QyFQQ>b;WX8ia5WYQDyDTM7(^BEPjvFAuh_BVlUY;1KXlO z-e{vGSn?fm*w;$TYheEa4!u)Mprjk(RTbY1@9x~VNBiLt1oeM<btc#`S<#G{l!F|hh`Y; zO7U(Q{0eKlN`;CKQR=aS_YqT?fbn!t4L5bF4Z!EMx|j6IXBT{4pW46VB#hg_c17bv z&0oa&@(N3qdxUr`Mc(;~j=~-6)87!exw9(*@3Nf$5w==Z)DomhrxOv}jSTh`s zq!ND@c?n$Z^`(|k)I!Ve+vOJyx`NolSJzKvA6xr=nWgeqw#p;Fx90!W?@vd#|EOJX z?(+J)ViMF9$v04-Sg`2L%ZaB$5{ppRJ{^2s+-e~tb8Y`) zeLT!rq{3tiO2Q_DcyVH0hZBEPT}QlZMBZJAlg4;`WnGq`L;E2mv64z2MPgoCM(d_r z2hcfp&@@xANHct(5K2;W&4J!z-|G|kH87Hvc}=$51=NHKws-nqyj1QgbLf8CxyCC^ zXaBS3?MmzFOOg5H(G9nqiMhtpBCk5ixwPxUP&=2iLqvVAkkCmxd*rM{*{dHkPvOdBlKxmIQYUy|q( zsThd!`((G;7Y|67AkwZWR4Kg~e$kr83m4=-!o+#!;)NR6dTQ{|HDgz(a6PIK{s7}e z7-t*fpTc+z$bXH0c-(SbUcpZ%sk?s!gFi)zOhG^9_ocI&AKfBng;ykT>f?23kP}R* ztNrd6uHWZL_J}l`Z(ED|f3J@;ZhLQE;jn=rY5L{HEyTPmUX#2eLA)NlWnrQZ;929f z>1x$J{bDnSFmWhL2}VEv@--x^8=xk$=E%FjOlY9ke0zwp8rZDm+~;>Vf^5fEzo`Y|wZDH^N7P}_3gfk^<2tVR z(qvs;1@)x|K82kE%ABu<6nWA8#6O=Wq{4Yi_#GC|9dX_M9nBx&k#Aakm5Z;h3wGOP zw$HW`{{H=c>udU_&f_{3H~95!pSrOE@qYW$D&f;l#7lEGV{o(*56oAQAM)0c1=TQ( zKKgnS7!#zeTUI637mxGEbOs9pq*!u)q6=yQJ^^RigZf#pB5xbrGf)i=XSGa!^E-iC z?fvVY7LC@vUwWA5Elu{FK1ztLu0C1}?AW zfCu9bnVdjcC9SLZ)AhCA^WS)UCL61G68V?*BE{S+7shMHDY7<&7e^uUi`fr-QDR;@ z^?2LPJJQ2@FOe%yQx6?|vITn2GQc{2j8W`;6@>FjU2$!2gh|C6m%PRd*Z$vUE&++B zXJpp&71g@M;pYugAZhMxIq@@h?R)%NUwm7eKg@Cj!pgLtZh1w73ALs6@QYO~d!YFeAn#c^yZ@q7p ze&Pa8*4umfAIgL1m2VSFM@`}KE2%``MWYcizSVGx++lJ8o-nwh!OG z46m1cPbqVOU-D(Yt`mrPHOO}E&1XRKw} zJyg*wFkPxjJDHOLQ_%}&7J~43G3qckR-eGnCrY0+2s|xt2GQZ2jF%q~$D?(2^T#8I zm({60&Zsc1HD04OW2>k^00>jYHEJXf>x~o^e&wF32lq{)+Z?7dV4GUz ztBRjhkWp8sBtmck=FEqr?U%54gyG%flRsqEc=5D+kfbDCpI5}`x8WB2r@>bJxu1nF z=J#}ic`f5jtZ1LA!tDAu1yr||i^>UE;^q?}-x!~^hT!X~sbH{S>XR>;f6&8ksz$6Y ztyYOMZy3>iInU92K1QUYS0NVXS4rW>wKz%Xv!X1V(Y z>w$XzJk9viOlTXpHc@fB8gzo#;&PmvVN3lpWiJg(UkZa0q!*FCI;dmHpT5l{P*jl` zCmWhVBLD4h_nv-S|EfQF;8o@LX|NDAIGtIB`jP(MpT$t4w>9||<1E1MD84VJJq2`S zn&Z61@OcF`xV!P+#P|F5o4h$^I^Do8k>JP7Ow6nN{P-Cr#B14I_Uu?E($ODYSBk~T zMn_D*H`IgPtCv_`vnxIN`b-Sa-F0fea!WmM{TRCLtC0!4A{mM@vejT*uDGY*trJ9i z$!?Hg#q^c`_0ld$L{@iDoH@HCUmbz_LFORH!zspD8GZpOJ3ES-Cp7BxV5hxnl zJ>4rz*5!5n;p5;}&A|}x!4EhxG5=aPPWyXkoE7dD#A&+Ir9yDAj_H?hd|sPb!xchl z@bj-lhZih-iC%DwZ|>%f8DjqmtL>o@L+j(`k)CrahcI4-8xJ(*rkM_ zpq28T8yCKRRf;{^@>cc)oOiSB+Bi+j%bAMG=OViQ)m^sGc~}|ab-mi?$T3kfkovfI zod(@M{_7Wi*GKgb(T}$U7~uN@udw(B4Y0k6`cP_S79>-@E^T0}2D`&Y>wadT{C+v9 zDe(x#>wKf>$z3QO|LUt(E|c$5*ZRETlgq!Xcm_e;-Hm#W4NyOZKmNtX_9BDFjTOvW zJ0BT%qxn`Pz3*MG;;)ZcFU1Nf1Mzv0wb@*hQ}6^k$8CHMNr>Z7!;0N$2(6D7w~CAC zo#KIxq6Ft^Rvs`EGKy=QHiOx%e&udq#Jm_fE|M3-F#x?N+rB{N2H>brA1yUc;|zHg12B{8olx@%v? z?HQompwaZXK?CS!D-UamXG6zN^EfG*YDn%-+CypX46XOY_m-W;c=ZWX>1F=T4G+8M6x#7B3=Y0T~1LNUI=3~O05>y4x#q_ zuP+2zz}x!n>$42R^}?Ux?t?s9jG$scb^#_EAVA`q@GrI;@bx(P>d@|L7;hwmd);#Z z({tA!CA|I*Uiux?ni4yKr&rr=@lzq;PD1>g?Eu<8&ZS2mGRFGa{5!uNX0FpqH4cM? z!Q^+^R#?5zbl_cz+dg)<{7`45J1iCCVvF;Ngz$NpMKKvvE#dEndIG2Cti2t;OrrW( z+mXV6+F2qFBHgsD(w>Aft}Yn_Gat#__IO&xl>;?om1iCYa0&!llZ(m=!a-N zsNnk-uUidwbmDO+oG5#8sD;??nG>}P{?S+Gtn-^RJCv9H;dSk^V4{Aq84L%jt8$$n z&R0bTBdt4X89>ve%IwaD1`t{;xpVbt7K|((3YJ|cq$G>)tw;bWmj?jW(e~87nK3lTYNBHQJf++DFtTm z^XgR4r4V9Xoplu1@tO3Xx510Mb58?I-fX;l>1Z}EXS>uo@2`f*g)nn>Gza#1CU?W3 zhNEkKv#h3+sQMrAI8Upq`c&N%2-71{WTx1=i+{&s2>=}aAot_U`P5Yr5Ah642SKkN73d?NOBz%+d-9f%AIi_W6?$p4SC z`;O=8{U66KBa-p7XR?wJB|=7dgci!)Te3&8_fDe7PWH;)vqD2dDk%+=kqQYBWmLcO z`P@D|PhaQ#d_RA;`txhWfHGajxp&LI&#eOyd@3+n0S%fno_U-GO zeBoLczgt@K@T~=qEtOO_a-q+UKCCC~XH!uxy`wT)nok%J^*T+b%3Ng_2v#gi)a{wb z{Q7^tPq~QNin%%~Fqtnc&pbRygugI4;(XuDMqBf?Htz2CgwczheaH`EJ{2ly3`0%)%%#G z$1*^Mp-IqYUo9vpE$eY=S%NE-vhKa}sMkblX8c$dvOd9+JsVJKr~IOVi`<8YxR?9! zQo}p8fgs&KzjQqdjjwmQTVx{kvx3E?4qe4>DNv&5zqj7Qm|%Rxcu(?%%n|nTP+!pZ zqOyf64fnlTZ(zNcPaEt>LcDNO?k8g{5SPT% z-&T0QO2Fc4^sqD}@0F4oGB<+4uXnGhA^pdH;{Q+nS|9xqryj%rDWVxgNx4n%fRoGC zuq6|s`b5m={A-bZlGbfpuPou3kn%yBL#UU9;`#Q6iHO%$gYiiI4CP9Cg^7_5`q1Uz zwm*&<{kw1UdO}&c&GbtUJTRb>f8d7Zd#ho$2WMHCA?J>QoXKK-K^P zX3KT=Z^P#M^aAf+;*mJ8VZBZJiM|QiI4jEgq%t7VYD9|~R}0k{&Zg(}%ppE6binA9 z4&M8buDkD?2I{31z`88Pi`4f-yyR$w=eBJMgpA?MZuUNC{(74>e>TFJ8ESu0O1z^$ zo`={p53RBh_ByL(%`1>iINy`d7S=Y)+kwkxZh!uMtXI2)?A6PNmlrS8a?Yb(<-05j z&TcjV!|R8pDb}%Gf?Y)dBUf>7zf|9B+p#8S=9XK89T{-%0lBFA-dc!CYu6{8H-r7^ zoA%A`N9!-6#N)nikp0a`6n$UTrFts&XRYl~AvY%KwbP!io7OS_%1>^i4R1#Ka}qlJ z`9l+#fyRO)Jo)q5v9Aq?dbN%OU)=jR z0QT;9`Dm{hIv%|&I_6z!n1M(3b9l~L3he4~zF%>Hu$OWAX_~be!tb|#yruFuL3auS zuT7+{#$mmNGpE)*Azt1-+gOFKpk98`v6;Aa69{hZjL$B{dQH4w3|2?RqnkWy#rqvi zF#hRDcnd`abi~Av_6ycR{fE!#zu%a_y-&TVv$Z@I9nIM??0}0WTIVeEvFoSg@Up$pF+E48wK# zxbQ#sC*M<62(6c|(7;{R+ug@?8vv(K5E6bd6x0wU3R7e%$)ojwwt*Gf1N*Gor(9JlW+2YXZ~lo z4zO{8)vXULK{I^drtj^!lfwX7+m3k#@pIyT?vGc7=4N)qZW>tLO!h-qqYkK#>)mm> zp9;Oy+K=Rf%3xXT_T{`GV{l9}+eM{{=KJGyflB(jk$mrAc%be2Ipsi`*$}!qT^Lmm zC*wUyxF2A;>F!7Q7GId2x2ex{lfwT#S)Jo`*CQC9$d0k~d{i<}cg?NXv=jDv>!$cq z-KvFPd=-4%axoV9etIpMSeH3HU856eQtT?qqmY5ecwA+*d`SBJbec4DR9 zu|EaKqxrw=d0;>=zS?|0OppB}obU5KUMr?I@`s%PHn)n7mg2AT$Lj^%Lk3SIzAoE- zzQuQz3*?@ZU%xec1fGjg(c9cNfWbDM(>x*A@u+F}!#M8<9f<2$r#G{*F9wl1ES(QjLawT*L|ir_i_Zu_f+p9iZ7gPP!@Gsc1!=Q2Oey1 zDOQF?@qhP?_1EHC%BuIZ{!q*HFuT)B3IG3d8>=;oyU7T9?e-UElO%(5UwdwXD&hDl z38R%P&LkXPOCAwa;?~G_d4fVqK7QEv%26Zx$$@w!@SNGNy^|B~m3F?+Fs)x7(&j9B zG$pa)@ynO1N-VN8&^Q*oOQ^mU_@19ltvHbiqVJ3*lak9Ijm(Z@^@R!Wo@(j3oQ}Rv ztK8*Oe_9&#qUaiXDGfSAy;dXIk1oafKx=-*o|}}4crQ<|3Z!ylg5O+>93nT9L2tCs zO8VQsduhKMx+!(U7WumbX3s8R^VeG|72&;zmzQ;;S>$KbOT*9W^l+vD(1i5`4=rK6 zCRbQ5wYt!P$Cp4ar@T5~(yrh-{UQ~}l^ledN6SHSDSyj_OQxVB(f=lV0riTS-j&O& ziF)-#&WcU1=o0mMS?HWQW9|by=?CGBAKEXsjr1UmivcsN46~B_W=Mu0O3%07%?OW2 zb|bL_Zi*}TINn&_2UI@L%ysew>a)Dbu`1Z{NJnwdMQqI*L`h5n^u@RVH+n7CZAl8w zR#8Yji!=mBH;2hY8SL|;Tgwgha8i1pm*+qJ6M5fA(RlK=BTG8cFDZL)2WbV|`mQ)4 z`@jtN8BctbJBmIJ<>-%WJ>e(`^VRpq?u!*D8>yFMTGb%+#tD_c5}QH%eQX?$rt!na zk8JY=lA4o>SDMiJ%iBCu?Zz)gsFq(QpVCc+n`cudSi}i?b(sqsi#7_tUuUD&`O=;u zQ)GX->jP{gl{Q!}rSYfEOo-RbqqnM+Wx4QPft@$0FZap2u zRSiTd5rmRq-l__{pyliemN+1?oj4YHif$@DG8ra(D<5~3e(90)T<->Ofr`} zaj$JU7w1-=Abmy`ukIRh*@qvGOG@zAB%cYUNmhM|&nClZqu;@ClK<{i==kY|il;5u z%YKq2iN$)!n!S}xMCM0vT|?@JzfmvBVBO%?ng-xrDLwv$8XI3nsSos5^U{MaeO4cT zO&z3-&s!inH$%J1=H10i6|n15wB}P}onR==P5Z4P8eetWB65NbpvT=$(NVqU|FJ+PLpE<2V=yjnYW?wC9G0%Xk8MiC&IE+ zdgn@&x7>=HX!@-SUDM90Bt_^vzi~V=#l-maZ1#pH-%qrD4??}h^c5!eaWcU#p6#F9 z%#tB=#oH`&-@nJ#=$*%(xkPNiAd}OXIutuUF4?Ivt0P|7)#{rH<4~{sHp44Z*Yu&2 zc`NVr6s%X8q|x3uZh9~n{{HjTwK|9&lfF6iJQW;nhVS~oQvs^D+JPhKSg+f~g_N!b zP_Gr)_hp}VBlSHIufW%zJ)ct|zz>$(BzRjVZTvwEEoa_>E_Fg}wE$IlB19^xbH zHG26c$u}+^g88e~M)Lc;W(ROO(Z4X&f%STwKT>=E@zOPli|eUDy#l;vE{1p+f!!C% z9n|euueaa1^9x?kLFDzO^D+~4z?mL=HHkV6UidtTTac`P$inU*!FO1%4m+tMld`B+ z*6;awIvL_#^>W+jcKdllhL-%tS8k|R!caj=84VMVDWn`92up_LT4j^P6NJ5#J9}vK z6YL0jEvUV?y3phV1zf476^mG}&BOM$B@wR$8IrT&N~l+%Dg(J0($~=8e4&B&2-Zve zxuhj>`JH?-)UV58k zvit0eLFiCYoYPyZ*N@2~JJPx6p_}Dhc7;eiG{o>4c88~d(7?GHz3~-r(O*pD1CJq1!hR8{@9imQr~(Mr_j&!~JoglQfX?X7r5;lS{QPLD=%V|;p9y+s zeg#apCWG9A{^yl%C(E^ay$#>M`NIQg;fbZeLq?_7WRkEs^cAClN1aQHoyC z9;9CVzkFZvW@daq!x&12h2n1^=OL5)??3c=_qGD{Wzq9=a5$~?>bLuK;Jk2sjBa}x zH1JHki1=6zABAHa=)Pj(t7yI;GZR@~_!D2E4Xp>*q!6!j<=7nudS(D>q7wb!V8$^TUnQ zYrAvTh5yN4G~(lo<~LEV{rxUy1P>U)VME*H^aoh4@l_YyG#NU0B3gA)71fjjRIpxD9C@<6dZ^co-!Yc`NyvEY(3A^YMf*TDdXdpN_1_%y zf*X020{XjAFPiGfs&76laOk7ighgR86Ye<^*zs5)HEa=zcyVV*kM8K<#>W@4h3(V%AyZ`iOJQ|>09${BzIDP?qx}G7>tBREC^5ms%u!s6$z?e1&$uKPK)C)=-llF< z8Jvy3&PFd^3#~XV&QRz}IjenC0$YD6hO#NvAzmykIjkDU{{Mf*V^{``rfA3vUhHm2 zV_3niuj*wqsrC9JUam8T#AWMY>D~hwKaNZo>>7>C9V&x7+ifhzF=epd+6-4|uX@b|IN%Sm2{GWoF&ux}ZC8g9aj&kx3zbs;jT&vVc9P!0SBv0KBD$RqNU_l%4(s&a^cAXN9FH&cv@3=Eou;-=1 zt3X<8e5La5y|L_wgXw_!Gp+j?z+-^4q~m7>EE|^;P$gFZ6^8*eZ4+^?t7i6@{>XcD zV(~Q<5MFgM#}6_jvSPD$pkB8eFBtT`V*#9u{9u1uGWhM9ns0agcdtmAwCl;&PJ@Gg zxGL#?@bXhOlrcoSR1>T}&O7tq^VcztV;j$kpMh?-nI8tI!QqS%7T7(;w(WIeGO}M3%bDzFdkD{uMDo}4sMgxp!_$CsvM6Y!z{VFYaB;^G z#LH#HZpRM?)Jrn)x%9z{HX!0oacsLc)+>50=0!O_4uTeoZhO!+z$|%q+{jV}{Mu{P z5tUO3mr2M?`0|K*wZ=sU7JWdTpH-Lkb}f1k%wMeSad#bCk@0wxfgzCu^;$Xl@=bs- zD+p>z++%*03~N=-cd9)m9A9B)6ECP1xZ|(0G2e?c515ISpMe2(#h}eNtk;(Bd^G

oUnta4eLw(zxJczxF~?0C$e$YM8g$AQ5a)~%iU8{p~1AnK%-naKWh zp9afms=$25gvIxp#K+_6qF>c6uxzCs&U6UIQ?!BNS31Dl0=A}LNmOC?H z#w~Jd^H3!Szn3`FT1MRK<>B*NW3!2SIqJFBYq0r4PS1svvkmT1kJ1;vH_&8yc zrrMtjkzG#X+;#u%#ePQRT~nS1j4PJjt=fh4+MdyTQNYs&$}Wdr9NL0<9jM8CaAL_C zF5P_k%bWr0l~#SWxK9=bGDZ9Mf85akzqhrXQ`&X~oEny^+TT|~T0-6Ywb#VQBjqT~ zFT-TwUO~#24!;rc2fx?*E*pMC`wMNODULi?V21YF<~AvJl7ahc!c^-7;pazU@#Vr8 zI+8!}p(8cw8M(uTN+A>m~17L2{qc9i1~+uX*hSE=zJ8 zw9EHV9NCJ@kCQfyu_>7#oXW^<@}Lq({TOch%Mtfl&^W(q2NiKI8tO+WtwR2=i~sjH zD-HU-19D69gda1*J{k}8Ye;>+t8%Y+Bho)iXg@w;UM%b;civuff?V$OMY9js=b?ue zS2I&{k@MZDUq8OWfqJ#<44@15vIL=m@wKgdSg#}WX>-@V&_iCX!^c|0E73iWcj#*- zu&y5wcAu|=;*JPEF64WMjqCT+f(AjA`QrHJ$4%7C8pi~Pd-004xYGsr0dD8ZwF_+M z`ouDg$A$eDnV~%9OZj|dGLX8r&h@?hcQ1~gZ#Vro=nk**GWK!xV!aefL?}-3dxP8b z_NESM)GM`u#1Ng>vdzFmcwZgdI+;{bJ{de4?)&6QSVDLp<|oObK3q& zNVP8x33`QHFWl(GNF|u^gpRn^wH2qqA!}ba;OLgu6M=ed|Mb%|{1P*?m`Hwqkedv7 zJ$J`=eEvOu^__hfudD3{0;hWg+V5e#s&u2*rmMVR+Q+{jG8^^!o^1BC`>Y|**clDQ z7h}ENUYS;YS4$5%-^(RfFV{oZO|r`^<(Z&q`Lk^5N+sN-C@frmhF$;K=#}R*Om{zq zxEEh`0c&59FN~Y`k66#3^+ai`a-du(8<6?Ck*(6CfZGl3{NEO}BN>SO3@(3>{WQ328g_$J6p-Z298&7eKT*X!YNj*Uz7^BkbIE!W?>UIokA)3i=DSg-r)VI#Y* z?Zf9UvotR`cV{DTnNp=1Ze7IR$40NXH>4j|#sZ*G*yrWc`eFR@Mv~ZsZ=nPy;I@_4 z3U5z=dvl@Y%_4yJuk&a9OT1x1^r5RC zG^YfYm&ps^y{3hpd}thFOT_@DdpJXC+%r)D}F+5by4E2%N75roXTHqW%= zLG|EM8{?E}c>T`mw%42uCcZX$#mT%d+P`7~VP7q|Mw({v_p#B7M(TsnKC2+8s<^#- z7dz?|`b{-QVHYPXEhZ`$Go`>Vr}wZ%G~xK-6~E#=MH7j?&PK08qgmV)-~B+D+bq#x zdl~+||9I)lk#oGt_J=@$SI)IfJMdm4Vxt93LAJ27{nsr0-5vO!`{TvLXJxbYm){O#gGb*oxr`t$j)V?$ujTpXh;{GZkw-_QsV1 z>s8;L&?s`*2hM&Hwbi0Qy}T^NKC<0A4Z7y3FUmQvUXgAGpH22N!59;BfKo&wJmurR z(WjROy`_ifQv<3%_=IkeKQnTk#fE->jqxS)M9<+TC(SR2*Qay**nVFR>P;Dk?)bo!=?9-{zN22;L!pj+tWKbl>64p9fgO)gt^O;kj4UuU zd3-%+s2TS9yj+S5&4-CsR;8b6YT(2XmfUrU{~z&{q`9Nn)(m*9Q$Fp>A-w;@cx}u? z%l0sc>$qGJ#enAfhXbbD+3s<}2kNqLHpw*T!yWk*lt?(f!dRc^XFdue=+z!ALV9J^ z7w(M>UwZxn8(%JwO}3dQ0^tZ%!!iDQMTg0V)!+R6}fhX-aX9=N(nX=lfm<+}^vdjA2hmMeXtw z;r&f6)kYm%77PU;qaP$6bO4_p*3^Q9ejxjaz6^N!O@SWSk8ktti!-|a?nS2NnUPcC z0Y94jb*C?3y{x-ekLHL40JF|-!>#0~*X~@ogEva8fhj}BI|_##kHY+CzbqCo!Rne_ zIFn5asP0)cs!GfQ4>biZQI1-m&)7pp7h;7Ok8*B{+bd^LFC|81RuO9AUT=Z|SVh!B z;Dx!2PoSb)L@@i#UOJcT2F|IY8V%DSVO#UP8-uzY(1e?v55jXIu^(d@olE&Y6Xk% zkI{-x^T4%!(y!f^zIXkOh9^SQ;0ir3p`oLDcPf(2U_|0)@E5XpcCff z*j9zDza*`0*n}pl;m0F`j<@u$Uq(d7qg5E$iePpygbC<9^UX%{ecRJ2l`0iZFgm~d zvhGNBVbPhbH3j)na!tx8`>c@G_FlEDE7sq8cMdTeXmLSG+^Qnn=>%=)R?g z?}rF`kyIs?-xYNvnD5I^>OUSdMZObvl9%XN!+PD-)pN=ivxBmtcnzyr)JsoxtBJ); zOL*VCiZfin#@Daq(SRGGOrYF9G9e?}0-Uchg}yE1!MMhu&K6`J6`o!DcHj6-+{@*b zvX@DKF_0t{mvHf-eeoOXuM6j9&m|TIBl`}Non^nEjL$3c#iRz}NPowklwYG?=2DS! zTTa|%B_~`@gwcB^GK8HX=*7tK+xt*&IPj; zjq$}^%G0shN!+V2w%^wsle{w!S@{52ZG@G z*I_lLaS6ix(VyCwG>vFI;q1E|5sLKK`SCP!2wflPX_&hdL2{o0ogW9h=pI`$S%7zT zzDp@0C$r3nP9^gMQxF`SC{I?+;5? zyzt}EO8ealmvhE2Dt=GYWQA})zhj~-|F&04C-}3 z;GLedvjO*2P|{bwe`rp z!W;9~Mz4imBy?n}Cg734l`7#uxL=MpeM~v!NGMF>NEd>(sNl!rQR{cN;+#1_aXG5| zxG{2GlSwdcVTN#gDK9^{^dS8#K`-iWBfTelec_#jOB=g8HojWk@bWXML;_g}`^Ty0 z=y+sLwvNqlGz41TB8REB*zrjBbFXFQRVK*X#Tybn*a{jZ3Z!46^C9fkBlphQS~xs^ z=SxBo@%T!mreqA-VFDx;@-p)(dHDO-n7`gnF`C|q48eGfUMldL5#)qPp7xqmTpA3f zF+WqE{CBUkUOUTPr2a}gsS~zk3H!Z!VMo`iYe|vdcEf%v#fR`;B?f-jqadcw{Mekf8*5xwV%8x%#X4075AfY zm?@YEB=u{qrL7?QN832qHrM4t->>6vZm1S6SYNzGB}&|@pL|;X;b9Z#x;}3B#<-ba zd^HWtPTgP+0jncj-^4u7di5u%DF65tPN3wT>ipG>^#3#8lQrccJU^1$rYqa4u8hCV z#^-I%iI-XmoStA1ZumTOPX)#+ZHi(qcXA{gr&boLo<;N5@RS&9o{JXTkks+qx`NGL zN&yv?#a2x4ZqI9t(55EPQ|(HA=9drocZ6fBt7?HYVD+oJI&rTO=3=Gk2oq4&xj84F z*NDH5jpOkk$=wf~M?+w}bKA4~ooIa3v?cTL-{(ZmgU!Frke3Q12Gp58orJx@jF^Mk zcSYf^v(f7S*>2ZK5--T9(yAEH$L6nsHl}m#Tv1T*#^;T%)lU5SS8oUH-t~DcIR7+2 zY{Ypd-uch^myU>(=$8Oyfqj=l7Tk)*>NlR!F8zUKsDSJ$>J$!7tj- zzT;O-(_vwZR~=o1Jm*g)czq}&xAq}&-nPCC@}XuvD4S1BW!|iXgLh(woa=~twaW}G z-l;JKUa#HnUVrPw|J^snmnMJaoHKnW@QE+<@{6Kg&x-g$AFXh}Z6m!iHrc7b|DxK8 zW0>&!G-6(CP7@s26$tNEFn_?UXTkhQ?Pv6&D{DIxhpxNA4 z9jk`*%64@ziMES`hnX2!`NQb?Jqra#4DOp2+*D)y)b$CwUg)Bn-Q0+L=Ts16m$XTs z5fnejuE=lBhtnEH@;`Rff;~mUTKyQde`4eOIIXC8+khF_Uyg{Eh`YPbN?|Z~z5n&= zU=Zqci`VG(D6*eB7pbLmqg^UgCP%b64gR}V!p@wd&qR@PG+wLF|7yb4t8!Zo?>9rd z%BboYnHN#7O)hWG3(jl6582s++`q714)!fv`?!()kWc;|p5-rUuW{5L#;NR zJEsPgTn-CfyM}$gz0u1|FyulKg&|QdG8rq=Ok01zoy_=DZGp~@3K#dk*;&U1l+rJC z7rax!?~AJT>4pE|l^=M`@~$p^+;7bHCVWRJR8KgAi-2pzksnyEbR6aHlB1E3#ajQ~ zU>2<>bgSy=&7!nHd6lC+m%A3I z-x>r`a%1~f&0>0f%R8a<*W=#eKsHuGz>Q1)dd)e6zmJXitH?g^8p$bt2wLF2YZ8Nc zomEeMK2pyHr@Q;9j$KNHK2Lc!t8IkeZF+X zOY%ve{6c9YJXW3YJ|&2Hc{{1Cp4_4Xv?+65O1rRL8O#-&UK&hrkMEF1opB>bj<`Jeu;>qWNHInOQN<8ph}qyITK`G5bR-w!RZr5y~7V}fg(&aKr*m|{<^7X67$B_MyCu+rVv>PE# z%eMG_T0R(OW;(v9sRrvCQ~Tdn5clGKTXc`No4D7>lY={ju7x4{{(W5Rzkqt}bllFD>%icH5zmhXM8~7M z(&HpapFkjIhB6Jsj_@zSifTU&$vi};P^}^9~lFF`xIQ+QZ znD3)JKSxWf1c6~5o2!N?)~i0mTC!tnB+wekCmN8U^~4X=tr|ii)9CpLf0k2kB$ujlVZ z_Tem1v9>9GgTIfB`Ac!PM%FggAUJ5Jbm$Z#Iv!Qn$9K1gutC&8fsAYqWPftpt==p; z!d}V`n=K_NeDT-W=(Y7_-lL;*USN^OYO2ACeSS2Z?pL}-5()Z|EaCe&QLk$L$MMy( zRzUvd#K&wQtk-h;wU)BoMr)WL|(kDIQajueh#gDbN!!X(wHK=bk3`OmF{ zz4GF>l;3WVe7AVVV*VT{GuUZtdGO85B1_1 zJL>fo`7ZbRkW^3Z5Z0^XpguUbGJ%iwU79y^O|af^TtfJ00X(@9*>$J67UJL9UExLM zS7P;qk$PrMf}IJ(?yi4E`Q-!tJ~ocW%j8V%j5oufWg*<;>l*sJz2==%o)^ys!>$hD zZy3`ctZ+bAK8A37g)vx<48QdF&o5FRr8OU;4PF-HvxYGlRpb*xqFxQ} z9Xabs!$9%*3{6=UT2EMPSDF|QWd-FRfnM@lB;UWQ{ARrW-@Pu_Ma1SfxI&`$Bd*|c z*zxFFrA;d`69u@~)ws!BsF&6^1!LZ1TVS{H_uVOn^$LwFQdp;8040{d>?pfNxKd}( zDhtT_4ky>Vf|J$oDq_p;m+r)$hY|*~6nK&OpV)Y;=-B&US6&DhNdF|6Nwwf750S6kpaecR^s47Ogqxk&Xe^b-y)i~YK_U%e5o*VwYmzCqR}@<=;#s-Wy$&nXvVb3t zd|*gyDikKy80aJMN~r!KA>Z=;kS^i(+nqVlLXOC}^t+_a*w|mjdX>db-xT(bhR z2}gFI@#TB1qfqU|8g57SG*I!g zdi3TMVK3g&t*YiM(fD~~WBqlP|G7{hyC?AI3GO+zggw8*M=SlV=I3Y_$ox>Npn`g- ze$-Y`ICBb)^X<4+t&R1HeR<(rzbyk4+e~u>{cHv5OU`?}I|{(LrCDfFwgzG*CW`3g zhV|Ju!($*%E6~QS7v3Hi8`{tB zfgg`)hpK8g@(qc4iJs~fuCxq>##0F+@m1(}v}!)teS-`6zV_NZMH=LNn%uj@hTUcV z?v+WR$QUN-0v{#c>YBG;y|x)0>lED*1Epp%=Z)IY{u~>ln4gapEn&LOkwNhRwm;|E z(R%^1Z8%tN^6V1MXoc&cyo(JE>qLRjX0|^xwVQ8>uWC5HFA3 zofapN_Yfri`w!iJLVkCG@wQztFs$IOQs|3%d1@cLvi#W+cE*2vAQ6rAayqv-DPDjB zN*V5x+Y4JE>%b=2&z<=osu*aXlU4;)wa*$7sj*(&QCd&S|MCi2O!c`nN8IZ=3-0r) z=fMy{w|QQ92=(&pzfw`J$PCPnOL&eW&yTMKIKDm;`**KR`?Vu#Ke_;&&%WPoN3mWx z6z4AnNkoDA`lF(?!>HE{(TKC+%jUq@BUL}6hxIyJ)63$!2?xIgJuHt+w!lm+ZSOf; zK2WNkc={=`5_HC$OgNDJcQ)4dx|}Xu-hX+07jl6cuNw}@5>#wnL)-f4>IbD z{TA9Qfwz0Jop2U*zqG)|6$}Ud^2+!cti1gIaWA20*_|3!ePPd2Gk0HMbicF$IZM~z zAV#3-p7}+>l?+)&jVFHc5Z<4O*!;Mj*+}2kyJBuFC zRH#??F*3F?q#u(l_}S;Z|KWUhgZ&{K9DH=(VDsT>o_P}l@K^_x=I4S&@gebp!c}mK zv)@dc1M5X=(iO({mlsKFj3}Fj0nvQF%2f1hQ-=@W+S0Zz`=MT^^I3-+Wtbscw4L>2 za58*?SqFC^!toVmdMldY@MZiwys@5m^T}0BgDe!@1$Ft$bYc5*T3Ff*b3aBvysvS! zbsIV!!?|9JT4`E?0O$H~$!_d;lwii?-`B<=-#^k; zWHg-6R3JV-?nxJNuSD*zR-AW`90u|Cv9X?r+ZrhVBeL8q*oo6lvC=N6FFqaSOJd_@-@tN8}6+es#=f+~jM9dpyhIvZeVzE}Ll^ zDPIk8`Y+XNknd79=C5Hd>H~>?#n(N-qKR>LWc`c7(YL^X@cvBSopc9M90I_Tc6TkQ zJ-VKwO~JGG4B|ywCiDFS5?|xHUvoX$LU=zbVqPuQ=f#ePy252c?KXd9tQV(4?7b5E zD0n{$s#LV7*B+ewwos&BWR5w z`qK&yyXWuA+lC#Ft%v7b1tf96%G?uBK#H8djZ+l%iO+{!9H}?WwpIa4?BK5vI&6ID zM&z8-{mYA!_hnzbG;yz^=eVb|R{Vf@Nj0=7M;`z86WC$K62r^_-ReuE9Y)E>`PLr= zygw55%F}sF^KmcX_Z{{7d^9hf@`1{DcbE7ate5}mtg>jW2&mG2WmfXn`h>8n+7C?= zYd9CBRL(Ys^~$DFvDOp)+^oj-H&p%2w-w4Jh_!0 zecn!Nod3Nm(i-NzJ^pxOH}-j3lC$3{pa=)EFuw0}MI*3@d})(VEC7?Qb@KZas^GYo zbQ(tlcK!Z?sDx%!| zQ&DjdaD6Sb`Su%hJkt1Z(1r?GL;d`w?*~6&r&9U6gXZ787Qd#t37vNWx^}X8Q){eOc!TS{-Hnm3b;9#1 zcOshai$YhMFDqKXKB?I2$F5@Y{aSIbAa@(GfAp!q0zTbFaN*Z--4dP;BXbM^Lf)02 zoM@dBosI4PqIxl;9QBu1j@gMl>gz^Cy#`V!)=iQ9tQup^pLw5v=KHB7?Y$$6%+T6U z`@FL|8H~Q)a8chz`2A2=K>}T(@@azcb=mE;*S@`HK`6oMiH*sBK*)%%e3oa&46?Bh+y5w>$N>vhQcp@E_L z82)N7sT>&cO~H;w8+M9BEgbrN$HK{-QD?WCf>y;vU5l%P*T39seLq8208pnW+kId_ z$K%lhc6&MzFJI>{M;2uN9NfVxDc!4t1fu6`}t*}UEa zE{}c}HD4=$&6hZ@&6QR`)TX=iM*^^3G>Oj2Dt~#E7TEi@Am0z?9|}%fIgtJd z9gk*pQ(OzV%*Z<8*@5YTWZ3DMq@tJb?_TC2_F+0k&S2wp?{L!rtk-1g{@=~~F@U?9 z&Lv`vdcCzZ?^NxuhWA^HvL@ZIUJklH`*)k*;7oF(*I+wx9;=3(VyJNeSkyXC9-*y* z2s0OT-wCW2ZPnUWn!mharPZ6gUKk|iS zeobD)j>oy@WqFE!dC4br_i@J}-^1WYXDqzYa|Sogj}5#Gl%aM3$oT|y!rqf;{yOnP zLgGXYGpPA8RD#8&8uJWbz8$enUriVWS^Kn|NlShUpq%m+Q&QNp!buOq~$^r zY%U3zpp7VipInZtD*Xh$+!xGJ&a%=usD$1)OgoW*<8OWh@SFYiTbpb$)Cj{Oh$e$&Ef%9V8-^i2t#u&-!>wOMmx zO7=oHH7_{B7f}t*uM2IPLY}8Ku1~ZKO<3*zE51x9CRJF%iF+Bjk{o|69tcN$Sf}`Y zqF(hoM&nXdSit6@%2%s}Wa#5mFfNQI?B&7a-Z9mZOfbG^w#6GQhJ-*^{@wed!r1tl z`*A`1LQ)LeDYr<^rE2;d%TB`cLKr02t5ylrM4(6#pMz7Vz^OsZgg`P zv5D)OiZCWSBaM&{PS-W!Qwq&ICb@fC6_C1->Nt-tgdv`+SQo54{lTa)rjsssVkNB*oGPAd5PP+WDm_lyKPj2K6NolT!Ua)YX`72ez>*zx%8Grvq4eIc~f_{$&Li+Vjhe&PnVgDsq2D1WgXuwGVa zJ=^+67(s(dZFBZ;69n$GxW{Ew1nwhZ9&N8`z$*E@vpFla-*=;z=#T6s`)6jLbv1&N zvzzeuRvIm>>|<|(;VKVFxA+~jpNV5oMZ~g_4WKKM|iK>V3x6^qh;_fq!cqSN(p z2S3eBXYyjK*LHq^UZI={q)&y+nQ|(FD{hrZYwO7TYQK9GUc8;wsjQ+v94HJM#UI-uLeg7ITg8{JGzY z`*g){BV1nXjd2Y)_XWODn!=8tJ}=?S4%1WJhJj1TP-bPT>av+|z9-r7Eg;>_o1j;Pk5q34SqR+D6B_Ca!p7H= z>wBAih9-irr?=|oaMY{x>7!ZBvQwbN#h=NTg!S^TuF*Y(d|w-)&BB~U-30YxH`VNr z^RtBAmNL`TYoM?3(AqI3J50Wh?stl7|I4drn+Ka!k10{FSo&#G+`ACy&Yk?avl{i1 zT;BKQH!UZ4Ci8sF-jf14$vfH^)+LeExCkIOr;-;0*8`95hQMf$F``_3zFZUPx5bICfp z5-4nZ^>OpT8aQS=rsi@C>($96*NkI8=f|$Nuiws9n}hJymH}n`ar}L3tncsKawymx z9tbRore%xns`&YF*LvlS^M;&IsFRtba3BTV?5=ZVeMk8HP@jqR65~h0{l3FHrAv*~ zy&ylO_lo);c7G`@8Vg{YS7|gj1MQDyI?HoZ_TjFLH1KnjGhsXs>gbL-E+-{>@Tkq>0*^n2aSk& z@gJIhAdlpg>p$pTCDo!{^(P*EHM8IZfmBAdPRA7JIx_zDR5oERW&fy{a4y38lg~=M zu%&r>25wh5$!))btyjB8zUhl(7eK)`8_MMs)Ju2o9kZDwOHlC6*ZoP2jjxy175Xz@ zkv_Gg&O1wkO;9E;a&fV<1TKEnnwT%Fg)2;LpDcn;WAfJ-GN)9*zq~S2)Gkq$BI~Ql z0WqaL83f}?e@}(TwND{1IpM>UyB!^mO?SHHt~}%hsy0_zpM(_X=W|iIahY)cMA!4p zx^71auNN{L+g!Ei1$*iPQb)YJ!A)K;C|;6k0jH&cC=^3oI`M#VPykFgrp~4j_L?7( z33sw5d|uOcj^jm|i{krp)SfV@9@nsj=T^o_ z_x6e6f4@KL6U^T&!|h+QLa&|hrTZyOaIPSM*}|t3^1~?J*qpBgzjIH!EM~D@P0l8V ztp4&!W6fXBFf@j+q+N8|-w>|9{y)ynJQ}OF`~Rdu$vj7fC`mFzNZeZ`37O}4ER{Jj zWTp(E%tM*SB=eY(F(QOWk`ziILMp@We4q9E+~--&=YFi!{eSCpUf%nC?Q5TX?d$!H zme|mEAHWIwTT&aNym-y}5B@m726Zw=uNfoysLVAkpcxW-4V?QiPNq!NduXB3x`@4sbdzb$IeTE6=ocgXQ>VU5# z5BneeM+DJ;xjGQk#&Rdh4uSWfAjN`}KcT}wKq_%EWM-Q1F?dfo(@hxHh+KSiN z2P`_bk?+YF5nAVabFjQ3j=e1H@W=%6D3zx0qlfVN8fjS3A&s>J=eVT{8FGg(`cfDW zqm1`sfI?Ccy5{Q@P7ZF6BXa(;U&Uc~-? z#8y~*VCU#{mv6mjet$i+zT>JAE8K9M)>rqB0U6p7eKY>Q*Oz^IXzl4m2jm>=i>!$I zrTF*$=X;0gksc5F$}D(CCS3P45ard&yo-z>0?Ai7LL}GXvAl>v=wnY7F(AF9A!pk^ zSAY@Y*Z7a3IWXZhXQ_Lm7`iRBZ}DYr&1+jnwA?y!kKb0j6y`kd*(&=%R{GU34Lx+f z?OR+G#fI<-kfA&g7!(7)0+IP(CkgW+diqTv_Y&dr?!{ep@5}OB;K2;lpUK-;UX}-? zcfG96gg&s@GwF@;(v{ntWqs%p6g$yaUA>Ft^)j2H_)Z4{ILbFoc8^uS(c=+0yaqWi z!zp?2rC~9=3d(;X5`fJoHtP$g_c4$D!q&W+mgvVL+WcYP)4_U{o9O!ZM$e+rt&|lO zob{YJ<6?l$;*^R@CShJUrSlrQtO@UjEEAV(_fvb*nPehk2iilm#5 z{s*rm^%vEpckn*Bxqk2dW1zie&=V#;R^_j-mSN&?hn@PzH;?YX8SCA*G{aH9FFE&h z&GWV`RI3a`ucTr9Uew`~+xP?ntQx&&*JwlP6XZX;&#vcypMK|CK`Uhb1j$j-PDbqd zxLIGiQnFWhS@fYNuK0v&IN|kCz9jc?u5JL3JvBN<4e0t<*Vq0g8=2=WCl-3t;)~lv)N;VM` z#qPWk_ubGRs^^k^8qA~X;~<4|Ny|rOc&_0d`Vh${_THDflEp>19}~BQzP>r>!APVt zczHEw>_P6KBKpq{?Z0;&ts(j}9R*~!pLgFqg7Q*RI&+{z-UKGjUraK>VR>~J9Zv}p zriTOhwa&d!6%hL1l}t5d4v1G-fsc4GxOh@%7_4s{kEI_UXqXx3ZpmvhAk4o)+8@TX zweMyLp?-g4@0n_5e`a|0j>3*BBnEbm%oG18|9gEIvaGtCm~{eHe_hx2Z?XCkC!Vyd z($0qZv*|JiZlk;e4Y>9yn%{pSPM8ja_{t^1#4CW4x~@EnI~#~Cs4Y~UJc5%?X;jDKu=PT_>7m<$ z4ru>Zg5sK$0{Paw&a-_|{Ho*!32S?e^2*TqSM4wIH_u}kq547wl{PU_pSU#gaJy}m6r6X?4|T^yxQUQrsPA12H7!SLu*tiB?amzmQ0m!pL=@SXUygZbeK znCJ|W5M0az?ITuR$Kr}1!DpiGnAO(2Hhf>~rzzFm(!VsVgT05oBIgL2ZMs(|(EfW% z&nKswk^A|6T>UgI$sGgV^!Fzg{vynasDf_FE|_q>TI;s$*MN%)R1J!BMw?*c@o~A= ze%;Van115As@jC|lA6er_>9~qEql{gzgPB)Q4|o4N1=XoItn?$=M@@Qu9-df;0OE)J1^aoz^;!4 zhR>f%p2~&*y5dJo7f@bhPo@;Ak?*MetdoJiO|ZNu4SpsFFVaGCj5p1qN+s0ipXL}U z$Oh{JH?JF0m4H#qhe5ZqSij#KkMZ<;cMm_*gYT)ajOjjv=bPLW6K7R8=LfEgM5GD* zX#VA=Z2i@g4(a!O#XJ=u7XzzDa#Q0?3G>1ck(k^oB0N9(Ih_JSaFG+_tjZkKL+;Vr zoKLvS`D@E|X2ESf_4@1P==Y9oi$vkP=M7-wBTZ|QBldfT?jb7w+vjLuI+wBb163u+ zJi57~Dklq`_tTIcYb=Hh^D<(c_O11`*O6MZ_f%s*2je9?x#Cm6Xr$K@cNbNQFFYGH|O`>Z#;(%iaA3^ zC^};Q9wX~^T zS*q{_Hj~{M54zFy@mc!|SJCJ6;Nn%dz?>ZeXS@}pcb_1fuL{W)GOd>qjz^W6uC|1S z0dS!WS7i4H%j*r9LGjbIOh`1DyE>hP#$#z4LjWm@2?#J#s!&y6Dx;`bxSJa?f) zDa#j$m2lN=Qt$Dl9Ju(4`7A?p322VLeDa6;*ERsWNy2kZCy_c9dOk>xk95INp3AUfNxwka1zV5aZqgvUJVpmj zWNJm(S`~08R4x6&P%bQ*mRb!^m%y1}DUvqRt$7W|#atIe=6^MCm+2H_685icD#~_$ zlmbA;@benQiL-cL;j=MeNbz6-Hy;k0rI#^4mKE|SgK0nE{+yw~x1x%P1bK0aYu@;9 zGZ45~4yQ?aV0rZ&PY%5qnFaEu-(>>K(R}qfr)0wXggM;RZB6iX!RD){ms6UcGjCv9HMYTGj7I9)2XOFPupK z{-8rK_~&f)ubrH^%L1YS&_EffUiSyf%iB0(SHY7ksBCu_cJxH`l^HlSf6mMtxd)Md zGC2&ZuXdws71jiLsQ0uMFh=^L)29mB3c2#ZcjNw}x1%K>!M@Euxo~S<-#^&SSl-kH z+<0mBLMGvPkK6MmhorduL0c#&BQX@^C9dowK#{@-QNp}Mthm5d)ZN+P@ z&(oDo`Z}!X{&Zj3kL5M@D~%$$D+}H4@4eZF%GhO1VcyIHNCR1@wi!EJ0sPVxsdwTR=kMbZOf>y z4glslCBVz>eK zN$-g1-@)=~8kkFuRLcV150jm`)@c1J;o0{E=UQ`+_}TH1)(czzddp8~&G?QUC>VZR zlety_;4Xj2*&!bkx>VeEJt~G%v@Ri1RM`60W?s=|-Bc;sHL zUwBq=6CUc^{65Qz<+UWS@U4Y48~T65nf-1-dBw|6A1*@XJ?by&^Y^^P@^YE<8NE2o z0Ey!7hRI%4fP;p>Mfu|epuL*H?YLYF7w3y(HG(f;>JyuJrP-1`4EU)J&o>18e)^5$ z-(xec`;963OYMR1Ji6Q8m-sy1zqIA{&AB~hg5#QkyvN*P!CZqe{3_C4K&T&cD_%mc z4l+KQvxm_ynTZwGvAn7hrb)c+=YY=yLsaJ})bFJzxM{xG7{eh`a$0FUY&@PG9j&Mq zV*oPhM`<_nD#4C1t1~Yj@%y0G`3{$2xW&J^N^OSKm)=O_*Bp9Jyx*5ut&I1yBKM1K z#mjXgOu#Gv*pg3_N?t_k(T$@WM^w6ydSMukpVac-&7+}=jB2g~b&(@w?4$MitCf4=>w zM+NY%lRb_@?&H+-ktt?I>i4ftThwYi$NJZs#%*FBHazgWXvL@wTPh%ZEnDy^p}D@8 zw&o8*V@IyoMWglA2Z^{V?ZHfNW|I3+oLDUI+IgwuivB&XV5wKE#%X z<>x`3;Q{SkN05G#Yc?{weqqw$Z{lW5F@$)q&kqJ+dGVaZek!t$`QH%iw)L%CBq2Tk^^n7%=`C z=?|q}Pf@n3qVrmF!T0wD3=!`4<$J=t^jYA<}JTTk?7? zInPgH7zo}6=(YAIqyF_F#^SiyAx5}A;9+P{69Z|I)wCh>gn5k>*;BVA-onTIX8)p| zy%{3j;|t+7snx<$Qj#?o;>?6rd#r| z`zk-rmV)H>m2b@z1yNqq?VMk3yD~!_^LZW;W#qoO4cd$1ZwbevE|VD}^XqhiyjBxp z)HQ5x!D`(Ji#vi?eNmol^VYnb1OCKZ5g)73d^PbQm(Xo7Q;1Bo7?tnA=BuMQ)sq!! z4A3z)6uYaU61=`pd>3kZ0PMo{6_*D~fG3bUv8W%LuQCYKOl*rndFed2$Zw`a=7K2w zUW?Fb!oSC6zh}AgbvTk6K%n?6dHQ*j7tg(mjoLqrpiDP# zRl*$0>(e>@h1gko@E`0d3Jh~kTRpB)P!vR!W@@oNdRmx$=y&$*1r@AYj~3sxFY zUdM4VBwiOxp_u5!>~i)fL0(U$*dLQU41`b~tFI;wig;dq!HqAfX_z6dThl<}2Ewau zb+Ac-FfXEObM2pN)$q^R9FO%D@$%kb*CBQ=%O9?>g>J$ou2=?!E13%eqZ-PftT zhR^R$9BdRPJ886~zN}>A>E?U=0jZtaTjiql=mf{`nzednxY4X%UO$K2FZz7Dbv@JH z>x)MC)YP9$M-aMVv)$MQt1r`ha>HMWIlwKqP-ZTI>T73Vc#+Q;Jy`Et9v3}>)z>X6 z@pOSF>$qx(9N$x<)xdg;XUFxx0+^jTpUsEld7~}^KUTM6{fnt%%)H?x%InjmjEnc( zO}6CqB$}>#aQ9BVpTJZKlMmgF^D|+2k@tPmy6^<$CB^+>e}k$q_!hpo%eG#R z_m$0lzwkcr#&PRF(0$UcB}|FdqqPkp!oQ#fzslEoLSL6SK*L!V?sO`7$ z&)Mu>GP@~y_y-&yg1EDW#TLu!knU*r)0RB2NU_#C{$ITOF7~o^&g+1wppazN7?#)Z z{gThU4A*geCZ`Rbb5wyhx#0J7k9?4Ro7HYjRSK*vw&{KEvGMr!FqctiFUqUBd|eSo zj_e+tWfnU{_&Oa(Fp!L(Z;zlu2OUpb+%$KA6S;j@h}6U)t5s3(2#;Zoq= z^DtAFu2_@xV#n1%bUtEIVupS6+7y#o(49=Y$N9z*-BRzRshnaw8W@gr|ITm~)^-2#o|2;49@7>voCJtb? zM?YTXb0vNq{wKe8eU@SKAC9Q1Qkc)Es* ze_%EjT2loyIegDTko$zof8=cFXp}+L?+g+M!18Jz1!1ool$TU!;Sa5+Tl4ZBVxm8h z8UU>6vTcEFC@+fRR0)(KOmIzNWH$6yEZAP+ppFY6%nPS0AFQ#HaR2Iy-+KK$%TADy zy>d?KU=@bf^by7@xP=@rKfD@xzzO9wP}h8GuZAXYUhJ}N^~dsBYFkr2H@1e$me`{s zre2Ml*EZej8I%wDwC>y*NoDY&iz}?H3Cl~8Zu>I%f9Wghad>0v!q&X%qwjY{%m#q` zdH=&%_tE_}oAHP|880JfH*`o8KSkC@vy2s9i@)df{aWJn;9U;DTrEg_H4>|@+}cn3 zFO}p1n^~&xS6_6$y-kNZ;e5FU_`5D0c@%}+Z{PM$$YUQ^#Wj*D4ck1cf|sdhb4?4= zU{`!{Xs>h`93zb+mXX2oa&1}P|D+i`&nI`G=uS(F!Itq@XP(ucH5>rn>f*R5qS5_! zgvVm5#WUpG#OxSf<9ICGlV7KHLhkV)RG(L*cNoV#B{eR4N~G)ncy zbu5JIh3pBey*o(*!BlFo-rf>jA9dPABdd&%ez}ZW#$F_G@c7`uvwITUUVkDkee%X;^3?_)!GMG>jf+4E^YgZ;BlrUF=A5zT{Y=Xuw0 zp94ybO|Ml$1@6>!uJ`GXpj>OT9m?RhfjX7kzvN%T{cci&?@(TgKhvY9zaaB{x8NmN zC;s;2g#dVbVCYD=5UMZouff5TF^ur-cK?Z**Rk;JL0p*^GG~NPJZ{A+yukX>duj)u z+oAj-F&E406P1i&Xm>sw?p+OZ6hV1CN<2)Wrmq1@o=LaX_!k3}oK1u=HXiRbtc~XGLHCdKC+_s^s5b$kG4Toi3H08B z&G9Hy5mYuW=?@aA8t!zYC@=L=R~k7}MhI8qJ*toFw}sp&3{PAAJ+Ezb&ho{X_AtM^ zC}#E(%ZuFfi)*bxE->b_I;ys#ejlVFUMTrn9f)=@+%(j;OB> zs-ZPnhH{5aI&i<4_ID>M1A#?z{m4WtuXF9?GWV9y^J(TI-rAF?2rnx(f1(q41pUi= zTKLCRS%2W)CO;)ngYwG6al05Jyk1F6R(My(LWM(QK*c4(yol_%552Rp#Xo1W-&f`G zKc4zx4-aYAZ)@(v@~VxhSm`Uy1*2r+daoIj7oT7cpVzWFL|iBS61axtb?m$TeY25u zoQtO1=(0f#bU)o2GUJdA;`w)YKOudL!qg-4?xp|2OJ8YT_`m8oWbP*7_lgl-0nW{Z{KE`Y0BRt8-s;oF~i+_p!Ll zs05uWxtUj8+6}YJVJGMqem2b5Q;qR^R%L|>%JMu&QCjDE8;|mmD>S$}wO<2tsh0J@T(CWR1#Nz=QX|eaoSikRnNi9Y4U*CuJ zpYUsEa7Mll&3V2^Ga)>G)sNvWe<K1 z!n(-+nAZ`$Kw>9=#{Fhqe|A}EoF2CakC6UGnp!NcN)nOR_8-z9z^+E0$q3~Y?0hQT<(yojcYUzu|euHUDzrQYhBwu7p-RP+vwSYC-e`J#5@nb7<3*oyyI zl-JqmqbF|lYJh#PS*58qmREU1qWiG!2Cj75eEK2M8f5P1q2FIK(t)pl0ayLI6yBs& zEI+u4)z{&gVlJmAC@-%BEvxd3t$8)I3YV|p0-)7Hr$=N8<)z%Co-Az31g2cnC2Z5N zu;cs1!&S?K^9j!L*?m`j6Yl@|Tx}o_m*fH_G`A9qr?LAXz7x95k{s#Kkr<@xa|-3f z3{*V@YMP*U`Kc+TI+jJGipbG|~e^=G;xRMzFk;YNi%+ zkn;*-Ph&}QX3>1&>qh!V!yQ_1H68bOY8{(T==*R;f2$z^jw=>P^*3r@XP_jr!cYdB z4i6PRTvG;ZiyfExX|VNZg(Dr7H1|+mUZck@pOn;tG1I3qeIE$>R|LDkpps@F2n^-% zr`eyy`*5}nMOw8iGl;(CZEI$VgOs16%{$r&&o6YC5h5$%F36+xRh+QBUVW-2#KfuG6#SR|oW6%wtoQX2fv@G<(^6!8 zT+)+zb2}&lYG>5<#v}1qnIN1~^a`u5h{IQpx(1{DISu*5tNX$YAx(Y<*DHoj{CjL( zA3fU6*NFDb1mb`mtMC`7z9s^~!xE@9VRV*`nfx7AUpZG?SO#N=py%gs zjl+o=NSh;(5XEJ{w~e2{En}3|M_UC4;cS^Ld3`xN7BZC+ z2()v|qQiw~JTlHbx)agMh}>f{TVnk?7F_(eVzrA2`~6nDMv^*$FIhUl58BXM(%Z3q z?@}J-wWA^pqz{*_;aE{#g(jTGexK2R@^o1p?!#DK9R_;q%6degA6z)3Ay5NfPHUW9 z=g0t}no~KY6{WzqmSvnLf%SX!*#6lVQIwbeHX41Wiw0Zrl5LrrrfCiY{veIycrUd6 zm3Uo5Yj!Uagc4Z?92|^=PZ2uI@BYVqnOpJFdQjD{XzT>iRmA*jzp(X*2l_vDymHBg z*XBECe*M?Hp}dr<@_9iTutzlM-I0HpH`Eg?C~5PU2(&qH2@ws|Kt3z;d~pHEC)i%< zu`ZVaX@~Q>#pl?3BIfswN6W~0@_*(}aL&%Pb3|;-i_xZv_uGd+sD%&p$0$*KUHJn? zY>;>iRXJj9I2j8sB|;R}`w9D3al6d{xdy`ds%(?hkaDTLeSPrF-SP)dYMzxWg zKm$tM8LQ! z|GNlP4U8sS2pkwohizK>H0?c0VPNtE3xzAzzt&up|1>{9c^&ADeqi8-%wOGtmzw93 zcy5DWP`blrDDnpNFNVhhIace85Fe9}p*|A}7j1&~(sKO0zP`P<(X%Avgxtp~u=l1U zR$pP?^{y-VWI~!w_2d98$_vEkOmPC5a0B$F>e;Zol;3C+awQQ#b;U8U;cyDs0`dCy# z9?NTQMU223|4fj(XV?|7g5C#4l(KxO1j#3=(v>bcZo~45KQlb@B8~_yo}PWkj_e;( ztr`!QizySWVJbcf$D*kgGm+f^1hi=(_sx6=wbUNb=QxE6KQhH`xVwP<-T zd?`{5&#UX zS|(Wz;BWmyV9^ZEN&inFZKz8P?OX)Yb@bOrzeV)@`Sq-8iUsO2#8;oTmsH6B_luc`t6V14rpNZ+d{sY8V7>x{&a+2-BqAhImYS3=H> z;r07YC9NLY2Cn_Jp?QUEHS`L-GJb292GP>9=d+ML|2nPa!JyOFemM!zo%9jzXg=Yv z`$^|$g+7Ltj3(8J{sokm;+}oDzJwqM-=|w;k%RV|EWDd4;d5sJenHg(Nd~dNxG_!6 zG*=M8f5CcpA{L^;>b&l^}hIAp@s3u>JR;yS&?PE~5Gp z!%2U(me#@Ws=ciK{DKk6>(O?fV>MpLJtH|iKP)m)|0-8;*A+$T6B)kSD0zR!K<>we zjdT8gudf4H2eudW+ki)qLdqFwEH4i?$se;$_aN8dSi`4LPJBE%fB9^o%&i6qR}K^# zeCEXHOU#4Sf2Z^sat)0Chccv}siB$X@`L^Nq3y)s0-4ehh;bk?l>Uu9UpxHb*1nsa zC@*QcE5rNtVf7WB@BSc@QyCwR=|(+NgN*@TOMgHuOb?Amd;g-fmn`hSHaUGHP$CwJ zym~BB-w-}Ow3WUreMCgoaxcQ-=a+%SF<4$tPv2M*K=wmc-jfhDg8G-BeR?x-vKkaG z+WnfC#QN9iMJ4UW_3OAw`kayP(N)m;%lV`ge=79$UAClHD}jcmd4aQqSbd4^n@TbJ zgz^fnKj(itSr3Q?q)u4pp!I0JsE2P1omKF>h`)5HPbLMzlO53`egkMce!q17iMi1} zpuYX9K2a_fEY&2`ij)ZJi<9q32Sibu@FXEc>;&b<00htV}pA&cHIlcKWB6O>q?D~Zq*w* zU?!c+ta*;*6`0=Pqr;p50^j9#nPj89HmF5}TF2EueqS2w4z)b&>QxE>KdLQbjIq4J3~ZuU zhtcy^F7xlSqWKa3a=mvg${-*A9_%flvx(Vg{`IkfB%|$C5YT#1iiLead5!I7RBu1c z26<8e2a>2`fyi|3T#oAh;I-%&wD8^%|D4UdYH6?D>=?fY8CE8pga8EOvg&Q#b%cQIz$ zjS~3S1h?=0%RD|lfz=d#CA7ZUojs8vAdBTC!n^xWt~DBucRSuRmF)yMhz^3o71C(NH-QNN1lYv|H)>yEqsgI6$kCdN>^oT(sI{nauNEJK?klIhung%&1jaVKXN9N~z`=Et8i{*8{FWKwj4)ps_ z7TK83sXPOa5L5cHGW7=k9(f0IeLKQX|4PlhW>=ON43M6rH9UaUC&uce{P*AHKq^kA z*Ls2w|7xxJ@-l=lui{&|j8g9C+~v*j=tZeb!zkbg1rf$VxI`?k7}8t9lE>3vH0dbK zg%MO=l8loU6ssD@{b~N!UQS{4MI|7@wdg_&8GE-MykJxX0>_KbQLLwd6G_g81m6;P zTz*yVY>F1WgP=AA1 z)w><77ZQnFk0)2kQUo1pR(1UPIqb?@p1mg~fvXrsuk`ymoen z6vrM;1JB6|YBIl3|2iU}SFon50aW1!(_U}K^6FXRuNM|2hVrCpnpxv2aC>Iky{jY* zsOKGYkA5tM!}jmlT@A6kzOCfB99Ko-F-zjqp@dVi7+$~2q;2M7P<`9JlOt;XP(bW zIRB)9cr;ghLkP-Ckw#_b!ESX(C2H($iN^Z<^=zj3m-57r!k#zQ##9Zq+6Vq5w5LJW z%k^+uBwrQo|M7t=6Ptf=JaV~LR)Xql{KlgBD?===5nvY-e~t3G#S$=e#61|cxgP4x zOF-90rWe$HD^#3d_?%`V80j~u-g(%%aEWj{;%IX=zUdIoC(QeV=IF^>!7h2+&gWlv zt$ivWc3DgZnjoWA=Y42A&U)SI6l2ze#`J*k20pC5+$#^uPIeQ6Lur@scGGGY+ELq0 zaW@?dxeJtbWS4@#HFkZO7HmEt=NU6s%k7S@Pq?h)ZLds6>Z^%N`jkVB_;{3%+9yz) zfX3r+dHCE$z)kSZ!#$8RMEy&J_1g!wbNi5Uw-3kZ{bFIPx%Cx;)!*|n))lE#&AAAl zn#g%hrDEgpdsvd9fqxp{>IQl5sG$0~u;b9d4WwVLeh=Gy6J4ynY}}^|_W&_S$Lo17 zI9EZ)dbL}>LmHI++I{F_NC||;sU8xZ#OjM}xZ&k@F*N^52_(ibO&CA}`N@->reK)9 z#t3%_%7}s{FFeQc+V)&gNKG~!HrA_}3fxg%PkX4M{8cod!}D2Sd;pf0lFq}SApv6G zTav0u_pXA$rb(qQFVdiSb-2Kxs07-y?ZdLfu=*Nx`P5igjr!NipE%7vq?NyZ5sZXw=zkcomMq+`o50*&(M7A6Gqg{mSIa|eJnV!ns zCasImdw^7Y-5<;A;rZiSC6wv#lOtJNmI~!{%!)VcSgAT3s-7iTXU6hMKBJ|d8jO5@ zRA`EEJ68ofMWoMW1k*szg12|;4pqpPsaa3l zhpkVPcGx&iXs+WlWNY}@1uCIx)>5T_I~gcy!&+`5_W(|rW^*p_Vdp&-_!#E(3!?q2 zQ{G=_AG#s)_%`thIFs3IK!NIO@1j8HtsZ}5PT0eyWO`A2K6p*mAp4sg3ly}anN0M= zK#FJUdf~yp=cPQ^@~7j7CCICMd?3q+-4E@4zWC);OcH#keZ|%ph1S1#wlPHPZ&C!k zVA;bY$=Lc=&=H4zEz4CL^Rvtr371Nk{k-@_I3yWVQtKptI2XgWU*jz!Y}orU6-+$_ zqu$u#{eEyhEaXWEGXE=q>s*&=8(v@Z`!B@sd!xKAT3%Yc>g)@}N$Xetq#noflKIVY zpx`w(RCtW!&?4upM8(c~{6YGM3GKHztG{PzoFcqFNAGF+#0a@8eZWOxPtU!zN#YGF6&SQ@B2HWnZ)k{ z=g8UGwCoai(oHp5;Do);u+{p>oH#!^uXb4h_~6ykcV}7mgOgJi7uw9wcqCJiDW&^#0Ct^`s%c7$1;dzX3!YxH^mIiE5_i}26c zoZoATd^J%L^Z-w7xB32DEHArDROcx#rGUZr9^d=BQU991-Pfe!q7LTgZV-htV*QI< zM{6mkLFxFPX2jm%v`Kqv1Qru<uUnNRiR6J+F!-e6so1~tn*`1aj#WQO0ZWOFU+R&l zzHSX(die6DCOmYyn6V=XtFM67yG^{~>o|F^Xq&oL14nkx#P2w9A8I~cvFd+V0-?il z0$2R7`nooh`u#R?f5AWf6G|P0?o8Qw7+w=XMhne{(eIB3v?FKqb^PGFQ=>unJvn?l z`m5a3aM0t1?Mn}Gjx59ilknBn!mPjNb+q|jiIS2fr%w+r?=)_FD z8u^C$*U)90sKN_X2zl4%etHS(UrVC1Mc;m{;ab189p*s#Cq#91Yls?CfDsuu#ywRE zwdr<3bH&*EShHCNy-0kJ`py5p@5EZ=ro-oRSY9eT!3h#)(0o-%!MCiN#~+HHo>2Y0 ziq7{{QCMZSbwl!dw#&)GuVO(ipBs15>+gBVkBO1zeKiM(l@&D|Bu^sx&kwCn$gQ5X zuA{gI6fXp24V+P4_svc?CrzjUsZsNMy9bsR>w`(fi2WAj?ZHQ{lH=S zp(hs_@WhDlYOX<7g{+4S3%R);r#n z2syUqiEKW|c|N0W-EEXuUfV-rR1}pQ@cN=X$1+y(>I8<@+D`u?zf(|N7b?#ygo*kh z=g?`S(}Gc6pA2?qbf4mcw)j$^@a|Zk5Ff3udQCWACEAvopJ%Q~(7!UMY-trkEdZwW zGZIT-c}-m0J^ZIB8Hn$fX64^Sd3p1uC)#JJgY+cT2Z|K_B zYUp$8(BLjkf{td=U|o(<=#o7^Wy*);#r!n&`FUi1+duIrp^dxpfBkK~+DXK|+n?ay zLrK>sKIR$9t6Av=sW^Cpg4TVcaRBu%`?^zM-rKpM-QPAw{||D$_NMZl9V>)+nYJ<& z#C;>&|JB*u{4tct9kwy`-FeuBZ_UOL$@mb8m{x2aj%L;4Y=qzk@&$qkcg?{IH6hwRS(xa(iE}zmuqY&kL3o+E03g? zl6*`Th{%U;pMQ+r*W|U~pHOod^{=BX@{2zN0^#hN zM4{)2%=OxQ|28h5UE$3V%Bx)I^W_uB{o)iKkzt{S4e4z&_l=Jg}}(Z&OhdOxG} z!sToGr_Ql+!LH2Dg+|Cch1Mlm>4xop&+9EudIGtyJ=|t-{6vGyZTXMg6QO>uTz!)v zQs+L5esz(!tc>!ybKjEWL8lSK4cly3XkmG^ne@!Lb*$n(e(F1)Z(0MAD@$&sg(=Wa z@{>1%whVF@&plq&!}2n3|7x#s%oZPyDKto4+Jx2D=!2wp(p_jgo;h=Zh;%UkltO4T zKaj}a-@`}t;-Kc*KHz!B!(TlT3+rrBgXDvR<8do}*>L4D8Vg$hmFkblFeU7Mh*x&W z)YvKo#FZ@LF2-}<`vGJ-Teh8lqzNr#Br|t%Iq?6#f9m(W78}WR7ucsB++2ER4Hr^oBL}0!Wg5k@~n^QhNf5XyhN%`q)e{pVD#0UcEt7Z3G{nM z@bhCC(g}W0en`dh8m~0||8sfzsPr3Q4rpxko!TQA2Um)0yo5&y^CGe_P|9N>JU=J< zN|PxQa*n$0;o5t;Ml3Hf!xQ?XWhua6?iP?Qh3ZREclQp?J}vmBJ2N&fht(Iu6PL$! zeXF>W!pCyX3D-dWa~TOa$z(VnU-KZdxD*vcvPbAQmFrIELlr zsHaDkqKE41tviYY7C`#}E=SFBxu4~Pu!LFH-PUp7d(HC^>?a(LoJBM) z-yRS?52N)*o|Q+>0Y=`{41T7>#^dGhms;YuGoVuJS;bXLloy*f<2zF1-Xl{f%An#a zSY89&PDXG2*KptUE!IceYG5qUgrd1O8Bzucy@HXs%R9+zUhR>_u8-{`qRJhzsDE9p zjyCcfME2Y1zg=G@pz}J~-N}7F{2os;vARe8K_6h3ikQ zXT(8i;rMAAVZyv{92;e`DZT{#eoch0gwg#XNbQEmGi_L2@!=eLd8`>gEO#}K(i7#? z$+C~)xQ8~pmtBf|^aOC@+73ar0fX{;=Qia@vZNHa;F3 zLoaBiN^?TXQR-le+&HLxPsd?3OqiEy$U=S}e;h$x+yB(}?T>H-m2HW-orkgY!tsu? z+WWFnp`kU~spUD!OD+8Rt|&PjSQ4*2E8L6a)$1EfIxw}4lTudybt(U(Pifu* zr@Cirf8LkE^4wh#u@NkGht^l{yjdwsm!VKM%PD4 z)@NC-n*E{W18z^iHS~Lj+rhS}!b==rSZ1JoAT*MgC*pW9P$uJaGXLIR3%Inm*qP5?wI4P0XmuG_q&1~{xZh--B20e4A5OTzb3;Qn!SbhZl1YsQ^~e2*TguLvhK z=`;OEK2aScRx6*2e-A~TBHQFjlvl$3-PNj3{Xyu7kk|`XR9~M)SG~LJ_W`ks&4|g# zI8aQ&EuCB-tgo$jRmu#WEfBPVCwAssZDZJY)K|%UoH3mS@jWxDF-9n_-L9o;hcvW- z#G*8%zzWNYTqam8&u$$T{B)d~p`r%fzVLNATbm4C@jFXcQjqUMuVuepk;U?Qs*pIf zKMwVK1Lv`$wy&_fG_;p2njWL^cv|Ox%plGmY(JkdI+Uh`|GQUR;>ncA+y`=o5>C;F z;y~~cXS{dx-}AahQ+($xlNF5c+WzY6#Pa(1@uB}9dn$bS637?Eh4P~9@XvbJs||dK ztyYnTu)OBBb!wcOT*JMd61!=hR0Hab>M#GaB!gJ+s3|)#FYQRW15y8FEHBQ7-(`Qc zT*9x9Sp}Bc`Zch;I0WkaYI9Luw2w;f-wgBz6NcT#oV`(APs+Uni=OUYbxt0LJBq7jdLKw+t@%_JPnS4$^+$BneuK)307KsmC1 z9G{rh<}~L~7zYiiuQwsZ9-I%gAo_dF zbv72Pz7oj2kI$8@;<)X!)FwG=;3~b(<2n8$=t({y78Y3wEXyx?w%<3xgNu<;)c}kL$cko8M-v_@#t_#(EXVQ8}qpEyzpVXO1+-^6K+WV6!Gl2j@C>|5|#K zSBb4!UuBUNuq(#@e$0mDb^PVNkc?NWxC3VuCpnSxe9hrE!sOm3K}5H}h0(jE@Q(9g z$}R`2zKDO)au-mT;(49?)=+8tQ5VMQK0cP<{EB~%BYQL-5451Xe%izrFp3oUK2yojiABN0>N$C0*&2RnpyyzTBFSQz&!|b-?;=#*UUTcrFgLbT? z0eyU1vHma`k3Og9^-fA@fw|2SCZTC;JboJ0`gDqB4JU15#rpj~4Ok_z6p?&Sg407Y z4b)ksFf#N!q1_Xk-@m$0Hl1dJ?jO0%eWmUj#`5}M{S5cD>%ZdBpW$}N34h>RI{Bt& z1J&0A`^F{zGGT($Hkn*=jH{t!zR{Bahu|4xanhjW=dm)ohgxwFlxOaB( z^5Q*s+jt}9cP%@9|7dQ1dS=E_9md4x9=5f!H1k}kU4gn;&C&tFotasJh~nj zUZmyw+L!jQfN|-IhkrDpp*dORt&*}P!Fu#o`uaujIx3CxDo{vq$j2aaWr_atL-WF=Vl*CahYhAJe^-Yex9l7bG-2bheMJY4kB=c_-k53jEUw)+%H_pg3{aSN0rb{@;tET@&DaNKUSIivKY2|Uy^#WAIrD`LjFYxcNnU|_nF?lAtKh)h!bX_`y@)C8;G@5Z^0hN&!D%JDRAkupD${OeY z;I-JjaM#NK|D4UdNbmHXIj`;r*I94N`dMT3^=fT)JNy15cz2R|W@$INK5p;HG>oRz zgp*1W`&y87^FOjg>Wu&J`ol`kWo<_W&pACq(~$21wjKNfbUn!V3Wp?HvF;KO^4DhB znS)&)bq`5$-9hGN{!_o_pH?oNk=6#)UmfE{kKf_nV>2(kxZmO(mI0t_9z@cRhVCD0 zgpV*EOJ|0lkb&G)gJ}5vJ5l<817Thbtb%{syS?$x+03i5f{~0o)Bz@64R>GH!t$Dp zt&-J=PJ|CoxZyv@e0Y8!e_DQCNLZkT1s+m7u(oO{g=PT@Za~Cxvor58&TpGq16Qu(T86 z0Wc_}?_y(t<`Y!E%;fqiSwPnfca$wP8oc(YR;7#*<~2roEhwRY@c9k_t11^qB3IB4 zpNo(9m-DqZm&mKMLJ|Rdb#Iv&qWu6n-^~u#Tk1pZl&7cvMJ%tge7_u>PLqTG=**l+ zP!+5*UE@f&8V3@63*X$4`_ua0JninU$LdQV=vJf1Aj<30XJ%IsQyrwA+@#bP$(;ZB zyKLrFFgL)PZXXD1`O0LCEolD)Y4(eA^W`iMKWFbl>GdeDJ7e}AR6KQ{ zp`dP3eX4>WuX_#+GHhLeP@<`rrrnP6I+j!rAKAeKFMI7z`b$KED9yB|$0x%2+KN{W zXT%NZT`sVYesb4=By2oNgiJ>0Mj`R2r4b&@hsL9HjYs=aQ$4WQzArgk6x%<+fn2a2KN38p2)j{+?I7ScM?nTPH}RI&$7T z0L#leTaBr1IuXVh^aAvLqkdm``j_eoogU2JrFTL`+hF}`?b|O#BQjV#(nDpjQVBUB z0jtHxd^a8(u14YW|0C_arbWr#QBo9QoECTOaviu9mxo>Hkq*Nvxth^A*=Y(&f|VeXUr1j?v?OlQRf2u$!i~+-b4JK2*zCxr z(yx(2kD18xl*bkD@aZLAey%8RZ00&d^|1)j?>!po%rL~ot9Ozcl0>92Ug?YOhr87E z{;01ac9t6B6fYo^47zZh9*e&;a`u=U-o^}P>+T2n|MWb+F7y3ye0HSM->ytny&JiPK-@r-ZA^wq%9=M@@=^13GK;Mt4U*T)7Xl1N$# zh}T_{Y5qVXV@$#q3a6XR9oY^9~%f$r;$3SR6xf#+40>=MBFjvRO^6<@!+ zUIJP#SpsO#_*$f`!=r^SA2O47sk#^#;q;Y0`Z!-h1k;zrqx>_YsK5Es`Y3It(%v}a z2{faG`oZ7k{U&_;xL3+*1*vSbnjPI13dAcRp|a}#o)__S`iTHvdtkj z(G&?f)m`}ck2CuoSd<-$gSV$=-U!!V{$8cMv^3pLAC&U@UpYL%`+M%}BKGDMO7I)L zZ8MJ6FU=itXXJbk33Wx%2lbEwa5Cl|YSY5=8oYMR>%$pYhbrIX3HwoVVMj;T?eu*dW2b;vuo^0%kF@4qC5nWp<%zvl9ps68>O zcln7tGY;Dp{k9nD;%aV%B}PHlCkZWKx_q$uVR!WaDV|qvX#Nw?qEqX<%F0i_|9KSk z_kZAZw3X`$+pH&$%07^aI*IX8c{1m;`xGN=%bLGQ#1;vZKWR!wB{$|Z$gNnN*s*c_ zS3IlSXt%s0hy>bO9?>pZ_mjWk`+aX`_#Y2PKudu*B0r1it7gwTW<9j-?10!U%aQMR zePwpB95~BI05bQMwokkzFfmdi7^xNw>ch`IeGAP8+lZ^Hil^}ToY*iyxyYE)>%8i6 z+VY}8C87TI$vmX8YD0Z(9edySOw$Vvn4CMLKY{Uz&*CC+-iOYG)hoFiLKO+7axOmR z0h$~7`#%MqzE#0=UJsZutiO?0 z;dy;hcAmT@P5_Ht)s^jBB~ZPub0AkT8b%5~iN24{2Xo*3K10KJ|MgL2?a+KJ#>?5a z>fjRvEg%Hb&O;9{UXA5r!7sTOfpAfxYd32oXp7$< zDM$NOY_#A06E7~J0LN$#2k5-^S@xtu0Zv~!Gs@cME8^kAbT#qO70e#%65AJh2KAuJ zd-}RR_3rhb`*q%7bDB1oVrUidJ`zCty1y94AmF0n@o0#@NzT$}oev}?*#xa@yuM^8 zWZrK(hw)nLDu6bnKl7@Jr|vk(>;+%ZpJRke^p#p3{_Brm0U8}Yo91UuRKzea$la+X^1DXc$0EB(VO*X;G6rnFtL`XZi} za6w%$o$M;oSSROb)m;oe)$Ro%gHcc~r{ka|kq>7*FDMnN;(5JYmDzoY2;-$c`s~ek zD>^Un5Bf4mPt6mH_Jkc*h81tCVDajU?njY1tqib*zEE|`u1HY7p4_x2lwtkqf>!+Uf+%PaV2+S@#<9a zc18zEbl&S9c!heLQ8=;F3k;uRnRJk0`ytvhCt~fr=z*;|ZlI(s0vZ&U`tG0o_q+~N z(1O@M{!l3i)+n7&ffUBcL`bzr~aogG>tczx~p zem2_j=?`SmQ2J!VwPJ9|8YIx8@v23{`2?$(JD?fR8za95UoZEwYawCeF;<`RY@%RN zDiH0r{0ClF-)YN;*So{)-k=ouH<&$Ea;HB3)XV^EtZhl%TO(ojIcpYs&yD>TGI!+W z$kW92b>;8zmz$lGyPfM5xXw>v@|Xr6PuRz7d*fdd3*TQqK{wVIFHT{hejz_yD48qc zSo6a3deyWkpX&5765W3!Fo?7S=1jF>I<=y~l`TyDwRb+0-uz(e7mB|x+AEn}a9}s4 zuTkQ}Ns4e?Fc-ZuE8>Oq2l%_cM@xs!#2{X9zU6d?`wV7}S_1R0z>g7@27Xqg(MCe0 zS`bTQ^~SvFw>TFV8HH@f%c9#^;TWY8Gk zWbZ-CXhqC_1$od)Jek!3j*5_b%?wz-p}+Iup!xVig~J=pn(*@=s+c`~-m7$?WSRk- zP1CM66C%N$(ff0TnQ1*Xz*Tb=DNUM~;4{o?vJ#^qW`FfbIJp)G(*v z;(g8=emn`bxPf5|ED4jD!J2K+i7 zTImfUS8@2WzUWEfCzp13f!%e6Du$bwJz7w0e({Qu0rH&G;)QA=pf^wbN{!yeya*~% zOy*kV8``7hM_uu<3`?+|bUNnxKk_-pqBW1?^~FIFy8~687_a?Z=bxOEUqU*MUwP!H8Ui0g zzjX4;{Ko6KX>-PNr}f`UxbysbzwgHR=v~bFma8Yu!IAGT=MU}7TVEeT?1rE72aF@b z?2c8W7vjOatx3vG6w}wLx8E_7W_<``f3Z@C;CW?xXzE^&SVjh219iMA3*kg>v9Ho* zG@h`_HQN2~4n#hxcfEn&?a_&RBr3uKTOVa=Pu-9-*_fB~_t8s(dg73M|MPunQZMKd zE9JJe#(1g5E@yZuEg?rdj+;)Yhro6gW%dSVj90y@Kxt=f2FB}$;J5L6n7-0O%O&=Y ztRnS4+FdjF?<3nu)ZcaK+C%omt>=_OFY6oSw`qlCX6te#0O{DvcM?h3NcY+gN5un6{MQe7c>76qeX*(vtMdC)*U zOd z#NQ$H(+iQ0*WMxnigpQ4^&BCq((cK4EVh6ADp+&CsAmcxln5ohcoq-Ux^G{`a$x?R z%81XT?XnS!rxNbm6vF#^5Ah$y!KaB}k~B=S_D%`RPsv;|LFdtXIW4uD#TJ73Q9u9N zVfgiNk>>3_{YlIo-7H?8@hV2=0Qw%dy?qnbPxKcrqUtY6ro2)x>$Br&lz|VZ1#6qN z|K{%vHmjLh++#rJ(?|)*5~KTVrjp8jQH&QMfx7dfXTq7g#0^$j6iB$}v4 z0?&*6DZP?U>nh^*#_+{_N(sDpRIi@a77Zk~GZmc=7s75qU*)&z`23efHnY0;Z}GkU zi_*R+Wt3M4mGAab|3hEf7J_eWk_3jipO3w6c|*U5*3w!iW{+F88&!MlV1UyKPH$N6 zM8G)xi2c)ImUVmlzv}DG%H>qWdp1znPm-DAiSc?CS0yGsKZa~-Q>fn?84m%UJ~rkL zWBHtS47BB16$bFJkbhJDJU*XukiYMg-uwy@*DgyMuT%mZimm&ZzefQLA16hbNdX*s zas9+=Z~XpoD*9(1^))QNN_^%m!(jrtf82oA9I3R4HmL;cIqbpnaK#Ipx4+;Fh{g6p zZk#n_aqjf+x{)tgv@`^mn(+!7Pv{u>eL?FE>=+=vY!O;0mO zvIeiO;I?`OrNmXFRFuHVh@R7_aYDR9$zmPIa~I_&Klzt zE}b%axfIRkY`}{yfoQ1gsRS%tiy8{M?v3MRD0|o4N$VRj=)EQ5{*Dk(-dAgHmGhgv z{*SzpJ$YtLifmwd-}S4Jl)>pO5s>MKqtY0WY7r!xr)N60lXbcTyHT-#+FQ$HA+H>C0#5Q|{Mlba1Y) zbm-2F2yl65Z=G@S-}CY?m#iPAvjHxBpGn7`SbTrz?yDC*YNN>PCccU`nK<~|c~-}s z9OE^Ov`o9P>I3s+xsEq|t$S!M@vT-wZcaz+ntdD z!?o)VDQzIR>Q-&#Jx?74!YpB?+VOBrLRCM$2jj)H{dNVli$3J>W+rzJ z;d$NJ`GWMv&mRayK*WnzUBwXXramrnFbc*eW6ou{=Yu}6?YpBkcwQmTddYoPFtUhn2oA zZ-G9?u#Oyy2=qSACWbV?cnzi*#;S1S{bPNNm`~Fh(4Pafofn6-8}6?E9jD}is5{Tb zk)D=E!sS;&VRC1`B+Cr8K3<^?8Jhm63!VovGM3P~e8OK37T>R#viTQlts)1{$(R`p z6hn8Kcks*c2=L+Ucoz904-%`QLMm?Id0md>nx{E~`7cTF-~x*^-VOETZJ+QZQbc)u|5@9~7_iN|zgI zdq9oFuOe(aj|8?&AsiLr)dgE0AZruzZ}d$Z;Pq|UWF9xH{wt8$R6N*p5)l~Ru{~$& z4R9S3vI&Z zRS{7afX|+BO#yGbJyvq>>$?9Lv&SZZ?P0f8`5-+m%7=Oun>Y6>zGv9tpp#}P31PXG z`vq=Y2kynr%)#xLJ$fl6yyCt}2NXNg?pGsEB;R&_90hZ4B_1^*d5nbap@7ptC~_v z_zv2yL!qF8c(4=C>(Kdv2BrQLguu1-JX^U0o(juN&pnESk6Ys;K2sF{IpvMeV`uQZ zcwQ!#KAY8Bw?~t)%_+=ip9um7u^3B+_PP#!@nY=Js!eZ~1cM`q(MK$=L-0PgpbA5b zSIf2M-Z!P_AhpM|dP`;mC>p(JJ`u|TFX?{HzYUfDKl2K@o>H`L^b7=gJgwPpgXznb zIEqWBY#g!KYoSloI!|xv_-iPYQ4lg4|K77=q zS1o}Gn$QXZ{zyd9~GV{hkN+p42Jycs&JC-6VkM}<79+;Y7&7q*k>o zOZ$WapUgLgT-gH9S89aFQvCk$aY7F7k#vmLnyS%Wl`%f>G8%t1^8uTC@JnCM4czok zs7ONo66eC_Iy8Tu5$E}x4dZ31K;vGJP6wwG#tWrT|Me>KW$7+=7HEEElPO(#0FVp8 zWil;U>wNyi>x5&Ft5x_}h^8Ijiwnkh5gXI6U8|i$9=aW;afwR+*{G;(=1(zu?3omd zwcM-^_vGjIFQENk34cA9zkd<1Ib$+#4Y6d}+d!*W0)>Z9T+Yr4hp`@B1L^nqFqlLe zqm_i`b*TKHG`}>)>v)ddp_1f{dHLP2eeAJW5=y?$dSf|=16uNb?Jt&`jJp~A2=>wX5Fm+IlKHbh!0 zNC7d0Sd?Z7xE?ueI&rJbC>61QM%v ziZ*pe5{!HbJ1ZlN@tV9|=Gf_L0Ijzo2NMqCc}0ah^m+Gw1<9cy^J-5iK>H)Qr-lTE z!(4lWOiN=v9BX^7ca9a$iy~*+a1(c+3po#G+dNDm<=a2gPcNRmd;rMzJp6G_FS~nzs;jk~YZkP{T)qfh3nuGBw zKFu>V-AD&Q9}&S7#R!o6SjmuT#R9b%rs;=_QGNY^SGsXH`;qHi2tmcR!XrNj(XBA` z=D2PJVxHQ=h1?jgDwg{JM-HOAf_bEitZgk8>a69NVpudDA(D)uMo!TJ4L(-%+Rd6Dw&?BqDHiWD_@ zQwJF415X+2c*2!%5bAzOq*7Q2uKV(p<4)pv-Kw(pDY=C4qImZ&`U-< zuagu<*!)IUkdtRRn~JUSLB8(smK?!wSW;5`Sb*va$y9qtQi4D4xPKxtviv!=A6kCf z6nSI_tzX?hUuiyuWaS+az~7w0$WrACq~@kvcgitdZ&Ey8<{Hug`y;l-JyDThcw6yc zP_hCPoNH~7e~R)_gtpmo?0kdp<6NbPc5ba1$O(vxb}g zZm(^gm#xPI=^u{=A3i>bgzV72*q|Q`NDX(WAT?wC$(I~)GI;~IxZbA%bj%jId} zHi{qwkocYb=bay}zsE0oq!5d#KYm;m%${xPv(*g)rNAx*WWacT@dS z;b5`jL%@x+60mwPpsUS==e6(pVV|G*=zOOC_S+84YaXbI+L+h8HS3OKWmyojTv=LT z3WI4<%MXm#F49q>`&wm2BwQeU`nn-h2NtELj#nI%hkEUkm&gxeb0rBy z0n;YE7VF=OxnolJ%=K}EP0^HcEu068G>$RN)mj1FyQf2bg&40DvgC|r>nY^&NW`;z zu0(M1>ud2u_tn4htH*mEo)Q|?1Bwr(GKNifeISoT?J^y|Mn8U~JrBVo`+z#6RCuR>TV=LDmcVNgQF~urKA#F^pH0rsXCs z(ivp`0IO>n0qt+nGoPsY z?bFH@L+=ks430%h%9Wz`r?$tA-8hZE-w~O6+&$;Sy1t5-(%E7s(fa68#=3;l`1@A( zx5P*_oj(9J-|jwbS}wfJ7E4eXO2H#7W!^&67ZO@6*VY6hWHwUD}A9Sq}T5BaWSW4s1M zCk%$rzFhr**^^K1;(5_9)gP*7`H4u^e)}2cTLF0r-j^)V?>AQc(L8BrDO7#jak|eC zKkwI=V~1dsGR7+@MeZ^M33{Jz1AR?*Dofq`CI?Il%P&Wb!(rs~$k(<$jMsj#uk#O_ z7$Dc{onH@i6nNBee_KQQvd9T0plpI+7YU6?; zU*(t)T;k7Yyu^>MU!CQ(8@XpefQM`<+L_fA=y?}Y5<_eh$P&#rPE(aZdSuQ;#B5_k?){Mm(RCpr|CPCGd_ zd7ZuBJ2Z10;}uxc5g3a0NhG|~I+U=0`Q=~!Uh|Q}wG_SsK*qmgppYs8W_qrwAO4M( zF%6$6rw{|E8&8wQ&P77CwSJ!BtPU)s=UqHvi|&X1A+HdV=Ox5?yg*-M+9e@m1GC*} zvjg`rdwk{?o+F0F_pcVN&dt1vgKDMhy~?_nJ#L{3B;-sRf+SJucRn+`Jtng-ZBBGu zMS`zMucii7p!osGxAmk^aF*pVUvN?>EE{uC-u!|0UvJt;vKlBcdn7V>@Qzw=V_t^L zdBUf;WWkm$x-{@?IM_Yp)0I=f>=EI4Ou z5ETBdsm&OtFXIOXn!f$u1?>sT5{W5m;EGB~8*apSC0?74+xc`7(V>}&$yH5(S|vQMkJsLdZ$<0xU0w{1zj3YrE1MNnv+hV(^K9I^E2b2@ za?`44-{N_h zixN#SUWO60gjSRnf7{}gAzD?Mw_ zxO%$2r3mBIr=+y}x_uH!wawy~R7ira(>!DjztuDG*(69P9?*j~%raTJ{CHj@`~m7S zT&sxK+~K7u^9p!5v$918?f-CK*Zs|+nx){ILXg!e!t**==Pa4ChQ$+%ahIMS57XU{ zm)9fm;cbO7@cDSr?2uwOXyz0h7iGtI#pd&w5{&3U`tY;ukI{I>W9QSA6GU3Te`wPj zhuA;o6%p{UWtxo_4BY4w_UyL?hZ#6^dK%+ZzBz)*Z2u%8#C&JgJ1rS3Je1$PSjOU2 zX0igR0#-dx@I5#d{Xgc5R&&H>G9O$;s$ARN`*>A=*W^8E0(#HQcFNTx!Jrhhdk(iG z-of*7v9uW%FT!{oaI60ic}i4?veKn7LUCWUtm-Br`#I@i`dptj6+0ghL9$yU*I=?6IAM>eimyJiy2$ zJ*+cs1Cp`P4$IV-zpwFWajY~&Oa7KjVDCrWBqsohBSY?+N|Af1#{}dH~-e zjMsTbPvwwn8}o`}K|Ed$%D}?abB;smX#Q5iUz1)C<0a6p(SCg&J*4|_yQ}r0{Q-&? zqLhSo!M4r=?yig|uZAG=NMpBkJ{$2GWy!KJ5$6S^CBBwE;6dvVEO@Atc4|Q_{YmQKQwL#^-_oD(>M_Qv zv%cZlD{NkXW^WzuF`WhE4N2b?+Q)oQ^Rb}NsPZB_^LQlT*MRA33nBA>bJ#rM*7h>Q z^F%8AFgT|%TZY9G6^uC|%r^RPpTGV@)dPGy(bUY&nRAQ;iY7MM_eEC%TS~u0N@*1A zo+20xg_MDPMZut*2!6ikA-;vp2?Q*jm|UoR&`F~W>EvAmq96GC3UaA;FAqn`L4F-u zaS3{#e%a;Tai_gl{AH=^U@Y#)0Ns?ah1r)k9q3E&qg{r`VF!&Y8Uk=3f z+g{@|1Drlsz3(qxBvdsG3Oo3rUUljIY`QI21n30Q)ndFj$h|x)!#^O5_l9i)gwi1V zlF8C74@_S<-sY;e`1Rr3YgsS(0KC3VGc}BJeLSF7!3=g zmB9~%;?5JZ`1tGUSWB+yHf(()P_e2#+qE&T-ZIPG*$Hw`=t}Xq9X&70>)7gX@esx< zu>ZTm%SL)=A#n(jB8!4b7OqDAEKN{6;a z#HGZV7PLQrVE9*`#7ZdNtY=dy83jI(-?(?3DTDFo=|?QpczYDMwf~2!6vm6>!M8HX zCan$iB}1bvD~IMMG`@$LIygpv@b@jMrrnr5mZp#+;~(ik!hw<|X;T!a@B1vB)twkU`#_LT|X7&DVw7$^lB1MKhHh2A3{I#kx z-ClKC7S2z^b!!?$Kz63koaPe7%i?(+Y5EZc$T>%8CWP*{i7(Js@6FHz9`5|0XH^PN zpTsj{VxNlnuPsC|725&(Tt&&yE*XSWakh-O>hQtn(@Xc;1{0xX_M+OszXwMX*C z&|#WL=sPVr!dYAj@r+;InsVXo@x$`|M;y;Fe;-6$SmCUK@-k$o{$7WjOZ?(hbPq5)d>TLtO%I&ZK)z@4)-}lEqqKH)o8Ozj^)S4k476!07O)JnUZLFJ3lE@ydnr zaxmcR&UGgw0tP1*UdJ+H{);)Z>ktVAJm?xo7^Pk`cf)Si}41VF`L~MN+L~7N1}J zUa3k__0ecO|8?Z+7kkxY%?){(x)pLKI?BLq*XqyGxuKxJ;@h!>8q?S6Y?`|Nz)33WuH3H?8A8F`lJGb z@)&Y%dNQ)_U<$Y<*CoBz!u)*{QSggv8U|pB+USEgXd&_dUe+!`a9givD^)+qH+}S>k>1#M;jM0`#0z`NOzAL}N$6v#Nt!Ez1 zOM`CC>ZDE4O>n87KgvUiop*#j{yc9+P6wzH(@L-Pf^9n+T!z*}A^#9*#a(LIf6UAH zdTkJMFgKJ9a7s~_S%I-wF5(i1@sd1uRd0BC92whwRoz`Z1x`k$pR#;}>FbWlu6|c( z18CQh&wJm7*B9Y{l&W>;3PP3Mlykni9Jta}2YBVeA^%cdT+P-JxR}dPp%#qS7xTf~ zTaU#tUX~BuJ!5<&z9BCUAI;XYAEhC?EaPm^U~6yvog?~9xDlL_R& zp8a?HmQz78!*IBx2;)^Jm?BAi&;TNQ!jK)m`3MVkC0`GpGGWpsB28Sid=ed-t~KLIcSB_*H@$G+pRx#DwLwU z9CVItca6sLV!YyVsY4IrHGTHrocN$NAaUH$@EVe?X(rdr=VMr2hEo%l?1NYrN}NX!OlJ@Z7KAlJfR*&`s>S zId>1nOY+8nPifsFh`}MQyb?kZJV9=9JN;H~GB#~Bad=b@2*19p8wPp`>})|eu?<{iSbuNbuJTVf9?OC*A{0A9f+Y3 z-;h`K82RHHtkPgPQg7na9Sr+I?F2^BF?(#(tk|#dgBET@*QT32^a9&=PG5N2wc%-3 z*pE4ofWa3AX3rx^*Y%a2NC-sI1J*yAJBKKZ$I9FNbX2L?qq-=*1REpB~4rk51i++stmQB19_j zr~2p0!6(`2;^66Upd7P3wsg4|n8zDESUK^$Xf$QGJpC|xlr1R=exHlp=Mz{SI8uwf z$1h%o&sIvY@XA6V*Ufvn+9BZ8$#e4kBg}uj%Um+&m!<=%v$i`9v3Wz^j6PwQUI!?* z(JG&6mca2M*nTS0vy{Z~;&{@T6==r|nKU;j&o5g;ZNzSc*5A&D48Kb@{7f51biU59 zwbG@+sQ{jj*;$zXS~T9J#DB^FSS8l>>6hU36;)DxJxXpB35yVIemhhS=Fht>zDN#- z-q-OrygZ8`<5do0jTWBQ+ztJFQxA++=%dS5lS(({`Sj%L0q4ka+43d4|MK3B z79Tb&BjltCby4)?F!$K?+gklixMg~bqbaKh-uM<)9`459@5oVKZ1Bv${MP{44M&rN z|LlkUx4t-12IwCpNkOKH$a`(ZU|3mw=y>in=D#>T80<-6-2z**U$GjZ_q#i6&rdaW zXaW5US%*#GKj$UrJDA+CTc$U=NiT#!GPDf%pASw9$EnQ4!Y;;CZ=U{`|gR z<{RR+Z6dMtV+l~?ImuIh4+5K0#GHx>dEmg>-VY7o)u9^-FFo4U=nxIlD0j8PMDI2i=gO)iZ+ z40)(sN^!n@iSI8b)Gu>vNPJ(@gccxThO9|WWqC>%%ppy~XV z4>`}I*Z&_H87~j>1S*K8iH#@n_ktr=1M1zSdW*=)f4x{G*pW8^oJr4hKUL+YD!mQ<-@XGBo z!>s0MaC`S8-R&;Mi{ydy&$gB^#PphJT0~7OSkRNFaJ68({71&P)LXT|=*j$Jiyl0$ zCv;yJ`L=#TxZ&zJ7Z2|MAy0$WuV)q}z)ZO}zQH&z<*SmbK z-H!pUFg{T2Z7+i zGfv*6c?T*q#~pg>@cp*6c}Sg^3o!pBMEfw8Z{Nne6cSHZcLa&SBk!vF#U=r8x?%pH znLS$n^w+cGSxePYXQfCAPHhYs-^ks-Sb9cM(oPdhw<&t8&Y|aY|Ii*=x}OhL>ahV$ zwMU_q^hua=ymdEU>(2V$A?)HGd~@bCQet|hD z=ne%+FuZh!FNT!wIE}TyZAqp#`lT??Q4SDoCT?E0^U0{rZ8_Rl-KSRyBbeWwMaYG} z`>dVd1QGTh{^4F1Ve%fiV6`IFA7Cz{K}Voq6q(B3ZmrWB3ms!i7oV_U`cmlp%9@7e zDKm_yM-T4B>x($*?bZAb1hhVu{*r;kU8tGg!~BLk2$pizlq=6=1J8vpBIKG5j@Qf+ zR>j6D)L;FV7h8*l1y!}ihP-+?R3^FI#egnhm*m)pKL`{)_SL1v^i?B2;wDZ*4r463 z&%S+g2PfGXe_t+jkU7ivuI0N3q;uNpr0mD;K@h&#ejBPbTIWZoPp_FeWs7{E+eve5 znghD)n2ZUg77!qI%Rdh7!}v?bg9fd9n^udyeNNuv?b+ zEE=Ew;w2b8bH=J#WkX*413|`_U&Y{Z3g4b9(g9Ez_TX$28OEzuA_!^wKn4@99+&e_ zcmQ+1{?xdk7G%*1tL$eI0YY(t)Fq$a`tNPP>zccvkx?ZFSTejwu0#9#7JO8yU90Bn0seI;%akSp`uV={Fka5m*0L+onjnyW`%TnCJg>q-PkOiQSV7Y3`D#pm+=b&z z1BsldKQ3q8F*Wc$8`!+FrgtW3b8BQKdnrds2(><}=NnS9mx6dX`2%lm-d zOZ-d5=p4tLubsS}>%Kx~>QeQ}x5mSq+46q1CTzZF=TgvTT26I%-#M-m{)&D5=YFk^ z9@53?@#4#fVV8=O6mv1WJAd-IVV)n<55*|WWafZF@cZF;OFS>4N~8DyO^jFH=__lM z+c)Nw@|KHPze^OlnBPuT*7?E1_}lkfUZQ#Czn&$}?e-DGvqWTIp*{DaF$ld+su?3Y zgz7c;`GHNAnSbWxMSGcHRiYmGHWL4WRe>Egk;Yw??l}z{l6v;d^;mzo+9qE z$+UPS%rFX^??#k3sy>{E;DA@t+ZgqUFN4m$TOWhIW4v;0?Dt(Mc!PXMUeshd zkpzMXBlgEVF?&4RdHze=vz*<;}2nc)Yo$e|%P%R<23146#p%5hvo`&m`SJuZ4D0$vwah&gdU^db$4gNB+434$(yraYOyGq2PGftPt1e(@t6x5Q7_-N|0w?y~ihiX0 z)P9>ge2FmTNpvlf6U$H7bg&n(pmk@}m4%#X0{Hxd!|T=ZtZo7@?arAUCM|{*KPP*Q zkD;KQ=w-$?o(pDMOui8D;MYf|lPBbI%P?N{Vh5&^-=Mq#-C9ypi`L)c7q7V%`kMIk z{Sbo0q*(c&>tlcVNA2@iJmEd4CFCbe25x+%Tu~3);e75gk=iE>_%e1mT-`(HAM;{~ zDve|sVTUlfbYc;3Li;&>s0|gu^rf2X5Y_tN6;gPfYHSkir(k!#HLi^g)7N6OXZVpd z4WM3`6d>A(*B9lDj?9?C?+D5L)b6dycOfZuTJWRXO%PfB@r|)12e|X}ZoJXNua7T$ zV(iW|V)j_?r_c8z3C-thz^neQdS+;nC=~C1)6EY4aE0mJ@+N6KyZ*fptmxCaj4Bbo zO|&Fm3)z8u^f0KWIYC$hv%uvVj29Pw(x*(_*T^x+v}xyJw4Xvl-l14_j8~rdnYd6i zz8`P@SyRe~=Or2?J$Uckcck}uPi-AdF}yC>=laz7CS1Idzk3H)F4zr{5lCF{ybgc8 zT$=bD_skHz?Gz>A&oZ0=)wcJOo%J3Vy9365A1KiSrU@tWNI zaN>sQTST^VO7*;K5*VG9e@>5b`Bm?mKCHy}ffV%vh41S$MDe`Hj8!?^*_M%y=@A;& zQUA4Abe_(u@Ftve>_^e=wl#Ouw#%>C5e~yJW6289d9h3RKB+huoVFs)J}XAw6HaP?~JtKjxKBTN>mb z%noWVue^IW^L07wDDNbG$>a&qvdxY%7}3C7dTu zL@$l&BjC{O^ROY@fbbD&iFS!>i0LU4Gd+RL_kApRS2I;mua-StS^_S_y}Dau&L?|| zfN~Tadst);WX>Ur~t3a$*k|9maY7Z@d#^_SLlk^N8-?15ea|6{Mj zEHf*eCkH^pd~CL0)B|qdOFz3Spm<+4>Ni;XYQLpbsg)S#WbfGK1z|=Ph z>!qC6HbGagiesNTcVAkm5YAMI;E&cvfTZN`!M1zZ&@xJ}&cA^5;u!fI)?|ZvIlriK z%J@v&>++Rr(e2wsL7+qA^WB9Y=n&W4yVV%=>MZU*7(-73QFqAtI3|5TL+M9|Zj=^$ z6V9#H^c4hrXxOJr3H-l4B>PfzM^(}NLnHz^<3U%$ainZPmG?>yz^QTk^~YMauy1BQ z*Zz-sVsuc|>e8z&WR4jXw4QeH*9itNM$g(Y->>&IQuGhtXS?{@de<19aM zdv9)=$TTPf(@%Z$vAQ>ak~Iz2)|U+m*AI=W@atjXODc1T&l&msje7Ozyh77;ZA~D# z=gcJe1zl&^s3%ObM3uMliNf&UO25dBU>L;H7*4uN463rWi^YBDeRKRVgITJs+Tg>>^& zKlaLRGPHYSydRvz6jJEUnnP(i<5wRAG=IICt$x-d*n^u@pys`5n+zxD&mNh3xc}eh zqwu{Er?a72u;=Vc#lRQ)F8%O2XK$( zx7Dnb4ko_Ng+|}EeUHq4{^RAuXG)?Zq(jte*E5DAk91%;^L!l|aVB5q&MbI;akeWeRc@2bhu`Thr9hEdvwNh5F%_Gk-KxFcRt zK`F1M%wfwaPs=zh>ealJD(>`rCoXe8{jt!KNw9n*hL`sa>XptgaO%@VEr{sKm56wN z^?JWkh+W)p87FzmWkDOMCyp(hR^eX{2g$pnd*^RG0KE4zxtRl4uRz=Meum3veD%Fi z@ZeM-?#2H|H2mWp5!mlJ^156y2nyr2bKP@8y{;bjF}du=+q2g+<6GCzgUHCotYTZ1wnalhQ1@%`iop) z=pmB}DcHC;)ax}B!oa?okFs(%pk7|>?zb~Ja7w-uEoNzAh#G~a8;NKp7IaRD4*_6gdt ziYsv3+hUnm2o5+YoRVn-Txnj2B~a!d_d>Yw(nMmt*zeLL`P!jg20ncss;iN8Tq0fp zJ7ujXzKcMQ{toY|YayWNRaAZcIhwz6P5jcxJ*dF+>2hXttuIuoGzX2DX@e+@7I&Nu z|Nq#_-nwtdLXi#n4R8AxrI|yJ<~L!VL3BJi)&|zBICtXYZs9U6sV9MO#ksK3Ks4XG zZj1X`UZn-QnsVzMYwv|aU&_7EqmQzoHWW`O ziu9}e*Zq$L)%D?81k`K0<<`4q3BOa0^k{{OMpi+AFKj9b{?*e^z{#Z^n-%3$23Q-FHa z?|aIz{I(N!s4v{>B!3bpm{L&h(?Gp6E7R{*UeJQ9i{3-7=2$O_Cq;>|aX)c9HR49g z0foRQbm8_ySvauK{{Bsh#Mk%w$Rl?RvGFBtCLXZ<1bsg0ZIRjQJgH7JzS{Cz_eAd& zhSh!HywMuLuvAa^D)a^#Uy@oo)@*yJz?!8d{f%oNJX6e)YB{0}k?1yKJk@aZh8jG$Sp~!iBZqk$uXnali*^0@DQ9&e^9Lsr52PKqCXt*Trbtp+}jT!0g!eS%w1j+8B@Lp11xMyRd@O^D$;H;3xtOx3}XB z`r+{8$9;z3ojE}1JyaMwiM=1$%&jt$@gV9o@`bIW&x*L$Nho-#a#I*`66~|hB7@<{ zm$cRHU8tAa?WFh0lgPc+?1b94?EawK%rJeCOC8$Cct{rtkoCf)%Mmf}-~1b2Hao`E zqOHRJ{Wbmj>}%dOqqx4J($PE%cDS-+6R-Hy9^y#-4K-NNdZKZrhH~!05boG7uezM! z6i5^g4{-XUpGjX`+R&6-8|GYQYSmV;UM}JyslsGrFgqq#qwZ4#aU(%4<%c7H*PCHz zYh5llY^6^5cGLuuzr?1kf3=RFUNL+ZuN*vt#KYSs)I2Y+_tz>k)G?o-5e2Op?X4l( zLc!K)TTC=FT2FYrOKcvUpoY1pAKNeO@`tm%i~C+Ss%_fqH2IgcZ%92g=&~h&Y|m;4v9(G=iIc`HWUijK$fA~hz4FfEvE;dZRNWe`p$`2Jy%{WzQP@VOMpo`My`1_lhb0@P4704d&x!VnYwv!(8ms?3H;m ze{ICq*KWa?*0B_DJ1;qT*%I}-t7j)F{%-Z_k-B~~-_r)#(x;TG!v6T?q94Q9dLlJTmv^F!1eoRdOJ92w0kfP*Bc3+^ z>?O~KeEz3zs+N^q;V{-qSCAxT=qu`VxH0Nrf(G&N_->AV?7*A|{4`Z>=3GJgGyY5d zs<-3pqO}H$NU3%0-;3P9A`}ec!*(p0tCHVTB4zPFSx}O_^~mB)(oN ztwqoM(O)QdpJgbWQU{W=Z$v8mrN0n2%C{oHLW=Y)cH#^9ECd?{Nh4m7p?)(72|~FvPg!NN;^>Y*`wp=h4o74)EO5*;!Ax&c!}@W*#7? z3)|_`WjS7By+)=|Oz53RVTmsONAG*&JbFo=H4+|m_v2u6o`&3q z8rwfXW|E-AdXW_Acl>NFy^7>}jr=?^2n2T8yRCP>*95%*d&V@gdR69uWkH3RTdmo=7u(gT@}E06 z!H_D;iiXi1WZH(QUmirykKV{%GakR5*&|*nSJZY=ok#mk3{K?*L>dB^&R=qBy@K@; zcCLS?PC^F44Nq``^u-`QKFw!R8VJ{%%YV(W5uiWIY(-20Ti^FMld`Ceqh7wG(n1H# zh#uKG12ef*jUMjIYSaXpkD(p1Ga4Bkm7n9tAUe4>uoQ>-m25A>o|B|HQJ^ zC&?a$ex#hj{c-;NMtr@$Cm+3aBn1YHJqp!=QLhLiossr@BT%f7j<}56o4oN{-B_Q{ z(|G~4eJ%|u~SM8c|N|RSyS0Vj(U;kH*HN^Hh^uvJ(h5ESg%Nvq`0Sfq%bL( zZB?R?rB0B&=8Zq~fb1v7Q~a*tT7mn*f$gnJ|ErR7=`{#%5&mx5{aK9e;O z8Mx46Hejx$hn*zXR8tz@~~b}VQILh`K0hsjq~@D{vx== z{~)z#JOI4j7`^Z5&IJ>WqyECruwJZ^wZ;?w?63OI``SlFxX3%Y#J!9JnmMzO{#7+! z`P#Ar*!VJ?o)3^i)+ZdM>lAIi`hkQSNfK4t=Dqq1Sm~^g`=RygpN?C**ui7(3geo4 zSg)g*l`L(D*YI*qN`N`)6&}d@<{h~q*ms{N%s67b*vc#S&t$Rc$f{@)?%B7D5fYNf} z!uj-E7>uDFxc(cvet*Zonp^iZ>gDX=KTTjI?v*5~r;*AzwtGu1eS0;4kx`%PU%hm+jQhbGbf> ziNz0$C*_y;G&k?1njmOC)`jGI+rpoo(RR@6Joo0Z0oF^iiH&I%S&tqqK9Zi}fsV)C zy|RPRiv|$b!k)|PhaHbUGWE-1tx1u7O*Yfg2IM|N3XXw8F#%xo-fEz42w5+j1rhOH zte1+z)Lp%gs8^f#*9Cet;$Cap&#YP_UUVl8Cp%MN$D?$V=zT#XzVu7UzrSYi1A$+A zHG<4H?=`x9>=HLJAJVY!V6=Z`2XU$ly#^LoucG>vd{-pC=&sK_8Vf_c_O1D7ZC5sg z(OG?Y1LU5t|5|^Yl%D$55J3v2J5KUUwG=`7l@*g}$p9E4(@Q*#%!f2Swv2YtVC%08 z@=w$k`cW^|yO9yu|4n`0?nS4-z9<4p>ZY#Wc(7jcWofnRt<)gjwN>O5y&rTN=aIZ8 z-Mm-xo6@=q_8f3w?~ox&i7mK#O74Fmi1nfzDId>5^4I0F>;;46X#N_1D#_C9U;xV# z%GN(?(ea4?BCgKQx(|A2gLq0tNq{mlGR?EJ2+q283`@sfL(Zj?E&cQ`7up%N(ntnk z6k;N$%;_($CNKvTKECN)zVKRaHuwM7%MM_^bQ-i0>y3{v1KWL)= zxuRRS`Sl!IW#RN1V%^sOEWPr71?@w!Fs*6ic@Pvyuz-QMEG%`UPtQj zUmmX-z@EB-GY5sRUJtUk120}7LB{Lo!|e}{^;Q2&8sRiNycnE{I1!Qy)QjE07ms4S z-ud~5k7c4>rQP+9h5CtmEe{Fxs1%5R@2bUfwq01SnTKo2(UvebSG_#EBM&c{V<;J>?1s)_%+w{45H=`Hp^Pm z{=dmz<9!FK3DHRZm+^xteMYR;ZN;G0d?erBC=$EM(B=ydvaD@{C^w()=O^C#R%CI& zuY`Hauf5iQV=)}sK8p3K&Kem`m_W`OdURC(+YQt!Zj?XPsl^a_NeXJyZ)3+}q!b$` zF9RuDE`M(E^mP${)LZ805_^*}51mj>>hv}s zp6|_j^)!ALi@>w{>QfARuwJfDy~g@msKJ`djpR9xFZ|p?mC?Vn`S^lZ1F5k-4){o0 zYEctp4Q9r5_1=qEujcLN8(k4EMb{%XxpksLb z^qhV9t*!`;c)a{H_Y^s4S0jg4YBU#Sb{{I(7mxLFYcHm!x{Z2GUgj2lZbjV7;O2YA zo>UQVOtPoV`^)pOOug{Sek9-fhBj8Ta{GdC73t4}rpO(m2)Lks8>WG+rl+^4ybL@fBdzQ z1WG^6KHK`N2sk)h82K;Y;iSytp3#rFpm*(n0|OVbC%4 zGu!HYtynMZtb5(2$b9=vE1tP*0`;n@xm@nzVF*sGqY)Q=pkArGp1svA9MHx*8ftcv z1ZXY0Sj(RjfuY<{!l*hP3Q`}{MIc@dE7oMf$FN==pY;vWqfxJ5hn;zT%EY}Gp9@Vc z28ck!41e8s2CP@y!G_5N#OuMy$|BBcA2{g3=Pveb^Ijdzl_uXpIlyn=efrx43vig) zFFDwT^$MI3FKl1*JpKa>K0rZrtt}jRpSjuouc+In-!V4W6YSuZ zD6Ms>lmyPxIOa_v&&TdEsb|VZ@$g}B!23~4F5K`~P6)EXdNB^PXp;WnwVRM0vNn$N z-xKNox_1Fb&gU-zaX&M6gAJD ze&jLozMFBSiZ;3zxt}jtL;l_?0X%VQh(fpO_&?>H}Pu%O&k=PU7qatvtA;$lMAl6GVdE0;y5?|Y%JKgnb^nvQY zbKM%dH}9qWM4?SUofBj?Ueug z>uXjzx43%)R-+}2ndk( z&^TPm4eK>-(r=)hg?fE{)f&@kNZd=jzo=TgLj)#I^*5USWj%VPv-#t@4r<7|ppjCB zcrl1I9`HccJ2soY^2za^(~fWgQ`fNMwFxt@5>pHF*1>um7-B9IN4)A;^&Kpce7E6s zD2NB3`1`c`i@z zqvx-F)sm3?tzitFJ-LNZlGyWC=lKnML}*E2`P5mnWU^wQ9C6JP495dq;MCjpa00Zo zl_xzf$Ic(09uCeAEJWjrI__qoYZq}ZwyKMKt3x7i+F^(+LkR2T7^1S$fs99^#)oI; zkofAMB^#}Nzj?3pOZSx@XK;e^&77w{OUsNG>5Wj?MuN!$uG7+`FH|o zEW25rYsY%2-gNT)R;K>%d}t!>==1Je;$Cxp*&o~6MIgpBOu%~JaxRiy|6SmVr zr9ySc^U;d$>!<&Z_q8RG4|FH_I3cm);DCdn8H|3Fyt!{L){8;&>zd;rvR)`-Eh>h* zPududrYRrqy`@6VUo}&U*?A8gk30i4S3ab3BKugsyx#th1QN~foR(WC0(rC7FHLyy z&?!kl;%q{IALa6%7frEVDd`<&KRiOc98!F#-VAFI9gp1dRvJ9dMPNuq+dbn1)~l<% zVl4$3k6){P22df-N1B>EN^i2w_fKRB7oAC-=LE_3S8HoJ%;9>IAXT3c)@$%s+(|*? z`MBL=Kwjtq8ej4YGPXU)x%=Ujvb0CM(fGPwn*O07nhWGY6Lw1>{n2En2apqiiowcN zPH21;4>%*?J1IE?sLf89+x-gb75U%@Q$-2tCH$)}@gAc#Q7`YH1F1vvB5*-qH%Oad zy_TqwXkH=f6D4`~HCJx>07r>ZI}h?cV6*xD4v)9{%zaLn@A-Z=P{#(mRNr|=D`CBK zSq%4IL%j5_vDpO)pJ?QP?9dd>1tVGkamQbiK#?K;iYe01 zjBpjN;9EG4K6c)OeT+!swz!2%)wR<`O*)iC8b@ zcS<+~Wd4}I(G**88uem+v(-0W-w>)!GPRW=akue&oM!37N5&ok*~MFACx(#qd#x|3 z5BC%US);{SiaXbU`SqLNA8`c0Z~1Iqw}L&d_VPE)M|IE9_$sctI~s}feG+><8k1Qa zBvluMD#nU-v3FRn1bNo31*BiD-Y@>;YMKwwjnQf*Mr_{eXHCNLWpOSL@4UrTEpG=i z7Uqu0saP+H^E_QENWPB|<$1~Pih9xbq=eq>JPJ20OXaw(pkA~gB{*r9Lm;2)v*4FT z0yFp2cbV@ihQr_9@ipuW0LR;*xOF@MX4Hgb{46Z7`(b)NFgJWhy`&$`y}XI+=O*TL z(^XjTCC9{jGN@Z5nyX@3jWi{f6x=sRRS)SgHm$?xO?dWt*VdwFl(t9Lat zY{Z5O1OmkCvftQ2F#gMEKs46NeqoU76_UR)d(V?CBKP`kJRe_sf9~=~kcI5BQr&K3 zoNjnMUSDv*TXKM2*^;2&I0;zY<#?9OUJOArlD(PNu0j3z1NO`*1egf)ovu8E_1bA` zH#+b~f8p%tlm=}Aaj(0H{1hicL?Q52(ZSnqv0j9zmbC>W-!uQX#Q#Op2Ryz9c10fD zyjLFAEcda;oDk;w#IccZ5?XPO)GURuUfiF=E}ccj<4n5ww(rP#?1q<}qN1e4FKL*R zayWKw9IYoDTrS1w_A&!)@ZIza*Gb@n$u-U1No4*Qn$p<((p|j8G&YdFe#dDa9{G4Ei3tWK;41 zyB|#1>~}ZsmHyhzlazxCQpv~Zi=1rW0ge6I6EUpUfggRx84xeh4Eg#WvS@rsc$y9N zILW}*4zYn8Wi-Av=0lOo4~MENNTA$In(EPhHyp{QT|OFglTWJk9IG9Qu=K1wdzf_l9d3G#ha zEd%x!Mv|gGV!f^po*_F~LjpLD*1<5OUbVGt>oyd^gW#DSNd-;_BnFDS591JlFS~g7jGM3k35pUu4IuuVJ`52_mNAjfwPNKRVZTP>p`-o1Odrkd7s{D{5p>2`RL>^$~< z1vi}?(&xUTUbm(2$4m-|do6vF4NY_tg@RxMeoud_*WA%<^rsLn-&P$$rUFv0ZdXpO zxU_jMQ}@WjmlrufmHs2$fo4;fsyouM{+IZg)swAzhInl`c7y3$G8$hDDkV2fWEH{t zG@tSL2WWhmpK(aKY46Jg~f}n$q!1G*m=#?=8yZW=8}2} z-`oT+;UwEf@d)+WB3&!MU(5p^ubUR1L-s!=FZB0aJyHzG1tF$&=kahrj1CUIB*2x% z#AMaK)Dth4ZO+R7aXzMVG`okuTjab)BI`L`jrV=`A?KSEaPUkmjAOmJx9hFRAzoMX zyS$P`ec;D5lYng5=HpBB%H&D0ubj{j?_}W>ZVtP51PM0sVZFYI$NbPh`oE6X8wj>(lkMkD86V17cb^3U) zrx=^hZzX`}o9?`zZ`l0Rw;Do6O^w!HHq`9}vuecSE4WhbNNBt$^e>ian#^FmBKprX zWg-3d?jmwO<@O@$ISluT9^XbL#YT{lIo!=HSAB)1)9Z?lI ze%N}#CS#>t5ZTYyFC_I)5b>f;YQ%*%Z$7^Acwbm;nc;+)L-w^QtX4oMcKF$U2h(zpSNJMl33#^^mQ5i0+%~+xvyCi)9Hol6zM$c!2?LY+|zSXDXQYA*ORL+#IIgZf~dF=;pcLwmwuZ-5@V-7#g z-H*(N0?4yT0rsDlO_I%@e(o5vj+B<*yZmt{?uP zqTnQLkRE7@^*SlpIK7Valb_f9t(C;!0~)`yL?b*l@0AgfmYSQw2?jw|+h3v+~n^RcW#1cB6 z^2W-Z#qPKI5iF4|{gpx4 z%M$Xgl*&d0V!h~JPo?-E@#X9q=w20%dO3Blc8J6&0-kig1VtL^)imeQ*m08wMuu>S z30skV@=HwZ2j3z0ZOd@T%I-$iSGg{*PAw2%!~w!nldxX&S`prqe~ibvOKzrnx{2p6 zpY#frd}C2C5;Ued^%t+b9aq}&hpFMMp8NHQA%o~#J%*Z<0>3) ziGu&4%sTD|Hog=;iq22>BkvzYsXm!A`oQApXSefoHt*%#pd~ke+?!3_q-`qGX9@W7 zLlTuYuwK=oeY`YC|JPNHVD2Jbw0}ad{$1+Py(&=X%fEO_80|ONm~S`tr}R%vk%EIs z%r1wUB_QCap+>S6jPz9ifIIy-4mul4ipOv(Bx@PSpY;Rgfi-DB*!Ep+SFS=A?H8};;>+}q>b-AGmbe110IDH)TI({vL+?|IP zIKnT99lK5ji7K5V*X>H+;h8(Ej+*|!FV_)LsF4RlXC~)a-eJ97?Y!{mJ}2sxdEHat z^9u2NKXxzJF+4&PL~m++@ifGG!O!QP77#B{lklex5wA4v$M=0nHy>YmSEY6a%X5Lc z%+q^PCRVT#l(t(q9_uCb$Sd=o@z^9-MZw^VdKv6IMiug01-fiVT11h3-5d3UYj3B` z_1}kKO=+UZR)-9nOzIC7nUuh(-)r6zw*DaSI&R>BVIIs#^IVxt!PfU#r;}Ol97Mft zEX(>7ZzUdI$)h326$GMS;jx9O|2Wnw#Lubh+$c4?G<>^*2Jw1XeBW@~eDm>@rF}F= z@FB83vtrSBA32XtZsMYRf+^Ok)-LniGo(MdQ#hYqRR;CiqS|v~znTjC6w^I&6i_c6 zX{!fo^1MJVW;R#1M1`EUe#NV{qryX=L=6NkNmeBY%e-m zpLO=Q__vSlw)_=E??p}W&-bp4&mlkYq&>&UN$lOuop-$Ge9_I_`l9FAzkPnb?7Pm! n=ePb}pBJp8A#t&GJay{-e2jwfpMSaq`D+UKdFALofA0SUth>AA literal 0 HcmV?d00001 diff --git a/results/full_sweep/daily_log_sharpe_excess_penalty1.0_aave_1m.json b/results/full_sweep/daily_log_sharpe_excess_penalty1.0_aave_1m.json new file mode 100644 index 00000000..3b0da195 --- /dev/null +++ b/results/full_sweep/daily_log_sharpe_excess_penalty1.0_aave_1m.json @@ -0,0 +1,9 @@ +{ + "daily_log_sharpe_excess": { + "price_ratio": 1.012011484075616, + "centeredness_margin": 0.026628408428241428, + "shift_exponent": 0.08827350039359076, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/full_sweep/daily_log_sharpe_excess_penalty1.0_cow_500k.json b/results/full_sweep/daily_log_sharpe_excess_penalty1.0_cow_500k.json new file mode 100644 index 00000000..b45c1797 --- /dev/null +++ b/results/full_sweep/daily_log_sharpe_excess_penalty1.0_cow_500k.json @@ -0,0 +1,9 @@ +{ + "daily_log_sharpe_excess": { + "price_ratio": 5.992515918267629, + "centeredness_margin": 0.4084847505974458, + "shift_exponent": 0.00014153838858700288, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_20m.json b/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_20m.json new file mode 100644 index 00000000..43b73df5 --- /dev/null +++ b/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_20m.json @@ -0,0 +1,8 @@ +{ + "returns_over_hodl": { + "centeredness_margin": "[0.01026839]", + "price_ratio": "[1.3478746]", + "shift_exponent": "[0.54106706]", + "subsidary_params": [] + } +} \ No newline at end of file diff --git a/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_5m.json b/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_5m.json new file mode 100644 index 00000000..edd7992b --- /dev/null +++ b/results/full_sweep/returns_over_hodl_penalty1.0_cmaes_aave_5m.json @@ -0,0 +1,8 @@ +{ + "returns_over_hodl": { + "centeredness_margin": "[0.0100561]", + "price_ratio": "[1.2961863]", + "shift_exponent": "[0.06752973]", + "subsidary_params": [] + } +} \ No newline at end of file diff --git a/results/full_sweep/returns_over_hodl_penalty1.0_cow_20m.json b/results/full_sweep/returns_over_hodl_penalty1.0_cow_20m.json new file mode 100644 index 00000000..e08407b8 --- /dev/null +++ b/results/full_sweep/returns_over_hodl_penalty1.0_cow_20m.json @@ -0,0 +1,9 @@ +{ + "returns_over_hodl": { + "price_ratio": 4.05509593366626, + "centeredness_margin": 0.3192831580248914, + "shift_exponent": 0.0013440567877262095, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/full_sweep/returns_over_hodl_penalty1.0_cow_2m.json b/results/full_sweep/returns_over_hodl_penalty1.0_cow_2m.json new file mode 100644 index 00000000..a2c913a8 --- /dev/null +++ b/results/full_sweep/returns_over_hodl_penalty1.0_cow_2m.json @@ -0,0 +1,9 @@ +{ + "returns_over_hodl": { + "price_ratio": 5.832161383230471, + "centeredness_margin": 0.7765835409120687, + "shift_exponent": 0.00018668202918720232, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/mm_noise/meta.json b/results/mm_noise/meta.json new file mode 100644 index 00000000..1c37a0da --- /dev/null +++ b/results/mm_noise/meta.json @@ -0,0 +1,227 @@ +{ + "model": "michaelis_menten", + "pool_ids": [ + "0x072f14b85add63", + "0x0b09dea16768f0", + "0x10f21c9bd8128a", + "0x1535d7ca00323a", + "0x21d4c792ea7e38", + "0x25ca5451cd5a50", + "0x260dbd54d87a10", + "0x272d6be442e30d", + "0x32df62dc3aed2c", + "0x36be1e97ea98ab", + "0x3de27efa2f1aa6", + "0x3e5fa9518ea95c", + "0x4683e340a80492", + "0x4cdabe9e07ca39", + "0x4fbb7870dbe7a7", + "0x571bea0e99e139", + "0x5c6ee304399dbd", + "0x5f1f4e50ba51d7", + "0x711af51a937e01", + "0x713fb5036dc700", + "0x9232a548dd9e81", + "0x92762b42a06dcd", + "0x96646936b91d6b", + "0x9d1fcf346ea1b0", + "0xa6f548df93de92", + "0xa83b8d30f61d75", + "0xb460daa847c45f", + "0xbc2acf5e821c5c", + "0xbda917a67c7d9a", + "0xcc65a812ce382a", + "0xcf354603a9aebd", + "0xcf7b51ce575551", + "0xd1d7fa8871d84d", + "0xd321300ef77067", + "0xdaba3d8ccf79ef", + "0xe99481dc77691d", + "0xf16aee6a71af1a", + "0xff028c1ec4559d" + ], + "pool_tokens": [ + [ + "SNX", + "WETH" + ], + [ + "DAI", + "WETH" + ], + [ + "USDC", + "WETH" + ], + [ + "WETH", + "ALCX" + ], + [ + "COW", + "GNO" + ], + [ + "scUSD", + "stS" + ], + [ + "wstETH", + "JitoSOL" + ], + [ + "waGnowstETH", + "waGnoGNO" + ], + [ + "RDNT", + "WETH" + ], + [ + "ACX", + "wstETH" + ], + [ + "wstETH", + "AAVE" + ], + [ + "WETH", + "USDT" + ], + [ + "wstETH", + "GNO" + ], + [ + "COW", + "wstETH" + ], + [ + "waBasUSDC", + "waBasWETH" + ], + [ + "WBTC", + "QNT" + ], + [ + "BAL", + "WETH" + ], + [ + "LDO", + "wstETH" + ], + [ + "QNT", + "waEthLidowstETH" + ], + [ + "USDC", + "stS" + ], + [ + "WETH", + "LIT" + ], + [ + "GNO", + "COW" + ], + [ + "USDC", + "WETH" + ], + [ + "AAVE", + "WETH" + ], + [ + "WBTC", + "WETH" + ], + [ + "WETH", + "ARB" + ], + [ + "WBTC", + "BADGER" + ], + [ + "wstETH", + "sDAI" + ], + [ + "WETH", + "EIGEN" + ], + [ + "BAL", + "WETH" + ], + [ + "WBTC", + "WETH" + ], + [ + "RDNT", + "WETH" + ], + [ + "waGnoGNO", + "sDAI" + ], + [ + "WETH", + "COW" + ], + [ + "waEthLidoWETH", + "TREE" + ], + [ + "LINK", + "WETH" + ], + [ + "WETH", + "ALCX" + ], + [ + "WETH", + "COW" + ] + ], + "market_names": [ + "xobs_0", + "xobs_2", + "xobs_3", + "btc_log_price", + "btc_log_return", + "btc_realized_vol_7d", + "btc_trend_7d", + "btc_volume_zscore", + "pair_realized_vol_7d", + "tok_a_log_return", + "tok_a_realized_vol_7d", + "tok_a_trend_7d", + "tok_a_volume_zscore", + "tok_b_log_return", + "tok_b_realized_vol_7d", + "tok_b_trend_7d", + "tok_b_volume_zscore", + "tok_a_realized_vol_7d\u00d7tok_b_realized_vol_7d" + ], + "n_pools": 38, + "n_market_feat": 18, + "per_pool_gamma": true, + "hparams": { + "epochs": 5000, + "lr": 0.0001, + "l2_alpha": 0.001, + "huber_delta": 0.5, + "init_log_K": 17.0 + } +} \ No newline at end of file diff --git a/results/mm_noise/model.npz b/results/mm_noise/model.npz new file mode 100644 index 0000000000000000000000000000000000000000..9e9e80b5fceed3f42ceb9ef939485a9ed6b3a18f GIT binary patch literal 7534 zcmd6M2{hGT+qSVphJ=hILnsPGDs}&oh*U&_W)&h+D5Qu6DM@Hh#=pp1=GkqY=VKnv zG0$_NsE_A;dcJD?pS9lad)9i_d!Mz>Ui-eU>zwnu@4ffhXJ49!Y3Vqq{>wH}Z9AhP z@rjy>YPm5}aZs6<8(!2ivACwU)y%?jNe^ zM$xQ!kE8k91$-qkFiHOn@qP1RK-1MoJigVBv{JARWfIok6g8$I7=`Wk-1M5q)Kd94 zl5#luxGqS`JX}ORffvrzF@4an-@cErb{$c>xNCc1umn+YGBedlxEW&NZ`Fa?s?SbXeIWvfEl*gt+wX_?@?ud3edrea-&uIwH)5;ljQf3lPS-+ittq zHbRz>#d`kY7vz^)OuuB5foEc6{COPsh##MghT{s^3C%(NH7n>uiGdFryZ8DH;>@}C zwIba)AhA=TS^ZWTC>Gbfo=6cV3Ipk@+1ojYHIK3-u6^GCn!lBM+U{KwnUjiYxvg3% zx1pY?>Ay<7>|C~={eN0&K^cYrq}q%LdYvJQP2xm@k(6^vS2pbbZaN*u=82&Z(MPWt z&SB&j<0EgkB#^3Mp&K6=g5DUTO>7_A;DV~Qg#vWp=d!VTPR7Y#LGrlGd8rXB=*#BQ z73$#I*;8M|Ue01={dg87x(;}g_b`?E4gzo42C-iHUToeMOWNnlOq5e{G|o@d!d>6) z_fIHQ==|}z__3qHgekSl8vEW{=-Va}PBHg`)3#KRW6q685) zseEu4*w%92ZkG$pMq)rfdw^2-F#N^ArCYc75wi=*+uw0f;Qm7|)m4KzP}jwE?f`u! zXzggcaDPP^)E21No0g;kgS1;sZ)pm|uPj`QobNNZ#Rb|fKG7d^esy7^p>md95c6aF^5_(U_vmSd&fx=Wl`SP?HbP#vlZPb>77M>06 zl(KltC6Z1*o9jhqF=gYIuR0+@SCTc_n*x_-!|tql(uvRFURu&`?g7Kq#J9}wX*h3V zpj1jdfJ7!ok2R?ivNNeT1(NDe8v-mCNkbqvYGGwc%7J57SM(f-$%ZuFSNM*q5=$G7 zhb2=rg3Vy8%E7^52v}%cu(;cVqpc>ay_0ER!89G`mX(8UPNElYat*_=9gcs63df<$ zmV-N-KYYnD7&?%*?oJ`H-??zWu}THLti9e(x!Zw4qbHu)cXXk)hh)^6=>*)_$`dQ~ ztro#=B&6ngA6#T6J^a%45!|0UkZ-;k!$i-|o6~&@P%)DCZh2b~@br6ot^7=Z3T0~6 zij_t1s;GTT_BJBvvehRV-*nLbD5=TYnT~;rlAW9m6qxv{sM@CW1N!n_eJkdgfM%8- zeQt5LVsOI4m!6%ijRjlM=j(Kn zVeMz*QGuagJeezEwy$FvPByXk*z^>EW8Y;N@1!QYX5bVM!TJQkPM^2ulud-sqnGp# zc-Nueslms?E*;>*5VzrD_!O30b}`loYJ^u`pB!#mgOFHL(AA)ljRS@wdCg6gNc)sM zV+B(z{AHzA`TS@P)YWaL(+((s1j$ea74}+qkmY%xAwC*Cnl`<8q)`rwSHE;?&GkbK zzrcNG{$8l?tFSpVmV-aMG{x#<5Ij_~d_w{!z(HB!DCgc9c)YiIox#Okq?G$#>Us7N z=^Nh}^;GsiB8fL*@7;RnDXC~~l}y5#y8H~^;d&$;O#S-NgbX4I;U|;TyTCL3WKl&|h^eOctJ7cHGF`F)+LA9A+P717|Dv5d; z-GQ=~<;E>cB5@}1HgCkCLHKUAVtqqHE$BKsGRYmG;P~Ohm+E@`Fg2f_;N?DovFm%#XL`g&LU3?DY)9Ixt?0bwEI;>L6F;mdXX8BTwm|Nd396&UV^rRJs;;N}-ew|5C9fbJUu{OD5+OwR=x()N18+K6qrKJt^W zzc;eN@k9gUdu%nRem?AM^nGK(T9?YzXE#XuXZGpdwty8jH`x1mXEm1M;`^ z1^7yU4Iam~gSPOc*<%?YXmG;xcnH>kMY=ZW%32CoKM1a~(CtHOX3Z&O|0y(P;j62t zF2f#ie^&c9wV>}9YFn?r0Q4-24VBq%z;Ba!!fO``F!E7#+q#<}JJRo4hqKBdta)49 zNq$s?(M-zdxp-p(qM)pPQl`*iK-C~|mI1HIP zOxh3db;4zTUcO$TDflcxqdwG~huzmS+vOGpU_n&w2&dR2q?AN5Sk`_)KgK;f7O56s ztA{mhYj8c}N-|Ar2=$}8eAt}y+gaFO9@wP8(T=SQi;~UP2Ve_U@`~tpVmnRz{8q0> zOxN~2dvv4(ywVd!obEK^Q!ibML$}G`xFdfkv%C(43tP@FrVPQCh1}>X-zeB|)ae|L zTRp@-PoR8$nbG*-J9HxvJR{p zDcUXQeV5rs+PxP~e~`VyN>I={YQ{BDs~=Bjw~G|2ea9;YqqBI9j``WG^m%fK; zFG3q_x6%^ad(~sbzA8dw(pKKYoTcwOm8+35iMU`W%R}ed58veOe`@y`LWW~~!&J{l zu|Kh6Gj=Q;=xS>*^_t<8N`8*=ljdts@Yg|@ibVr zSirkShpI+$K(ThHs<%ktdC>h zVWDi0yX53JZ7>Ax$|Wy^LfgS;&!oqB-vPM5nRzU>Ck!Or$f6r-DG)%JXyCaT3_tEA z+*}`3g*s<F08@UtNXbgd~JY^bq6(8>@ULushhQA7ZDzr^lVez zJ_0;*iTuj_)!25Jy@YIvaE56c!XeNBUU>2>Vy-?tCOQz`K-E+s5m4_+(UIYmadU3RtyRJ>Nx=NbsYH4A3 zQoLZ}$sHpgH{}=TZqSXnnj_*hGHp1+Q{EZAnG8=Ec5Qs?5&$xi1>v6}$02KehMgy^ z2l!Gq$Pc%Wp5o^iblWQGrV$kY5$oCp0tU9U6~@Irl&gKaz}ZqP;@@hO-{(Pr`O@N zA4001`D!7;N{p5CAq50$)k4A++OamOD)haTC?Q$s)2$F(2`|*k_Q_%vHgo!i)U!^a zRq_eTlj@^5a4>h1;=vqnc5ZXwTpzqPk1qn#?iPQsbaM>7`j_kz z6^`s;x!9%{)JAKrY$XHcA`Gflx0T#5=AgZbXRoj89bbM;7LHBw75#*mSWcT+iMwU6{dl@hbP9fSk(jV39k zNfJ_lln|^U12tC4da*YGziFruLqID$I_W0f8+{oW{u+H1X19$nd1O+4%PJ1s(>S#b z>g3_DP1UfWFKzZdhrn|@hkbDCr`pUH7r#NQ>n-00&&$AflkSbe`uA|YJgnQ1HV1y< zl0K^!sC(07|A`ZEx?*XdU%j;ctf*WM6rqT;^-3@5&k!4&J@@<`1qP1X-;_jY#BKT> zV)U{<(A*-Cuj^pj-X_(OH!QZFr04caF#6_#~%mDjuM_lK+wi2rikX_ z`28B6H|f*3n6Yn7C0i-3uPk}{bXzDEGLG-r6447r^^q|tr!z6e#k0*qf`pT^y4k+1 zSrER1(Myt;0CF_Hv6yrY@b5hL_L_SR#xw2>1g9=AAnMIByFTI8hKECcoo#}fRa*pK zOmsnWD=T%2NFT0E>Sf;J(g(LD?MKAvDq(VlZts}F1iA(edg}9#uqNdCGX>)YJi_eh zeI+ah^fE;Vm9x3PVZPX-$IMUM-e#$D%)kX7kNEY*+zQ4s(k$ZA=O}3AII`Z1W&{K0 zG6lad7lE{w$KcJzF1+I`8#m4#vh;uZF@s-JH9S_fuhz1h02aSC-SXaC97+&&$=@^% z0)n)5H>t8Q<|S{y!1)RoIQ**o#PuF@I#Zj*nOG0bcc_^&&1%8zg#6bVx2n*;L+Qyw z_qD{C%AuL|P%>!xF{?fFZAR-$*V&~qOOY`m?5oW7ejM=Lk|jYq2}#$Pn`w?yXWwx- zn;XR%hU<>amoi9Az)#%NLg&y(G6ny{84r%^O;SvP3+}5cq}GYzXH_a_2~k39)e#5N!A6{pf0^4blZLh*)zcihK4UD)e%gGg3()%F zeFewkA<$X<(tqc-NQjCn9QO0G!RjUIWwI_i(+gXXtyA4*D3Db>#6G9DbR})ehSgyJC!FACo2u&BwOx!hqh6jSgAB85v0)OFT?X^K*F1V11C)rY>Mm9J91+-?X?DY z%J==a|K%bu_vf`a+}H#k0y+)v#}8t6;Oq}ACjDE((5@OS^_Z>!{P{Bc?)#LYyV;A>z8NwKSxd1N zyLO@3nqnm(|5DVzupRn*OXrdA_6qe6>!5nUQukBA80@yrVV`9#gA=m85%U8QM1G@A z+xL_t2tIsA)rb;_eXB%g%Z6ugALmiWGujLI$Q}hgc(lTRZ0uB&T?;UsX4^gepa2_6 z+jYNUJ%qHsAiQ=CgF12bwn$+q;`O^iG1=9qb?8MsYj_c~&TOh&7|p{`Wl}zOR0vpD z%YFKBoSu-LRNzi>eUG04^;0wzm(FkfD(i2;dwx)J$#B&&K=06A|n>NG9vzp`ll5ODE zC;e1mZ#B>w$?d9>DnnK3%d=6`6|mDJs=aWs69P2k4$!Tv$8)XSr+pvP0>$;%!Zn(B zC=^pBnNfOR;_>|%bB+#pM{D!3mrQ}g+xOE`){lT7=V;|PDFA=J8W!DfoXov+$-mrK z|EUCA(YvZ|c18cE7O)&zwx9jKxgg&0KcxlmF|rY#t_!458w;f6rppio=M;%EDgtS@ zl~svOy8A@9$PQwPnVoo<76dxd(nPPjE)m!*K!jWqNE7?9jj*^ZPtdS$C#F@85L{OT z(v~?It!%{krTE+Yk8s+MX`hFo4zyC8Cjt!x(i9C?iKDCU6JihF!7}y(PbnzAv;g75 zY(zufC|tBwCwjns2DYyO|!$wP8J@HY&>G!Fk?{n_1tzm5N-kj2jj{fhrH{ji(7 Date: Tue, 19 May 2026 12:40:49 +0100 Subject: [PATCH 110/115] add theming, csv exports and plots md --- RECLAMM_PLOTS.md | 143 ++++++++++++++++++++++++++ scripts/plot_reclamm_optuna_result.py | 126 +++++++++++++++++------ scripts/run_final_sims.py | 88 +++++++++++++++- 3 files changed, 325 insertions(+), 32 deletions(-) create mode 100644 RECLAMM_PLOTS.md diff --git a/RECLAMM_PLOTS.md b/RECLAMM_PLOTS.md new file mode 100644 index 00000000..1339fc4d --- /dev/null +++ b/RECLAMM_PLOTS.md @@ -0,0 +1,143 @@ +# Reproducing the AAVE/ETH and COW/ETH plots + +This branch (`noise-modelling-fixes-data`) carries the minimum data needed to +produce the final train/test panels and PR-sweep heatmaps for both pairs +without re-running the training sweep. + +## What's on the branch + +- `results/mm_noise/model.npz` + `meta.json` — the frozen MM noise model. +- `results/competitor_tvl/competitor_tvl.npz` — DeFi Llama competitor TVL. +- `results/full_sweep/<6 files>` — sweep summaries for the 6 winning configs. +- `results/run_<6 hashes>.json` — trial trajectories for the same 6 winners + (read by `scripts/run_final_sims.py`). + +What's **not** included and must be pulled locally: + +- Binance minute parquets for `BTC`, `ETH`, `AAVE`, `COW`. + +## Prerequisites + +Conda env per the README: + +``` +conda activate qsim +``` + +## 1. Pull Binance price data + +``` +python scripts/download_data.py BTC ETH AAVE COW +``` + +Writes `quantammsim/data/_USD.parquet` per token. `BTC` is required by +the MM noise model's market features; `AAVE`/`COW`/`ETH` are the pair tokens +the simulator reads prices from. + +## 2. Final-sims: train/test forward passes and selection + +``` +python scripts/run_final_sims.py --all +``` + +Loads the 6 committed `run_.json` files, picks the single candidate per +`(pair, TVL)` tier, runs train and test forward passes at the picked params, +and writes: + +- `results/final_sims/{aave,cow}_{train,test}.png` — share price, fee revenue, + cumulative volume per TVL tier. +- `results/final_sims/{aave,cow}_{train,test}_weights.png` — effective weight + trajectories. +- `results/final_sims/{aave,cow}_sim_results.pkl` — per-tier results cache. + +Per-tier picks for this branch's data (printed to stdout as `Params:` lines): + +| Tier | PR | margin | shift | +|-----------|---------|--------|--------| +| AAVE 1m | 1.068 | 0.0240 | 0.0225 | +| AAVE 5m | 1.296 | 0.0101 | 0.0675 | +| AAVE 20m | 1.348 | 0.0103 | 0.5411 | +| COW 500k | 99.665 | 0.8235 | 0.1418 | +| COW 2m | 54.549 | 0.8351 | 0.1422 | +| COW 20m | 172.463 | 0.9265 | 0.0521 | + +## 3. PR-sweep heatmaps + +One `run_pr_sweep.py` invocation per (pair, TVL), with that tier's +`margin`, `shift`, `initial-pool-value`, and the selected PR as a marker. + +### AAVE/ETH + +Override the pair-specific flags (`run_pr_sweep.py` defaults are COW). + +``` +mkdir -p results/final_sims/for_fabio/aave/price_ratio_sweep + +python scripts/run_pr_sweep.py --tokens AAVE ETH --pool-id 0x9d1fcf346ea1b0 \ + --gas-cost 1.0 --fees 0.0025 \ + --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/aave/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 4.0 5.0 7.5 10.0 \ + --margin 0.0240 --shift 0.0225 --initial-pool-value 1000000 --selected-pr 1.068 + +python scripts/run_pr_sweep.py --tokens AAVE ETH --pool-id 0x9d1fcf346ea1b0 \ + --gas-cost 1.0 --fees 0.0025 \ + --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/aave/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 4.0 5.0 7.5 10.0 \ + --margin 0.0101 --shift 0.0675 --initial-pool-value 5000000 --selected-pr 1.296 + +python scripts/run_pr_sweep.py --tokens AAVE ETH --pool-id 0x9d1fcf346ea1b0 \ + --gas-cost 1.0 --fees 0.0025 \ + --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/aave/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 4.0 5.0 7.5 10.0 \ + --margin 0.0103 --shift 0.5411 --initial-pool-value 20000000 --selected-pr 1.348 +``` + +### COW/ETH + +The COW selections are at PR ≈ 50–200, so the `--prs` list extends past 10. +The script's defaults already match the COW pair, so the pair-specific flags +can be omitted. + +``` +mkdir -p results/final_sims/for_fabio/cow/price_ratio_sweep + +python scripts/run_pr_sweep.py --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/cow/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 5.0 10.0 30.0 50.0 80.0 100.0 120.0 150.0 200.0 \ + --margin 0.8235 --shift 0.1418 --initial-pool-value 500000 --selected-pr 99.665 + +python scripts/run_pr_sweep.py --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/cow/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 5.0 10.0 30.0 50.0 80.0 100.0 120.0 150.0 200.0 \ + --margin 0.8351 --shift 0.1422 --initial-pool-value 2000000 --selected-pr 54.549 + +python scripts/run_pr_sweep.py --multi-period --period-months 3 --onchain-pr 2.02 \ + --output-dir results/final_sims/for_fabio/cow/price_ratio_sweep \ + --prs 1.01 1.1 1.2 1.4 1.6 1.8 2.0 2.5 3.0 5.0 10.0 30.0 50.0 80.0 100.0 120.0 150.0 200.0 \ + --margin 0.9265 --shift 0.0521 --initial-pool-value 20000000 --selected-pr 172.463 +``` + +### Outputs + +Per-period single-config PNGs and one heatmap per invocation, named +`pr_heatmap__3mo_m_s.png`, under +`results/final_sims/for_fabio/{aave,cow}/price_ratio_sweep/`. + +## Gotchas + +- **zsh and shell variables**. `python ... $PRS_ARGS` with + `PRS_ARGS="--prs 1.01 1.1 ..."` does not word-split in zsh by default; + argparse receives the whole string as one token and emits + `unrecognized arguments: --prs ...`. Either inline the list (as above) + or use `${=PRS_ARGS}`. +- **COW early-period failures**. The COW pool's data starts around 2024-12. + `run_pr_sweep.py --multi-period` tries each period from `2024-01-01` + onwards; the early periods raise inside `run_single_period`, the script + catches the exception and the final heatmap only contains the surviving + rows. Expected. +- **Re-running with the same `(start, end)` reuses the cached noise array** + in `results/mm_noise/_sim_arrays/___mm.npz` — + rebuilding only the first time each period is touched. diff --git a/scripts/plot_reclamm_optuna_result.py b/scripts/plot_reclamm_optuna_result.py index c9bf96ca..9a0862c8 100644 --- a/scripts/plot_reclamm_optuna_result.py +++ b/scripts/plot_reclamm_optuna_result.py @@ -19,6 +19,7 @@ import argparse import json +import os import sys import jax.numpy as jnp @@ -39,14 +40,56 @@ "price_ratio": 4.0, "centeredness_margin": 0.1, "shift_exponent": 0.001, } -BG = "#162536" -TEXT_COLOR = "#E6CE97" -# Extended palette for multi-file comparison -COLORS = [ - "#3498db", "#2ecc71", "#e74c3c", "#f39c12", "#9b59b6", - "#1abc9c", "#e67e22", "#2980b9", "#c0392b", "#8e44ad", - "#27ae60", "#d35400", "#16a085", "#f1c40f", "#7f8c8d", -] +THEME_ORDER = ("light", "dark") +THEMES = { + "light": { + "bg": "#F7FAFC", + "text": "#171923", + "subtle_text": "#4A5568", + "reference": "#112055", + "grid": "#4F5764", + "colors": [ + "#2048e9", "#008361", "#b43821", "#c2410c", "#5c38c9", + "#6D4F2C", "#457dff", "#00a474", "#d7462b", "#ea580c", + "#7f6ae8", "#92693A", "#183bbb", "#00674e", "#718096", + ], + }, + "dark": { + "bg": "#171923", + "text": "#EDF2F7", + "subtle_text": "#CBD5E0", + "reference": "#F7FAFC", + "grid": "#4F5764", + "colors": [ + "#457dff", "#00d395", "#ea6249", "#f97316", "#7f6ae8", + "#B68449", "#2554ff", "#00a474", "#d7462b", "#fb923c", + "#6c4add", "#92693A", "#2048e9", "#008361", "#A0AEC0", + ], + }, +} +BG = THEMES["dark"]["bg"] +TEXT_COLOR = THEMES["dark"]["text"] +SUBTLE_TEXT_COLOR = THEMES["dark"]["subtle_text"] +REFERENCE_COLOR = THEMES["dark"]["reference"] +GRID_COLOR = THEMES["dark"]["grid"] +COLORS = THEMES["dark"]["colors"] + + +def _apply_theme(theme_name): + """Apply one of the chart themes derived from colors.ts.""" + global BG, TEXT_COLOR, SUBTLE_TEXT_COLOR, REFERENCE_COLOR, GRID_COLOR, COLORS + theme = THEMES[theme_name] + BG = theme["bg"] + TEXT_COLOR = theme["text"] + SUBTLE_TEXT_COLOR = theme["subtle_text"] + REFERENCE_COLOR = theme["reference"] + GRID_COLOR = theme["grid"] + COLORS = theme["colors"] + + +def _themed_output_path(output, theme_name): + root, ext = os.path.splitext(output) + return f"{root}_{theme_name}{ext or '.png'}" # Short labels for objectives _OBJ_SHORT = { @@ -159,7 +202,7 @@ def run_full_period(params, config, fees_override=None): return do_run_on_historic_data(run_fingerprint=fp, params=jax_params) -def plot_results(configs, time_series, hodl_values, ref_config, args): +def _plot_results_once(configs, time_series, hodl_values, ref_config, args, theme_name): """Two-panel plot: value-over-time + cumulative fee revenue.""" train_end_str = ref_config["endDateString"] train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") @@ -217,22 +260,22 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): hodl_daily = hodl_values[::step] * val_scale ax_val.plot(dates_daily[:len(hodl_daily)], hodl_daily, linewidth=2, - color="white", alpha=0.7, linestyle="--", label="HODL") + color=REFERENCE_COLOR, alpha=0.7, linestyle="--", label="HODL") if train_end_dt > start_dt and train_end_dt < dates[-1]: - ax_val.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ax_val.axvline(x=train_end_dt, color=REFERENCE_COLOR, linestyle=":", alpha=0.5, linewidth=1.5) ylims = ax_val.get_ylim() ax_val.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", - color="white", alpha=0.6, fontsize=11, ha="right", va="top") + color=SUBTLE_TEXT_COLOR, alpha=0.8, fontsize=11, ha="right", va="top") ax_val.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", - color="white", alpha=0.6, fontsize=11, ha="left", va="top") + color=SUBTLE_TEXT_COLOR, alpha=0.8, fontsize=11, ha="left", va="top") _style_axis(ax_val) ax_val.set_ylabel(val_ylabel, color=TEXT_COLOR, fontsize=12) tokens_str = "/".join(ref_config["tokens"]) date_range_str = f"{start_dt.strftime('%b %Y')} — {dates[-1].strftime('%b %Y')}" ax_val.set_title( - f"reCLAMM {tokens_str} — {date_range_str}", + f"Auto-range {tokens_str} — {date_range_str}", color=TEXT_COLOR, fontsize=13, pad=15, ) ax_val.legend(loc="upper left", fontsize=8, facecolor=BG, @@ -255,7 +298,7 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): zorder=3 if is_optimized else 2) if train_end_dt > start_dt and train_end_dt < dates[-1]: - ax_fee.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ax_fee.axvline(x=train_end_dt, color=REFERENCE_COLOR, linestyle=":", alpha=0.5, linewidth=1.5) _style_axis(ax_fee) ax_fee.set_ylabel(fee_ylabel, color=TEXT_COLOR, fontsize=12) ax_fee.legend(loc="upper left", fontsize=8, facecolor=BG, @@ -294,7 +337,7 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): zorder=3 if is_optimized else 2) if train_end_dt > start_dt and train_end_dt < dates[-1]: - ax_vol.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ax_vol.axvline(x=train_end_dt, color=REFERENCE_COLOR, linestyle=":", alpha=0.5, linewidth=1.5) _style_axis(ax_vol) ax_vol.set_ylabel(vol_ylabel, color=TEXT_COLOR, fontsize=12) ax_vol.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) @@ -306,13 +349,22 @@ def plot_results(configs, time_series, hodl_values, ref_config, args): fig.patch.set_facecolor(BG) plt.tight_layout() - output = args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png" + output = _themed_output_path( + args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png", + theme_name, + ) plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) print(f"\nSaved plot to {output}") plt.close() -def plot_test_only(configs, time_series, hodl_values, ref_config, args): +def plot_results(configs, time_series, hodl_values, ref_config, args): + for theme_name in THEME_ORDER: + _apply_theme(theme_name) + _plot_results_once(configs, time_series, hodl_values, ref_config, args, theme_name) + + +def _plot_test_only_once(configs, time_series, hodl_values, ref_config, args, theme_name): """Test-period plot with all curves normalised to start at 1.0.""" train_end_str = ref_config["endDateString"] train_end_dt = datetime.strptime(train_end_str, "%Y-%m-%d %H:%M:%S") @@ -348,9 +400,9 @@ def plot_test_only(configs, time_series, hodl_values, ref_config, args): if len(hodl_test) > 0: hodl_norm = hodl_test / hodl_test[0] ax.plot(test_dates[:len(hodl_norm)], hodl_norm, linewidth=2, - color="white", alpha=0.7, linestyle="--", label="HODL") + color=REFERENCE_COLOR, alpha=0.7, linestyle="--", label="HODL") - ax.axhline(1.0, color="white", linestyle=":", alpha=0.3, linewidth=1) + ax.axhline(1.0, color=REFERENCE_COLOR, linestyle=":", alpha=0.3, linewidth=1) _style_axis(ax) tokens_str = "/".join(ref_config["tokens"]) ax.set_title(f"Test Period Only (normalised) — {tokens_str}", @@ -363,13 +415,19 @@ def plot_test_only(configs, time_series, hodl_values, ref_config, args): fig.patch.set_facecolor(BG) plt.tight_layout() base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") - output = base.replace(".png", "_test_only.png") + output = _themed_output_path(base.replace(".png", "_test_only.png"), theme_name) plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) print(f"Saved plot to {output}") plt.close() -def plot_weights(configs, time_series, ref_config, args): +def plot_test_only(configs, time_series, hodl_values, ref_config, args): + for theme_name in THEME_ORDER: + _apply_theme(theme_name) + _plot_test_only_once(configs, time_series, hodl_values, ref_config, args, theme_name) + + +def _plot_weights_once(configs, time_series, ref_config, args, theme_name): """Effective weight (value fraction) of token 0 over time.""" start_dt = datetime.strptime(ref_config["startDateString"], "%Y-%m-%d %H:%M:%S") train_end_dt = datetime.strptime(ref_config["endDateString"], "%Y-%m-%d %H:%M:%S") @@ -395,19 +453,19 @@ def plot_weights(configs, time_series, ref_config, args): alpha=0.9 if is_optimized else 0.7, zorder=3 if is_optimized else 2) - ax.axhline(0.5, color="white", linestyle="--", alpha=0.3, linewidth=1) + ax.axhline(0.5, color=REFERENCE_COLOR, linestyle="--", alpha=0.3, linewidth=1) if train_end_dt > start_dt and train_end_dt < dates[-1]: - ax.axvline(x=train_end_dt, color="white", linestyle=":", alpha=0.5, linewidth=1.5) + ax.axvline(x=train_end_dt, color=REFERENCE_COLOR, linestyle=":", alpha=0.5, linewidth=1.5) ylims = ax.get_ylim() ax.text(train_end_dt - pd.Timedelta(days=5), ylims[1] * 0.97, "Train", - color="white", alpha=0.6, fontsize=11, ha="right", va="top") + color=SUBTLE_TEXT_COLOR, alpha=0.8, fontsize=11, ha="right", va="top") ax.text(train_end_dt + pd.Timedelta(days=5), ylims[1] * 0.97, "Test", - color="white", alpha=0.6, fontsize=11, ha="left", va="top") + color=SUBTLE_TEXT_COLOR, alpha=0.8, fontsize=11, ha="left", va="top") _style_axis(ax) tokens_str = "/".join(ref_config["tokens"]) date_range_str = f"{start_dt.strftime('%b %Y')} — {dates[-1].strftime('%b %Y')}" - ax.set_title(f"Effective {token_name} Weight — reCLAMM {tokens_str} — {date_range_str}", + ax.set_title(f"Effective {token_name} Weight — Auto-range {tokens_str} — {date_range_str}", color=TEXT_COLOR, fontsize=13, pad=15) ax.set_ylabel(f"{token_name} weight (value fraction)", color=TEXT_COLOR, fontsize=12) ax.set_xlabel("Date", color=TEXT_COLOR, fontsize=12) @@ -417,21 +475,27 @@ def plot_weights(configs, time_series, ref_config, args): fig.patch.set_facecolor(BG) plt.tight_layout() base = (args.output or f"reclamm_optuna_{tokens_str.replace('/', '_')}.png") - output = base.replace(".png", "_weights.png") + output = _themed_output_path(base.replace(".png", "_weights.png"), theme_name) plt.savefig(output, dpi=200, bbox_inches="tight", facecolor=BG) print(f"Saved plot to {output}") plt.close() +def plot_weights(configs, time_series, ref_config, args): + for theme_name in THEME_ORDER: + _apply_theme(theme_name) + _plot_weights_once(configs, time_series, ref_config, args, theme_name) + + def _style_axis(ax): ax.set_facecolor(BG) ax.tick_params(colors=TEXT_COLOR) for spine in ax.spines.values(): - spine.set_color(TEXT_COLOR) - spine.set_alpha(0.3) + spine.set_color(GRID_COLOR) + spine.set_alpha(0.8) ax.spines["top"].set_visible(False) ax.spines["right"].set_visible(False) - ax.grid(True, alpha=0.15, color=TEXT_COLOR) + ax.grid(True, alpha=0.35, color=GRID_COLOR) def main(): diff --git a/scripts/run_final_sims.py b/scripts/run_final_sims.py index 7f3b6119..3bd09e4f 100644 --- a/scripts/run_final_sims.py +++ b/scripts/run_final_sims.py @@ -24,6 +24,7 @@ matplotlib.use("Agg") import matplotlib.pyplot as plt import numpy as np +import pandas as pd from datetime import datetime from quantammsim.runners.jax_runners import do_run_on_historic_data @@ -164,6 +165,78 @@ def run_forward(pair_cfg, tvl, params, start, end, noise_path): return result +def _trial_hash(best): + """Extract the originating run hash from a selected trial.""" + study_id = str(best.get("study_id", "unknown")) + return study_id.removeprefix("run_") + + +def _unix_values_for_result(result): + """Return the unix timestamp vector aligned to result['value'].""" + value_len = len(np.asarray(result["value"])) + if "unix_values" in result: + unix_values = np.asarray(result["unix_values"]) + else: + data_dict = result.get("data_dict") + if data_dict is None or "unix_values" not in data_dict: + raise KeyError("Forward result has no unix_values or data_dict['unix_values']") + start_idx = int(data_dict.get("start_idx", 0)) + unix_values = np.asarray(data_dict["unix_values"])[start_idx:start_idx + value_len] + + if len(unix_values) != value_len: + raise ValueError( + f"Timestamp/value length mismatch: unix={len(unix_values)} value={value_len}" + ) + return unix_values.astype(np.int64) + + +def export_forward_csvs(result, run_fingerprint, output_dir, identifier, source_hash): + """Write value, reserves, and per-token value CSVs for one forward pass.""" + value = np.asarray(result["value"], dtype=np.float64) + reserves = np.asarray(result["reserves"], dtype=np.float64) + prices = np.asarray(result["prices"], dtype=np.float64) + unix_values = _unix_values_for_result(result) + + tokens = list(run_fingerprint["tokens"]) + if tokens != sorted(tokens): + raise ValueError( + "Final-sim CSV export assumes run_fingerprint['tokens'] is alphabetically " + f"ordered to match runner price/reserve arrays; got {tokens}" + ) + + if reserves.shape != prices.shape: + raise ValueError(f"Reserves/prices shape mismatch: {reserves.shape} vs {prices.shape}") + if reserves.shape[0] != value.shape[0]: + raise ValueError(f"Reserves/value length mismatch: {reserves.shape[0]} vs {value.shape[0]}") + + token_values = reserves * prices + value_from_tokens = token_values.sum(axis=1) + if not np.allclose(value_from_tokens, value, rtol=1e-8, atol=1e-6): + max_diff = float(np.max(np.abs(value_from_tokens - value))) + raise ValueError( + f"Token value sanity check failed for {identifier}: max_diff={max_diff:.6g}" + ) + + output_path = Path(output_dir) + output_path.mkdir(parents=True, exist_ok=True) + file_stem = f"{identifier}_{source_hash}" + + pd.DataFrame({"unix": unix_values, "value": value}).to_csv( + output_path / f"run_Value_{file_stem}.csv", index=False, + ) + + reserve_df = pd.DataFrame({"unix": unix_values}) + for idx, token in enumerate(tokens): + reserve_df[f"reserve_{token}"] = reserves[:, idx] + reserve_df.to_csv(output_path / f"run_Reserves_{file_stem}.csv", index=False) + + token_value_df = pd.DataFrame({"unix": unix_values}) + for idx, token in enumerate(tokens): + token_value_df[f"{token}_value"] = token_values[:, idx] + token_value_df.to_csv(output_path / f"run_TokenValues_{file_stem}.csv", index=False) + print(f" Saved CSVs: run_{{Value,Reserves,TokenValues}}_{file_stem}.csv") + + def make_plot_data(all_results, pair_cfg, start, end): """Convert our results into the format expected by plot_reclamm_optuna_result functions. @@ -212,7 +285,7 @@ def make_plot_data(all_results, pair_cfg, start, end): pr = float(jnp.asarray(p.get("price_ratio", 0)).flatten()[0]) margin = float(jnp.asarray(p.get("centeredness_margin", 0)).flatten()[0]) shift = float(jnp.asarray(p.get("shift_exponent", 0)).flatten()[0]) - name = f"reCLAMM ${tvl_label} (PR={pr:.2f}, m={margin:.2f}, s={shift:.3f})" + name = f"Auto-range ${tvl_label} (PR={pr:.2f}, m={margin:.2f}, s={shift:.3f})" configs[name] = { "tvl_label": tvl_label, "params": p, @@ -269,6 +342,13 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): # Train period forward pass print(f" Running train period...") train_result = run_forward(pair_cfg, tvl, params, TRAIN_START, TRAIN_END, train_noise) + source_hash = _trial_hash(best) + train_fp = build_fingerprint(pair_cfg, tvl, TRAIN_START, TRAIN_END, train_noise) + export_forward_csvs( + train_result, train_fp, output_dir, + identifier=f"{pair_name}_{tvl_label}_train", + source_hash=source_hash, + ) train_results[tvl_label] = {"result": train_result, "params": params, "best": best} # Clear JIT caches between runs to manage memory @@ -277,6 +357,12 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): # Test period forward pass print(f" Running test period...") test_result = run_forward(pair_cfg, tvl, params, TEST_START, TEST_END, test_noise) + test_fp = build_fingerprint(pair_cfg, tvl, TEST_START, TEST_END, test_noise) + export_forward_csvs( + test_result, test_fp, output_dir, + identifier=f"{pair_name}_{tvl_label}_test", + source_hash=source_hash, + ) test_results[tvl_label] = {"result": test_result, "params": params, "best": best} jax.clear_caches() From 40c6bdd0e600a1ff3da7e0b1f63f09d86665014d Mon Sep 17 00:00:00 2001 From: MatthewWilletts <16085750+MatthewWilletts@users.noreply.github.com> Date: Tue, 19 May 2026 15:05:40 +0100 Subject: [PATCH 111/115] add scripts/run_pr_sweep.py The PR-sweep entry point referenced by the training guide (RECLAMM_TRAINING.md) and the plot-reproduction runbook (RECLAMM_PLOTS.md, added in dd20627). --- scripts/run_pr_sweep.py | 505 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 505 insertions(+) create mode 100644 scripts/run_pr_sweep.py diff --git a/scripts/run_pr_sweep.py b/scripts/run_pr_sweep.py new file mode 100644 index 00000000..b737e4ec --- /dev/null +++ b/scripts/run_pr_sweep.py @@ -0,0 +1,505 @@ +#!/usr/bin/env python3 +"""Price Ratio sweep for reClAMM pools. + +Sweeps PR with fixed margin/shift over a single run period. +Plots RoH, fee revenue, and final value. +Optionally runs across multiple periods and produces a summary heatmap. + +Usage: + # Single period + python scripts/run_pr_sweep.py --tokens COW ETH --start 2025-01-01 --end 2026-01-01 + + # Multi-period with heatmap + python scripts/run_pr_sweep.py --tokens COW ETH --multi-period +""" + +import argparse +import json +import os + +import jax.numpy as jnp +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +import numpy as np +from datetime import datetime + +from quantammsim.runners.jax_runners import do_run_on_historic_data +from quantammsim.pools.reCLAMM.reclamm_reserves import set_blessed_arb + + +BG = "#162536" +TC = "#E6CE97" + +DEFAULT_PRS = [1.01, 1.1, 1.2, 1.4, 1.6, 1.8, 2.0, 2.2, 2.5, 3.0, 3.5, 4.0, 5.0, 6.0, 7.5, 10.0] + + +def _style_ax(ax): + ax.set_facecolor(BG) + ax.tick_params(colors=TC) + for s in ax.spines.values(): + s.set_color(TC) + s.set_alpha(0.3) + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.grid(True, alpha=0.15, color=TC) + + +def build_noise_arrays(pool_id, tokens, start, end): + """Build or load cached MM noise arrays.""" + from quantammsim.calibration.noise_model_arrays import build_mm_simulator_arrays + + cache_dir = os.path.join("results", "mm_noise", "_sim_arrays") + os.makedirs(cache_dir, exist_ok=True) + arrays_path = os.path.join(cache_dir, f"{pool_id}_{start}_{end}_mm.npz") + + if not os.path.exists(arrays_path): + print(f" Building noise arrays for {pool_id} {start} -> {end}...") + arrays = build_mm_simulator_arrays( + token_a=tokens[0], token_b=tokens[1], + start_date=start, end_date=end, + mm_artifact_dir="results/mm_noise", + competitor_tvl_path="results/competitor_tvl/competitor_tvl.npz", + pool_id=pool_id, + ) + np.savez(arrays_path, + noise_base=arrays["noise_base"], + competitor_tvl=arrays["competitor_tvl"]) + return arrays_path + + +def run_single_period(args, start, end): + """Run PR sweep for a single period. Returns dict of metrics per PR.""" + tokens = args.tokens + margin = args.margin + shift = args.shift + price_ratios = np.array(args.prs or DEFAULT_PRS) + + # Noise arrays + noise_path = None + if args.noise_model == "mm_observed" and args.pool_id: + noise_path = build_noise_arrays(args.pool_id, tokens, start, end) + + fp = { + "rule": "reclamm", + "tokens": tokens, + "startDateString": f"{start} 00:00:00", + "endDateString": f"{end} 00:00:00", + "initial_pool_value": args.initial_pool_value, + "do_arb": True, + "arb_frequency": args.arb_frequency, + "fees": args.fees, + "gas_cost": args.gas_cost, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + "reclamm_interpolation_method": "geometric", + "reclamm_centeredness_scaling": False, + "reclamm_use_shift_exponent": True, + } + if noise_path: + fp["noise_model"] = "mm_observed" + fp["noise_arrays_path"] = noise_path + else: + fp["noise_trader_ratio"] = 0.0 + + params_list = [ + { + "price_ratio": jnp.array(float(pr)), + "centeredness_margin": jnp.array(margin), + "shift_exponent": jnp.array(shift), + } + for pr in price_ratios + ] + + tok_str = "/".join(tokens) + print(f" PR sweep: {tok_str}, {start} -> {end}, {len(price_ratios)} PRs") + + results = do_run_on_historic_data( + run_fingerprint=fp, params=params_list, verbose=False, + ) + + # Compute metrics + rohs, fee_revs, final_vals = [], [], [] + hodl_final = None + + for i, pr in enumerate(price_ratios): + out = results[i] + val = np.array(out["value"]) + prices = np.array(out["prices"]) + reserves_0 = out["reserves"][0] + hodl = np.sum(np.array(reserves_0) * prices, axis=1) + if hodl_final is None: + hodl_final = float(hodl[-1]) + + roh = val[-1] / hodl[-1] - 1 + fr = np.array(out.get("fee_revenue", np.zeros(len(val)))) + + rohs.append(roh) + fee_revs.append(float(fr.sum())) + final_vals.append(float(val[-1])) + + # Run a base Balancer 50/50 pool as comparator (arb-only, same fees/gas) + bal_fp = { + "rule": "balancer", + "tokens": tokens, + "startDateString": f"{start} 00:00:00", + "endDateString": f"{end} 00:00:00", + "initial_pool_value": args.initial_pool_value, + "do_arb": True, + "arb_frequency": args.arb_frequency, + "fees": args.fees, + "gas_cost": args.gas_cost, + "arb_fees": 0.0, + "protocol_fee_split": 0.25, + "noise_trader_ratio": 0.0, + } + bal_params = {"initial_weights_logits": jnp.array([0.0, 0.0])} + print(f" Running Balancer 50/50 comparator...") + bal_result = do_run_on_historic_data( + run_fingerprint=bal_fp, params=bal_params, verbose=False, + ) + bal_val = np.array(bal_result["value"]) + bal_hodl = np.sum(np.array(bal_result["reserves"][0]) * np.array(bal_result["prices"]), axis=1) + bal_roh = bal_val[-1] / bal_hodl[-1] - 1 + bal_fee = float(np.array(bal_result.get("fee_revenue", np.zeros(1))).sum()) + print(f" Balancer 50/50: RoH={bal_roh:+.2%}, fee=${bal_fee:,.0f}") + + return { + "price_ratios": price_ratios, + "rohs": rohs, + "fee_revs": fee_revs, + "final_vals": final_vals, + "hodl_final": hodl_final, + "balancer_roh": bal_roh, + "balancer_fee": bal_fee, + "balancer_final": float(bal_val[-1]), + "start": start, + "end": end, + } + + +def plot_single_period(data, args, outpath): + """3-panel plot for a single period.""" + price_ratios = data["price_ratios"] + rohs = data["rohs"] + fee_revs = data["fee_revs"] + final_vals = data["final_vals"] + hodl_final = data["hodl_final"] + start, end = data["start"], data["end"] + onchain_pr = args.onchain_pr + tok_str = "/".join(args.tokens) + + fig, axes = plt.subplots(1, 2, figsize=(14, 5)) + for ax in axes: + _style_ax(ax) + + # Panel 1: RoH vs PR + axes[0].plot(price_ratios, [r * 100 for r in rohs], + "o-", color="#e74c3c", markersize=5, label="reClAMM") + axes[0].axhline(0, color="white", ls=":", alpha=0.3) + bal_roh = data.get("balancer_roh") + if bal_roh is not None: + axes[0].axhline(bal_roh * 100, color="#3498db", ls="-", alpha=0.7, + label=f"Balancer 50/50 ({bal_roh*100:+.1f}%)") + if onchain_pr: + axes[0].axvline(onchain_pr, color="#f39c12", ls="--", alpha=0.7, + label=f"On-chain PR={onchain_pr}") + selected_pr = getattr(args, "selected_pr", None) + if selected_pr: + axes[0].axvline(selected_pr, color="#9b59b6", ls="-.", alpha=0.8, linewidth=2, + label=f"Selected PR={selected_pr:.2f}") + best_pr_idx = int(np.argmax(rohs)) + best_pr = price_ratios[best_pr_idx] + axes[0].axvline(best_pr, color="#2ecc71", ls="-", alpha=0.8, linewidth=2, + label=f"Best PR={best_pr:.1f} ({rohs[best_pr_idx]*100:+.1f}%)") + axes[0].set_xlabel("Price Ratio", color=TC) + axes[0].set_ylabel("Returns over HODL (%)", color=TC) + axes[0].set_title("RoH vs Price Ratio", color=TC) + axes[0].legend(fontsize=8, facecolor=BG, edgecolor=TC, labelcolor=TC) + + # Panel 2: Fee revenue + axes[1].plot(price_ratios, [f / 1000 for f in fee_revs], + "o-", color="#2ecc71", markersize=5) + if onchain_pr: + axes[1].axvline(onchain_pr, color="#f39c12", ls="--", alpha=0.7) + if selected_pr: + axes[1].axvline(selected_pr, color="#9b59b6", ls="-.", alpha=0.8, linewidth=2) + axes[1].set_xlabel("Price Ratio", color=TC) + axes[1].set_ylabel("Fee Revenue ($K)", color=TC) + axes[1].set_title("Cumulative Fee Revenue", color=TC) + + fig.suptitle( + f"{tok_str} reCLAMM PR Sweep (margin={args.margin}, shift={args.shift}, " + f"gas=${args.gas_cost})\n{start} -> {end}", + color=TC, fontsize=12, fontweight="bold", + ) + fig.patch.set_facecolor(BG) + plt.tight_layout() + fig.savefig(outpath, dpi=200, bbox_inches="tight", facecolor=BG) + plt.close() + print(f" Saved: {outpath}") + + +def plot_heatmap(all_data, args, outpath): + """Heatmap: RoH as function of PR (x) and start date (y).""" + tok_str = "/".join(args.tokens) + price_ratios = all_data[0]["price_ratios"] + n_pr = len(price_ratios) + n_periods = len(all_data) + + roh_matrix = np.zeros((n_periods, n_pr)) + fee_matrix = np.zeros((n_periods, n_pr)) + period_labels = [] + + for j, data in enumerate(all_data): + roh_matrix[j, :] = data["rohs"] + fee_matrix[j, :] = data["fee_revs"] + period_labels.append(f"{data['start']} -> {data['end']}") + + fig, axes = plt.subplots(1, 2, figsize=(18, max(4, n_periods * 0.6 + 2))) + for ax in axes: + ax.set_facecolor(BG) + ax.tick_params(colors=TC) + + # RoH heatmap + vmax = max(abs(roh_matrix.min()), abs(roh_matrix.max())) + im0 = axes[0].imshow( + roh_matrix * 100, aspect="auto", cmap="RdYlGn", vmin=-vmax * 100, vmax=vmax * 100, + ) + axes[0].set_xticks(range(n_pr)) + axes[0].set_xticklabels([f"{p:.1f}" if p < 10 else f"{p:.0f}" for p in price_ratios], + rotation=45, fontsize=8, color=TC) + axes[0].set_yticks(range(n_periods)) + axes[0].set_yticklabels(period_labels, fontsize=8, color=TC) + axes[0].set_xlabel("Price Ratio", color=TC) + axes[0].set_title("Returns over HODL (%)", color=TC, fontsize=12) + cb0 = fig.colorbar(im0, ax=axes[0], shrink=0.8) + cb0.ax.tick_params(colors=TC) + + # Annotate cells + for j in range(n_periods): + for i in range(n_pr): + val = roh_matrix[j, i] * 100 + color = "black" if abs(val) < vmax * 50 else "white" + axes[0].text(i, j, f"{val:+.1f}", ha="center", va="center", + fontsize=6, color=color) + + # Mark on-chain PR with white cross-hatching + if args.onchain_pr: + pr_arr = np.array(price_ratios) + nearest = np.argmin(np.abs(np.log(pr_arr) - np.log(args.onchain_pr))) + for j in range(n_periods): + axes[0].add_patch(plt.Rectangle( + (nearest - 0.5, j - 0.5), 1, 1, + fill=False, edgecolor="white", linewidth=2, + hatch="//", label="On-chain PR" if j == 0 else None, + )) + + # Mark best PR per row with green border + for j in range(n_periods): + best_i = int(np.argmax(roh_matrix[j, :])) + axes[0].add_patch(plt.Rectangle( + (best_i - 0.5, j - 0.5), 1, 1, + fill=False, edgecolor="#2ecc71", linewidth=3, + label="Best RoH" if j == 0 else None, + )) + + # Best overall PR: two metrics + # 1) Geometric mean = product of (1+RoH) — penalises variance, correct for compounding + compounded = np.prod(1 + roh_matrix, axis=0) + geo_mean_roh = np.sign(compounded) * np.abs(compounded) ** (1.0 / n_periods) - 1 + best_geo_i = int(np.argmax(geo_mean_roh)) + # 2) Median — robust to outlier months + median_roh = np.median(roh_matrix, axis=0) + best_median_i = int(np.argmax(median_roh)) + + # Print both for diagnostics + print(f" Best geo-mean PR={price_ratios[best_geo_i]:.2f} " + f"(geo mean {geo_mean_roh[best_geo_i]*100:+.3f}%)") + print(f" Best median PR={price_ratios[best_median_i]:.2f} " + f"(median {median_roh[best_median_i]*100:+.3f}%)") + for i, pr in enumerate(price_ratios): + print(f" PR={pr:5.2f}: geo={geo_mean_roh[i]*100:+.3f}% " + f"median={median_roh[i]*100:+.3f}%") + + # Use median for the column shading (robust to outlier months) + best_overall_i = best_median_i + best_overall_pr = price_ratios[best_overall_i] + best_median = median_roh[best_overall_i] + for j in range(n_periods): + axes[0].add_patch(plt.Rectangle( + (best_overall_i - 0.5, j - 0.5), 1, 1, + fill=True, facecolor="#3498db", alpha=0.25, edgecolor="#3498db", + linewidth=2, linestyle="--", + label=(f"Best overall PR={best_overall_pr:.2f} " + f"(median {best_median*100:+.2f}%)") if j == 0 else None, + )) + + # Legend for markings + from matplotlib.patches import Patch + legend_elements = [] + if args.onchain_pr: + legend_elements.append(Patch(facecolor="none", edgecolor="white", + hatch="//", label="On-chain PR")) + legend_elements.append(Patch(facecolor="none", edgecolor="#2ecc71", + linewidth=3, label="Best RoH (per period)")) + legend_elements.append(Patch(facecolor="#3498db", alpha=0.25, edgecolor="#3498db", + linestyle="--", linewidth=2, + label=f"Best overall PR={best_overall_pr:.2f} " + f"(median {best_median*100:+.2f}%)")) + axes[0].legend(handles=legend_elements, loc="lower right", fontsize=7, + facecolor=BG, edgecolor=TC, labelcolor=TC) + + # Fee revenue heatmap + im1 = axes[1].imshow( + fee_matrix / 1000, aspect="auto", cmap="YlGn", + ) + axes[1].set_xticks(range(n_pr)) + axes[1].set_xticklabels([f"{p:.1f}" if p < 10 else f"{p:.0f}" for p in price_ratios], + rotation=45, fontsize=8, color=TC) + axes[1].set_yticks(range(n_periods)) + axes[1].set_yticklabels(period_labels, fontsize=8, color=TC) + axes[1].set_xlabel("Price Ratio", color=TC) + axes[1].set_title("Fee Revenue ($K)", color=TC, fontsize=12) + cb1 = fig.colorbar(im1, ax=axes[1], shrink=0.8) + cb1.ax.tick_params(colors=TC) + + for j in range(n_periods): + for i in range(n_pr): + val = fee_matrix[j, i] / 1000 + axes[1].text(i, j, f"{val:.0f}", ha="center", va="center", + fontsize=6, color="black") + + if args.onchain_pr: + for j in range(n_periods): + axes[1].add_patch(plt.Rectangle( + (nearest - 0.5, j - 0.5), 1, 1, + fill=False, edgecolor="white", linewidth=2, + hatch="//", + )) + # Best fee revenue per row + for j in range(n_periods): + best_i = int(np.argmax(fee_matrix[j, :])) + axes[1].add_patch(plt.Rectangle( + (best_i - 0.5, j - 0.5), 1, 1, + fill=False, edgecolor="#2ecc71", linewidth=3, + )) + # Best overall PR for fee revenue: maximise total fee across all periods + total_fees = np.sum(fee_matrix, axis=0) + best_fee_overall_i = int(np.argmax(total_fees)) + best_fee_overall_pr = price_ratios[best_fee_overall_i] + for j in range(n_periods): + axes[1].add_patch(plt.Rectangle( + (best_fee_overall_i - 0.5, j - 0.5), 1, 1, + fill=True, facecolor="#3498db", alpha=0.25, edgecolor="#3498db", + linewidth=2, linestyle="--", + )) + + fig.suptitle( + f"{tok_str} reClAMM: PR x Period Summary (margin={args.margin}, " + f"shift={args.shift}, gas=${args.gas_cost})", + color=TC, fontsize=13, fontweight="bold", + ) + fig.patch.set_facecolor(BG) + plt.tight_layout() + fig.savefig(outpath, dpi=200, bbox_inches="tight", facecolor=BG) + plt.close() + print(f" Saved heatmap: {outpath}") + + +def main(): + p = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + p.add_argument("--tokens", nargs=2, default=["COW", "ETH"]) + p.add_argument("--start", default=None, help="Start date (YYYY-MM-DD)") + p.add_argument("--end", default=None, help="End date (YYYY-MM-DD)") + p.add_argument("--multi-period", action="store_true", + help="Run multiple periods starting at regular intervals") + p.add_argument("--period-months", type=int, default=6, + help="Length of each period in months (default: 6)") + p.add_argument("--margin", type=float, default=0.5) + p.add_argument("--shift", type=float, default=0.1) + p.add_argument("--fees", type=float, default=0.003) + p.add_argument("--gas-cost", type=float, default=3.0) + p.add_argument("--arb-frequency", type=int, default=3) + p.add_argument("--initial-pool-value", type=float, default=600_000.0) + p.add_argument("--pool-id", default="0xd321300ef77067") + p.add_argument("--noise-model", default="mm_observed", + choices=["mm_observed", "none"]) + p.add_argument("--onchain-pr", type=float, default=None, + help="On-chain PR to mark on plot") + p.add_argument("--selected-pr", type=float, default=None, + help="Optimiser-selected PR to mark on plot") + p.add_argument("--prs", type=float, nargs="+", default=None) + p.add_argument("--output-dir", default="results/cow_sweep/pr_sweeps") + args = p.parse_args() + + os.makedirs(args.output_dir, exist_ok=True) + tok_tag = "_".join(args.tokens) + + if args.multi_period: + from dateutil.relativedelta import relativedelta + from datetime import date + period_months = args.period_months + step_months = 1 + starts = [] + d = date(2024, 1, 1) + data_end = date(2026, 4, 1) + while d + relativedelta(months=period_months) <= data_end: + starts.append(d.isoformat()) + d += relativedelta(months=step_months) + periods = [(s, (datetime.strptime(s, "%Y-%m-%d") + + relativedelta(months=period_months)).strftime("%Y-%m-%d")) + for s in starts] + + all_data = [] + for start, end in periods: + print(f"\n{'='*60}") + try: + data = run_single_period(args, start, end) + all_data.append(data) + + outpath = os.path.join( + args.output_dir, + f"pr_sweep_{tok_tag}_{start}_{end}_{args.period_months}mo_m{args.margin}_s{args.shift}.png", + ) + plot_single_period(data, args, outpath) + + # Print table + prs = data["price_ratios"] + print(f" {'PR':>8s} {'RoH':>8s} {'Fee $K':>8s} {'Final $K':>10s}") + for i, pr in enumerate(prs): + print(f" {pr:>8.2f} {data['rohs'][i]:>+7.2%} " + f"{data['fee_revs'][i]/1000:>8.0f} " + f"{data['final_vals'][i]/1000:>10.0f}") + except Exception as e: + print(f" FAILED: {e}") + + if all_data: + heatmap_path = os.path.join( + args.output_dir, + f"pr_heatmap_{tok_tag}_{args.period_months}mo_m{args.margin}_s{args.shift}.png", + ) + plot_heatmap(all_data, args, heatmap_path) + + elif args.start and args.end: + data = run_single_period(args, args.start, args.end) + outpath = os.path.join( + args.output_dir, + f"pr_sweep_{tok_tag}_{args.start}_{args.end}_m{args.margin}_s{args.shift}.png", + ) + plot_single_period(data, args, outpath) + + print(f"\n{'PR':>8s} {'RoH':>8s} {'Fee $K':>8s} {'Final $K':>10s}") + print("-" * 40) + for i, pr in enumerate(data["price_ratios"]): + print(f"{pr:>8.2f} {data['rohs'][i]:>+7.2%} " + f"{data['fee_revs'][i]/1000:>8.0f} " + f"{data['final_vals'][i]/1000:>10.0f}") + else: + print("ERROR: provide --start/--end or --multi-period") + + +if __name__ == "__main__": + main() From 65cb66f692be6ff33c3c699efb012b1be029e3a4 Mon Sep 17 00:00:00 2001 From: mendesfabio Date: Wed, 8 Jul 2026 10:40:09 -0300 Subject: [PATCH 112/115] fix data download for US users and pandas 3 - Skip the BinanceDataDumper exchangeInfo ticker validation: the endpoint is geo-switched to binance.us for US IPs, which hides pairs (e.g. RPLUSDT) that exist in the binance.vision archive but aren't listed on Binance.US, making download_data.py report 'no data' for those tokens. - Convert the reindex grid to milliseconds explicitly in forward_fill_ohlcv_data: int64 cast of a DatetimeIndex is unit-dependent on pandas>=3, which collapsed every timestamp to the same value and made update_historic_data fail its minute-grid check for every token. - Use lowercase 'h' resample alias ('H' removed in pandas 3). Verified end-to-end: update_historic_data('RPL') completes on pandas 2.3.3 (including with US geo-detection simulated) and pandas 3.0.3. Co-Authored-By: Claude Fable 5 --- .../utils/data_processing/amalgamated_data_utils.py | 4 +++- .../utils/data_processing/historic_data_utils.py | 10 ++++++++-- 2 files changed, 11 insertions(+), 3 deletions(-) diff --git a/quantammsim/utils/data_processing/amalgamated_data_utils.py b/quantammsim/utils/data_processing/amalgamated_data_utils.py index 61cd65f2..76ab9c69 100644 --- a/quantammsim/utils/data_processing/amalgamated_data_utils.py +++ b/quantammsim/utils/data_processing/amalgamated_data_utils.py @@ -43,7 +43,9 @@ def forward_fill_ohlcv_data(df, token): end=pd.to_datetime(df.index.max(), unit="ms"), freq="1min", ) - full_index = full_index.astype(np.int64) // 10**6 + # int64 cast of a DatetimeIndex is unit-dependent on pandas>=3 (this index + # is datetime64[ms] there, not [ns]), so convert to ms explicitly. + full_index = (full_index - pd.Timestamp(0)) // pd.Timedelta(milliseconds=1) # Reindex with the complete minute-level index df = df.reindex(full_index) diff --git a/quantammsim/utils/data_processing/historic_data_utils.py b/quantammsim/utils/data_processing/historic_data_utils.py index 3010bf09..e1a1b322 100644 --- a/quantammsim/utils/data_processing/historic_data_utils.py +++ b/quantammsim/utils/data_processing/historic_data_utils.py @@ -641,7 +641,13 @@ def get_binance_vision_data(token, numeraire, root): data_type="klines", data_frequency="1m" ) - + + # The dumper validates tickers against the live exchangeInfo endpoint, which + # is geo-switched to binance.us for US IPs and so hides pairs (e.g. RPLUSDT) + # that exist in the vision archive but aren't listed on Binance.US. Skip the + # check; a pair with no archive simply downloads nothing. + data_dumper.get_list_all_trading_pairs = lambda: [f"{token}{numeraire}"] + # Download all available data data_dumper.dump_data( tickers=[f"{token}{numeraire}"], @@ -1079,7 +1085,7 @@ def update_historic_data(token, root): agg_dict = {k: v for k, v in agg_dict.items() if k in concated_df_hourly.columns} # Perform resampling - hourly_data = concated_df_hourly.resample("1H").agg(agg_dict).reset_index() + hourly_data = concated_df_hourly.resample("1h").agg(agg_dict).reset_index() # Save hourly data hourly_data.to_csv(hourlyPath, index=False) From 955208fcd415deb13e80ef470739196a76fd8187 Mon Sep 17 00:00:00 2001 From: mendesfabio Date: Wed, 8 Jul 2026 11:04:03 -0300 Subject: [PATCH 113/115] fix noise_calibration import on Python 3.9 cli.py uses PEP 604 unions (str | None) in signatures, which are evaluated at import time and raise TypeError on Python 3.9 (the CI version). This broke collection of all seven tests/noise files. Defer annotation evaluation with the __future__ import. Verified on a Python 3.9.x env with .[dev,calibration]: full-suite collection is clean (2198 tests) and tests/noise passes 176/176. Co-Authored-By: Claude Fable 5 --- quantammsim/noise_calibration/cli.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/quantammsim/noise_calibration/cli.py b/quantammsim/noise_calibration/cli.py index e7f0233a..30fc823b 100644 --- a/quantammsim/noise_calibration/cli.py +++ b/quantammsim/noise_calibration/cli.py @@ -1,5 +1,7 @@ """CLI entry point for noise calibration.""" +from __future__ import annotations + import argparse import json import os From 0acd60f867538d06aae7b7b35c89ea4a48dc03f4 Mon Sep 17 00:00:00 2001 From: mendesfabio Date: Wed, 8 Jul 2026 11:24:29 -0300 Subject: [PATCH 114/115] update test baselines for protocol_fee_split=0.25 default 4c9eae7 changed the default protocol_fee_split from 0.0 to 0.25 to match reClAMM production configuration, but the stored forward-pass baselines (captured with split=0.0) were not re-captured, failing 12 baseline tests on any fingerprint that charges fees. Re-capture forward_pass_test_1/2 final values and returns with the production default; weights are unaffected. Also update test_excludes_training_fields: startDateString is now intentionally kept in the static dict (needed by the calibrated noise model, see _TRAINING_ONLY_FIELDS in jax_runner_utils). Co-Authored-By: Claude Fable 5 --- tests/integration/test_baseline_values.py | 8 ++++---- tests/integration/test_float32_forward_pass.py | 13 ++++++++----- tests/integration/test_gpu_path_baselines.py | 9 ++++++--- tests/integration/test_run_pool_simulation_path.py | 8 ++++---- tests/unit/test_jax_runner_utils.py | 6 ++++-- 5 files changed, 26 insertions(+), 18 deletions(-) diff --git a/tests/integration/test_baseline_values.py b/tests/integration/test_baseline_values.py index 96bb6cad..5b4dea35 100644 --- a/tests/integration/test_baseline_values.py +++ b/tests/integration/test_baseline_values.py @@ -89,8 +89,8 @@ "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1500094.138254407, - "return_pct": 50.00941382544071, + "final_value": 1489697.5771969256, + "return_pct": 48.969757719692566, "first_weights": [0.5, 0.5], "last_weights": [0.05000921, 0.94999079], }, @@ -116,8 +116,8 @@ "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1368731.4974473487, - "return_pct": 36.87314974473486, + "final_value": 1352951.4582811554, + "return_pct": 35.29514582811555, "first_weights": [0.5, 0.5], "last_weights": [0.05, 0.95], }, diff --git a/tests/integration/test_float32_forward_pass.py b/tests/integration/test_float32_forward_pass.py index b4d878a0..75d311b9 100644 --- a/tests/integration/test_float32_forward_pass.py +++ b/tests/integration/test_float32_forward_pass.py @@ -89,8 +89,8 @@ def override_backend(backend): "logit_lamb": jnp.array([-0.22066515, -0.22066515]), "initial_weights_logits": jnp.array([0.0, 0.0]), }, - "expected_final_value": 1500094.138254407, - "expected_return_pct": 50.00941382544071, + "expected_final_value": 1489697.5771969256, + "expected_return_pct": 48.969757719692566, "expected_first_weights": [0.5, 0.5], "expected_last_weights": [0.05000921, 0.94999079], }, @@ -114,8 +114,8 @@ def override_backend(backend): "logit_lamb": jnp.array([2.02840786, 2.02840786]), "initial_weights_logits": jnp.array([0.0, 0.0]), }, - "expected_final_value": 1368731.4974473487, - "expected_return_pct": 36.87314974473486, + "expected_final_value": 1352951.4582811554, + "expected_return_pct": 35.29514582811555, "expected_first_weights": [0.5, 0.5], "expected_last_weights": [0.05, 0.95], }, @@ -262,7 +262,10 @@ def test_final_value_matches_baseline(self, config_name): actual = float(result["final_value"]) expected = config["expected_final_value"] rel_diff = abs(actual - expected) / expected - assert rel_diff < 0.006, ( + # forward_pass_test_2 carries an inherent conv-vs-scan estimator + # divergence (~1.3% in f64 with protocol_fee_split=0.25) on top of + # float32 noise, so the GPU-path tolerance is looser than the CPU one. + assert rel_diff < 0.015, ( f"{config_name} f32 GPU: final value {actual:.2f} vs " f"f64 baseline {expected:.2f} ({rel_diff*100:.4f}%)" ) diff --git a/tests/integration/test_gpu_path_baselines.py b/tests/integration/test_gpu_path_baselines.py index faf20ecf..46fdb93f 100644 --- a/tests/integration/test_gpu_path_baselines.py +++ b/tests/integration/test_gpu_path_baselines.py @@ -85,7 +85,7 @@ def override_backend(backend): "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1500094.138254407, + "final_value": 1489697.5771969256, "first_weights": [0.5, 0.5], "last_weights": [0.05000921, 0.94999079], }, @@ -111,7 +111,7 @@ def override_backend(backend): "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1368731.4974473487, + "final_value": 1352951.4582811554, "first_weights": [0.5, 0.5], "last_weights": [0.05, 0.95], }, @@ -141,7 +141,10 @@ def test_gpu_final_value_matches_baseline(self, config_name): actual_final = float(result["final_value"]) relative_diff = abs(actual_final - expected_final) / expected_final - assert relative_diff < 0.01, ( + # forward_pass_test_2 (log_k=7, weights pinned at bounds) has an + # inherent conv-vs-scan estimator divergence: 0.82% with + # protocol_fee_split=0.0, 1.33% with the 0.25 production default. + assert relative_diff < 0.015, ( f"{config_name} GPU: Final value {actual_final:.2f} vs " f"baseline {expected_final:.2f} ({relative_diff*100:.4f}%)" ) diff --git a/tests/integration/test_run_pool_simulation_path.py b/tests/integration/test_run_pool_simulation_path.py index bfdf754f..f724ac1d 100644 --- a/tests/integration/test_run_pool_simulation_path.py +++ b/tests/integration/test_run_pool_simulation_path.py @@ -82,8 +82,8 @@ "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1500094.138254407, - "return_pct": 50.00941382544071, + "final_value": 1489697.5771969256, + "return_pct": 48.969757719692566, "first_weights": [0.5, 0.5], "last_weights": [0.05000921, 0.94999079], }, @@ -111,8 +111,8 @@ "initial_weights_logits": jnp.array([0.0, 0.0]), }, "expected": { - "final_value": 1368731.4974473487, - "return_pct": 36.87314974473486, + "final_value": 1352951.4582811554, + "return_pct": 35.29514582811555, "first_weights": [0.5, 0.5], "last_weights": [0.05, 0.95], }, diff --git a/tests/unit/test_jax_runner_utils.py b/tests/unit/test_jax_runner_utils.py index 2e014f57..78ba3773 100644 --- a/tests/unit/test_jax_runner_utils.py +++ b/tests/unit/test_jax_runner_utils.py @@ -1052,7 +1052,7 @@ def test_excludes_training_fields(self): fp = { "tokens": ["BTC", "ETH"], "optimisation_settings": {"lr": 0.01}, # Should be excluded - "startDateString": "2023-01-01", # Should be excluded + "startDateString": "2023-01-01", # Kept — needed by calibrated noise model "chunk_period": 60, "weight_interpolation_period": 60, "initial_pool_value": 1000000.0, @@ -1068,7 +1068,9 @@ def test_excludes_training_fields(self): result = create_static_dict(fp, bout_length=1440) assert "optimisation_settings" not in result - assert "startDateString" not in result + # startDateString stays in the static dict — the calibrated noise + # model needs it in forward passes + assert "startDateString" in result def test_with_overrides(self): """Test that overrides are applied.""" From 98ddcce4abc898ea9fa2b1562de3fb9aae7be83b Mon Sep 17 00:00:00 2001 From: mendesfabio Date: Wed, 8 Jul 2026 13:06:49 -0300 Subject: [PATCH 115/115] onboard BTC/ETH and BOLD/USDC pairs; generalize run_final_sims; add onboarding skill - run_final_sims.py: PAIR_CONFIGS entries for btceth (WBTC/WETH pool 0xa6f548df93de92 from the MM artifact) and boldusdc (median-fallback noise); --pair choices derived from PAIR_CONFIGS; optional per-pair train/test window overrides (BOLD data starts 2025-07-09, CoinGecko free-tier limit) with the trial filter applied per pair; thread the --metric flag into select_best_params (was hardcoded). - scripts/prepare_bold_usdc_data.py: CoinGecko data prep for BOLD (liquity-bold-2) + USDC $1-peg rebuild over the union grid. - run_full_sweep.sh: btceth_5m config line. - Sweep artifacts: 50-trial returns_over_hodl runs for btceth_5m and boldusdc_1m (demo scale; production is 300 trials x 4 objectives). - .claude/skills/reclamm-pair-onboarding: step-by-step runbook for onboarding new pairs (data routes, pool-id lookup and median-fallback caveats, sweep command, registration, outputs). Co-Authored-By: Claude Fable 5 --- .../skills/reclamm-pair-onboarding/SKILL.md | 102 ++++++++++++ .../returns_over_hodl_boldusdc_1m.json | 9 + .../returns_over_hodl_btceth_5m.json | 9 + ...4fb0c8f0eb6d75a9a37c3261d5f96a09db477.json | 1 + ...110a1c66faca6e4cf16da5854516f8de280cd.json | 1 + scripts/prepare_bold_usdc_data.py | 154 ++++++++++++++++++ scripts/run_final_sims.py | 84 +++++++--- scripts/run_full_sweep.sh | 1 + 8 files changed, 337 insertions(+), 24 deletions(-) create mode 100644 .claude/skills/reclamm-pair-onboarding/SKILL.md create mode 100644 results/full_sweep/returns_over_hodl_boldusdc_1m.json create mode 100644 results/full_sweep/returns_over_hodl_btceth_5m.json create mode 100644 results/run_44f5e094f9ee304409c9656580e4fb0c8f0eb6d75a9a37c3261d5f96a09db477.json create mode 100644 results/run_cc1d5e6467381c91004a58f57a1110a1c66faca6e4cf16da5854516f8de280cd.json create mode 100644 scripts/prepare_bold_usdc_data.py diff --git a/.claude/skills/reclamm-pair-onboarding/SKILL.md b/.claude/skills/reclamm-pair-onboarding/SKILL.md new file mode 100644 index 00000000..880930cc --- /dev/null +++ b/.claude/skills/reclamm-pair-onboarding/SKILL.md @@ -0,0 +1,102 @@ +--- +name: reclamm-pair-onboarding +description: Onboard a new token pair for reCLAMM simulations — fetch price data, run the Optuna parameter sweep, and produce final-sim CSVs/plots with candidate pool params. Use when asked to simulate a reCLAMM pool for a pair, gather candidate price_ratio/margin/shift params, or add a pair to run_final_sims.py. +--- + +# reCLAMM pair onboarding + +Pipeline: **price data → Optuna sweep → register pair → `run_final_sims.py`**. +Worked examples on branch `btc-eth-pair-onboarding`: BTC/ETH (Binance data, +real pool coefficients) and BOLD/USDC (CoinGecko data, median-fallback noise). + +**Never retrain the noise model.** The checked-in artifacts +(`results/mm_noise/{model.npz,meta.json}`, `results/competitor_tvl/competitor_tvl.npz`) +are fit across 38 pools and reused for every pair; per-pair arrays are built +and cached automatically under `results/mm_noise/_sim_arrays/`. + +## Step 1 — price data (`quantammsim/data/_USD.parquet`) + +`BTC_USD.parquet` is required for **every** pair (market features), plus one +parquet per pool token (`TOKEN_MAP` in `quantammsim/calibration/market_features.py` +maps WBTC→BTC, WETH→ETH, USDT→USDC, …). + +- **Binance-listed token**: `python scripts/download_data.py `. +- **Not on Binance**: copy `scripts/prepare_bold_usdc_data.py` (CoinGecko): + daily close+volume → minute grid by forward-fill, daily volume spread /1440 + (so the daily resample recovers real volume), stables as flat $1.00 peg. + CoinGecko free tier = **last 365 days only** — this bounds the earliest + simulation start and usually forces a per-pair train window (Step 4). + +Check coverage: the parquet must span train start → test end. + +## Step 2 — pool id and the noise fallback + +Look the pair up in `results/mm_noise/meta.json` (`pool_ids` / `pool_tokens`). + +- **Pair present** (e.g. WBTC/WETH → `0xa6f548df93de92`): use that id; the + noise model gets real per-pool coefficients and competitor TVL. +- **Pair absent**: any placeholder id works — the builder falls back to + median coefficients + K=$10M (`noise_model_arrays.py`). Median alpha is + tiny, so organic volume (and fee revenue) will be near zero. If realistic + fees matter, re-anchor: after `build_mm_simulator_arrays`, shift + `noise_base += log(V_max_daily) - mean(noise_base)` and optionally set a + constant competitor K — see `scripts/run_rpl_eth_sweep.py` (~lines 124-134) + for the pattern and the pair's real daily volume for V_max. + +## Step 3 — Optuna sweep + +Demo scale (~50 trials, minutes; the team's production scale is +`scripts/run_full_sweep.sh`: 300 trials × 4 objectives × 3 penalties per TVL): + +``` +python scripts/tune_reclamm_calibrated_noise.py \ + --noise-model mm_observed --artifact-dir results/mm_noise \ + --tokens BTC ETH --pool-id 0xa6f548df93de92 \ + --gas-cost 1.0 --fees 0.0025 --initial-pool-value 5000000 \ + --objective returns_over_hodl --n-trials 50 --pr-max 5.0 \ + --start-date "2025-01-01 00:00:00" --end-date "2025-10-05 00:00:00" \ + --output results/full_sweep/returns_over_hodl__.json +``` + +- One sweep per (pair, TVL). Search space: price_ratio 1.01–200 (log), + margin 0.01–0.99, shift_exponent 1e-5–125 (log). Cap `--pr-max` sensibly: + ~5 for correlated majors, ~1.05 for stable/stable. +- Dates: default train window is 2025-01-01 → 2025-10-05 (pre flash-crash); + keep it unless data starts later. `--end-test-date` sets the OOS span used + for validation metrics. +- Side effect (the part `run_final_sims.py` actually reads): a trajectory + file `results/run_.json` written by `train_on_historic_data`. + +## Step 4 — register the pair and run final sims + +Add an entry to `PAIR_CONFIGS` in `scripts/run_final_sims.py` (tokens, +pool_id, gas_cost, fees, the swept TVLs; `--pair` choices follow the dict). +If the pair's data can't cover the default windows, add per-pair overrides: + +```python +"train": ("2025-07-15 00:00:00", "2025-10-05 00:00:00"), # optional +"test": ("2025-10-25 00:00:00", "2026-05-01 00:00:00"), # optional +``` + +The sweep's train window **must match** the pair's train window — trials are +filtered by it (`filter_trials_to_window`). + +``` +python scripts/run_final_sims.py --pair # or --all +``` + +Selection: best trial per TVL by OOS `returns_over_hodl` (override with +`--metric`). Outputs in `results/final_sims/`: +`run_{Value,Reserves,TokenValues}___{train,test}_.csv`, +themed plots `_{train,test}[_weights]_{light,dark}.png`, and +`_sim_results.pkl`. The printed summary gives train/test RoH and fees. + +## Gotchas + +- `--method` on `run_final_sims.py` is currently a no-op. +- First run per (pool, window) builds noise arrays (needs network-free local + parquets only); subsequent runs hit the `_sim_arrays` cache. +- Stable pairs: expect PR near the 1.01 floor and near-zero fees under the + median-fallback noise — re-anchor (Step 2) before trusting fee numbers. +- Sweeps and final sims are pure JAX forward passes — a laptop handles demo + scale; production sweeps want the parallel `run_full_sweep.sh` machinery. diff --git a/results/full_sweep/returns_over_hodl_boldusdc_1m.json b/results/full_sweep/returns_over_hodl_boldusdc_1m.json new file mode 100644 index 00000000..6308f84a --- /dev/null +++ b/results/full_sweep/returns_over_hodl_boldusdc_1m.json @@ -0,0 +1,9 @@ +{ + "returns_over_hodl": { + "price_ratio": 1.0102394435218087, + "centeredness_margin": 0.03713493968302037, + "shift_exponent": 2.2665545348679854e-05, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/full_sweep/returns_over_hodl_btceth_5m.json b/results/full_sweep/returns_over_hodl_btceth_5m.json new file mode 100644 index 00000000..497ac07d --- /dev/null +++ b/results/full_sweep/returns_over_hodl_btceth_5m.json @@ -0,0 +1,9 @@ +{ + "returns_over_hodl": { + "price_ratio": 1.0135446536102861, + "centeredness_margin": 0.26645620813345267, + "shift_exponent": 1.0439828724530492e-05, + "subsidary_params": [], + "initial_weights_logits": "[0. 0.]" + } +} \ No newline at end of file diff --git a/results/run_44f5e094f9ee304409c9656580e4fb0c8f0eb6d75a9a37c3261d5f96a09db477.json b/results/run_44f5e094f9ee304409c9656580e4fb0c8f0eb6d75a9a37c3261d5f96a09db477.json new file mode 100644 index 00000000..42c426c9 --- /dev/null +++ b/results/run_44f5e094f9ee304409c9656580e4fb0c8f0eb6d75a9a37c3261d5f96a09db477.json @@ -0,0 +1 @@ +"[{\"alphabetic\": true, \"arb_fees\": 0.0, \"arb_frequency\": 5, \"arb_quality\": 1.0, \"bout_offset\": 10080, \"checkpoint_fused\": \"scan\", \"chunk_period\": 1440, \"do_arb\": true, \"do_trades\": false, \"endDateString\": \"2025-10-05 00:00:00\", \"endTestDateString\": \"2026-05-01 00:00:00\", \"ensemble_init_method\": \"gaussian\", \"ensemble_init_scale\": 0.5, \"ensemble_init_seed\": 42, \"evaluation_starts\": [7200, 8973, 9071, 9363, 9932, 10746, 12272, 12520, 14121, 14293, 15077, 16067, 17840, 19614, 21387, 21850, 22135, 22630, 23161, 24121, 24289, 24934, 25630, 26183, 26708, 27957, 28443, 28481, 30255, 31352, 32028, 33802, 34669, 35575, 37349, 37603, 39122, 39303, 39430, 40896], \"fees\": 0.0005, \"freq\": \"minute\", \"gas_cost\": 1.0, \"initial_arc_length_speed\": 0.0001, \"initial_centeredness_margin\": 0.2, \"initial_daily_price_shift_base\": 0.999991935483871, \"initial_k_per_day\": 20, \"initial_log_amplitude\": 0.0, \"initial_memory_length\": 10.0, \"initial_memory_length_delta\": 0.0, \"initial_pool_value\": 1000000.0, \"initial_pre_exp_scaling\": 0.5, \"initial_price_ratio\": 4.0, \"initial_raw_exponents\": 0.0, \"initial_raw_width\": 0.0, \"initial_shift_exponent\": 1.0, \"initial_weights_logits\": 1.0, \"learnable_bounds_settings\": {\"freeze_bounds\": false, \"max_weights_per_asset\": null, \"min_weights_per_asset\": null}, \"max_memory_days\": 365, \"maximum_change\": 0.0003, \"minimum_weight\": null, \"n_ensemble_members\": 1, \"noise_arrays_path\": \"results/mm_noise/_sim_arrays/boldusdc_2025-07-15_2026-05-01_mm.npz\", \"noise_model\": \"mm_observed\", \"noise_trader_ratio\": 0.0, \"numeraire\": null, \"optimisation_settings\": {\"base_lr\": 0.1, \"batch_size\": 8, \"bfgs_settings\": {\"compute_dtype\": \"float32\", \"maxiter\": 100, \"n_evaluation_points\": 20, \"tol\": 1e-06}, \"checkpoint_interval\": 10, \"clip_norm\": 10.0, \"cma_es_settings\": {\"compute_dtype\": \"float32\", \"memory_budget\": null, \"n_evaluation_points\": 20, \"n_generations\": 300, \"population_size\": null, \"sigma0\": 0.5, \"tol\": 1e-08}, \"decay_lr_plateau\": 100, \"decay_lr_ratio\": 0.8, \"early_stopping\": true, \"early_stopping_metric\": \"daily_log_sharpe\", \"early_stopping_patience\": 200, \"force_scalar\": false, \"include_flipped_training_data\": false, \"initial_random_key\": 0, \"lr_decay_ratio\": 1000, \"lr_schedule_type\": \"constant\", \"max_mc_version\": 9, \"method\": \"optuna\", \"min_lr\": 1e-06, \"n_cycles\": 5, \"n_iterations\": 1000, \"n_parameter_sets\": 1, \"noise_scale\": 0.1, \"optimiser\": \"adamw\", \"optuna_settings\": {\"early_stopping\": {\"enabled\": false, \"min_improvement\": 0.001, \"patience\": 100}, \"expand_around\": false, \"make_scalar\": true, \"min_train_returns_over_hodl\": -0.5, \"multi_objective\": false, \"n_jobs\": 4, \"n_startup_trials\": 10, \"n_trials\": 50, \"overfitting_penalty\": 0.2, \"parameter_config\": {\"centeredness_margin\": {\"high\": 0.99, \"low\": 0.01, \"scalar\": true}, \"k_per_day\": {\"high\": 1000, \"log_scale\": true, \"low\": 0.1, \"scalar\": true}, \"log_amplitude\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"log_k\": {\"high\": 10.0, \"log_scale\": false, \"low\": -10.0, \"scalar\": true}, \"logit_lamb\": {\"high\": 4.602848117654388, \"log_scale\": false, \"low\": -5.955817303419269, \"scalar\": true}, \"memory_days_1\": {\"high\": 200, \"log_scale\": true, \"low\": 0.5, \"scalar\": true}, \"memory_days_2\": {\"high\": 200, \"log_scale\": true, \"low\": 0.5, \"scalar\": true}, \"memory_length\": {\"high\": 200, \"log_scale\": true, \"low\": 1, \"scalar\": true}, \"memory_length_delta\": {\"high\": 100, \"log_scale\": true, \"low\": 0.1, \"scalar\": true}, \"price_ratio\": {\"high\": 1.05, \"log_scale\": true, \"low\": 1.01, \"scalar\": true}, \"raw_exponents\": {\"high\": 10, \"log_scale\": false, \"low\": 0, \"scalar\": true}, \"raw_pre_exp_scaling\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"raw_width\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"shift_exponent\": {\"high\": 125.0, \"log_scale\": true, \"low\": 1e-05, \"scalar\": true}, \"weights_logits\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}}, \"storage\": {\"type\": \"sqlite\", \"url\": null}, \"study_name\": null, \"timeout\": 7200}, \"parameter_init_method\": \"gaussian\", \"sample_method\": \"uniform\", \"swa_freq\": 10, \"swa_start_frac\": 0.75, \"track_checkpoints\": false, \"train_on_hessian_trace\": false, \"training_data_kind\": \"historic\", \"use_gradient_clipping\": true, \"use_plateau_decay\": false, \"use_swa\": false, \"val_fraction\": 0.2, \"warmup_steps\": 100, \"weight_decay\": 0.01}, \"price_noise_sigma\": 0.0, \"protocol_fee_split\": 0.25, \"reclamm_arc_length_speed\": null, \"reclamm_centeredness_scaling\": false, \"reclamm_interpolation_method\": \"geometric\", \"reclamm_learn_arc_length_speed\": false, \"reclamm_learn_fees\": false, \"reclamm_use_shift_exponent\": true, \"return_val\": \"returns_over_hodl\", \"rule\": \"reclamm\", \"startDateString\": \"2025-07-15 00:00:00\", \"ste_max_change\": false, \"ste_min_max_weight\": false, \"ste_temperature\": 10.0, \"subsidary_pools\": [], \"tokens\": [\"BOLD\", \"USDC\"], \"training_method\": \"optuna\", \"turnover_penalty\": 0.0, \"use_alt_lamb\": false, \"use_fused_reserves\": true, \"use_pre_exp_scaling\": true, \"weight_calculation_method\": \"auto\", \"weight_interpolation_method\": \"linear\", \"weight_interpolation_period\": 1440}, {\"centeredness_margin\": 0.9705315135199268, \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -Infinity, \"optuna_trial_number\": 0, \"price_ratio\": 1.0269216889261827, \"shift_exponent\": 3.828046963309817, \"step\": 0, \"test_objective\": -Infinity, \"train_objective\": -Infinity, \"train_return\": -Infinity, \"train_returns_over_hodl\": -Infinity, \"train_sharpe\": -Infinity, \"validation_return\": -Infinity, \"validation_returns_over_hodl\": -Infinity, \"validation_sharpe\": -Infinity}, {\"centeredness_margin\": 0.8483785433995226, \"continuous_test_metrics\": [{\"annualised_returns\": -0.0070318614084021736, \"annualised_returns_over_hodl\": -0.008709152139531495, \"annualised_returns_over_uniform_hodl\": -0.008655929042722388, \"calmar\": -1.0791695185673866, \"daily_log_sharpe\": -0.5960824551368603, \"daily_returns\": -4.623807273241882e-05, \"fee_revenue_over_value\": 6.715119112712034e-08, \"jax_sharpe\": -0.6148668676552378, \"return\": -0.004013264976805986, \"returns_over_hodl\": -0.004972341987827367, \"returns_over_uniform_hodl\": -0.004941898233937314, \"sharpe\": -0.5905177992314403, \"sterling\": -2.884276277807359, \"ulcer\": -0.0012501156699691526}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.000753162225985557, \"optuna_trial_number\": 1, \"price_ratio\": 1.0186778147600801, \"shift_exponent\": 0.017753754612275772, \"step\": 1, \"test_objective\": [{\"annualised_returns\": -0.0070318614084021736, \"annualised_returns_over_hodl\": -0.008709152139531495, \"annualised_returns_over_uniform_hodl\": -0.008655929042722388, \"calmar\": -1.0791695185673866, \"daily_log_sharpe\": -0.5960824551368603, \"daily_returns\": -4.623807273241882e-05, \"fee_revenue_over_value\": 6.715119112712034e-08, \"jax_sharpe\": -0.6148668676552378, \"return\": -0.004013264976805986, \"returns_over_hodl\": -0.004972341987827367, \"returns_over_uniform_hodl\": -0.004941898233937314, \"sharpe\": -0.5905177992314403, \"sterling\": -2.884276277807359, \"ulcer\": -0.0012501156699691526}], \"train_objective\": [{\"annualised_returns\": -0.005288023588112978, \"annualised_returns_over_hodl\": -0.004160891421715829, \"annualised_returns_over_uniform_hodl\": -0.004160891421716384, \"calmar\": -2.682904787719356, \"daily_log_sharpe\": -0.6213801860346686, \"daily_returns\": -0.000418286869733153, \"fee_revenue_over_value\": 2.1322309985610545e-08, \"jax_sharpe\": -0.6133529860890407, \"return\": -0.000952453263642683, \"returns_over_hodl\": -0.000749091972206628, \"returns_over_uniform_hodl\": -0.0007490919722067391, \"sharpe\": -0.6170857988632001, \"sterling\": -3.1639515729484975, \"ulcer\": -0.000793052049761743}], \"train_return\": -0.000952453263642683, \"train_returns_over_hodl\": -0.000749091972206628, \"train_sharpe\": -0.6133529860890405, \"validation_return\": -0.00030256492034297366, \"validation_returns_over_hodl\": -0.0001461082643764433, \"validation_sharpe\": -0.8653224960839583}, {\"centeredness_margin\": 0.1435054076939128, \"continuous_test_metrics\": [{\"annualised_returns\": -0.999999999999258, \"annualised_returns_over_hodl\": -0.9999999999992593, \"annualised_returns_over_uniform_hodl\": -0.9999999999992593, \"calmar\": -1.0000001218291144, \"daily_log_sharpe\": -1.3991374576685274, \"daily_returns\": -4.669304346907343e-05, \"fee_revenue_over_value\": 5.922504137937057e-08, \"jax_sharpe\": -10.620242459966331, \"return\": -0.9999998775947065, \"returns_over_hodl\": -0.9999998777137343, \"returns_over_uniform_hodl\": -0.9999998777088343, \"sharpe\": -1.9896363763128277, \"sterling\": -9.982804138233442, \"ulcer\": -0.06221563445122143}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00014473387453358688, \"optuna_trial_number\": 2, \"price_ratio\": 1.0219520108601237, \"shift_exponent\": 47.71175613598654, \"step\": 2, \"test_objective\": [{\"annualised_returns\": -0.999999999999258, \"annualised_returns_over_hodl\": -0.9999999999992593, \"annualised_returns_over_uniform_hodl\": -0.9999999999992593, \"calmar\": -1.0000001218291144, \"daily_log_sharpe\": -1.3991374576685274, \"daily_returns\": -4.669304346907343e-05, \"fee_revenue_over_value\": 5.922504137937057e-08, \"jax_sharpe\": -10.620242459966331, \"return\": -0.9999998775947065, \"returns_over_hodl\": -0.9999998777137343, \"returns_over_uniform_hodl\": -0.9999998777088343, \"sharpe\": -1.9896363763128277, \"sterling\": -9.982804138233442, \"ulcer\": -0.06221563445122143}], \"train_objective\": [{\"annualised_returns\": 0.0006440676706649384, \"annualised_returns_over_hodl\": 0.0017779216329476544, \"annualised_returns_over_uniform_hodl\": 0.0017779216329489866, \"calmar\": 0.33955936933294467, \"daily_log_sharpe\": 0.07215639722852649, \"daily_returns\": -0.0004228490253137523, \"fee_revenue_over_value\": 2.1448033367835388e-08, \"jax_sharpe\": 0.0763556891089936, \"return\": 0.00011572393305936401, \"returns_over_hodl\": 0.000319302657484144, \"returns_over_uniform_hodl\": 0.00031930265748436604, \"sharpe\": 0.0766488170731237, \"sterling\": 0.3776382675021919, \"ulcer\": -0.0007618429251113228}], \"train_return\": 0.00011572393305936401, \"train_returns_over_hodl\": 0.000319302657484144, \"train_sharpe\": 0.0763556891089936, \"validation_return\": -8.727929765606213e-05, \"validation_returns_over_hodl\": 6.730638289131896e-05, \"validation_sharpe\": -0.2714382894508191}, {\"centeredness_margin\": 0.31170056013703984, \"continuous_test_metrics\": [{\"annualised_returns\": 0.002830117296601875, \"annualised_returns_over_hodl\": 0.0011324956510567752, \"annualised_returns_over_uniform_hodl\": 0.001189919718326049, \"calmar\": 0.6735477633193868, \"daily_log_sharpe\": 0.3221162688379301, \"daily_returns\": -4.63384482852676e-05, \"fee_revenue_over_value\": 6.734964862492636e-08, \"jax_sharpe\": 0.29373733784833866, \"return\": 0.0016117934594890304, \"returns_over_hodl\": 0.0006452081263255138, \"returns_over_uniform_hodl\": 0.00067791553779184, \"sharpe\": 0.3270288360203714, \"sterling\": 1.5738818488926838, \"ulcer\": -0.0007359265434141715}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0001793904798752355, \"optuna_trial_number\": 3, \"price_ratio\": 1.0269459875320377, \"shift_exponent\": 0.07697056285540975, \"step\": 3, \"test_objective\": [{\"annualised_returns\": 0.002830117296601875, \"annualised_returns_over_hodl\": 0.0011324956510567752, \"annualised_returns_over_uniform_hodl\": 0.001189919718326049, \"calmar\": 0.6735477633193868, \"daily_log_sharpe\": 0.3221162688379301, \"daily_returns\": -4.63384482852676e-05, \"fee_revenue_over_value\": 6.734964862492636e-08, \"jax_sharpe\": 0.29373733784833866, \"return\": 0.0016117934594890304, \"returns_over_hodl\": 0.0006452081263255138, \"returns_over_uniform_hodl\": 0.00067791553779184, \"sharpe\": 0.3270288360203714, \"sterling\": 1.5738818488926838, \"ulcer\": -0.0007359265434141715}], \"train_objective\": [{\"annualised_returns\": 0.0003191878620236732, \"annualised_returns_over_hodl\": 0.0014526736951496755, \"annualised_returns_over_uniform_hodl\": 0.0014526736951496755, \"calmar\": 0.16717580233768653, \"daily_log_sharpe\": 0.03621157872891251, \"daily_returns\": -0.00041828687828537365, \"fee_revenue_over_value\": 2.1204372415326673e-08, \"jax_sharpe\": 0.04052638427086134, \"return\": 5.7358250794337096e-05, \"returns_over_hodl\": 0.00026092509458308655, \"returns_over_uniform_hodl\": 0.00026092509458308655, \"sharpe\": 0.040655193886243654, \"sterling\": 0.18875666484065598, \"ulcer\": -0.0007650362747108154}], \"train_return\": 5.7358250794337096e-05, \"train_returns_over_hodl\": 0.00026092509458308655, \"train_sharpe\": 0.04052638427086134, \"validation_return\": -9.908251163703863e-05, \"validation_returns_over_hodl\": 5.500418138204566e-05, \"validation_sharpe\": -0.302056642897357}, {\"centeredness_margin\": 0.19757177606387216, \"continuous_test_metrics\": [{\"annualised_returns\": 0.003865701438038549, \"annualised_returns_over_hodl\": 0.002142360654313835, \"annualised_returns_over_uniform_hodl\": 0.0022238100907268077, \"calmar\": 0.9536129196775749, \"daily_log_sharpe\": 0.44954967268727886, \"daily_returns\": -4.6992831588591404e-05, \"fee_revenue_over_value\": 6.666161766360154e-08, \"jax_sharpe\": 0.420784483850652, \"return\": 0.002201084794766217, \"returns_over_hodl\": 0.0012202860831742601, \"returns_over_uniform_hodl\": 0.0012666574324862179, \"sharpe\": 0.4541851842347319, \"sterling\": 2.371793011448105, \"ulcer\": -0.0006118864857495677}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00011543041626387307, \"optuna_trial_number\": 4, \"price_ratio\": 1.0189779037348938, \"shift_exponent\": 0.00024176679153565904, \"step\": 4, \"test_objective\": [{\"annualised_returns\": 0.003865701438038549, \"annualised_returns_over_hodl\": 0.002142360654313835, \"annualised_returns_over_uniform_hodl\": 0.0022238100907268077, \"calmar\": 0.9536129196775749, \"daily_log_sharpe\": 0.44954967268727886, \"daily_returns\": -4.6992831588591404e-05, \"fee_revenue_over_value\": 6.666161766360154e-08, \"jax_sharpe\": 0.420784483850652, \"return\": 0.002201084794766217, \"returns_over_hodl\": 0.0012202860831742601, \"returns_over_uniform_hodl\": 0.0012666574324862179, \"sharpe\": 0.4541851842347319, \"sterling\": 2.371793011448105, \"ulcer\": -0.0006118864857495677}], \"train_objective\": [{\"annualised_returns\": 0.0009188433926954342, \"annualised_returns_over_hodl\": 0.0020530087099863703, \"annualised_returns_over_uniform_hodl\": 0.0020530087099863703, \"calmar\": 0.4876643723482636, \"daily_log_sharpe\": 0.10300122664048747, \"daily_returns\": -0.0004182868782575554, \"fee_revenue_over_value\": 2.159654808318211e-08, \"jax_sharpe\": 0.10724093194713924, \"return\": 0.0001650761264666567, \"returns_over_hodl\": 0.00036866489678555325, \"returns_over_uniform_hodl\": 0.00036866489678555325, \"sharpe\": 0.10748441790553563, \"sterling\": 0.5345848958384085, \"ulcer\": -0.0007571702044217817}], \"train_return\": 0.0001650761264666567, \"train_returns_over_hodl\": 0.00036866489678555325, \"train_sharpe\": 0.10724093194713921, \"validation_return\": -7.729995583372062e-05, \"validation_returns_over_hodl\": 7.770761833181261e-05, \"validation_sharpe\": -0.24405315973159467}, {\"centeredness_margin\": 0.06801178262232609, \"continuous_test_metrics\": [{\"annualised_returns\": 0.00481054963208849, \"annualised_returns_over_hodl\": 0.003077037527198012, \"annualised_returns_over_uniform_hodl\": 0.0031671129206185533, \"calmar\": 1.1942402650225643, \"daily_log_sharpe\": 0.5626879601356506, \"daily_returns\": -4.7226064538751335e-05, \"fee_revenue_over_value\": 6.855408719099562e-08, \"jax_sharpe\": 0.5368489055656892, \"return\": 0.00273851544174164, \"returns_over_hodl\": 0.0017523253082134538, \"returns_over_uniform_hodl\": 0.0018035869924946102, \"sharpe\": 0.5671760105596919, \"sterling\": 3.124398064621498, \"ulcer\": -0.0005589561953890088}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -9.263216184786927e-05, \"optuna_trial_number\": 5, \"price_ratio\": 1.0171682170362146, \"shift_exponent\": 1.67421801873129e-05, \"step\": 5, \"test_objective\": [{\"annualised_returns\": 0.00481054963208849, \"annualised_returns_over_hodl\": 0.003077037527198012, \"annualised_returns_over_uniform_hodl\": 0.0031671129206185533, \"calmar\": 1.1942402650225643, \"daily_log_sharpe\": 0.5626879601356506, \"daily_returns\": -4.7226064538751335e-05, \"fee_revenue_over_value\": 6.855408719099562e-08, \"jax_sharpe\": 0.5368489055656892, \"return\": 0.00273851544174164, \"returns_over_hodl\": 0.0017523253082134538, \"returns_over_uniform_hodl\": 0.0018035869924946102, \"sharpe\": 0.5671760105596919, \"sterling\": 3.124398064621498, \"ulcer\": -0.0005589561953890088}], \"train_objective\": [{\"annualised_returns\": 0.0011326690015966978, \"annualised_returns_over_hodl\": 0.0022670766098495942, \"annualised_returns_over_uniform_hodl\": 0.0022670766098495942, \"calmar\": 0.605496792288272, \"daily_log_sharpe\": 0.1258990644488, \"daily_returns\": -0.0004241031213781333, \"fee_revenue_over_value\": 2.1688196698789577e-08, \"jax_sharpe\": 0.129709397898856, \"return\": 0.00020347344235505105, \"returns_over_hodl\": 0.00040707002864603936, \"returns_over_uniform_hodl\": 0.00040707002864603936, \"sharpe\": 0.13041663440512247, \"sterling\": 0.6550504890496187, \"ulcer\": -0.0007548145538433133}], \"train_return\": 0.00020347344235505105, \"train_returns_over_hodl\": 0.00040707002864603936, \"train_sharpe\": 0.129709397898856, \"validation_return\": -6.95364798308784e-05, \"validation_returns_over_hodl\": 8.57993148781766e-05, \"validation_sharpe\": -0.21994038890444897}, {\"centeredness_margin\": 0.8812651525819237, \"continuous_test_metrics\": [{\"annualised_returns\": -0.000429606968937013, \"annualised_returns_over_hodl\": -0.0021402725316576054, \"annualised_returns_over_uniform_hodl\": -0.002064473044145698, \"calmar\": -0.08846378116808873, \"daily_log_sharpe\": -0.009759723843075774, \"daily_returns\": -4.6847473055543624e-05, \"fee_revenue_over_value\": 6.094335546949726e-08, \"jax_sharpe\": -0.032462254888176785, \"return\": -0.0002448389291770381, \"returns_over_hodl\": -0.0012202200739555025, \"returns_over_uniform_hodl\": -0.0011769857730177247, \"sharpe\": -0.004115272870416158, \"sterling\": -0.19934918943491012, \"ulcer\": -0.001011987238040047}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0004127901621728353, \"optuna_trial_number\": 6, \"price_ratio\": 1.0460008670372156, \"shift_exponent\": 0.0029750485212631174, \"step\": 6, \"test_objective\": [{\"annualised_returns\": -0.000429606968937013, \"annualised_returns_over_hodl\": -0.0021402725316576054, \"annualised_returns_over_uniform_hodl\": -0.002064473044145698, \"calmar\": -0.08846378116808873, \"daily_log_sharpe\": -0.009759723843075774, \"daily_returns\": -4.6847473055543624e-05, \"fee_revenue_over_value\": 6.094335546949726e-08, \"jax_sharpe\": -0.032462254888176785, \"return\": -0.0002448389291770381, \"returns_over_hodl\": -0.0012202200739555025, \"returns_over_uniform_hodl\": -0.0011769857730177247, \"sharpe\": -0.004115272870416158, \"sterling\": -0.19934918943491012, \"ulcer\": -0.001011987238040047}], \"train_objective\": [{\"annualised_returns\": -0.002276692544529335, \"annualised_returns_over_hodl\": -0.001146148166193517, \"annualised_returns_over_uniform_hodl\": -0.0011461481661929618, \"calmar\": -1.1262244546847484, \"daily_log_sharpe\": -0.26327236878176946, \"daily_returns\": -0.0004204960053340095, \"fee_revenue_over_value\": 1.96117727753207e-08, \"jax_sharpe\": -0.2577564808232864, \"return\": -0.00040955917556440014, \"returns_over_hodl\": -0.0002060873752312009, \"returns_over_uniform_hodl\": -0.0002060873752310899, \"sharpe\": -0.2588990851454688, \"sterling\": -1.3430634060714324, \"ulcer\": -0.0008036865888934341}], \"train_return\": -0.00040955917556440014, \"train_returns_over_hodl\": -0.0002060873752312009, \"train_sharpe\": -0.2577564808232864, \"validation_return\": -0.0001232785946706505, \"validation_returns_over_hodl\": 3.306510600564749e-05, \"validation_sharpe\": -0.36223388360320563}, {\"centeredness_margin\": 0.5209450519058768, \"continuous_test_metrics\": [{\"annualised_returns\": -0.9999999999999113, \"annualised_returns_over_hodl\": -0.9999999999999115, \"annualised_returns_over_uniform_hodl\": -0.9999999999999115, \"calmar\": -1.0000000362726984, \"daily_log_sharpe\": -2.6770243876566022, \"daily_returns\": -4.728268897071485e-05, \"fee_revenue_over_value\": 9.573876148025981e-09, \"jax_sharpe\": -13.918601370968403, \"return\": -0.9999999635235529, \"returns_over_hodl\": -0.9999999635594704, \"returns_over_uniform_hodl\": -0.9999999635575627, \"sharpe\": -2.7213052132537277, \"sterling\": -3.003539250738222, \"ulcer\": -0.1929264346185054}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -8.709599430423432e-05, \"optuna_trial_number\": 7, \"price_ratio\": 1.0167796553153785, \"shift_exponent\": 12.96009542042159, \"step\": 7, \"test_objective\": [{\"annualised_returns\": -0.9999999999999113, \"annualised_returns_over_hodl\": -0.9999999999999115, \"annualised_returns_over_uniform_hodl\": -0.9999999999999115, \"calmar\": -1.0000000362726984, \"daily_log_sharpe\": -2.6770243876566022, \"daily_returns\": -4.728268897071485e-05, \"fee_revenue_over_value\": 9.573876148025981e-09, \"jax_sharpe\": -13.918601370968403, \"return\": -0.9999999635235529, \"returns_over_hodl\": -0.9999999635594704, \"returns_over_uniform_hodl\": -0.9999999635575627, \"sharpe\": -2.7213052132537277, \"sterling\": -3.003539250738222, \"ulcer\": -0.1929264346185054}], \"train_objective\": [{\"annualised_returns\": 0.001184599707077183, \"annualised_returns_over_hodl\": 0.0023190661592666917, \"annualised_returns_over_uniform_hodl\": 0.0023190661592666917, \"calmar\": 0.6340936925702964, \"daily_log_sharpe\": 0.1299089493195694, \"daily_returns\": -0.00042423637398822617, \"fee_revenue_over_value\": 2.170800304254294e-08, \"jax_sharpe\": 0.13438967376621594, \"return\": 0.0002127977820214344, \"returns_over_hodl\": 0.00041639626632994364, \"returns_over_uniform_hodl\": 0.00041639626632994364, \"sharpe\": 0.13448897077532876, \"sterling\": 0.6840904120897559, \"ulcer\": -0.0007543952108578901}], \"train_return\": 0.0002127977820214344, \"train_returns_over_hodl\": 0.00041639626632994364, \"train_sharpe\": 0.1343896737662159, \"validation_return\": -6.765129859032104e-05, \"validation_returns_over_hodl\": 8.776419825395898e-05, \"validation_sharpe\": -0.21409600002417173}, {\"centeredness_margin\": 0.988031719881667, \"continuous_test_metrics\": [{\"annualised_returns\": -0.006294838604110953, \"annualised_returns_over_hodl\": -0.007934588116751073, \"annualised_returns_over_uniform_hodl\": -0.00792011168988116, \"calmar\": -1.0607676773184975, \"daily_log_sharpe\": -0.531520522328535, \"daily_returns\": -5.108568823890045e-05, \"fee_revenue_over_value\": 2.791160595633111e-08, \"jax_sharpe\": -0.5507456651995571, \"return\": -0.0035920547782578582, \"returns_over_hodl\": -0.00452935860432202, \"returns_over_uniform_hodl\": -0.004521080761301088, \"sharpe\": -0.5259142133319055, \"sterling\": -2.602928055857092, \"ulcer\": -0.0012034107269803108}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0009698913735971898, \"optuna_trial_number\": 8, \"price_ratio\": 1.0253141250717166, \"shift_exponent\": 0.0022437951549727107, \"step\": 8, \"test_objective\": [{\"annualised_returns\": -0.006294838604110953, \"annualised_returns_over_hodl\": -0.007934588116751073, \"annualised_returns_over_uniform_hodl\": -0.00792011168988116, \"calmar\": -1.0607676773184975, \"daily_log_sharpe\": -0.531520522328535, \"daily_returns\": -5.108568823890045e-05, \"fee_revenue_over_value\": 2.791160595633111e-08, \"jax_sharpe\": -0.5507456651995571, \"return\": -0.0035920547782578582, \"returns_over_hodl\": -0.00452935860432202, \"returns_over_uniform_hodl\": -0.004521080761301088, \"sharpe\": -0.5259142133319055, \"sterling\": -2.602928055857092, \"ulcer\": -0.0012034107269803108}], \"train_objective\": [{\"annualised_returns\": -0.005921079438325094, \"annualised_returns_over_hodl\": -0.004794664602802867, \"annualised_returns_over_uniform_hodl\": -0.004794664602802867, \"calmar\": -2.6458973753486816, \"daily_log_sharpe\": -0.6810606219585663, \"daily_returns\": -0.0004222512624125535, \"fee_revenue_over_value\": 1.3801822985193785e-08, \"jax_sharpe\": -0.6704708792988321, \"return\": -0.0010667544145916974, \"returns_over_hodl\": -0.0008634163897457414, \"returns_over_uniform_hodl\": -0.0008634163897457414, \"sharpe\": -0.6766706384707122, \"sterling\": -3.3998616153343226, \"ulcer\": -0.0008456789882654904}], \"train_return\": -0.0010667544145916974, \"train_returns_over_hodl\": -0.0008634163897457414, \"train_sharpe\": -0.6704708792988321, \"validation_return\": -0.0003787145716238616, \"validation_returns_over_hodl\": -0.00022544350989706086, \"validation_sharpe\": -1.1075541558843474}, {\"centeredness_margin\": 0.08206634412143726, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0029880374681980904, \"annualised_returns_over_hodl\": 0.0013084038725106328, \"annualised_returns_over_uniform_hodl\": 0.0013475816006278674, \"calmar\": 0.6952538195894145, \"daily_log_sharpe\": 0.32525683181282616, \"daily_returns\": -4.5839579527839575e-05, \"fee_revenue_over_value\": 6.549944912055727e-08, \"jax_sharpe\": 0.30197685741426056, \"return\": 0.001701673649341151, \"returns_over_hodl\": 0.0007453988028258696, \"returns_over_uniform_hodl\": 0.0007677119255904419, \"sharpe\": 0.330301654756731, \"sterling\": 1.620534201407616, \"ulcer\": -0.000778700962511635}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0002415284538617252, \"optuna_trial_number\": 9, \"price_ratio\": 1.0396269633686013, \"shift_exponent\": 2.375032934430731e-05, \"step\": 9, \"test_objective\": [{\"annualised_returns\": 0.0029880374681980904, \"annualised_returns_over_hodl\": 0.0013084038725106328, \"annualised_returns_over_uniform_hodl\": 0.0013475816006278674, \"calmar\": 0.6952538195894145, \"daily_log_sharpe\": 0.32525683181282616, \"daily_returns\": -4.5839579527839575e-05, \"fee_revenue_over_value\": 6.549944912055727e-08, \"jax_sharpe\": 0.30197685741426056, \"return\": 0.001701673649341151, \"returns_over_hodl\": 0.0007453988028258696, \"returns_over_uniform_hodl\": 0.0007677119255904419, \"sharpe\": 0.330301654756731, \"sterling\": 1.620534201407616, \"ulcer\": -0.000778700962511635}], \"train_objective\": [{\"annualised_returns\": -0.00021980321425263405, \"annualised_returns_over_hodl\": 0.0009130718750649525, \"annualised_returns_over_uniform_hodl\": 0.0009130718750649525, \"calmar\": -0.11287095697324459, \"daily_log_sharpe\": -0.0247565322089569, \"daily_returns\": -0.0004208415028742001, \"fee_revenue_over_value\": 2.0615999641970387e-08, \"jax_sharpe\": -0.02020473147248405, \"return\": -3.950750205283793e-05, \"returns_over_hodl\": 0.00016403962421107643, \"returns_over_uniform_hodl\": 0.00016403962421107643, \"sharpe\": -0.020273660316453068, \"sterling\": -0.13134687685092708, \"ulcer\": -0.0007734945922629643}], \"train_return\": -3.950750205283793e-05, \"train_returns_over_hodl\": 0.00016403962421107643, \"train_sharpe\": -0.02020473147248405, \"validation_return\": -0.00011568802416239699, \"validation_returns_over_hodl\": 3.769668164022022e-05, \"validation_sharpe\": -0.3503802510554694}, {\"centeredness_margin\": 0.40568382859472657, \"continuous_test_metrics\": [{\"annualised_returns\": 0.002029640340732941, \"annualised_returns_over_hodl\": 0.00034428911411543694, \"annualised_returns_over_uniform_hodl\": 0.00039075199753257905, \"calmar\": 0.4756624358551505, \"daily_log_sharpe\": 0.2299694720035385, \"daily_returns\": -4.6039871825105047e-05, \"fee_revenue_over_value\": 6.62789010719208e-08, \"jax_sharpe\": 0.2022991237858996, \"return\": 0.001156108710249848, \"returns_over_hodl\": 0.00019618245188235406, \"returns_over_uniform_hodl\": 0.00022265565767787265, \"sharpe\": 0.23514231296087845, \"sterling\": 1.0812682924056973, \"ulcer\": -0.0007954916049290155}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00021853728120247164, \"optuna_trial_number\": 10, \"price_ratio\": 1.033329966564406, \"shift_exponent\": 0.008034540706451312, \"step\": 10, \"test_objective\": [{\"annualised_returns\": 0.002029640340732941, \"annualised_returns_over_hodl\": 0.00034428911411543694, \"annualised_returns_over_uniform_hodl\": 0.00039075199753257905, \"calmar\": 0.4756624358551505, \"daily_log_sharpe\": 0.2299694720035385, \"daily_returns\": -4.6039871825105047e-05, \"fee_revenue_over_value\": 6.62789010719208e-08, \"jax_sharpe\": 0.2022991237858996, \"return\": 0.001156108710249848, \"returns_over_hodl\": 0.00019618245188235406, \"returns_over_uniform_hodl\": 0.00022265565767787265, \"sharpe\": 0.23514231296087845, \"sterling\": 1.0812682924056973, \"ulcer\": -0.0007954916049290155}], \"train_objective\": [{\"annualised_returns\": -1.881226433431138e-05, \"annualised_returns_over_hodl\": 0.0011142905726839736, \"annualised_returns_over_uniform_hodl\": 0.0011142905726839736, \"calmar\": -0.009718040957239523, \"daily_log_sharpe\": -0.002122773859529543, \"daily_returns\": -0.00042131257717104985, \"fee_revenue_over_value\": 2.0902902109334014e-08, \"jax_sharpe\": 0.0023238900183436347, \"return\": -3.3810438301307144e-06, \"returns_over_hodl\": 0.00020017343616118843, \"returns_over_uniform_hodl\": 0.00020017343616118843, \"sharpe\": 0.002349369553174303, \"sterling\": -0.011194789012938255, \"ulcer\": -0.0007695073013547083}], \"train_return\": -3.3810438301307144e-06, \"train_returns_over_hodl\": 0.00020017343616118843, \"train_sharpe\": 0.0023238900183436347, \"validation_return\": -0.00010902095929965494, \"validation_returns_over_hodl\": 4.464558852346201e-05, \"validation_sharpe\": -0.331758933961017}, {\"centeredness_margin\": 0.926013817890981, \"continuous_test_metrics\": [{\"annualised_returns\": -0.0021749770095473853, \"annualised_returns_over_hodl\": -0.0038542432856114583, \"annualised_returns_over_uniform_hodl\": -0.003806988412100698, \"calmar\": -0.43317324635706567, \"daily_log_sharpe\": -0.16517557024543592, \"daily_returns\": -4.60669845974308e-05, \"fee_revenue_over_value\": 5.2087581719107485e-08, \"jax_sharpe\": -0.18547701439100156, \"return\": -0.0012400151890110678, \"returns_over_hodl\": -0.002198207362261395, \"returns_over_uniform_hodl\": -0.0021712341552615477, \"sharpe\": -0.15948554695625605, \"sterling\": -0.9794037108072988, \"ulcer\": -0.001053210190594978}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0005688438266606505, \"optuna_trial_number\": 11, \"price_ratio\": 1.0318670667859244, \"shift_exponent\": 0.001725866789080287, \"step\": 11, \"test_objective\": [{\"annualised_returns\": -0.0021749770095473853, \"annualised_returns_over_hodl\": -0.0038542432856114583, \"annualised_returns_over_uniform_hodl\": -0.003806988412100698, \"calmar\": -0.43317324635706567, \"daily_log_sharpe\": -0.16517557024543592, \"daily_returns\": -4.60669845974308e-05, \"fee_revenue_over_value\": 5.2087581719107485e-08, \"jax_sharpe\": -0.18547701439100156, \"return\": -0.0012400151890110678, \"returns_over_hodl\": -0.002198207362261395, \"returns_over_uniform_hodl\": -0.0021712341552615477, \"sharpe\": -0.15948554695625605, \"sterling\": -0.9794037108072988, \"ulcer\": -0.001053210190594978}], \"train_objective\": [{\"annualised_returns\": -0.003210735127432751, \"annualised_returns_over_hodl\": -0.0020812491353078277, \"annualised_returns_over_uniform_hodl\": -0.0020812491353071616, \"calmar\": -1.5514498568404178, \"daily_log_sharpe\": -0.37418467853756565, \"daily_returns\": -0.0004214486635937713, \"fee_revenue_over_value\": 1.745243988709705e-08, \"jax_sharpe\": -0.36772798843328464, \"return\": -0.0005778079249524337, \"returns_over_hodl\": -0.0003743703725216374, \"returns_over_uniform_hodl\": -0.0003743703725215264, \"sharpe\": -0.3698397801668877, \"sterling\": -1.8967239145105879, \"ulcer\": -0.0008039499426406654}], \"train_return\": -0.0005778079249524337, \"train_returns_over_hodl\": -0.0003743703725216374, \"train_sharpe\": -0.36772798843328464, \"validation_return\": -0.0002789823917117573, \"validation_returns_over_hodl\": -0.0001238997416656007, \"validation_sharpe\": -0.805481688664809}, {\"centeredness_margin\": 0.6025676727517822, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0026312691238983277, \"annualised_returns_over_hodl\": 0.0009527070320891617, \"annualised_returns_over_uniform_hodl\": 0.0009913967754751063, \"calmar\": 0.605538134246308, \"daily_log_sharpe\": 0.28488365483738204, \"daily_returns\": -4.5826624608761324e-05, \"fee_revenue_over_value\": 5.8239553453583705e-08, \"jax_sharpe\": 0.26124489236597775, \"return\": 0.001498610457635996, \"returns_over_hodl\": 0.0005427994969551264, \"returns_over_uniform_hodl\": 0.0005648380649543316, \"sharpe\": 0.290037532931639, \"sterling\": 1.3915258803016708, \"ulcer\": -0.000815778880112569}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00026014075897213277, \"optuna_trial_number\": 12, \"price_ratio\": 1.0498800109374409, \"shift_exponent\": 1.879273073793736e-05, \"step\": 12, \"test_objective\": [{\"annualised_returns\": 0.0026312691238983277, \"annualised_returns_over_hodl\": 0.0009527070320891617, \"annualised_returns_over_uniform_hodl\": 0.0009913967754751063, \"calmar\": 0.605538134246308, \"daily_log_sharpe\": 0.28488365483738204, \"daily_returns\": -4.5826624608761324e-05, \"fee_revenue_over_value\": 5.8239553453583705e-08, \"jax_sharpe\": 0.26124489236597775, \"return\": 0.001498610457635996, \"returns_over_hodl\": 0.0005427994969551264, \"returns_over_uniform_hodl\": 0.0005648380649543316, \"sharpe\": 0.290037532931639, \"sterling\": 1.3915258803016708, \"ulcer\": -0.000815778880112569}], \"train_objective\": [{\"annualised_returns\": -0.00040282794222457063, \"annualised_returns_over_hodl\": 0.0007298397573538562, \"annualised_returns_over_uniform_hodl\": 0.0007298397573538562, \"calmar\": -0.20619905738623137, \"daily_log_sharpe\": -0.04583811980605102, \"daily_returns\": -0.0004203289501581347, \"fee_revenue_over_value\": 2.0169693814951087e-08, \"jax_sharpe\": -0.041298759207707036, \"return\": -7.240986351009226e-05, \"returns_over_hodl\": 0.00013113056530822398, \"returns_over_uniform_hodl\": 0.00013113056530822398, \"sharpe\": -0.041399430061006486, \"sterling\": -0.24228063136173536, \"ulcer\": -0.0007744077994164249}], \"train_return\": -7.240986351009226e-05, \"train_returns_over_hodl\": 0.00013113056530822398, \"train_sharpe\": -0.041298759207707036, \"validation_return\": -0.0001225104313521408, \"validation_returns_over_hodl\": 3.056765605524703e-05, \"validation_sharpe\": -0.36668421150467173}, {\"centeredness_margin\": 0.44409206762800946, \"continuous_test_metrics\": [{\"annualised_returns\": -0.012421048342217023, \"annualised_returns_over_hodl\": -0.014473354468504951, \"annualised_returns_over_uniform_hodl\": -0.014036301590924838, \"calmar\": -1.1978104135540362, \"daily_log_sharpe\": -0.944164425622122, \"daily_returns\": -5.690236401136156e-05, \"fee_revenue_over_value\": 7.038586586680694e-08, \"jax_sharpe\": -0.9605788092152842, \"return\": -0.0070972940376407, \"returns_over_hodl\": -0.008273654132566977, \"returns_over_uniform_hodl\": -0.008023051822789284, \"sharpe\": -0.9380216428703629, \"sterling\": -4.685585217420836, \"ulcer\": -0.0014031108667419572}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0009982817353654306, \"optuna_trial_number\": 13, \"price_ratio\": 1.010274847611044, \"shift_exponent\": 0.27895610784089786, \"step\": 13, \"test_objective\": [{\"annualised_returns\": -0.012421048342217023, \"annualised_returns_over_hodl\": -0.014473354468504951, \"annualised_returns_over_uniform_hodl\": -0.014036301590924838, \"calmar\": -1.1978104135540362, \"daily_log_sharpe\": -0.944164425622122, \"daily_returns\": -5.690236401136156e-05, \"fee_revenue_over_value\": 7.038586586680694e-08, \"jax_sharpe\": -0.9605788092152842, \"return\": -0.0070972940376407, \"returns_over_hodl\": -0.008273654132566977, \"returns_over_uniform_hodl\": -0.008023051822789284, \"sharpe\": -0.9380216428703629, \"sterling\": -4.685585217420836, \"ulcer\": -0.0014031108667419572}], \"train_objective\": [{\"annualised_returns\": -0.007414654557348799, \"annualised_returns_over_hodl\": -0.006289932127869546, \"annualised_returns_over_uniform_hodl\": -0.0062899321278702125, \"calmar\": -3.4447263547077265, \"daily_log_sharpe\": -0.8301100520821049, \"daily_returns\": -0.0004279639297458482, \"fee_revenue_over_value\": 2.2034788587600973e-08, \"jax_sharpe\": -0.82662277761531, \"return\": -0.0013366630832210014, \"returns_over_hodl\": -0.0011333799996795513, \"returns_over_uniform_hodl\": -0.0011333799996796623, \"sharpe\": -0.8255789037755825, \"sterling\": -3.769768431248404, \"ulcer\": -0.0008596241042636524}], \"train_return\": -0.0013366630832210014, \"train_returns_over_hodl\": -0.0011333799996795513, \"train_sharpe\": -0.8266227776153099, \"validation_return\": -4.230402514937559e-05, \"validation_returns_over_hodl\": 0.00014261619528976865, \"validation_sharpe\": -0.11602809447596746}, {\"centeredness_margin\": 0.5946097415736847, \"continuous_test_metrics\": [{\"annualised_returns\": -0.9999999999998803, \"annualised_returns_over_hodl\": -0.9999999999998805, \"annualised_returns_over_uniform_hodl\": -0.9999999999998805, \"calmar\": -1.0000000430168712, \"daily_log_sharpe\": -1.6116984329198079, \"daily_returns\": -4.6779937337003405e-05, \"fee_revenue_over_value\": 2.6157192192010916e-09, \"jax_sharpe\": -11.772774536397659, \"return\": -0.9999999567305403, \"returns_over_hodl\": -0.999999956772694, \"returns_over_uniform_hodl\": -0.9999999567708837, \"sharpe\": -1.884975213816705, \"sterling\": -5.898848180685208, \"ulcer\": -0.14369952100917222}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00013624071440709873, \"optuna_trial_number\": 14, \"price_ratio\": 1.0209982614877327, \"shift_exponent\": 39.759793332730524, \"step\": 14, \"test_objective\": [{\"annualised_returns\": -0.9999999999998803, \"annualised_returns_over_hodl\": -0.9999999999998805, \"annualised_returns_over_uniform_hodl\": -0.9999999999998805, \"calmar\": -1.0000000430168712, \"daily_log_sharpe\": -1.6116984329198079, \"daily_returns\": -4.6779937337003405e-05, \"fee_revenue_over_value\": 2.6157192192010916e-09, \"jax_sharpe\": -11.772774536397659, \"return\": -0.9999999567305403, \"returns_over_hodl\": -0.999999956772694, \"returns_over_uniform_hodl\": -0.9999999567708837, \"sharpe\": -1.884975213816705, \"sterling\": -5.898848180685208, \"ulcer\": -0.14369952100917222}], \"train_objective\": [{\"annualised_returns\": 0.0007236998854187604, \"annualised_returns_over_hodl\": 0.0018576440808879546, \"annualised_returns_over_uniform_hodl\": 0.001857644080889287, \"calmar\": 0.38289543979853696, \"daily_log_sharpe\": 0.08128124443243961, \"daily_returns\": -0.00041828686973826993, \"fee_revenue_over_value\": 2.1495377959234005e-08, \"jax_sharpe\": 0.08566609630028973, \"return\": 0.00013002773960679725, \"returns_over_hodl\": 0.00033360937564541615, \"returns_over_uniform_hodl\": 0.0003336093756456382, \"sharpe\": 0.08576064277857685, \"sterling\": 0.4229952467186126, \"ulcer\": -0.0007596662594185313}], \"train_return\": 0.00013002773960679725, \"train_returns_over_hodl\": 0.00033360937564541615, \"train_sharpe\": 0.08566609630028973, \"validation_return\": -8.438687291956182e-05, \"validation_returns_over_hodl\": 7.032108863258557e-05, \"validation_sharpe\": -0.26251928331900426}, {\"centeredness_margin\": 0.534816964042462, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0053140887878768694, \"annualised_returns_over_hodl\": 0.0035330550041474673, \"annualised_returns_over_uniform_hodl\": 0.0036698285035110523, \"calmar\": 1.28569135124944, \"daily_log_sharpe\": 0.6285295796667285, \"daily_returns\": -4.849816404114671e-05, \"fee_revenue_over_value\": 2.317922780966333e-08, \"jax_sharpe\": 0.5870917563343523, \"return\": 0.0030248402509460703, \"returns_over_hodl\": 0.0020118233725410217, \"returns_over_uniform_hodl\": 0.0020896448395690825, \"sharpe\": 0.6329654627898057, \"sterling\": 3.6209885979905083, \"ulcer\": -0.0004755698948777609}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 6.147640318139026e-05, \"optuna_trial_number\": 15, \"price_ratio\": 1.0103016632854462, \"shift_exponent\": 1.6001296106285563e-05, \"step\": 15, \"test_objective\": [{\"annualised_returns\": 0.0053140887878768694, \"annualised_returns_over_hodl\": 0.0035330550041474673, \"annualised_returns_over_uniform_hodl\": 0.0036698285035110523, \"calmar\": 1.28569135124944, \"daily_log_sharpe\": 0.6285295796667285, \"daily_returns\": -4.849816404114671e-05, \"fee_revenue_over_value\": 2.317922780966333e-08, \"jax_sharpe\": 0.5870917563343523, \"return\": 0.0030248402509460703, \"returns_over_hodl\": 0.0020118233725410217, \"returns_over_uniform_hodl\": 0.0020896448395690825, \"sharpe\": 0.6329654627898057, \"sterling\": 3.6209885979905083, \"ulcer\": -0.0004755698948777609}], \"train_objective\": [{\"annualised_returns\": 0.002525217288763537, \"annualised_returns_over_hodl\": 0.0036612028271154617, \"annualised_returns_over_uniform_hodl\": 0.003661202827116572, \"calmar\": 1.2681141156769806, \"daily_log_sharpe\": 0.27040598068503113, \"daily_returns\": -0.00041828685263915744, \"fee_revenue_over_value\": 1.9624346080748328e-08, \"jax_sharpe\": 0.27222979488977916, \"return\": 0.0004533731471603186, \"returns_over_hodl\": 0.0006570206018277069, \"returns_over_uniform_hodl\": 0.000657020601827929, \"sharpe\": 0.27504760569589587, \"sterling\": 1.408853440521711, \"ulcer\": -0.0007384166559925903}], \"train_return\": 0.0004533731471603186, \"train_returns_over_hodl\": 0.0006570206018277069, \"train_sharpe\": 0.27222979488977916, \"validation_return\": -1.4074940065889052e-05, \"validation_returns_over_hodl\": 0.0001423413506898008, \"validation_sharpe\": -0.04511386984529566}, {\"centeredness_margin\": 0.6271372983636601, \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -Infinity, \"optuna_trial_number\": 16, \"price_ratio\": 1.017340534351543, \"shift_exponent\": 59.1318877787967, \"step\": 16, \"test_objective\": -Infinity, \"train_objective\": -Infinity, \"train_return\": -Infinity, \"train_returns_over_hodl\": -Infinity, \"train_sharpe\": -Infinity, \"validation_return\": -Infinity, \"validation_returns_over_hodl\": -Infinity, \"validation_sharpe\": -Infinity}, {\"centeredness_margin\": 0.05624495749185493, \"continuous_test_metrics\": [{\"annualised_returns\": 0.004340377455966937, \"annualised_returns_over_hodl\": 0.0026218921280118934, \"annualised_returns_over_uniform_hodl\": 0.002697709743404886, \"calmar\": 1.058187849940496, \"daily_log_sharpe\": 0.5337379843464682, \"daily_returns\": -4.683808303846063e-05, \"fee_revenue_over_value\": 6.836923196238105e-08, \"jax_sharpe\": 0.48919646351279544, \"return\": 0.0024711081972059734, \"returns_over_hodl\": 0.0014932728981795762, \"returns_over_uniform_hodl\": 0.0015364290718222762, \"sharpe\": 0.5380641070929417, \"sterling\": 2.638217887135953, \"ulcer\": -0.0006299640173647679}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0001305573672417737, \"optuna_trial_number\": 17, \"price_ratio\": 1.0204050102203064, \"shift_exponent\": 3.4678646479472374e-05, \"step\": 17, \"test_objective\": [{\"annualised_returns\": 0.004340377455966937, \"annualised_returns_over_hodl\": 0.0026218921280118934, \"annualised_returns_over_uniform_hodl\": 0.002697709743404886, \"calmar\": 1.058187849940496, \"daily_log_sharpe\": 0.5337379843464682, \"daily_returns\": -4.683808303846063e-05, \"fee_revenue_over_value\": 6.836923196238105e-08, \"jax_sharpe\": 0.48919646351279544, \"return\": 0.0024711081972059734, \"returns_over_hodl\": 0.0014932728981795762, \"returns_over_uniform_hodl\": 0.0015364290718222762, \"sharpe\": 0.5380641070929417, \"sterling\": 2.638217887135953, \"ulcer\": -0.0006299640173647679}], \"train_objective\": [{\"annualised_returns\": 0.0007769905332926097, \"annualised_returns_over_hodl\": 0.0019109951136822012, \"annualised_returns_over_uniform_hodl\": 0.0019109951136822012, \"calmar\": 0.4109587512353234, \"daily_log_sharpe\": 0.08656140122088254, \"daily_returns\": -0.00042319025128732503, \"fee_revenue_over_value\": 2.1524960868392252e-08, \"jax_sharpe\": 0.0907220346205863, \"return\": 0.00013959946375408094, \"returns_over_hodl\": 0.0003431830481666065, \"returns_over_uniform_hodl\": 0.0003431830481666065, \"sharpe\": 0.09107776151288284, \"sterling\": 0.45386528116858443, \"ulcer\": -0.0007584719418487777}], \"train_return\": 0.00013959946375408094, \"train_returns_over_hodl\": 0.0003431830481666065, \"train_sharpe\": 0.0907220346205863, \"validation_return\": -8.245138650397887e-05, \"validation_returns_over_hodl\": 7.233840110743017e-05, \"validation_sharpe\": -0.2576288429466494}, {\"centeredness_margin\": 0.02618915085725712, \"continuous_test_metrics\": [{\"annualised_returns\": 0.005292692734730098, \"annualised_returns_over_hodl\": 0.003508722954396193, \"annualised_returns_over_uniform_hodl\": 0.0036484674450798504, \"calmar\": 1.3642817873350332, \"daily_log_sharpe\": 0.661567006454807, \"daily_returns\": -4.8579259647227754e-05, \"fee_revenue_over_value\": 5.876939201489565e-08, \"jax_sharpe\": 0.6230169375053684, \"return\": 0.003012675181399027, \"returns_over_hodl\": 0.0019979784304047232, \"returns_over_uniform_hodl\": 0.002077491112430385, \"sharpe\": 0.665739949829398, \"sterling\": 3.860355481417872, \"ulcer\": -0.00044212692594195466}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 3.937557194134134e-05, \"optuna_trial_number\": 18, \"price_ratio\": 1.01105160769199, \"shift_exponent\": 3.441813935232187e-05, \"step\": 18, \"test_objective\": [{\"annualised_returns\": 0.005292692734730098, \"annualised_returns_over_hodl\": 0.003508722954396193, \"annualised_returns_over_uniform_hodl\": 0.0036484674450798504, \"calmar\": 1.3642817873350332, \"daily_log_sharpe\": 0.661567006454807, \"daily_returns\": -4.8579259647227754e-05, \"fee_revenue_over_value\": 5.876939201489565e-08, \"jax_sharpe\": 0.6230169375053684, \"return\": 0.003012675181399027, \"returns_over_hodl\": 0.0019979784304047232, \"returns_over_uniform_hodl\": 0.002077491112430385, \"sharpe\": 0.665739949829398, \"sterling\": 3.860355481417872, \"ulcer\": -0.00044212692594195466}], \"train_objective\": [{\"annualised_returns\": 0.0023744460883270424, \"annualised_returns_over_hodl\": 0.0035102607841892564, \"annualised_returns_over_uniform_hodl\": 0.0035102607841905886, \"calmar\": 1.2083105155945268, \"daily_log_sharpe\": 0.257524491322361, \"daily_returns\": -0.0004272881078812173, \"fee_revenue_over_value\": 2.2005370044795188e-08, \"jax_sharpe\": 0.2603363815412292, \"return\": 0.0004263302568363603, \"returns_over_hodl\": 0.0006299722067835134, \"returns_over_uniform_hodl\": 0.0006299722067837354, \"sharpe\": 0.2621144011599262, \"sterling\": 1.3250245735561978, \"ulcer\": -0.0007490185918167388}], \"train_return\": 0.0004263302568363603, \"train_returns_over_hodl\": 0.0006299722067835134, \"train_sharpe\": 0.2603363815412292, \"validation_return\": -2.4489305784802795e-05, \"validation_returns_over_hodl\": 0.00013275108007859693, \"validation_sharpe\": -0.07881706779011492}, {\"centeredness_margin\": 0.38734475472873575, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0036597031457017426, \"annualised_returns_over_hodl\": 0.0019570368493122547, \"annualised_returns_over_uniform_hodl\": 0.0020181487227548534, \"calmar\": 0.8751314675056431, \"daily_log_sharpe\": 0.4164883637586103, \"daily_returns\": -4.6437865305643344e-05, \"fee_revenue_over_value\": 5.686097393809706e-08, \"jax_sharpe\": 0.3897927398075544, \"return\": 0.0020838838638146395, \"returns_over_hodl\": 0.0011147702279721283, \"returns_over_uniform_hodl\": 0.0011495657767675027, \"sharpe\": 0.4212319456644117, \"sterling\": 2.1260112498506376, \"ulcer\": -0.0006830336064526534}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0001696741961512066, \"optuna_trial_number\": 19, \"price_ratio\": 1.0253304462525643, \"shift_exponent\": 1.208845726379127e-05, \"step\": 19, \"test_objective\": [{\"annualised_returns\": 0.0036597031457017426, \"annualised_returns_over_hodl\": 0.0019570368493122547, \"annualised_returns_over_uniform_hodl\": 0.0020181487227548534, \"calmar\": 0.8751314675056431, \"daily_log_sharpe\": 0.4164883637586103, \"daily_returns\": -4.6437865305643344e-05, \"fee_revenue_over_value\": 5.686097393809706e-08, \"jax_sharpe\": 0.3897927398075544, \"return\": 0.0020838838638146395, \"returns_over_hodl\": 0.0011147702279721283, \"returns_over_uniform_hodl\": 0.0011495657767675027, \"sharpe\": 0.4212319456644117, \"sterling\": 2.1260112498506376, \"ulcer\": -0.0006830336064526534}], \"train_objective\": [{\"annualised_returns\": 0.0004102606181257684, \"annualised_returns_over_hodl\": 0.0015438496479902586, \"annualised_returns_over_uniform_hodl\": 0.0015438496479902586, \"calmar\": 0.21507750077032106, \"daily_log_sharpe\": 0.04633078153769644, \"daily_returns\": -0.00042224874648489816, \"fee_revenue_over_value\": 2.128242650618217e-08, \"jax_sharpe\": 0.05061396930735433, \"return\": 7.372132693261868e-05, \"returns_over_hodl\": 0.000277291501509902, \"returns_over_uniform_hodl\": 0.000277291501509902, \"sharpe\": 0.05079224207273589, \"sterling\": 0.2421545282332039, \"ulcer\": -0.0007618383820661472}], \"train_return\": 7.372132693261868e-05, \"train_returns_over_hodl\": 0.000277291501509902, \"train_sharpe\": 0.05061396930735434, \"validation_return\": -9.577327957388526e-05, \"validation_returns_over_hodl\": 5.84533109946328e-05, \"validation_sharpe\": -0.29542921300769825}, {\"centeredness_margin\": 0.7373481422994814, \"continuous_test_metrics\": [{\"annualised_returns\": 0.004782093169693136, \"annualised_returns_over_hodl\": 0.003026698124901639, \"annualised_returns_over_uniform_hodl\": 0.0031387030007232752, \"calmar\": 1.1100643994894577, \"daily_log_sharpe\": 0.6013137820363557, \"daily_returns\": -4.7824391934017327e-05, \"fee_revenue_over_value\": 1.3958815046599188e-08, \"jax_sharpe\": 0.5247348398830253, \"return\": 0.002722332551617246, \"returns_over_hodl\": 0.0017236764160108997, \"returns_over_uniform_hodl\": 0.0017874191908944237, \"sharpe\": 0.6055379446010655, \"sterling\": 3.0077104151859615, \"ulcer\": -0.0005582786389718081}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -5.580956007588401e-05, \"optuna_trial_number\": 20, \"price_ratio\": 1.0140283399050525, \"shift_exponent\": 1.3260684493289513e-05, \"step\": 20, \"test_objective\": [{\"annualised_returns\": 0.004782093169693136, \"annualised_returns_over_hodl\": 0.003026698124901639, \"annualised_returns_over_uniform_hodl\": 0.0031387030007232752, \"calmar\": 1.1100643994894577, \"daily_log_sharpe\": 0.6013137820363557, \"daily_returns\": -4.7824391934017327e-05, \"fee_revenue_over_value\": 1.3958815046599188e-08, \"jax_sharpe\": 0.5247348398830253, \"return\": 0.002722332551617246, \"returns_over_hodl\": 0.0017236764160108997, \"returns_over_uniform_hodl\": 0.0017874191908944237, \"sharpe\": 0.6055379446010655, \"sterling\": 3.0077104151859615, \"ulcer\": -0.0005582786389718081}], \"train_objective\": [{\"annualised_returns\": 0.0014706620635589474, \"annualised_returns_over_hodl\": 0.002605452659913521, \"annualised_returns_over_uniform_hodl\": 0.002605452659913521, \"calmar\": 0.7794720378050325, \"daily_log_sharpe\": 0.15899699107659154, \"daily_returns\": -0.00042539115931780424, \"fee_revenue_over_value\": 1.4870108493759951e-08, \"jax_sharpe\": 0.16390578864204156, \"return\": 0.00026415416729630437, \"returns_over_hodl\": 0.000467763105462371, \"returns_over_uniform_hodl\": 0.000467763105462371, \"sharpe\": 0.1636353176151234, \"sterling\": 0.8361008211352461, \"ulcer\": -0.0007568930890868834}], \"train_return\": 0.00026415416729630437, \"train_returns_over_hodl\": 0.000467763105462371, \"train_sharpe\": 0.16390578864204156, \"validation_return\": -5.283757565455183e-05, \"validation_returns_over_hodl\": 0.00010310259159451718, \"validation_sharpe\": -0.16975069228808268}, {\"centeredness_margin\": 0.7538875716302246, \"continuous_test_metrics\": [{\"annualised_returns\": 0.005197126651665718, \"annualised_returns_over_hodl\": 0.003392477313631126, \"annualised_returns_over_uniform_hodl\": 0.0035530576669122738, \"calmar\": 1.185623301453069, \"daily_log_sharpe\": 0.5415542220766973, \"daily_returns\": -4.9147861767092814e-05, \"fee_revenue_over_value\": 1.239812894067734e-08, \"jax_sharpe\": 0.5193061199201056, \"return\": 0.00295833819081448, \"returns_over_hodl\": 0.0019318326400725727, \"returns_over_uniform_hodl\": 0.002023204784304289, \"sharpe\": 0.5465301734284792, \"sterling\": 3.273956908615545, \"ulcer\": -0.0005473226807016279}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -1.788220353519665e-05, \"optuna_trial_number\": 21, \"price_ratio\": 1.0121109456606732, \"shift_exponent\": 1.3400509094066617e-05, \"step\": 21, \"test_objective\": [{\"annualised_returns\": 0.005197126651665718, \"annualised_returns_over_hodl\": 0.003392477313631126, \"annualised_returns_over_uniform_hodl\": 0.0035530576669122738, \"calmar\": 1.185623301453069, \"daily_log_sharpe\": 0.5415542220766973, \"daily_returns\": -4.9147861767092814e-05, \"fee_revenue_over_value\": 1.239812894067734e-08, \"jax_sharpe\": 0.5193061199201056, \"return\": 0.00295833819081448, \"returns_over_hodl\": 0.0019318326400725727, \"returns_over_uniform_hodl\": 0.002023204784304289, \"sharpe\": 0.5465301734284792, \"sterling\": 3.273956908615545, \"ulcer\": -0.0005473226807016279}], \"train_objective\": [{\"annualised_returns\": 0.0018454959920195524, \"annualised_returns_over_hodl\": 0.002980711321753482, \"annualised_returns_over_uniform_hodl\": 0.002980711321753482, \"calmar\": 0.9555370822886182, \"daily_log_sharpe\": 0.20244257285434616, \"daily_returns\": -0.0004265061429431668, \"fee_revenue_over_value\": 1.3917455223831695e-08, \"jax_sharpe\": 0.20579984862091202, \"return\": 0.00033142938142627365, \"returns_over_hodl\": 0.0005350520138101, \"returns_over_uniform_hodl\": 0.0005350520138101, \"sharpe\": 0.20699753495730236, \"sterling\": 1.0334889780710468, \"ulcer\": -0.0007543554274812417}], \"train_return\": 0.00033142938142627365, \"train_returns_over_hodl\": 0.0005350520138101, \"train_sharpe\": 0.20579984862091202, \"validation_return\": -4.597954498597456e-05, \"validation_returns_over_hodl\": 0.00011184782116058223, \"validation_sharpe\": -0.14856773787181476}, {\"centeredness_margin\": 0.7799016066430687, \"continuous_test_metrics\": [{\"annualised_returns\": 0.003925107046516718, \"annualised_returns_over_hodl\": 0.0020591653590120718, \"annualised_returns_over_uniform_hodl\": 0.0022831185372500507, \"calmar\": 0.7731006064259414, \"daily_log_sharpe\": 0.4252770611423177, \"daily_returns\": -5.0884357319968274e-05, \"fee_revenue_over_value\": 1.5853270623910297e-08, \"jax_sharpe\": 0.3664790613725467, \"return\": 0.002234881175519954, \"returns_over_hodl\": 0.0011729190959048896, \"returns_over_uniform_hodl\": 0.0013004223023354022, \"sharpe\": 0.4303442461978311, \"sterling\": 2.100490720181369, \"ulcer\": -0.0006559718181975635}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -8.052585025898063e-05, \"optuna_trial_number\": 22, \"price_ratio\": 1.012580616461133, \"shift_exponent\": 3.6058655726715554e-05, \"step\": 22, \"test_objective\": [{\"annualised_returns\": 0.003925107046516718, \"annualised_returns_over_hodl\": 0.0020591653590120718, \"annualised_returns_over_uniform_hodl\": 0.0022831185372500507, \"calmar\": 0.7731006064259414, \"daily_log_sharpe\": 0.4252770611423177, \"daily_returns\": -5.0884357319968274e-05, \"fee_revenue_over_value\": 1.5853270623910297e-08, \"jax_sharpe\": 0.3664790613725467, \"return\": 0.002234881175519954, \"returns_over_hodl\": 0.0011729190959048896, \"returns_over_uniform_hodl\": 0.0013004223023354022, \"sharpe\": 0.4303442461978311, \"sterling\": 2.100490720181369, \"ulcer\": -0.0006559718181975635}], \"train_objective\": [{\"annualised_returns\": 0.0012684766867847586, \"annualised_returns_over_hodl\": 0.0024030381820063784, \"annualised_returns_over_uniform_hodl\": 0.0024030381820063784, \"calmar\": 0.6621768820311253, \"daily_log_sharpe\": 0.13668661093081927, \"daily_returns\": -0.0004262016082916001, \"fee_revenue_over_value\": 1.2184330301164086e-08, \"jax_sharpe\": 0.14104596748887735, \"return\": 0.0002278573490410718, \"returns_over_hodl\": 0.00043145889880236155, \"returns_over_uniform_hodl\": 0.00043145889880236155, \"sharpe\": 0.14134368931719468, \"sterling\": 0.7022914109257686, \"ulcer\": -0.000771026240043555}], \"train_return\": 0.0002278573490410718, \"train_returns_over_hodl\": 0.00043145889880236155, \"train_sharpe\": 0.14104596748887735, \"validation_return\": -6.975919262730557e-05, \"validation_returns_over_hodl\": 9.08768992251563e-05, \"validation_sharpe\": -0.22056278078680802}, {\"centeredness_margin\": 0.5917285669570868, \"continuous_test_metrics\": [{\"annualised_returns\": 0.004244596569223713, \"annualised_returns_over_hodl\": 0.0025189582687681344, \"annualised_returns_over_uniform_hodl\": 0.0026020855128845444, \"calmar\": 0.9737097663128215, \"daily_log_sharpe\": 0.45520422117382675, \"daily_returns\": -4.703778857915465e-05, \"fee_revenue_over_value\": 3.0738572170540334e-08, \"jax_sharpe\": 0.4279323949028855, \"return\": 0.0024166268643539546, \"returns_over_hodl\": 0.0014346796304323117, \"returns_over_uniform_hodl\": 0.001481998536009721, \"sharpe\": 0.46016378782042083, \"sterling\": 2.592253491759179, \"ulcer\": -0.0005813242790474491}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -4.92481134893169e-05, \"optuna_trial_number\": 23, \"price_ratio\": 1.0143081880034814, \"shift_exponent\": 2.595676316182535e-05, \"step\": 23, \"test_objective\": [{\"annualised_returns\": 0.004244596569223713, \"annualised_returns_over_hodl\": 0.0025189582687681344, \"annualised_returns_over_uniform_hodl\": 0.0026020855128845444, \"calmar\": 0.9737097663128215, \"daily_log_sharpe\": 0.45520422117382675, \"daily_returns\": -4.703778857915465e-05, \"fee_revenue_over_value\": 3.0738572170540334e-08, \"jax_sharpe\": 0.4279323949028855, \"return\": 0.0024166268643539546, \"returns_over_hodl\": 0.0014346796304323117, \"returns_over_uniform_hodl\": 0.001481998536009721, \"sharpe\": 0.46016378782042083, \"sterling\": 2.592253491759179, \"ulcer\": -0.0005813242790474491}], \"train_objective\": [{\"annualised_returns\": 0.001467992634155646, \"annualised_returns_over_hodl\": 0.0026027802057158045, \"annualised_returns_over_uniform_hodl\": 0.0026027802057158045, \"calmar\": 0.7795539064928655, \"daily_log_sharpe\": 0.16000491131020933, \"daily_returns\": -0.00042525341283996453, \"fee_revenue_over_value\": 1.9822780163937054e-08, \"jax_sharpe\": 0.16303159095552494, \"return\": 0.0002636749838944574, \"returns_over_hodl\": 0.0004672838245203259, \"returns_over_uniform_hodl\": 0.0004672838245203259, \"sharpe\": 0.16459569597398324, \"sterling\": 0.8459313995310164, \"ulcer\": -0.0007453486046716789}], \"train_return\": 0.0002636749838944574, \"train_returns_over_hodl\": 0.0004672838245203259, \"train_sharpe\": 0.16303159095552497, \"validation_return\": -5.0968287999331174e-05, \"validation_returns_over_hodl\": 0.00010275795415592981, \"validation_sharpe\": -0.16625019117402895}, {\"centeredness_margin\": 0.7162404842489385, \"continuous_test_metrics\": [{\"annualised_returns\": 0.00153424384996792, \"annualised_returns_over_hodl\": -0.0002669075566078538, \"annualised_returns_over_uniform_hodl\": -0.00010383423822457605, \"calmar\": 0.26335448582254956, \"daily_log_sharpe\": 0.1576348274256455, \"daily_returns\": -4.923211709403038e-05, \"fee_revenue_over_value\": 3.288830519231097e-08, \"jax_sharpe\": 0.1310814615132343, \"return\": 0.0008740176220431994, \"returns_over_hodl\": -0.00015210896899386928, \"returns_over_uniform_hodl\": -5.9172415814989776e-05, \"sharpe\": 0.16372722808455661, \"sterling\": 0.6969510405496425, \"ulcer\": -0.0008137795385833267}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00024298164924708987, \"optuna_trial_number\": 24, \"price_ratio\": 1.0139447466678104, \"shift_exponent\": 0.00013324512500718927, \"step\": 24, \"test_objective\": [{\"annualised_returns\": 0.00153424384996792, \"annualised_returns_over_hodl\": -0.0002669075566078538, \"annualised_returns_over_uniform_hodl\": -0.00010383423822457605, \"calmar\": 0.26335448582254956, \"daily_log_sharpe\": 0.1576348274256455, \"daily_returns\": -4.923211709403038e-05, \"fee_revenue_over_value\": 3.288830519231097e-08, \"jax_sharpe\": 0.1310814615132343, \"return\": 0.0008740176220431994, \"returns_over_hodl\": -0.00015210896899386928, \"returns_over_uniform_hodl\": -5.9172415814989776e-05, \"sharpe\": 0.16372722808455661, \"sterling\": 0.6969510405496425, \"ulcer\": -0.0008137795385833267}], \"train_objective\": [{\"annualised_returns\": -0.00026150120856704984, \"annualised_returns_over_hodl\": 0.0008713266317457169, \"annualised_returns_over_uniform_hodl\": 0.0008713266317470492, \"calmar\": -0.1343311459936119, \"daily_log_sharpe\": -0.02899715208795857, \"daily_returns\": -0.00041828686971696335, \"fee_revenue_over_value\": 1.4784614126879526e-08, \"jax_sharpe\": -0.024033254443118303, \"return\": -4.700311726357764e-05, \"returns_over_hodl\": 0.00015654248322904962, \"returns_over_uniform_hodl\": 0.00015654248322927167, \"sharpe\": -0.02444183573880608, \"sterling\": -0.14524777126627783, \"ulcer\": -0.0007825017723650301}], \"train_return\": -4.700311726357764e-05, \"train_returns_over_hodl\": 0.00015654248322904962, \"train_sharpe\": -0.024033254443118303, \"validation_return\": -5.561012968779977e-05, \"validation_returns_over_hodl\": 0.00010541038246958401, \"validation_sharpe\": -0.17290456888609157}, {\"centeredness_margin\": 0.04079298187588158, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0014319915938789674, \"annualised_returns_over_hodl\": -0.0003445642504747992, \"annualised_returns_over_uniform_hodl\": -0.00020591925372126507, \"calmar\": 0.3689592777014508, \"daily_log_sharpe\": 0.19317868733721943, \"daily_returns\": -4.856385278199563e-05, \"fee_revenue_over_value\": 6.966838753928606e-08, \"jax_sharpe\": 0.14708915213685245, \"return\": 0.0008157851712671249, \"returns_over_hodl\": -0.0001963683198586974, \"returns_over_uniform_hodl\": -0.00011735057210249256, \"sharpe\": 0.19819042322149732, \"sterling\": 0.8907234181056677, \"ulcer\": -0.0005643576907442256}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 3.787024112540949e-05, \"optuna_trial_number\": 25, \"price_ratio\": 1.0110966317923087, \"shift_exponent\": 0.012268486459997585, \"step\": 25, \"test_objective\": [{\"annualised_returns\": 0.0014319915938789674, \"annualised_returns_over_hodl\": -0.0003445642504747992, \"annualised_returns_over_uniform_hodl\": -0.00020591925372126507, \"calmar\": 0.3689592777014508, \"daily_log_sharpe\": 0.19317868733721943, \"daily_returns\": -4.856385278199563e-05, \"fee_revenue_over_value\": 6.966838753928606e-08, \"jax_sharpe\": 0.14708915213685245, \"return\": 0.0008157851712671249, \"returns_over_hodl\": -0.0001963683198586974, \"returns_over_uniform_hodl\": -0.00011735057210249256, \"sharpe\": 0.19819042322149732, \"sterling\": 0.8907234181056677, \"ulcer\": -0.0005643576907442256}], \"train_objective\": [{\"annualised_returns\": 0.0023602979247219213, \"annualised_returns_over_hodl\": 0.003496096588959219, \"annualised_returns_over_uniform_hodl\": 0.003496096588959219, \"calmar\": 1.202004695022952, \"daily_log_sharpe\": 0.257143437676534, \"daily_returns\": -0.000427251835689078, \"fee_revenue_over_value\": 2.200299257498476e-08, \"jax_sharpe\": 0.2573521927741892, \"return\": 0.0004237924176924146, \"returns_over_hodl\": 0.0006274338510494637, \"returns_over_uniform_hodl\": 0.0006274338510494637, \"sharpe\": 0.261705445306399, \"sterling\": 1.3197588687789321, \"ulcer\": -0.000745039813924573}], \"train_return\": 0.0004237924176924146, \"train_returns_over_hodl\": 0.0006274338510494637, \"train_sharpe\": 0.2573521927741892, \"validation_return\": -2.500216041989578e-05, \"validation_returns_over_hodl\": 0.00013221653881045903, \"validation_sharpe\": -0.07997015278183044}, {\"centeredness_margin\": 0.06566039870454846, \"continuous_test_metrics\": [{\"annualised_returns\": 0.003975732548218147, \"annualised_returns_over_hodl\": 0.002183539292742065, \"annualised_returns_over_uniform_hodl\": 0.0023336612374638133, \"calmar\": 0.9738436823587432, \"daily_log_sharpe\": 0.5523495835948069, \"daily_returns\": -4.886761643057823e-05, \"fee_revenue_over_value\": 6.623548374621087e-08, \"jax_sharpe\": 0.4486266401713105, \"return\": 0.0022636817962884415, \"returns_over_hodl\": 0.0012437303893979568, \"returns_over_uniform_hodl\": 0.0013291960701213856, \"sharpe\": 0.5563109263262898, \"sterling\": 2.582982781401223, \"ulcer\": -0.00048682058156571665}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 6.755312619437515e-05, \"optuna_trial_number\": 26, \"price_ratio\": 1.0102715408891025, \"shift_exponent\": 0.000390822635876792, \"step\": 26, \"test_objective\": [{\"annualised_returns\": 0.003975732548218147, \"annualised_returns_over_hodl\": 0.002183539292742065, \"annualised_returns_over_uniform_hodl\": 0.0023336612374638133, \"calmar\": 0.9738436823587432, \"daily_log_sharpe\": 0.5523495835948069, \"daily_returns\": -4.886761643057823e-05, \"fee_revenue_over_value\": 6.623548374621087e-08, \"jax_sharpe\": 0.4486266401713105, \"return\": 0.0022636817962884415, \"returns_over_hodl\": 0.0012437303893979568, \"returns_over_uniform_hodl\": 0.0013291960701213856, \"sharpe\": 0.5563109263262898, \"sterling\": 2.582982781401223, \"ulcer\": -0.00048682058156571665}], \"train_objective\": [{\"annualised_returns\": 0.0026393063509124737, \"annualised_returns_over_hodl\": 0.003775421166337223, \"annualised_returns_over_uniform_hodl\": 0.003775421166337223, \"calmar\": 1.3246540581694939, \"daily_log_sharpe\": 0.2744903859854288, \"daily_returns\": -0.0004182868696941536, \"fee_revenue_over_value\": 2.20466638062847e-08, \"jax_sharpe\": 0.2764078998497612, \"return\": 0.0004738343737722417, \"returns_over_hodl\": 0.0006774859934282063, \"returns_over_uniform_hodl\": 0.0006774859934282063, \"sharpe\": 0.27925577149274183, \"sterling\": 1.4647349508895144, \"ulcer\": -0.0007450453040258635}], \"train_return\": 0.0004738343737722417, \"train_returns_over_hodl\": 0.0006774859934282063, \"train_sharpe\": 0.2764078998497612, \"validation_return\": -1.4889615703173043e-05, \"validation_returns_over_hodl\": 0.0001427566683709358, \"validation_sharpe\": -0.04840182128056577}, {\"centeredness_margin\": 0.3397766488086633, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0009744417279229367, \"annualised_returns_over_hodl\": -0.0006617701861841851, \"annualised_returns_over_uniform_hodl\": -0.0006627207654253953, \"calmar\": 0.18957347981897052, \"daily_log_sharpe\": 0.15353718298347274, \"daily_returns\": -4.474293306456692e-05, \"fee_revenue_over_value\": 5.4010352285153336e-08, \"jax_sharpe\": 0.10157867859092669, \"return\": 0.0005551801248719901, \"returns_over_hodl\": -0.0003771707843120975, \"returns_over_uniform_hodl\": -0.0003777126368346151, \"sharpe\": 0.15819158743554515, \"sterling\": 0.542293299858641, \"ulcer\": -0.0006057439504529188}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 3.8043081162841833e-06, \"optuna_trial_number\": 27, \"price_ratio\": 1.010527397112252, \"shift_exponent\": 0.0002502943906104179, \"step\": 27, \"test_objective\": [{\"annualised_returns\": 0.0009744417279229367, \"annualised_returns_over_hodl\": -0.0006617701861841851, \"annualised_returns_over_uniform_hodl\": -0.0006627207654253953, \"calmar\": 0.18957347981897052, \"daily_log_sharpe\": 0.15353718298347274, \"daily_returns\": -4.474293306456692e-05, \"fee_revenue_over_value\": 5.4010352285153336e-08, \"jax_sharpe\": 0.10157867859092669, \"return\": 0.0005551801248719901, \"returns_over_hodl\": -0.0003771707843120975, \"returns_over_uniform_hodl\": -0.0003777126368346151, \"sharpe\": 0.15819158743554515, \"sterling\": 0.542293299858641, \"ulcer\": -0.0006057439504529188}], \"train_objective\": [{\"annualised_returns\": 0.00147521181771193, \"annualised_returns_over_hodl\": 0.002610007569504136, \"annualised_returns_over_uniform_hodl\": 0.002610007569504136, \"calmar\": 0.7439174148272298, \"daily_log_sharpe\": 0.17557848891176794, \"daily_returns\": -0.0004277332400712069, \"fee_revenue_over_value\": 2.0630878263721186e-08, \"jax_sharpe\": 0.17986839438856753, \"return\": 0.00026497088124433077, \"returns_over_hodl\": 0.00046857998565696946, \"returns_over_uniform_hodl\": 0.00046857998565696946, \"sharpe\": 0.17978200839402353, \"sterling\": 0.8735601237399009, \"ulcer\": -0.000680586846793901}], \"train_return\": 0.00026497088124433077, \"train_returns_over_hodl\": 0.00046857998565696946, \"train_sharpe\": 0.1798683943885675, \"validation_return\": -4.542159874554308e-06, \"validation_returns_over_hodl\": 0.0001393041549961893, \"validation_sharpe\": -0.0139899694319445}, {\"centeredness_margin\": 0.39637713313093076, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0016311388299181662, \"annualised_returns_over_hodl\": -5.911620740606249e-05, \"annualised_returns_over_uniform_hodl\": -7.0977366731783675e-06, \"calmar\": 0.3046783234524405, \"daily_log_sharpe\": 0.17962254365405236, \"daily_returns\": -4.619239327115717e-05, \"fee_revenue_over_value\": 4.754278972408896e-08, \"jax_sharpe\": 0.1492555877603308, \"return\": 0.0009291967552280678, \"returns_over_hodl\": -3.3688455961744523e-05, \"returns_over_uniform_hodl\": -4.044730281260733e-06, \"sharpe\": 0.18532124881643003, \"sterling\": 0.8646606260342938, \"ulcer\": -0.0006639807167283642}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -1.1320370173364958e-05, \"optuna_trial_number\": 28, \"price_ratio\": 1.0117615377376021, \"shift_exponent\": 0.00014637161538734857, \"step\": 28, \"test_objective\": [{\"annualised_returns\": 0.0016311388299181662, \"annualised_returns_over_hodl\": -5.911620740606249e-05, \"annualised_returns_over_uniform_hodl\": -7.0977366731783675e-06, \"calmar\": 0.3046783234524405, \"daily_log_sharpe\": 0.17962254365405236, \"daily_returns\": -4.619239327115717e-05, \"fee_revenue_over_value\": 4.754278972408896e-08, \"jax_sharpe\": 0.1492555877603308, \"return\": 0.0009291967552280678, \"returns_over_hodl\": -3.3688455961744523e-05, \"returns_over_uniform_hodl\": -4.044730281260733e-06, \"sharpe\": 0.18532124881643003, \"sterling\": 0.8646606260342938, \"ulcer\": -0.0006639807167283642}], \"train_objective\": [{\"annualised_returns\": 0.00160830553473712, \"annualised_returns_over_hodl\": 0.002743252098234672, \"annualised_returns_over_uniform_hodl\": 0.002743252098234672, \"calmar\": 0.8275917637126533, \"daily_log_sharpe\": 0.18248471786448806, \"daily_returns\": -0.00042674850702263103, \"fee_revenue_over_value\": 2.0572327696605314e-08, \"jax_sharpe\": 0.18732558034969485, \"return\": 0.00028886082443047023, \"returns_over_hodl\": 0.0004924747917645078, \"returns_over_uniform_hodl\": 0.0004924747917645078, \"sharpe\": 0.18689465010752546, \"sterling\": 0.9350442431182815, \"ulcer\": -0.0007167075169005768}], \"train_return\": 0.00028886082443047023, \"train_returns_over_hodl\": 0.0004924747917645078, \"train_sharpe\": 0.18732558034969485, \"validation_return\": -2.4802009538138492e-05, \"validation_returns_over_hodl\": 0.00012479396871523107, \"validation_sharpe\": -0.08362609613215669}, {\"centeredness_margin\": 0.05923328143566958, \"continuous_test_metrics\": [{\"annualised_returns\": 0.003453705988319733, \"annualised_returns_over_hodl\": 0.0017170190898099236, \"annualised_returns_over_uniform_hodl\": 0.001812488487881403, \"calmar\": 0.8609350999606568, \"daily_log_sharpe\": 0.43435552870562766, \"daily_returns\": -4.737673597317784e-05, \"fee_revenue_over_value\": 6.903587328110494e-08, \"jax_sharpe\": 0.3779542984414848, \"return\": 0.0019666732311560686, \"returns_over_hodl\": 0.0009781013767886648, \"returns_over_uniform_hodl\": 0.0010324644283872253, \"sharpe\": 0.4386956978651518, \"sterling\": 2.140723853931127, \"ulcer\": -0.0005945912247394703}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -7.790305874251502e-05, \"optuna_trial_number\": 29, \"price_ratio\": 1.0161718778721502, \"shift_exponent\": 0.2117870592058916, \"step\": 29, \"test_objective\": [{\"annualised_returns\": 0.003453705988319733, \"annualised_returns_over_hodl\": 0.0017170190898099236, \"annualised_returns_over_uniform_hodl\": 0.001812488487881403, \"calmar\": 0.8609350999606568, \"daily_log_sharpe\": 0.43435552870562766, \"daily_returns\": -4.737673597317784e-05, \"fee_revenue_over_value\": 6.903587328110494e-08, \"jax_sharpe\": 0.3779542984414848, \"return\": 0.0019666732311560686, \"returns_over_hodl\": 0.0009781013767886648, \"returns_over_uniform_hodl\": 0.0010324644283872253, \"sharpe\": 0.4386956978651518, \"sterling\": 2.140723853931127, \"ulcer\": -0.0005945912247394703}], \"train_objective\": [{\"annualised_returns\": 0.001270837315665574, \"annualised_returns_over_hodl\": 0.002405401485772396, \"annualised_returns_over_uniform_hodl\": 0.002405401485772396, \"calmar\": 0.6817496172851633, \"daily_log_sharpe\": 0.13741577125751217, \"daily_returns\": -0.0004182868782966659, \"fee_revenue_over_value\": 2.1739075278199572e-08, \"jax_sharpe\": 0.1411313094218755, \"return\": 0.0002282811696550091, \"returns_over_hodl\": 0.00043188280568706716, \"returns_over_uniform_hodl\": 0.00043188280568706716, \"sharpe\": 0.14205380379308527, \"sterling\": 0.731295592794189, \"ulcer\": -0.0007555632081561704}], \"train_return\": 0.0002282811696550091, \"train_returns_over_hodl\": 0.00043188280568706716, \"train_sharpe\": 0.1411313094218755, \"validation_return\": -6.452097496012499e-05, \"validation_returns_over_hodl\": 9.102686769324464e-05, \"validation_sharpe\": -0.20455153325072056}, {\"centeredness_margin\": 0.060439804670370745, \"continuous_test_metrics\": [{\"annualised_returns\": 0.002801552121672879, \"annualised_returns_over_hodl\": 0.0011194031944683491, \"annualised_returns_over_uniform_hodl\": 0.0011614012637035653, \"calmar\": 0.6469033678721214, \"daily_log_sharpe\": 0.303500038326635, \"daily_returns\": -4.591686607268633e-05, \"fee_revenue_over_value\": 6.462497504999784e-08, \"jax_sharpe\": 0.280258353208181, \"return\": 0.0015955349536906915, \"returns_over_hodl\": 0.0006377508549744171, \"returns_over_uniform_hodl\": 0.0006616721910197576, \"sharpe\": 0.3086055846828765, \"sterling\": 1.4971933375788202, \"ulcer\": -0.0008012003479584448}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00025390615978918566, \"optuna_trial_number\": 30, \"price_ratio\": 1.0459018055123313, \"shift_exponent\": 2.411910430988103, \"step\": 30, \"test_objective\": [{\"annualised_returns\": 0.002801552121672879, \"annualised_returns_over_hodl\": 0.0011194031944683491, \"annualised_returns_over_uniform_hodl\": 0.0011614012637035653, \"calmar\": 0.6469033678721214, \"daily_log_sharpe\": 0.303500038326635, \"daily_returns\": -4.591686607268633e-05, \"fee_revenue_over_value\": 6.462497504999784e-08, \"jax_sharpe\": 0.280258353208181, \"return\": 0.0015955349536906915, \"returns_over_hodl\": 0.0006377508549744171, \"returns_over_uniform_hodl\": 0.0006616721910197576, \"sharpe\": 0.3086055846828765, \"sterling\": 1.4971933375788202, \"ulcer\": -0.0008012003479584448}], \"train_objective\": [{\"annualised_returns\": -0.00034152333147152714, \"annualised_returns_over_hodl\": 0.0007912138338408425, \"annualised_returns_over_uniform_hodl\": 0.0007912138338408425, \"calmar\": -0.17500465352084887, \"daily_log_sharpe\": -0.0388980084280256, \"daily_returns\": -0.0004205006410805963, \"fee_revenue_over_value\": 2.0339901538702867e-08, \"jax_sharpe\": -0.03439850173956965, \"return\": -6.138858122972657e-05, \"returns_over_hodl\": 0.00014215409102735777, \"returns_over_uniform_hodl\": 0.00014215409102735777, \"sharpe\": -0.034464594324305924, \"sterling\": -0.20496280309634524, \"ulcer\": -0.0007727658041891594}], \"train_return\": -6.138858122972657e-05, \"train_returns_over_hodl\": 0.00014215409102735777, \"train_sharpe\": -0.03439850173956965, \"validation_return\": -0.0001200438756556732, \"validation_returns_over_hodl\": 3.313694906892373e-05, \"validation_sharpe\": -0.3616501260938264}, {\"centeredness_margin\": 0.03713493968302037, \"continuous_test_metrics\": [{\"annualised_returns\": 0.005549580158815237, \"annualised_returns_over_hodl\": 0.0037541077319436233, \"annualised_returns_over_uniform_hodl\": 0.0039049347121276057, \"calmar\": 1.442804594841702, \"daily_log_sharpe\": 0.7032212890779918, \"daily_returns\": -4.888042175697694e-05, \"fee_revenue_over_value\": 5.623706226762951e-08, \"jax_sharpe\": 0.6619125397975072, \"return\": 0.0031587252934024423, \"returns_over_hodl\": 0.002137595813663795, \"returns_over_uniform_hodl\": 0.0022234050509422065, \"sharpe\": 0.7072989940818653, \"sterling\": 4.187262235716605, \"ulcer\": -0.0004178507530500047}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 6.880483519998426e-05, \"optuna_trial_number\": 31, \"price_ratio\": 1.0102394435218087, \"shift_exponent\": 2.2665545348679854e-05, \"step\": 31, \"test_objective\": [{\"annualised_returns\": 0.005549580158815237, \"annualised_returns_over_hodl\": 0.0037541077319436233, \"annualised_returns_over_uniform_hodl\": 0.0039049347121276057, \"calmar\": 1.442804594841702, \"daily_log_sharpe\": 0.7032212890779918, \"daily_returns\": -4.888042175697694e-05, \"fee_revenue_over_value\": 5.623706226762951e-08, \"jax_sharpe\": 0.6619125397975072, \"return\": 0.0031587252934024423, \"returns_over_hodl\": 0.002137595813663795, \"returns_over_uniform_hodl\": 0.0022234050509422065, \"sharpe\": 0.7072989940818653, \"sterling\": 4.187262235716605, \"ulcer\": -0.0004178507530500047}], \"train_objective\": [{\"annualised_returns\": 0.0026510725367892007, \"annualised_returns_over_hodl\": 0.0037872006847634587, \"annualised_returns_over_uniform_hodl\": 0.0037872006847634587, \"calmar\": 1.3297487089491116, \"daily_log_sharpe\": 0.2783135636778685, \"daily_returns\": -0.00042799717640505474, \"fee_revenue_over_value\": 2.2048367093862865e-08, \"jax_sharpe\": 0.28445382649925594, \"return\": 0.0004759444638764432, \"returns_over_hodl\": 0.0006795965130521608, \"returns_over_uniform_hodl\": 0.0006795965130521608, \"sharpe\": 0.2830543866773521, \"sterling\": 1.4682642788410833, \"ulcer\": -0.0007544569784971218}], \"train_return\": 0.0004759444638764432, \"train_returns_over_hodl\": 0.0006795965130521608, \"train_sharpe\": 0.28445382649925594, \"validation_return\": -1.4463214276116965e-05, \"validation_returns_over_hodl\": 0.00014320109528020986, \"validation_sharpe\": -0.04634676013012011}, {\"centeredness_margin\": 0.12764043674062714, \"continuous_test_metrics\": [{\"annualised_returns\": 0.003712286907352702, \"annualised_returns_over_hodl\": 0.0019267647721501469, \"annualised_returns_over_uniform_hodl\": 0.0020706464800495095, \"calmar\": 0.8738981726210676, \"daily_log_sharpe\": 0.39349875858477007, \"daily_returns\": -4.869823927946053e-05, \"fee_revenue_over_value\": 6.582495294841682e-08, \"jax_sharpe\": 0.3601093380239897, \"return\": 0.0021138019196078606, \"returns_over_hodl\": 0.0010975337379033334, \"returns_over_uniform_hodl\": 0.001179455937709406, \"sharpe\": 0.39866766599678544, \"sterling\": 2.251227932570609, \"ulcer\": -0.0005538215095634818}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 5.1003826411924474e-05, \"optuna_trial_number\": 32, \"price_ratio\": 1.0107157848439716, \"shift_exponent\": 0.00038617580664964475, \"step\": 32, \"test_objective\": [{\"annualised_returns\": 0.003712286907352702, \"annualised_returns_over_hodl\": 0.0019267647721501469, \"annualised_returns_over_uniform_hodl\": 0.0020706464800495095, \"calmar\": 0.8738981726210676, \"daily_log_sharpe\": 0.39349875858477007, \"daily_returns\": -4.869823927946053e-05, \"fee_revenue_over_value\": 6.582495294841682e-08, \"jax_sharpe\": 0.3601093380239897, \"return\": 0.0021138019196078606, \"returns_over_hodl\": 0.0010975337379033334, \"returns_over_uniform_hodl\": 0.001179455937709406, \"sharpe\": 0.39866766599678544, \"sterling\": 2.251227932570609, \"ulcer\": -0.0005538215095634818}], \"train_objective\": [{\"annualised_returns\": 0.0024837388946323813, \"annualised_returns_over_hodl\": 0.0036196774328138837, \"annualised_returns_over_uniform_hodl\": 0.003619677432814994, \"calmar\": 1.2567089342579687, \"daily_log_sharpe\": 0.2664576899182112, \"daily_returns\": -0.00042756826848956784, \"fee_revenue_over_value\": 2.2023123365450877e-08, \"jax_sharpe\": 0.26838888993890536, \"return\": 0.0004459337588433865, \"returns_over_hodl\": 0.0006495796991847769, \"returns_over_uniform_hodl\": 0.0006495796991849989, \"sharpe\": 0.27109066415886507, \"sterling\": 1.3818759254872135, \"ulcer\": -0.000746687877925331}], \"train_return\": 0.0004459337588433865, \"train_returns_over_hodl\": 0.0006495796991847769, \"train_sharpe\": 0.2683888899389054, \"validation_return\": -2.0527690301030965e-05, \"validation_returns_over_hodl\": 0.00013688019639435112, \"validation_sharpe\": -0.06622453146245177}, {\"centeredness_margin\": 0.01755190133917555, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0050538322498927535, \"annualised_returns_over_hodl\": 0.0032866322919455904, \"annualised_returns_over_uniform_hodl\": 0.003409997632981776, \"calmar\": 1.3145695853158386, \"daily_log_sharpe\": 0.6158218609193377, \"daily_returns\": -4.813341911575088e-05, \"fee_revenue_over_value\": 6.426366708649339e-08, \"jax_sharpe\": 0.5830468750753665, \"return\": 0.0028768596552486425, \"returns_over_hodl\": 0.0018716021156941487, \"returns_over_uniform_hodl\": 0.0019418022172983385, \"sharpe\": 0.6201135489304256, \"sterling\": 3.594222358391935, \"ulcer\": -0.00046898935357678325}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -4.186639123987535e-06, \"optuna_trial_number\": 33, \"price_ratio\": 1.012521701210514, \"shift_exponent\": 7.529807211612141e-05, \"step\": 33, \"test_objective\": [{\"annualised_returns\": 0.0050538322498927535, \"annualised_returns_over_hodl\": 0.0032866322919455904, \"annualised_returns_over_uniform_hodl\": 0.003409997632981776, \"calmar\": 1.3145695853158386, \"daily_log_sharpe\": 0.6158218609193377, \"daily_returns\": -4.813341911575088e-05, \"fee_revenue_over_value\": 6.426366708649339e-08, \"jax_sharpe\": 0.5830468750753665, \"return\": 0.0028768596552486425, \"returns_over_hodl\": 0.0018716021156941487, \"returns_over_uniform_hodl\": 0.0019418022172983385, \"sharpe\": 0.6201135489304256, \"sterling\": 3.594222358391935, \"ulcer\": -0.00046898935357678325}], \"train_objective\": [{\"annualised_returns\": 0.0019651051467659553, \"annualised_returns_over_hodl\": 0.0031004560085221566, \"annualised_returns_over_uniform_hodl\": 0.0031004560085221566, \"calmar\": 1.021996750686056, \"daily_log_sharpe\": 0.21712753745963503, \"daily_returns\": -0.0004262385578838912, \"fee_revenue_over_value\": 2.192807505783539e-08, \"jax_sharpe\": 0.2219020516821627, \"return\": 0.0003528924937410416, \"returns_over_hodl\": 0.00055651949505231, \"returns_over_uniform_hodl\": 0.00055651949505231, \"sharpe\": 0.22165775010476566, \"sterling\": 1.1106377494800925, \"ulcer\": -0.0007473346754069304}], \"train_return\": 0.0003528924937410416, \"train_returns_over_hodl\": 0.00055651949505231, \"train_sharpe\": 0.22190205168216268, \"validation_return\": -3.9331453638546954e-05, \"validation_returns_over_hodl\": 0.00011728138721722736, \"validation_sharpe\": -0.1259202114912143}, {\"centeredness_margin\": 0.20793366613398886, \"continuous_test_metrics\": [{\"annualised_returns\": 0.005547262519520979, \"annualised_returns_over_hodl\": 0.003760123746447741, \"annualised_returns_over_uniform_hodl\": 0.0039026208634918014, \"calmar\": 1.4442539869606827, \"daily_log_sharpe\": 0.6220247243767859, \"daily_returns\": -4.865333788281044e-05, \"fee_revenue_over_value\": 4.8061490875708824e-08, \"jax_sharpe\": 0.6039671069286064, \"return\": 0.003157407700498549, \"returns_over_hodl\": 0.0021410185808126148, \"returns_over_uniform_hodl\": 0.002222088686529178, \"sharpe\": 0.6265012825914998, \"sterling\": 4.132300912410128, \"ulcer\": -0.0004405675946661247}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 4.6614269793840424e-05, \"optuna_trial_number\": 34, \"price_ratio\": 1.010840126199725, \"shift_exponent\": 1.0136043927253024e-05, \"step\": 34, \"test_objective\": [{\"annualised_returns\": 0.005547262519520979, \"annualised_returns_over_hodl\": 0.003760123746447741, \"annualised_returns_over_uniform_hodl\": 0.0039026208634918014, \"calmar\": 1.4442539869606827, \"daily_log_sharpe\": 0.6220247243767859, \"daily_returns\": -4.865333788281044e-05, \"fee_revenue_over_value\": 4.8061490875708824e-08, \"jax_sharpe\": 0.6039671069286064, \"return\": 0.003157407700498549, \"returns_over_hodl\": 0.0021410185808126148, \"returns_over_uniform_hodl\": 0.002222088686529178, \"sharpe\": 0.6265012825914998, \"sterling\": 4.132300912410128, \"ulcer\": -0.0004405675946661247}], \"train_objective\": [{\"annualised_returns\": 0.002442480880674891, \"annualised_returns_over_hodl\": 0.0035783726684055495, \"annualised_returns_over_uniform_hodl\": 0.0035783726684055495, \"calmar\": 1.2385035086359657, \"daily_log_sharpe\": 0.26825490381378325, \"daily_returns\": -0.0004182868696987984, \"fee_revenue_over_value\": 2.201654584429279e-08, \"jax_sharpe\": 0.27152824830562566, \"return\": 0.0004385336464927114, \"returns_over_hodl\": 0.000642178080503264, \"returns_over_uniform_hodl\": 0.000642178080503264, \"sharpe\": 0.27279964582280136, \"sterling\": 1.3626882779382392, \"ulcer\": -0.0007439342779081001}], \"train_return\": 0.0004385336464927114, \"train_returns_over_hodl\": 0.000642178080503264, \"train_sharpe\": 0.2715282483056256, \"validation_return\": -2.2023145073046813e-05, \"validation_returns_over_hodl\": 0.00013532151365525102, \"validation_sharpe\": -0.07043907165185614}, {\"centeredness_margin\": 0.18253885019969876, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0030690541989675246, \"annualised_returns_over_hodl\": 0.0012853953183897815, \"annualised_returns_over_uniform_hodl\": 0.0014284658229666292, \"calmar\": 0.7172291900491324, \"daily_log_sharpe\": 0.3779003404813584, \"daily_returns\": -4.8678597815575645e-05, \"fee_revenue_over_value\": 6.395395399395585e-08, \"jax_sharpe\": 0.3111119048560068, \"return\": 0.0017477819202837974, \"returns_over_hodl\": 0.0007322944301442202, \"returns_over_uniform_hodl\": 0.0008137772063283588, \"sharpe\": 0.3825080407325892, \"sterling\": 1.8756648931084423, \"ulcer\": -0.0005469221712075237}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 4.9083635108712614e-05, \"optuna_trial_number\": 35, \"price_ratio\": 1.010769826672939, \"shift_exponent\": 0.0003170014875022116, \"step\": 35, \"test_objective\": [{\"annualised_returns\": 0.0030690541989675246, \"annualised_returns_over_hodl\": 0.0012853953183897815, \"annualised_returns_over_uniform_hodl\": 0.0014284658229666292, \"calmar\": 0.7172291900491324, \"daily_log_sharpe\": 0.3779003404813584, \"daily_returns\": -4.8678597815575645e-05, \"fee_revenue_over_value\": 6.395395399395585e-08, \"jax_sharpe\": 0.3111119048560068, \"return\": 0.0017477819202837974, \"returns_over_hodl\": 0.0007322944301442202, \"returns_over_uniform_hodl\": 0.0008137772063283588, \"sharpe\": 0.3825080407325892, \"sterling\": 1.8756648931084423, \"ulcer\": -0.0005469221712075237}], \"train_objective\": [{\"annualised_returns\": 0.002465690332946302, \"annualised_returns_over_hodl\": 0.0036016084198666753, \"annualised_returns_over_uniform_hodl\": 0.0036016084198666753, \"calmar\": 1.2487545461691338, \"daily_log_sharpe\": 0.2652142063873801, \"daily_returns\": -0.0004275220044953868, \"fee_revenue_over_value\": 2.202026400502721e-08, \"jax_sharpe\": 0.2694946710403502, \"return\": 0.0004426965667641003, \"returns_over_hodl\": 0.0006463418481583716, \"returns_over_uniform_hodl\": 0.0006463418481583716, \"sharpe\": 0.2698396306180884, \"sterling\": 1.3725109117933663, \"ulcer\": -0.0007477358194414582}], \"train_return\": 0.0004426965667641003, \"train_returns_over_hodl\": 0.0006463418481583716, \"train_sharpe\": 0.26949467104035013, \"validation_return\": -2.118187081456835e-05, \"validation_returns_over_hodl\": 0.00013619835616118792, \"validation_sharpe\": -0.06800897002618747}, {\"centeredness_margin\": 0.7167545801273484, \"continuous_test_metrics\": [{\"annualised_returns\": -0.7655144360507082, \"annualised_returns_over_hodl\": -0.7659074083669326, \"annualised_returns_over_uniform_hodl\": -0.7658979533060208, \"calmar\": -1.3580434029202966, \"daily_log_sharpe\": -2.3433536628328935, \"daily_returns\": -4.587415786029793e-05, \"fee_revenue_over_value\": 6.065562800019587e-08, \"jax_sharpe\": -6.308333415184516, \"return\": -0.5624238257209404, \"returns_over_hodl\": -0.5628418727722524, \"returns_over_uniform_hodl\": -0.5628318108613795, \"sharpe\": -2.414547275359577, \"sterling\": -6.258215242736426, \"ulcer\": -0.06958038711341538}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0002568566961449692, \"optuna_trial_number\": 36, \"price_ratio\": 1.0477022961511124, \"shift_exponent\": 7.963829054799163, \"step\": 36, \"test_objective\": [{\"annualised_returns\": -0.7655144360507082, \"annualised_returns_over_hodl\": -0.7659074083669326, \"annualised_returns_over_uniform_hodl\": -0.7658979533060208, \"calmar\": -1.3580434029202966, \"daily_log_sharpe\": -2.3433536628328935, \"daily_returns\": -4.587415786029793e-05, \"fee_revenue_over_value\": 6.065562800019587e-08, \"jax_sharpe\": -6.308333415184516, \"return\": -0.5624238257209404, \"returns_over_hodl\": -0.5628418727722524, \"returns_over_uniform_hodl\": -0.5628318108613795, \"sharpe\": -2.414547275359577, \"sterling\": -6.258215242736426, \"ulcer\": -0.06958038711341538}], \"train_objective\": [{\"annualised_returns\": -0.00037053615667781425, \"annualised_returns_over_hodl\": 0.0007621681335034936, \"annualised_returns_over_uniform_hodl\": 0.0007621681335021613, \"calmar\": -0.18977589216208365, \"daily_log_sharpe\": -0.04209883387684427, \"daily_returns\": -0.00042041938888845166, \"fee_revenue_over_value\": 2.026241280189369e-08, \"jax_sharpe\": -0.037541668593456444, \"return\": -6.660440914119103e-05, \"returns_over_hodl\": 0.00013693720140750543, \"returns_over_uniform_hodl\": 0.00013693720140728338, \"sharpe\": -0.03765378656613765, \"sterling\": -0.22260348483058348, \"ulcer\": -0.0007730494278323287}], \"train_return\": -6.660440914119103e-05, \"train_returns_over_hodl\": 0.00013693720140750543, \"train_sharpe\": -0.03754166859345644, \"validation_return\": -0.0001212111725367171, \"validation_returns_over_hodl\": 3.19210322878849e-05, \"validation_sharpe\": -0.3632205534117394}, {\"centeredness_margin\": 0.10026923619101107, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0018131914131431781, \"annualised_returns_over_hodl\": 6.333202469432919e-05, \"annualised_returns_over_uniform_hodl\": 0.00017465708704000882, \"calmar\": 0.45747819768107223, \"daily_log_sharpe\": 0.23839998602899917, \"daily_returns\": -4.781485465954429e-05, \"fee_revenue_over_value\": 6.902802048432257e-08, \"jax_sharpe\": 0.1932571007092214, \"return\": 0.0010328646862087787, \"returns_over_hodl\": 3.608996639581363e-05, \"returns_over_uniform_hodl\": 9.952654329925537e-05, \"sharpe\": 0.24315841177354103, \"sterling\": 1.0923705436665925, \"ulcer\": -0.0006036450630362139}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -3.531003624021034e-05, \"optuna_trial_number\": 37, \"price_ratio\": 1.0138366641191046, \"shift_exponent\": 0.005044486667651876, \"step\": 37, \"test_objective\": [{\"annualised_returns\": 0.0018131914131431781, \"annualised_returns_over_hodl\": 6.333202469432919e-05, \"annualised_returns_over_uniform_hodl\": 0.00017465708704000882, \"calmar\": 0.45747819768107223, \"daily_log_sharpe\": 0.23839998602899917, \"daily_returns\": -4.781485465954429e-05, \"fee_revenue_over_value\": 6.902802048432257e-08, \"jax_sharpe\": 0.1932571007092214, \"return\": 0.0010328646862087787, \"returns_over_hodl\": 3.608996639581363e-05, \"returns_over_uniform_hodl\": 9.952654329925537e-05, \"sharpe\": 0.24315841177354103, \"sterling\": 1.0923705436665925, \"ulcer\": -0.0006036450630362139}], \"train_objective\": [{\"annualised_returns\": 0.0016727432087593197, \"annualised_returns_over_hodl\": 0.002807762788140211, \"annualised_returns_over_uniform_hodl\": 0.002807762788140211, \"calmar\": 0.8838340876085745, \"daily_log_sharpe\": 0.18249562880929757, \"daily_returns\": -0.0004182868697812086, \"fee_revenue_over_value\": 2.185951281637059e-08, \"jax_sharpe\": 0.18636306133989408, \"return\": 0.0003004262678791836, \"returns_over_hodl\": 0.0005040425894187184, \"returns_over_uniform_hodl\": 0.0005040425894187184, \"sharpe\": 0.1870811625516593, \"sterling\": 0.9530068109112524, \"ulcer\": -0.0007536239651735832}], \"train_return\": 0.0003004262678791836, \"train_returns_over_hodl\": 0.0005040425894187184, \"train_sharpe\": 0.18636306133989408, \"validation_return\": -4.993649661222399e-05, \"validation_returns_over_hodl\": 0.00010622795847736732, \"validation_sharpe\": -0.1618573335146774}, {\"centeredness_margin\": 0.10154444412604416, \"continuous_test_metrics\": [{\"annualised_returns\": -0.0007337648574956557, \"annualised_returns_over_hodl\": -0.0025075579915646573, \"annualised_returns_over_uniform_hodl\": -0.0023681334615739402, \"calmar\": -0.15615285310518576, \"daily_log_sharpe\": -0.020363722098825934, \"daily_returns\": -4.8593463951476684e-05, \"fee_revenue_over_value\": 6.805925893037306e-08, \"jax_sharpe\": -0.06812463423896471, \"return\": -0.0004182100675601541, \"returns_over_hodl\": -0.0014297313425932767, \"returns_over_uniform_hodl\": -0.0013501952644640047, \"sharpe\": -0.015358649440981783, \"sterling\": -0.4138211245268561, \"ulcer\": -0.0006538222601764989}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 4.076328634789739e-05, \"optuna_trial_number\": 38, \"price_ratio\": 1.0110104201389132, \"shift_exponent\": 0.0020031941669168716, \"step\": 38, \"test_objective\": [{\"annualised_returns\": -0.0007337648574956557, \"annualised_returns_over_hodl\": -0.0025075579915646573, \"annualised_returns_over_uniform_hodl\": -0.0023681334615739402, \"calmar\": -0.15615285310518576, \"daily_log_sharpe\": -0.020363722098825934, \"daily_returns\": -4.8593463951476684e-05, \"fee_revenue_over_value\": 6.805925893037306e-08, \"jax_sharpe\": -0.06812463423896471, \"return\": -0.0004182100675601541, \"returns_over_hodl\": -0.0014297313425932767, \"returns_over_uniform_hodl\": -0.0013501952644640047, \"sharpe\": -0.015358649440981783, \"sterling\": -0.4138211245268561, \"ulcer\": -0.0006538222601764989}], \"train_objective\": [{\"annualised_returns\": 0.002387489330532011, \"annualised_returns_over_hodl\": 0.00352331880600798, \"annualised_returns_over_uniform_hodl\": 0.0035233188060093124, \"calmar\": 1.214115516501658, \"daily_log_sharpe\": 0.25162525286710985, \"daily_returns\": -0.000427321549218856, \"fee_revenue_over_value\": 2.2007545493836164e-08, \"jax_sharpe\": 0.25348335560121427, \"return\": 0.0004286698737134831, \"returns_over_hodl\": 0.0006323122999019049, \"returns_over_uniform_hodl\": 0.0006323122999021269, \"sharpe\": 0.2563370368433179, \"sterling\": 1.3339835586799673, \"ulcer\": -0.0007463977135836316}], \"train_return\": 0.0004286698737134831, \"train_returns_over_hodl\": 0.0006323122999019049, \"train_sharpe\": 0.2534833556012142, \"validation_return\": -2.401650208305739e-05, \"validation_returns_over_hodl\": 0.00013324387773083757, \"validation_sharpe\": -0.07847385925291371}, {\"centeredness_margin\": 0.05017472797500756, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0033375458069369035, \"annualised_returns_over_hodl\": 0.0016141491350913917, \"annualised_returns_over_uniform_hodl\": 0.0016965182944581603, \"calmar\": 0.8209576554494095, \"daily_log_sharpe\": 0.3943935289479762, \"daily_returns\": -4.7019137576601266e-05, \"fee_revenue_over_value\": 6.834879561398725e-08, \"jax_sharpe\": 0.3627373325037736, \"return\": 0.0019005745110411976, \"returns_over_hodl\": 0.0009195217385338239, \"returns_over_uniform_hodl\": 0.0009664273370746379, \"sharpe\": 0.3990316980465303, \"sterling\": 1.9987418977928635, \"ulcer\": -0.000636752212281181}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00011285979398463764, \"optuna_trial_number\": 39, \"price_ratio\": 1.0187550009948039, \"shift_exponent\": 0.0026263795026912227, \"step\": 39, \"test_objective\": [{\"annualised_returns\": 0.0033375458069369035, \"annualised_returns_over_hodl\": 0.0016141491350913917, \"annualised_returns_over_uniform_hodl\": 0.0016965182944581603, \"calmar\": 0.8209576554494095, \"daily_log_sharpe\": 0.3943935289479762, \"daily_returns\": -4.7019137576601266e-05, \"fee_revenue_over_value\": 6.834879561398725e-08, \"jax_sharpe\": 0.3627373325037736, \"return\": 0.0019005745110411976, \"returns_over_hodl\": 0.0009195217385338239, \"returns_over_uniform_hodl\": 0.0009664273370746379, \"sharpe\": 0.3990316980465303, \"sterling\": 1.9987418977928635, \"ulcer\": -0.000636752212281181}], \"train_objective\": [{\"annualised_returns\": 0.000942950900067574, \"annualised_returns_over_hodl\": 0.002077143534156445, \"annualised_returns_over_uniform_hodl\": 0.002077143534156445, \"calmar\": 0.500752983081683, \"daily_log_sharpe\": 0.10418562994474882, \"daily_returns\": -0.00042361623391010524, \"fee_revenue_over_value\": 2.1607783822496637e-08, \"jax_sharpe\": 0.10780356131698463, \"return\": 0.00016940552151356592, \"returns_over_hodl\": 0.0003729951731030745, \"returns_over_uniform_hodl\": 0.0003729951731030745, \"sharpe\": 0.1087311067695089, \"sterling\": 0.54823904324049, \"ulcer\": -0.000757228458884499}], \"train_return\": 0.00016940552151356592, \"train_returns_over_hodl\": 0.0003729951731030745, \"train_sharpe\": 0.10780356131698461, \"validation_return\": -7.642461809842516e-05, \"validation_returns_over_hodl\": 7.861996431257623e-05, \"validation_sharpe\": -0.2399164335610029}, {\"centeredness_margin\": 0.029897104197839197, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0034372793712811323, \"annualised_returns_over_hodl\": 0.001745044106981286, \"annualised_returns_over_uniform_hodl\": 0.0017960887377037604, \"calmar\": 0.8122604894881919, \"daily_log_sharpe\": 0.4028203190354952, \"daily_returns\": -4.6163246864228886e-05, \"fee_revenue_over_value\": 6.684561658872628e-08, \"jax_sharpe\": 0.36633399076751716, \"return\": 0.0019573261800507336, \"returns_over_hodl\": 0.0009940598676314583, \"returns_over_uniform_hodl\": 0.0010231260922397567, \"sharpe\": 0.4074470102405224, \"sterling\": 1.9458125933710533, \"ulcer\": -0.0007194149747759313}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00019651341379152144, \"optuna_trial_number\": 40, \"price_ratio\": 1.0303580697675268, \"shift_exponent\": 0.016157101088207254, \"step\": 40, \"test_objective\": [{\"annualised_returns\": 0.0034372793712811323, \"annualised_returns_over_hodl\": 0.001745044106981286, \"annualised_returns_over_uniform_hodl\": 0.0017960887377037604, \"calmar\": 0.8122604894881919, \"daily_log_sharpe\": 0.4028203190354952, \"daily_returns\": -4.6163246864228886e-05, \"fee_revenue_over_value\": 6.684561658872628e-08, \"jax_sharpe\": 0.36633399076751716, \"return\": 0.0019573261800507336, \"returns_over_hodl\": 0.0009940598676314583, \"returns_over_uniform_hodl\": 0.0010231260922397567, \"sharpe\": 0.4074470102405224, \"sterling\": 1.9458125933710533, \"ulcer\": -0.0007194149747759313}], \"train_objective\": [{\"annualised_returns\": 0.00015870963605779664, \"annualised_returns_over_hodl\": 0.0012920136274279237, \"annualised_returns_over_uniform_hodl\": 0.0012920136274279237, \"calmar\": 0.08279397286608325, \"daily_log_sharpe\": 0.01791682026735541, \"daily_returns\": -0.00042160277482474146, \"fee_revenue_over_value\": 2.104188016676114e-08, \"jax_sharpe\": 0.02234081940083584, \"return\": 2.852209391313032e-05, \"returns_over_hodl\": 0.00023208306795274858, \"returns_over_uniform_hodl\": 0.00023208306795274858, \"sharpe\": 0.022384123301706218, \"sterling\": 0.09435517653620895, \"ulcer\": -0.0007657137620339987}], \"train_return\": 2.852209391313032e-05, \"train_returns_over_hodl\": 0.00023208306795274858, \"train_sharpe\": 0.02234081940083584, \"validation_return\": -0.00010491450625127463, \"validation_returns_over_hodl\": 4.892563953884377e-05, \"validation_sharpe\": -0.3192004196086638}, {\"centeredness_margin\": 0.01784824749364847, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0036575034090213787, \"annualised_returns_over_hodl\": 0.001959243760971674, \"annualised_returns_over_uniform_hodl\": 0.0020159525838954195, \"calmar\": 0.8695962718176155, \"daily_log_sharpe\": 0.43744830603271456, \"daily_returns\": -4.6317621028669154e-05, \"fee_revenue_over_value\": 6.730351565608381e-08, \"jax_sharpe\": 0.39750684984600954, \"return\": 0.0020826322869924585, \"returns_over_hodl\": 0.0011160268033136855, \"returns_over_uniform_hodl\": 0.0011483153668845336, \"sharpe\": 0.4419604899357063, \"sterling\": 2.1062789932691617, \"ulcer\": -0.0006968276551681361}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00018142563983193362, \"optuna_trial_number\": 41, \"price_ratio\": 1.0273108845310133, \"shift_exponent\": 0.001945215955222029, \"step\": 41, \"test_objective\": [{\"annualised_returns\": 0.0036575034090213787, \"annualised_returns_over_hodl\": 0.001959243760971674, \"annualised_returns_over_uniform_hodl\": 0.0020159525838954195, \"calmar\": 0.8695962718176155, \"daily_log_sharpe\": 0.43744830603271456, \"daily_returns\": -4.6317621028669154e-05, \"fee_revenue_over_value\": 6.730351565608381e-08, \"jax_sharpe\": 0.39750684984600954, \"return\": 0.0020826322869924585, \"returns_over_hodl\": 0.0011160268033136855, \"returns_over_uniform_hodl\": 0.0011483153668845336, \"sharpe\": 0.4419604899357063, \"sterling\": 2.1062789932691617, \"ulcer\": -0.0006968276551681361}], \"train_objective\": [{\"annualised_returns\": 0.0003001120260703871, \"annualised_returns_over_hodl\": 0.0014335762439050548, \"annualised_returns_over_uniform_hodl\": 0.0014335762439050548, \"calmar\": 0.1571101662383602, \"daily_log_sharpe\": 0.03384215805906122, \"daily_returns\": -0.000418286869748287, \"fee_revenue_over_value\": 2.11868429257616e-08, \"jax_sharpe\": 0.0381270898407905, \"return\": 5.3930733432183686e-05, \"returns_over_hodl\": 0.00025749687953191547, \"returns_over_uniform_hodl\": 0.00025749687953191547, \"sharpe\": 0.038312343990436565, \"sterling\": 0.17757441604020915, \"ulcer\": -0.0007634660789973377}], \"train_return\": 5.3930733432183686e-05, \"train_returns_over_hodl\": 0.00025749687953191547, \"train_sharpe\": 0.0381270898407905, \"validation_return\": -9.97756653795534e-05, \"validation_returns_over_hodl\": 5.428172021137989e-05, \"validation_sharpe\": -0.3052853610875176}, {\"centeredness_margin\": 0.1978833125752275, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0038689900261628107, \"annualised_returns_over_hodl\": 0.002153346331731365, \"annualised_returns_over_uniform_hodl\": 0.002227093300138705, \"calmar\": 0.944356006883743, \"daily_log_sharpe\": 0.45741237702712434, \"daily_returns\": -4.678251125472e-05, \"fee_revenue_over_value\": 6.765043759992352e-08, \"jax_sharpe\": 0.42411074338101956, \"return\": 0.0022029557243876674, \"returns_over_hodl\": 0.0012265406186351413, \"returns_over_uniform_hodl\": 0.0012685266176992727, \"sharpe\": 0.46196043314328417, \"sterling\": 2.337774014123153, \"ulcer\": -0.0006333973690166398}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0001359886853739195, \"optuna_trial_number\": 42, \"price_ratio\": 1.0209712270928826, \"shift_exponent\": 0.00044904777560741987, \"step\": 42, \"test_objective\": [{\"annualised_returns\": 0.0038689900261628107, \"annualised_returns_over_hodl\": 0.002153346331731365, \"annualised_returns_over_uniform_hodl\": 0.002227093300138705, \"calmar\": 0.944356006883743, \"daily_log_sharpe\": 0.45741237702712434, \"daily_returns\": -4.678251125472e-05, \"fee_revenue_over_value\": 6.765043759992352e-08, \"jax_sharpe\": 0.42411074338101956, \"return\": 0.0022029557243876674, \"returns_over_hodl\": 0.0012265406186351413, \"returns_over_uniform_hodl\": 0.0012685266176992727, \"sharpe\": 0.46196043314328417, \"sterling\": 2.337774014123153, \"ulcer\": -0.0006333973690166398}], \"train_objective\": [{\"annualised_returns\": 0.0007260629335521518, \"annualised_returns_over_hodl\": 0.001860009806647911, \"annualised_returns_over_uniform_hodl\": 0.0018600098066492432, \"calmar\": 0.3841685037027495, \"daily_log_sharpe\": 0.08025922774995593, \"daily_returns\": -0.0004230595159673963, \"fee_revenue_over_value\": 2.1496723813767543e-08, \"jax_sharpe\": 0.0843807378575879, \"return\": 0.00013045218400509206, \"returns_over_hodl\": 0.0003340339064414888, \"returns_over_uniform_hodl\": 0.0003340339064417108, \"sharpe\": 0.08480914924898583, \"sterling\": 0.4247276425303744, \"ulcer\": -0.0007585591494525196}], \"train_return\": 0.00013045218400509206, \"train_returns_over_hodl\": 0.0003340339064414888, \"train_sharpe\": 0.0843807378575879, \"validation_return\": -8.430104765755342e-05, \"validation_returns_over_hodl\": 7.041054205325636e-05, \"validation_sharpe\": -0.2637815389980875}, {\"centeredness_margin\": 0.18207656872046918, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0030019674628973814, \"annualised_returns_over_hodl\": 0.0012869996812108209, \"annualised_returns_over_uniform_hodl\": 0.0013614888118635982, \"calmar\": 0.7335511808164116, \"daily_log_sharpe\": 0.3465015737003752, \"daily_returns\": -4.680452991950479e-05, \"fee_revenue_over_value\": 6.768193313072317e-08, \"jax_sharpe\": 0.3158903363471671, \"return\": 0.0017096016069739761, \"returns_over_hodl\": 0.0007332081887780895, \"returns_over_uniform_hodl\": 0.0007756324913927859, \"sharpe\": 0.3513003538466716, \"sterling\": 1.7425143493036, \"ulcer\": -0.0006736855057536668}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.00013383613599512456, \"optuna_trial_number\": 43, \"price_ratio\": 1.020743102926963, \"shift_exponent\": 0.0013379065021733422, \"step\": 43, \"test_objective\": [{\"annualised_returns\": 0.0030019674628973814, \"annualised_returns_over_hodl\": 0.0012869996812108209, \"annualised_returns_over_uniform_hodl\": 0.0013614888118635982, \"calmar\": 0.7335511808164116, \"daily_log_sharpe\": 0.3465015737003752, \"daily_returns\": -4.680452991950479e-05, \"fee_revenue_over_value\": 6.768193313072317e-08, \"jax_sharpe\": 0.3158903363471671, \"return\": 0.0017096016069739761, \"returns_over_hodl\": 0.0007332081887780895, \"returns_over_uniform_hodl\": 0.0007756324913927859, \"sharpe\": 0.3513003538466716, \"sterling\": 1.7425143493036, \"ulcer\": -0.0006736855057536668}], \"train_objective\": [{\"annualised_returns\": 0.0007462463847038858, \"annualised_returns_over_hodl\": 0.0018802161281570307, \"annualised_returns_over_uniform_hodl\": 0.0018802161281556984, \"calmar\": 0.395048248271198, \"daily_log_sharpe\": 0.0841103194099251, \"daily_returns\": -0.000423111336489351, \"fee_revenue_over_value\": 2.150808902258349e-08, \"jax_sharpe\": 0.08842082725747732, \"return\": 0.00013407744812843347, \"returns_over_hodl\": 0.0003376599085063159, \"returns_over_uniform_hodl\": 0.00033765990850609384, \"sharpe\": 0.08857667771447617, \"sterling\": 0.4358910160942372, \"ulcer\": -0.0007590476648240692}], \"train_return\": 0.00013407744812843347, \"train_returns_over_hodl\": 0.0003376599085063159, \"train_sharpe\": 0.08842082725747732, \"validation_return\": -8.356798516251374e-05, \"validation_returns_over_hodl\": 7.11745964894206e-05, \"validation_sharpe\": -0.26110148371879016}, {\"centeredness_margin\": 0.22379375720670616, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0034894038827557594, \"annualised_returns_over_hodl\": 0.001718985728732969, \"annualised_returns_over_uniform_hodl\": 0.0018481279959574604, \"calmar\": 0.8256114605841, \"daily_log_sharpe\": 0.3905458328681284, \"daily_returns\": -4.8296476542139516e-05, \"fee_revenue_over_value\": 6.195881326762525e-08, \"jax_sharpe\": 0.35444753229487835, \"return\": 0.001986985771176286, \"returns_over_hodl\": 0.000979221260663854, \"returns_over_uniform_hodl\": 0.0010527580295003336, \"sharpe\": 0.395451955698583, \"sterling\": 2.145952049915885, \"ulcer\": -0.0005523573502262288}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 1.174480851457993e-05, \"optuna_trial_number\": 44, \"price_ratio\": 1.0119408165818633, \"shift_exponent\": 0.00019424637369423584, \"step\": 44, \"test_objective\": [{\"annualised_returns\": 0.0034894038827557594, \"annualised_returns_over_hodl\": 0.001718985728732969, \"annualised_returns_over_uniform_hodl\": 0.0018481279959574604, \"calmar\": 0.8256114605841, \"daily_log_sharpe\": 0.3905458328681284, \"daily_returns\": -4.8296476542139516e-05, \"fee_revenue_over_value\": 6.195881326762525e-08, \"jax_sharpe\": 0.35444753229487835, \"return\": 0.001986985771176286, \"returns_over_hodl\": 0.000979221260663854, \"returns_over_uniform_hodl\": 0.0010527580295003336, \"sharpe\": 0.395451955698583, \"sterling\": 2.145952049915885, \"ulcer\": -0.0005523573502262288}], \"train_objective\": [{\"annualised_returns\": 0.0021147896400453003, \"annualised_returns_over_hodl\": 0.0032503101129151, \"annualised_returns_over_uniform_hodl\": 0.0032503101129151, \"calmar\": 1.0910683871748286, \"daily_log_sharpe\": 0.23702637638688673, \"daily_returns\": -0.0004266223887720975, \"fee_revenue_over_value\": 2.1958535229796643e-08, \"jax_sharpe\": 0.23980273979316957, \"return\": 0.0003797494762460829, \"returns_over_hodl\": 0.0005833819444347466, \"returns_over_uniform_hodl\": 0.0005833819444347466, \"sharpe\": 0.24148184220860666, \"sterling\": 1.1903731927435766, \"ulcer\": -0.0007462041001008655}], \"train_return\": 0.0003797494762460829, \"train_returns_over_hodl\": 0.0005833819444347466, \"train_sharpe\": 0.23980273979316957, \"validation_return\": -3.39032632327152e-05, \"validation_returns_over_hodl\": 0.00012293908549487753, \"validation_sharpe\": -0.11166993296707672}, {\"centeredness_margin\": 0.21414675741214684, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0034643109343532874, \"annualised_returns_over_hodl\": 0.0016744908964296812, \"annualised_returns_over_uniform_hodl\": 0.0018230760887967268, \"calmar\": 0.7767702838031456, \"daily_log_sharpe\": 0.3692270473809525, \"daily_returns\": -4.882772264938876e-05, \"fee_revenue_over_value\": 5.3759872653025926e-08, \"jax_sharpe\": 0.3384539435596129, \"return\": 0.0019727076080466865, \"returns_over_hodl\": 0.0009538838692231266, \"returns_over_uniform_hodl\": 0.0010384931789748642, \"sharpe\": 0.3744075836644456, \"sterling\": 2.1289775358701952, \"ulcer\": -0.0005349634056046745}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 6.365502815966206e-05, \"optuna_trial_number\": 45, \"price_ratio\": 1.0103728388855158, \"shift_exponent\": 0.00010711543433090381, \"step\": 45, \"test_objective\": [{\"annualised_returns\": 0.0034643109343532874, \"annualised_returns_over_hodl\": 0.0016744908964296812, \"annualised_returns_over_uniform_hodl\": 0.0018230760887967268, \"calmar\": 0.7767702838031456, \"daily_log_sharpe\": 0.3692270473809525, \"daily_returns\": -4.882772264938876e-05, \"fee_revenue_over_value\": 5.3759872653025926e-08, \"jax_sharpe\": 0.3384539435596129, \"return\": 0.0019727076080466865, \"returns_over_hodl\": 0.0009538838692231266, \"returns_over_uniform_hodl\": 0.0010384931789748642, \"sharpe\": 0.3744075836644456, \"sterling\": 2.1289775358701952, \"ulcer\": -0.0005349634056046745}], \"train_objective\": [{\"annualised_returns\": 0.002602660550143021, \"annualised_returns_over_hodl\": 0.0037387338413255033, \"annualised_returns_over_uniform_hodl\": 0.0037387338413255033, \"calmar\": 1.3087476082263887, \"daily_log_sharpe\": 0.27088480646661833, \"daily_returns\": -0.000427873076246114, \"fee_revenue_over_value\": 2.2041290468517963e-08, \"jax_sharpe\": 0.2730292691570092, \"return\": 0.00046726236533078946, \"returns_over_hodl\": 0.0006709126472204119, \"returns_over_uniform_hodl\": 0.0006709126472204119, \"sharpe\": 0.27564838090526267, \"sterling\": 1.4458171480047046, \"ulcer\": -0.0007452143558767711}], \"train_return\": 0.00046726236533078946, \"train_returns_over_hodl\": 0.0006709126472204119, \"train_sharpe\": 0.2730292691570092, \"validation_return\": -1.621761442505143e-05, \"validation_returns_over_hodl\": 0.00014137251320089916, \"validation_sharpe\": -0.05185131100111336}, {\"centeredness_margin\": 0.17889727751843418, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0026902553204901647, \"annualised_returns_over_hodl\": 0.0009024802009374167, \"annualised_returns_over_uniform_hodl\": 0.001050286496088626, \"calmar\": 0.5841471978146826, \"daily_log_sharpe\": 0.3291015743421234, \"daily_returns\": -4.8809560994011376e-05, \"fee_revenue_over_value\": 6.642589550874385e-08, \"jax_sharpe\": 0.26645054396430784, \"return\": 0.0015321859991446196, \"returns_over_hodl\": 0.0005141885917119282, \"returns_over_uniform_hodl\": 0.0005983823014632517, \"sharpe\": 0.3338663121633946, \"sterling\": 1.5711840820642817, \"ulcer\": -0.0005805916220782738}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 6.188025767016603e-05, \"optuna_trial_number\": 46, \"price_ratio\": 1.010419619078799, \"shift_exponent\": 0.0006195004630886502, \"step\": 46, \"test_objective\": [{\"annualised_returns\": 0.0026902553204901647, \"annualised_returns_over_hodl\": 0.0009024802009374167, \"annualised_returns_over_uniform_hodl\": 0.001050286496088626, \"calmar\": 0.5841471978146826, \"daily_log_sharpe\": 0.3291015743421234, \"daily_returns\": -4.8809560994011376e-05, \"fee_revenue_over_value\": 6.642589550874385e-08, \"jax_sharpe\": 0.26645054396430784, \"return\": 0.0015321859991446196, \"returns_over_hodl\": 0.0005141885917119282, \"returns_over_uniform_hodl\": 0.0005983823014632517, \"sharpe\": 0.3338663121633946, \"sterling\": 1.5711840820642817, \"ulcer\": -0.0005805916220782738}], \"train_objective\": [{\"annualised_returns\": 0.002585976909580845, \"annualised_returns_over_hodl\": 0.0037220312961281365, \"annualised_returns_over_uniform_hodl\": 0.0037220312961268043, \"calmar\": 1.3014857850399315, \"daily_log_sharpe\": 0.2752959139701365, \"daily_returns\": -0.00042783031410257883, \"fee_revenue_over_value\": 2.203881013088893e-08, \"jax_sharpe\": 0.27982078044550845, \"return\": 0.00046427027848072733, \"returns_over_hodl\": 0.0006679199513157652, \"returns_over_uniform_hodl\": 0.0006679199513155432, \"sharpe\": 0.2799697061436621, \"sterling\": 1.4371920957183675, \"ulcer\": -0.0007455556902863912}], \"train_return\": 0.00046427027848072733, \"train_returns_over_hodl\": 0.0006679199513157652, \"train_sharpe\": 0.2798207804455084, \"validation_return\": -1.6822239322977772e-05, \"validation_returns_over_hodl\": 0.00014074232317340396, \"validation_sharpe\": -0.05265691074375665}, {\"centeredness_margin\": 0.3464214401208171, \"continuous_test_metrics\": [{\"annualised_returns\": 0.002784032057486119, \"annualised_returns_over_hodl\": 0.0011025533388790976, \"annualised_returns_over_uniform_hodl\": 0.0011439098547860738, \"calmar\": 0.6424366504558281, \"daily_log_sharpe\": 0.3105868353018095, \"daily_returns\": -4.589935471822786e-05, \"fee_revenue_over_value\": 6.452628002678005e-08, \"jax_sharpe\": 0.2824468438964424, \"return\": 0.001585562920216832, \"returns_over_hodl\": 0.000628153362496553, \"returns_over_uniform_hodl\": 0.0006517094552218605, \"sharpe\": 0.315558070313521, \"sterling\": 1.483370351279434, \"ulcer\": -0.0008060150400301569}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0002551162913233346, \"optuna_trial_number\": 47, \"price_ratio\": 1.0466235354996332, \"shift_exponent\": 0.0022650409013889993, \"step\": 47, \"test_objective\": [{\"annualised_returns\": 0.002784032057486119, \"annualised_returns_over_hodl\": 0.0011025533388790976, \"annualised_returns_over_uniform_hodl\": 0.0011439098547860738, \"calmar\": 0.6424366504558281, \"daily_log_sharpe\": 0.3105868353018095, \"daily_returns\": -4.589935471822786e-05, \"fee_revenue_over_value\": 6.452628002678005e-08, \"jax_sharpe\": 0.2824468438964424, \"return\": 0.001585562920216832, \"returns_over_hodl\": 0.000628153362496553, \"returns_over_uniform_hodl\": 0.0006517094552218605, \"sharpe\": 0.315558070313521, \"sterling\": 1.483370351279434, \"ulcer\": -0.0008060150400301569}], \"train_objective\": [{\"annualised_returns\": -0.0003534226016318476, \"annualised_returns_over_hodl\": 0.0007793010803311962, \"annualised_returns_over_uniform_hodl\": 0.0007793010803311962, \"calmar\": -0.18106470296128932, \"daily_log_sharpe\": -0.04036852203070881, \"daily_returns\": -0.00041828689116965893, \"fee_revenue_over_value\": 2.03087489500672e-08, \"jax_sharpe\": -0.035920117128405725, \"return\": -6.352777688567457e-05, \"returns_over_hodl\": 0.00014001445992728456, \"returns_over_uniform_hodl\": 0.00014001445992728456, \"sharpe\": -0.035947321268062646, \"sterling\": -0.21219352922590118, \"ulcer\": -0.0007740477824145461}], \"train_return\": -6.352777688567457e-05, \"train_returns_over_hodl\": 0.00014001445992728456, \"train_sharpe\": -0.035920117128405725, \"validation_return\": -0.00012052262817152659, \"validation_returns_over_hodl\": 3.263825635424489e-05, \"validation_sharpe\": -0.36187586151526707}, {\"centeredness_margin\": 0.15658225193937902, \"continuous_test_metrics\": [{\"annualised_returns\": 0.004304359873642083, \"annualised_returns_over_hodl\": 0.0025341744817795053, \"annualised_returns_over_uniform_hodl\": 0.0026617510703115244, \"calmar\": 1.0831966678622105, \"daily_log_sharpe\": 0.5136699828614266, \"daily_returns\": -4.825087780376818e-05, \"fee_revenue_over_value\": 6.099330818912243e-08, \"jax_sharpe\": 0.47778518847928153, \"return\": 0.0024506212206905076, \"returns_over_hodl\": 0.0014433413519159277, \"returns_over_uniform_hodl\": 0.0015159611968542652, \"sharpe\": 0.5181690284688911, \"sterling\": 2.8574053661297243, \"ulcer\": -0.0005094393100740077}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 7.289951932948347e-06, \"optuna_trial_number\": 48, \"price_ratio\": 1.0120977640380395, \"shift_exponent\": 0.00010952351431486546, \"step\": 48, \"test_objective\": [{\"annualised_returns\": 0.004304359873642083, \"annualised_returns_over_hodl\": 0.0025341744817795053, \"annualised_returns_over_uniform_hodl\": 0.0026617510703115244, \"calmar\": 1.0831966678622105, \"daily_log_sharpe\": 0.5136699828614266, \"daily_returns\": -4.825087780376818e-05, \"fee_revenue_over_value\": 6.099330818912243e-08, \"jax_sharpe\": 0.47778518847928153, \"return\": 0.0024506212206905076, \"returns_over_hodl\": 0.0014433413519159277, \"returns_over_uniform_hodl\": 0.0015159611968542652, \"sharpe\": 0.5181690284688911, \"sterling\": 2.8574053661297243, \"ulcer\": -0.0005094393100740077}], \"train_objective\": [{\"annualised_returns\": 0.002072930620242719, \"annualised_returns_over_hodl\": 0.0032084036616470968, \"annualised_returns_over_uniform_hodl\": 0.0032084036616470968, \"calmar\": 1.0718639444958282, \"daily_log_sharpe\": 0.22221277223132552, \"daily_returns\": -0.0004182868526391776, \"fee_revenue_over_value\": 2.1950294792357377e-08, \"jax_sharpe\": 0.22610420268302178, \"return\": 0.00037223929717589144, \"returns_over_hodl\": 0.0005758702366289725, \"returns_over_uniform_hodl\": 0.0005758702366289725, \"sharpe\": 0.22686916250852318, \"sterling\": 1.1681408257045796, \"ulcer\": -0.0007518247513979362}], \"train_return\": 0.00037223929717589144, \"train_returns_over_hodl\": 0.0005758702366289725, \"train_sharpe\": 0.22610420268302175, \"validation_return\": -3.5421122636436486e-05, \"validation_returns_over_hodl\": 0.00012135704408566816, \"validation_sharpe\": -0.11707461807361987}, {\"centeredness_margin\": 0.4137649298733668, \"continuous_test_metrics\": [{\"annualised_returns\": 0.0028184401346791343, \"annualised_returns_over_hodl\": 0.00114382039844374, \"annualised_returns_over_uniform_hodl\": 0.0011782616552042935, \"calmar\": 0.6512752546917528, \"daily_log_sharpe\": 0.3125959610128578, \"daily_returns\": -4.571030313124183e-05, \"fee_revenue_over_value\": 6.473109105506831e-08, \"jax_sharpe\": 0.2849972498280691, \"return\": 0.0016051471666631567, \"returns_over_hodl\": 0.0006516585025928556, \"returns_over_uniform_hodl\": 0.00067127544180412, \"sharpe\": 0.3175830126919182, \"sterling\": 1.5092785472188925, \"ulcer\": -0.000798339058678693}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0002525677773292157, \"optuna_trial_number\": 49, \"price_ratio\": 1.0451291387715875, \"shift_exponent\": 0.0288908954949865, \"step\": 49, \"test_objective\": [{\"annualised_returns\": 0.0028184401346791343, \"annualised_returns_over_hodl\": 0.00114382039844374, \"annualised_returns_over_uniform_hodl\": 0.0011782616552042935, \"calmar\": 0.6512752546917528, \"daily_log_sharpe\": 0.3125959610128578, \"daily_returns\": -4.571030313124183e-05, \"fee_revenue_over_value\": 6.473109105506831e-08, \"jax_sharpe\": 0.2849972498280691, \"return\": 0.0016051471666631567, \"returns_over_hodl\": 0.0006516585025928556, \"returns_over_uniform_hodl\": 0.00067127544180412, \"sharpe\": 0.3175830126919182, \"sterling\": 1.5092785472188925, \"ulcer\": -0.000798339058678693}], \"train_objective\": [{\"annualised_returns\": -0.0003283626058199207, \"annualised_returns_over_hodl\": 0.000804389472228495, \"annualised_returns_over_uniform_hodl\": 0.0008043894722296052, \"calmar\": -0.1682992510849246, \"daily_log_sharpe\": -0.03748345297605414, \"daily_returns\": -0.00042053749337691784, \"fee_revenue_over_value\": 2.0373388793740123e-08, \"jax_sharpe\": -0.03296299386299222, \"return\": -5.902263127610663e-05, \"returns_over_hodl\": 0.0001445205225822921, \"returns_over_uniform_hodl\": 0.00014452052258251413, \"sharpe\": -0.033059985196192174, \"sterling\": -0.19697263618377084, \"ulcer\": -0.000772670900495662}], \"train_return\": -5.902263127610663e-05, \"train_returns_over_hodl\": 0.0001445205225822921, \"train_sharpe\": -0.03296299386299222, \"validation_return\": -0.00011999097976722606, \"validation_returns_over_hodl\": 3.321182613635898e-05, \"validation_sharpe\": -0.3618455017180548}]" \ No newline at end of file diff --git a/results/run_cc1d5e6467381c91004a58f57a1110a1c66faca6e4cf16da5854516f8de280cd.json b/results/run_cc1d5e6467381c91004a58f57a1110a1c66faca6e4cf16da5854516f8de280cd.json new file mode 100644 index 00000000..f153bb6f --- /dev/null +++ b/results/run_cc1d5e6467381c91004a58f57a1110a1c66faca6e4cf16da5854516f8de280cd.json @@ -0,0 +1 @@ +"[{\"alphabetic\": true, \"arb_fees\": 0.0, \"arb_frequency\": 1, \"arb_quality\": 1.0, \"bout_offset\": 10080, \"checkpoint_fused\": \"scan\", \"chunk_period\": 1440, \"do_arb\": true, \"do_trades\": false, \"endDateString\": \"2025-10-05 00:00:00\", \"endTestDateString\": \"2026-03-01 00:00:00\", \"ensemble_init_method\": \"gaussian\", \"ensemble_init_scale\": 0.5, \"ensemble_init_seed\": 42, \"evaluation_starts\": [86400, 91129, 93321, 95858, 100587, 103489, 105317, 105383, 107643, 110046, 114775, 118630, 119504, 124234, 125912, 128393, 128963, 129013, 129967, 132291, 133692, 135000, 138421, 139020, 141426, 143151, 147880, 152609, 154668, 157338, 162068, 166586, 166797, 166871, 167366, 168857, 171526, 172693, 176088, 176256], \"fees\": 0.0025, \"freq\": \"minute\", \"gas_cost\": 1.0, \"initial_arc_length_speed\": 0.0001, \"initial_centeredness_margin\": 0.2, \"initial_daily_price_shift_base\": 0.999991935483871, \"initial_k_per_day\": 20, \"initial_log_amplitude\": 0.0, \"initial_memory_length\": 10.0, \"initial_memory_length_delta\": 0.0, \"initial_pool_value\": 5000000.0, \"initial_pre_exp_scaling\": 0.5, \"initial_price_ratio\": 4.0, \"initial_raw_exponents\": 0.0, \"initial_raw_width\": 0.0, \"initial_shift_exponent\": 1.0, \"initial_weights_logits\": 1.0, \"learnable_bounds_settings\": {\"freeze_bounds\": false, \"max_weights_per_asset\": null, \"min_weights_per_asset\": null}, \"max_memory_days\": 365, \"maximum_change\": 0.0003, \"minimum_weight\": null, \"n_ensemble_members\": 1, \"noise_arrays_path\": \"results/mm_noise/_sim_arrays/0xa6f548df93de92_2025-01-01_2026-03-01_mm.npz\", \"noise_model\": \"mm_observed\", \"noise_trader_ratio\": 0.0, \"numeraire\": null, \"optimisation_settings\": {\"base_lr\": 0.1, \"batch_size\": 8, \"bfgs_settings\": {\"compute_dtype\": \"float32\", \"maxiter\": 100, \"n_evaluation_points\": 20, \"tol\": 1e-06}, \"checkpoint_interval\": 10, \"clip_norm\": 10.0, \"cma_es_settings\": {\"compute_dtype\": \"float32\", \"memory_budget\": null, \"n_evaluation_points\": 20, \"n_generations\": 300, \"population_size\": null, \"sigma0\": 0.5, \"tol\": 1e-08}, \"decay_lr_plateau\": 100, \"decay_lr_ratio\": 0.8, \"early_stopping\": true, \"early_stopping_metric\": \"daily_log_sharpe\", \"early_stopping_patience\": 200, \"force_scalar\": false, \"include_flipped_training_data\": false, \"initial_random_key\": 0, \"lr_decay_ratio\": 1000, \"lr_schedule_type\": \"constant\", \"max_mc_version\": 9, \"method\": \"optuna\", \"min_lr\": 1e-06, \"n_cycles\": 5, \"n_iterations\": 1000, \"n_parameter_sets\": 1, \"noise_scale\": 0.1, \"optimiser\": \"adamw\", \"optuna_settings\": {\"early_stopping\": {\"enabled\": false, \"min_improvement\": 0.001, \"patience\": 100}, \"expand_around\": false, \"make_scalar\": true, \"min_train_returns_over_hodl\": -0.5, \"multi_objective\": false, \"n_jobs\": 4, \"n_startup_trials\": 10, \"n_trials\": 50, \"overfitting_penalty\": 0.2, \"parameter_config\": {\"centeredness_margin\": {\"high\": 0.99, \"low\": 0.01, \"scalar\": true}, \"k_per_day\": {\"high\": 1000, \"log_scale\": true, \"low\": 0.1, \"scalar\": true}, \"log_amplitude\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"log_k\": {\"high\": 10.0, \"log_scale\": false, \"low\": -10.0, \"scalar\": true}, \"logit_lamb\": {\"high\": 4.602848117654388, \"log_scale\": false, \"low\": -5.955817303419269, \"scalar\": true}, \"memory_days_1\": {\"high\": 200, \"log_scale\": true, \"low\": 0.5, \"scalar\": true}, \"memory_days_2\": {\"high\": 200, \"log_scale\": true, \"low\": 0.5, \"scalar\": true}, \"memory_length\": {\"high\": 200, \"log_scale\": true, \"low\": 1, \"scalar\": true}, \"memory_length_delta\": {\"high\": 100, \"log_scale\": true, \"low\": 0.1, \"scalar\": true}, \"price_ratio\": {\"high\": 5.0, \"log_scale\": true, \"low\": 1.01, \"scalar\": true}, \"raw_exponents\": {\"high\": 10, \"log_scale\": false, \"low\": 0, \"scalar\": true}, \"raw_pre_exp_scaling\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"raw_width\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}, \"shift_exponent\": {\"high\": 125.0, \"log_scale\": true, \"low\": 1e-05, \"scalar\": true}, \"weights_logits\": {\"high\": 10, \"log_scale\": false, \"low\": -10, \"scalar\": true}}, \"storage\": {\"type\": \"sqlite\", \"url\": null}, \"study_name\": null, \"timeout\": 7200}, \"parameter_init_method\": \"gaussian\", \"sample_method\": \"uniform\", \"swa_freq\": 10, \"swa_start_frac\": 0.75, \"track_checkpoints\": false, \"train_on_hessian_trace\": false, \"training_data_kind\": \"historic\", \"use_gradient_clipping\": true, \"use_plateau_decay\": false, \"use_swa\": false, \"val_fraction\": 0.2, \"warmup_steps\": 100, \"weight_decay\": 0.01}, \"price_noise_sigma\": 0.0, \"protocol_fee_split\": 0.25, \"reclamm_arc_length_speed\": null, \"reclamm_centeredness_scaling\": false, \"reclamm_interpolation_method\": \"geometric\", \"reclamm_learn_arc_length_speed\": false, \"reclamm_learn_fees\": false, \"reclamm_use_shift_exponent\": true, \"return_val\": \"returns_over_hodl\", \"rule\": \"reclamm\", \"startDateString\": \"2025-01-01 00:00:00\", \"ste_max_change\": false, \"ste_min_max_weight\": false, \"ste_temperature\": 10.0, \"subsidary_pools\": [], \"tokens\": [\"BTC\", \"ETH\"], \"training_method\": \"optuna\", \"turnover_penalty\": 0.0, \"use_alt_lamb\": false, \"use_fused_reserves\": true, \"use_pre_exp_scaling\": true, \"weight_calculation_method\": \"auto\", \"weight_interpolation_method\": \"linear\", \"weight_interpolation_period\": 1440}, {\"centeredness_margin\": 0.3147276387836944, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8312010829673651, \"annualised_returns_over_hodl\": -0.050291612805094976, \"annualised_returns_over_uniform_hodl\": -0.019829514774061363, \"calmar\": -1.4270626658998706, \"daily_log_sharpe\": -2.8170377408060348, \"daily_returns\": 0.007170686746859631, \"fee_revenue_over_value\": 8.142442901228541e-05, \"jax_sharpe\": -2.38056974836613, \"return\": -0.5115360276937015, \"returns_over_hodl\": -0.02056694836138162, \"returns_over_uniform_hodl\": -0.008033892765772155, \"sharpe\": -2.511741106407347, \"sterling\": -3.5271395653440982, \"ulcer\": -0.11506745878585406}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.19857763893855068, \"optuna_trial_number\": 0, \"price_ratio\": 1.5973063737576239, \"shift_exponent\": 1.7924533302109447e-05, \"step\": 0, \"test_objective\": [{\"annualised_returns\": -0.8312010829673651, \"annualised_returns_over_hodl\": -0.050291612805094976, \"annualised_returns_over_uniform_hodl\": -0.019829514774061363, \"calmar\": -1.4270626658998706, \"daily_log_sharpe\": -2.8170377408060348, \"daily_returns\": 0.007170686746859631, \"fee_revenue_over_value\": 8.142442901228541e-05, \"jax_sharpe\": -2.38056974836613, \"return\": -0.5115360276937015, \"returns_over_hodl\": -0.02056694836138162, \"returns_over_uniform_hodl\": -0.008033892765772155, \"sharpe\": -2.511741106407347, \"sterling\": -3.5271395653440982, \"ulcer\": -0.11506745878585406}], \"train_objective\": [{\"annualised_returns\": 0.5272210556783168, \"annualised_returns_over_hodl\": 0.03704959988797518, \"annualised_returns_over_uniform_hodl\": 0.03704959988797518, \"calmar\": 0.8861208828137276, \"daily_log_sharpe\": 0.5550835136937159, \"daily_returns\": 0.008891717964706726, \"fee_revenue_over_value\": 3.378234615161209e-05, \"jax_sharpe\": 0.9378368424237751, \"return\": 0.2931555797944905, \"returns_over_hodl\": 0.022332651201108167, \"returns_over_uniform_hodl\": 0.022332651201108167, \"sharpe\": 0.9241895950308691, \"sterling\": 2.26539077335143, \"ulcer\": -0.12013865522083667}], \"train_return\": 0.2931555797944905, \"train_returns_over_hodl\": 0.022332651201108167, \"train_sharpe\": 0.9378368424237751, \"validation_return\": 0.062224629817726695, \"validation_returns_over_hodl\": 0.012812595298003604, \"validation_sharpe\": 1.2384715601320442}, {\"centeredness_margin\": 0.9652144107952862, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8320821203191187, \"annualised_returns_over_hodl\": -0.02216070527745706, \"annualised_returns_over_uniform_hodl\": -0.0249454647087608, \"calmar\": -1.4323673458067954, \"daily_log_sharpe\": -2.9927164031276545, \"daily_returns\": 0.006880705763984296, \"fee_revenue_over_value\": 4.271804038365441e-05, \"jax_sharpe\": -2.5994622279287585, \"return\": -0.5125644178438137, \"returns_over_hodl\": -0.00898472552402696, \"returns_over_uniform_hodl\": -0.010122333739439493, \"sharpe\": -2.710784701218345, \"sterling\": -3.6171965824460166, \"ulcer\": -0.11621888572376676}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.11389781063292784, \"optuna_trial_number\": 1, \"price_ratio\": 2.5010347770244694, \"shift_exponent\": 0.03727729494907189, \"step\": 1, \"test_objective\": [{\"annualised_returns\": -0.8320821203191187, \"annualised_returns_over_hodl\": -0.02216070527745706, \"annualised_returns_over_uniform_hodl\": -0.0249454647087608, \"calmar\": -1.4323673458067954, \"daily_log_sharpe\": -2.9927164031276545, \"daily_returns\": 0.006880705763984296, \"fee_revenue_over_value\": 4.271804038365441e-05, \"jax_sharpe\": -2.5994622279287585, \"return\": -0.5125644178438137, \"returns_over_hodl\": -0.00898472552402696, \"returns_over_uniform_hodl\": -0.010122333739439493, \"sharpe\": -2.710784701218345, \"sterling\": -3.6171965824460166, \"ulcer\": -0.11621888572376676}], \"train_objective\": [{\"annualised_returns\": 0.37461722607337045, \"annualised_returns_over_hodl\": -0.06657504557167804, \"annualised_returns_over_uniform_hodl\": -0.0665750455716777, \"calmar\": 0.7702994666555929, \"daily_log_sharpe\": 0.5447249952492602, \"daily_returns\": 0.008885152955737372, \"fee_revenue_over_value\": 6.540766484472939e-05, \"jax_sharpe\": 0.832573234044386, \"return\": 0.21309025753211586, \"returns_over_hodl\": -0.040964754352191934, \"returns_over_uniform_hodl\": -0.04096475435219171, \"sharpe\": 0.8304399373634881, \"sterling\": 1.995810141468259, \"ulcer\": -0.09068721991285439}], \"train_return\": 0.21309025753211586, \"train_returns_over_hodl\": -0.040964754352191934, \"train_sharpe\": 0.832573234044386, \"validation_return\": 0.042722666182133384, \"validation_returns_over_hodl\": -0.005663016323258052, \"validation_sharpe\": 0.8640206087892123}, {\"centeredness_margin\": 0.3192511805019692, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8232590984070586, \"annualised_returns_over_hodl\": -0.022813229731224016, \"annualised_returns_over_uniform_hodl\": 0.026287480506349326, \"calmar\": -1.4377078235276923, \"daily_log_sharpe\": -2.923628045329693, \"daily_returns\": 0.007217742692687175, \"fee_revenue_over_value\": 4.925402218633479e-05, \"jax_sharpe\": -2.52883588938604, \"return\": -0.5024071076750863, \"returns_over_hodl\": -0.009251115967999213, \"returns_over_uniform_hodl\": 0.010504996011146295, \"sharpe\": -2.6416875036014646, \"sterling\": -3.638912287077602, \"ulcer\": -0.11275966196417472}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.05185558916873091, \"optuna_trial_number\": 2, \"price_ratio\": 3.1008699292083852, \"shift_exponent\": 0.09196989731553974, \"step\": 2, \"test_objective\": [{\"annualised_returns\": -0.8232590984070586, \"annualised_returns_over_hodl\": -0.022813229731224016, \"annualised_returns_over_uniform_hodl\": 0.026287480506349326, \"calmar\": -1.4377078235276923, \"daily_log_sharpe\": -2.923628045329693, \"daily_returns\": 0.007217742692687175, \"fee_revenue_over_value\": 4.925402218633479e-05, \"jax_sharpe\": -2.52883588938604, \"return\": -0.5024071076750863, \"returns_over_hodl\": -0.009251115967999213, \"returns_over_uniform_hodl\": 0.010504996011146295, \"sharpe\": -2.6416875036014646, \"sterling\": -3.638912287077602, \"ulcer\": -0.11275966196417472}], \"train_objective\": [{\"annualised_returns\": 0.21596099152163895, \"annualised_returns_over_hodl\": -0.17430953754312817, \"annualised_returns_over_uniform_hodl\": -0.17430953754312817, \"calmar\": 0.4105513242031547, \"daily_log_sharpe\": 0.2964677152594782, \"daily_returns\": 0.008886135358035285, \"fee_revenue_over_value\": 8.312177476178387e-05, \"jax_sharpe\": 0.6257881339255866, \"return\": 0.12604703447287013, \"returns_over_hodl\": -0.10977869320817224, \"returns_over_uniform_hodl\": -0.10977869320817224, \"sharpe\": 0.5993503147931558, \"sterling\": 1.1163989220355095, \"ulcer\": -0.09648867809062449}], \"train_return\": 0.12604703447287013, \"train_returns_over_hodl\": -0.10977869320817224, \"train_sharpe\": 0.6257881339255865, \"validation_return\": 0.038056353982299784, \"validation_returns_over_hodl\": -0.004077361497044518, \"validation_sharpe\": 0.8401553603729407}, {\"centeredness_margin\": 0.43904752389527396, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8302554459604553, \"annualised_returns_over_hodl\": -0.032502794304975824, \"annualised_returns_over_uniform_hodl\": -0.014338451796855667, \"calmar\": -1.4297706981119163, \"daily_log_sharpe\": -2.8769956070017852, \"daily_returns\": 0.007046338047447419, \"fee_revenue_over_value\": 5.49679630866417e-05, \"jax_sharpe\": -2.4650732822376678, \"return\": -0.5104357944849082, \"returns_over_hodl\": -0.013219407963077634, \"returns_over_uniform_hodl\": -0.005799553868629959, \"sharpe\": -2.5812596183232963, \"sterling\": -3.5637359367353283, \"ulcer\": -0.11523391588983409}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.1891746288681631, \"optuna_trial_number\": 3, \"price_ratio\": 2.1459015884497172, \"shift_exponent\": 1.7586329132467026e-05, \"step\": 3, \"test_objective\": [{\"annualised_returns\": -0.8302554459604553, \"annualised_returns_over_hodl\": -0.032502794304975824, \"annualised_returns_over_uniform_hodl\": -0.014338451796855667, \"calmar\": -1.4297706981119163, \"daily_log_sharpe\": -2.8769956070017852, \"daily_returns\": 0.007046338047447419, \"fee_revenue_over_value\": 5.49679630866417e-05, \"jax_sharpe\": -2.4650732822376678, \"return\": -0.5104357944849082, \"returns_over_hodl\": -0.013219407963077634, \"returns_over_uniform_hodl\": -0.005799553868629959, \"sharpe\": -2.5812596183232963, \"sterling\": -3.5637359367353283, \"ulcer\": -0.11523391588983409}], \"train_objective\": [{\"annualised_returns\": 0.5458365547452653, \"annualised_returns_over_hodl\": 0.04969033436928405, \"annualised_returns_over_uniform_hodl\": 0.04969033436928405, \"calmar\": 0.9468949471181128, \"daily_log_sharpe\": 0.5889449400502048, \"daily_returns\": 0.008887869680937063, \"fee_revenue_over_value\": 2.923775177214017e-05, \"jax_sharpe\": 0.9606766258895345, \"return\": 0.30270251145088856, \"returns_over_hodl\": 0.029880188484032955, \"returns_over_uniform_hodl\": 0.029880188484032955, \"sharpe\": 0.9461201410552752, \"sterling\": 2.4179436540857426, \"ulcer\": -0.11497577257842463}], \"train_return\": 0.30270251145088856, \"train_returns_over_hodl\": 0.029880188484032955, \"train_sharpe\": 0.9606766258895345, \"validation_return\": 0.05712998897596511, \"validation_returns_over_hodl\": 0.008106234025711867, \"validation_sharpe\": 1.1223431898795002}, {\"centeredness_margin\": 0.7267397929465383, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8300509273017959, \"annualised_returns_over_hodl\": -0.0007192203736909875, \"annualised_returns_over_uniform_hodl\": -0.013150866257682248, \"calmar\": -1.4312325382740037, \"daily_log_sharpe\": -2.954739156786953, \"daily_returns\": 0.00680163530443928, \"fee_revenue_over_value\": 4.1915702943146726e-05, \"jax_sharpe\": -2.5726897913478473, \"return\": -0.51019832217831, \"returns_over_hodl\": -0.0002897194847445439, \"returns_over_uniform_hodl\": -0.005317298281918181, \"sharpe\": -2.6708523037388474, \"sterling\": -3.606549912723399, \"ulcer\": -0.11546241042602062}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.10555596391480204, \"optuna_trial_number\": 4, \"price_ratio\": 3.924720766515448, \"shift_exponent\": 68.14716110641697, \"step\": 4, \"test_objective\": [{\"annualised_returns\": -0.8300509273017959, \"annualised_returns_over_hodl\": -0.0007192203736909875, \"annualised_returns_over_uniform_hodl\": -0.013150866257682248, \"calmar\": -1.4312325382740037, \"daily_log_sharpe\": -2.954739156786953, \"daily_returns\": 0.00680163530443928, \"fee_revenue_over_value\": 4.1915702943146726e-05, \"jax_sharpe\": -2.5726897913478473, \"return\": -0.51019832217831, \"returns_over_hodl\": -0.0002897194847445439, \"returns_over_uniform_hodl\": -0.005317298281918181, \"sharpe\": -2.6708523037388474, \"sterling\": -3.606549912723399, \"ulcer\": -0.11546241042602062}], \"train_objective\": [{\"annualised_returns\": 0.33476607090112154, \"annualised_returns_over_hodl\": -0.09363571525848924, \"annualised_returns_over_uniform_hodl\": -0.09363571525848902, \"calmar\": 0.6724433193541911, \"daily_log_sharpe\": 0.4889383386791241, \"daily_returns\": 0.008885605720337782, \"fee_revenue_over_value\": 7.433492378876747e-05, \"jax_sharpe\": 0.7840365651051812, \"return\": 0.1916155271094535, \"returns_over_hodl\": -0.05794207589792455, \"returns_over_uniform_hodl\": -0.05794207589792444, \"sharpe\": 0.7754574964769436, \"sterling\": 1.7661806122697914, \"ulcer\": -0.09275358137776338}], \"train_return\": 0.1916155271094535, \"train_returns_over_hodl\": -0.05794207589792455, \"train_sharpe\": 0.784036565105181, \"validation_return\": 0.041158583364731216, \"validation_returns_over_hodl\": -0.00600624373543579, \"validation_sharpe\": 0.8419043474148417}, {\"centeredness_margin\": 0.8072246801368033, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8395473352006171, \"annualised_returns_over_hodl\": -0.04834785359183391, \"annualised_returns_over_uniform_hodl\": -0.06829398507456108, \"calmar\": -1.4268436568296685, \"daily_log_sharpe\": -3.0604488026966497, \"daily_returns\": 0.006860060528582505, \"fee_revenue_over_value\": 0.00013053192193210805, \"jax_sharpe\": -2.67130644932066, \"return\": -0.5214105135300485, \"returns_over_hodl\": -0.019760114474260848, \"returns_over_uniform_hodl\": -0.02808686664176241, \"sharpe\": -2.7795832775533937, \"sterling\": -3.61848222782676, \"ulcer\": -0.11815069910943962}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.049670643846860374, \"optuna_trial_number\": 5, \"price_ratio\": 1.344691468277918, \"shift_exponent\": 0.8210237531445104, \"step\": 5, \"test_objective\": [{\"annualised_returns\": -0.8395473352006171, \"annualised_returns_over_hodl\": -0.04834785359183391, \"annualised_returns_over_uniform_hodl\": -0.06829398507456108, \"calmar\": -1.4268436568296685, \"daily_log_sharpe\": -3.0604488026966497, \"daily_returns\": 0.006860060528582505, \"fee_revenue_over_value\": 0.00013053192193210805, \"jax_sharpe\": -2.67130644932066, \"return\": -0.5214105135300485, \"returns_over_hodl\": -0.019760114474260848, \"returns_over_uniform_hodl\": -0.02808686664176241, \"sharpe\": -2.7795832775533937, \"sterling\": -3.61848222782676, \"ulcer\": -0.11815069910943962}], \"train_objective\": [{\"annualised_returns\": 0.1861018112330297, \"annualised_returns_over_hodl\": -0.19458522118182198, \"annualised_returns_over_uniform_hodl\": -0.19458522118182175, \"calmar\": 0.3640734320676346, \"daily_log_sharpe\": 0.2877503809960303, \"daily_returns\": 0.008897474612630109, \"fee_revenue_over_value\": 0.00023765181164023852, \"jax_sharpe\": 0.5853368568307571, \"return\": 0.10917745234615506, \"returns_over_hodl\": -0.12311531324812164, \"returns_over_uniform_hodl\": -0.12311531324812153, \"sharpe\": 0.5757657097668546, \"sterling\": 0.9693219607576165, \"ulcer\": -0.0948926367451816}], \"train_return\": 0.10917745234615506, \"train_returns_over_hodl\": -0.12311531324812164, \"train_sharpe\": 0.5853368568307572, \"validation_return\": 0.028668282357580566, \"validation_returns_over_hodl\": -0.019987181550927713, \"validation_sharpe\": 0.6514927813354684}, {\"centeredness_margin\": 0.9217468093793482, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8341536983640978, \"annualised_returns_over_hodl\": -0.2586272566646154, \"annualised_returns_over_uniform_hodl\": -0.03697456829085399, \"calmar\": -1.4320358651525755, \"daily_log_sharpe\": -2.851774580486881, \"daily_returns\": 0.00869052495023827, \"fee_revenue_over_value\": 2.4558903531054318e-05, \"jax_sharpe\": -2.3741025222145917, \"return\": -0.514995231314694, \"returns_over_hodl\": -0.11354064162063693, \"returns_over_uniform_hodl\": -0.01505879725124526, \"sharpe\": -2.5523705468775657, \"sterling\": -3.7316887871060036, \"ulcer\": -0.10386452366738227}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.06967069839345606, \"optuna_trial_number\": 6, \"price_ratio\": 1.034144221115526, \"shift_exponent\": 0.004792596495273959, \"step\": 6, \"test_objective\": [{\"annualised_returns\": -0.8341536983640978, \"annualised_returns_over_hodl\": -0.2586272566646154, \"annualised_returns_over_uniform_hodl\": -0.03697456829085399, \"calmar\": -1.4320358651525755, \"daily_log_sharpe\": -2.851774580486881, \"daily_returns\": 0.00869052495023827, \"fee_revenue_over_value\": 2.4558903531054318e-05, \"jax_sharpe\": -2.3741025222145917, \"return\": -0.514995231314694, \"returns_over_hodl\": -0.11354064162063693, \"returns_over_uniform_hodl\": -0.01505879725124526, \"sharpe\": -2.5523705468775657, \"sterling\": -3.7316887871060036, \"ulcer\": -0.10386452366738227}], \"train_objective\": [{\"annualised_returns\": -0.19549879576637252, \"annualised_returns_over_hodl\": -0.4537086501931974, \"annualised_returns_over_uniform_hodl\": -0.4537086501931974, \"calmar\": -0.31384982089182767, \"daily_log_sharpe\": -0.3542281975164275, \"daily_returns\": 0.008839590140184822, \"fee_revenue_over_value\": 1.3974557576498594e-05, \"jax_sharpe\": 0.05281052889202954, \"return\": -0.12371931644978074, \"returns_over_hodl\": -0.3072369880253817, \"returns_over_uniform_hodl\": -0.3072369880253817, \"sharpe\": 0.00019063716483089637, \"sterling\": -0.8913172187805238, \"ulcer\": -0.11562654867221975}], \"train_return\": -0.12371931644978074, \"train_returns_over_hodl\": -0.3072369880253817, \"train_sharpe\": 0.052810528892029536, \"validation_return\": 0.03048627045142238, \"validation_returns_over_hodl\": -7.37649941129348e-11, \"validation_sharpe\": 0.80172657449514}, {\"centeredness_margin\": 0.9236160379223946, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8249718761415619, \"annualised_returns_over_hodl\": -0.05171974449509065, \"annualised_returns_over_uniform_hodl\": 0.01634183504472775, \"calmar\": -1.4378006442812294, \"daily_log_sharpe\": -2.9469976970766574, \"daily_returns\": 0.0073428575715866725, \"fee_revenue_over_value\": 1.3570641025066518e-05, \"jax_sharpe\": -2.5520224414974653, \"return\": -0.5043548077045156, \"returns_over_hodl\": -0.021160379454276068, \"returns_over_uniform_hodl\": 0.006549632820017637, \"sharpe\": -2.6663307259572075, \"sterling\": -3.6388650310490256, \"ulcer\": -0.11331995215535867}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.10904287626162008, \"optuna_trial_number\": 7, \"price_ratio\": 4.575957116754659, \"shift_exponent\": 0.0020693780440319354, \"step\": 7, \"test_objective\": [{\"annualised_returns\": -0.8249718761415619, \"annualised_returns_over_hodl\": -0.05171974449509065, \"annualised_returns_over_uniform_hodl\": 0.01634183504472775, \"calmar\": -1.4378006442812294, \"daily_log_sharpe\": -2.9469976970766574, \"daily_returns\": 0.0073428575715866725, \"fee_revenue_over_value\": 1.3570641025066518e-05, \"jax_sharpe\": -2.5520224414974653, \"return\": -0.5043548077045156, \"returns_over_hodl\": -0.021160379454276068, \"returns_over_uniform_hodl\": 0.006549632820017637, \"sharpe\": -2.6663307259572075, \"sterling\": -3.6388650310490256, \"ulcer\": -0.11331995215535867}], \"train_objective\": [{\"annualised_returns\": 0.3607713620517323, \"annualised_returns_over_hodl\": -0.07597699016270854, \"annualised_returns_over_uniform_hodl\": -0.07597699016270831, \"calmar\": 0.7058206789555646, \"daily_log_sharpe\": 0.4810265457872633, \"daily_returns\": 0.008884062769254119, \"fee_revenue_over_value\": 4.936396397743963e-06, \"jax_sharpe\": 0.8080512977926353, \"return\": 0.2056571669739602, \"returns_over_hodl\": -0.04684114795530703, \"returns_over_uniform_hodl\": -0.04684114795530692, \"sharpe\": 0.7822611933489498, \"sterling\": 1.871380431271233, \"ulcer\": -0.09471819034152562}], \"train_return\": 0.2056571669739602, \"train_returns_over_hodl\": -0.04684114795530703, \"train_sharpe\": 0.8080512977926353, \"validation_return\": 0.04356371175316398, \"validation_returns_over_hodl\": 0.0013166816662713021, \"validation_sharpe\": 0.949929565920463}, {\"centeredness_margin\": 0.3953683725711484, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8161142628554787, \"annualised_returns_over_hodl\": -0.14689661798470455, \"annualised_returns_over_uniform_hodl\": 0.0677756431827583, \"calmar\": -1.4446277391253322, \"daily_log_sharpe\": -2.847682605463989, \"daily_returns\": 0.008532967236451807, \"fee_revenue_over_value\": 0.00010412289508181116, \"jax_sharpe\": -2.402415845784038, \"return\": -0.49440161222175616, \"returns_over_hodl\": -0.061980730226495084, \"returns_over_uniform_hodl\": 0.02676244919408388, \"sharpe\": -2.563643500709178, \"sterling\": -3.660092349923256, \"ulcer\": -0.10776864273884076}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0513275499510886, \"optuna_trial_number\": 8, \"price_ratio\": 1.3934693702999432, \"shift_exponent\": 0.004726758995977553, \"step\": 8, \"test_objective\": [{\"annualised_returns\": -0.8161142628554787, \"annualised_returns_over_hodl\": -0.14689661798470455, \"annualised_returns_over_uniform_hodl\": 0.0677756431827583, \"calmar\": -1.4446277391253322, \"daily_log_sharpe\": -2.847682605463989, \"daily_returns\": 0.008532967236451807, \"fee_revenue_over_value\": 0.00010412289508181116, \"jax_sharpe\": -2.402415845784038, \"return\": -0.49440161222175616, \"returns_over_hodl\": -0.061980730226495084, \"returns_over_uniform_hodl\": 0.02676244919408388, \"sharpe\": -2.563643500709178, \"sterling\": -3.660092349923256, \"ulcer\": -0.10776864273884076}], \"train_objective\": [{\"annualised_returns\": -0.03216682805387361, \"annualised_returns_over_hodl\": -0.34279913180004584, \"annualised_returns_over_uniform_hodl\": -0.34279913180004573, \"calmar\": -0.054883861578959865, \"daily_log_sharpe\": -0.09482569417275354, \"daily_returns\": 0.008901066315454974, \"fee_revenue_over_value\": 8.451801994683857e-05, \"jax_sharpe\": 0.29822131286662, \"return\": -0.019654449927912432, \"returns_over_hodl\": -0.2249662136881605, \"returns_over_uniform_hodl\": -0.2249662136881604, \"sharpe\": 0.2452731718659548, \"sterling\": -0.154616010268441, \"ulcer\": -0.10847806923527921}], \"train_return\": -0.019654449927912432, \"train_returns_over_hodl\": -0.2249662136881605, \"train_sharpe\": 0.29822131286661996, \"validation_return\": 0.033117291145951855, \"validation_returns_over_hodl\": 0.0025531835815058024, \"validation_sharpe\": 0.8555351812359692}, {\"centeredness_margin\": 0.7631307873790555, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8313675633384072, \"annualised_returns_over_hodl\": -0.06130615871652745, \"annualised_returns_over_uniform_hodl\": -0.020796222078430016, \"calmar\": -1.4263315894310478, \"daily_log_sharpe\": -2.7843546937168826, \"daily_returns\": 0.007267905027056986, \"fee_revenue_over_value\": 2.6793880138222024e-05, \"jax_sharpe\": -2.336123076444815, \"return\": -0.5117301056880721, \"returns_over_hodl\": -0.02515770627185976, \"returns_over_uniform_hodl\": -0.008428023763124348, \"sharpe\": -2.474224365710967, \"sterling\": -3.5123499689900815, \"ulcer\": -0.11466630103778763}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.21062453340428577, \"optuna_trial_number\": 9, \"price_ratio\": 1.4561284677861681, \"shift_exponent\": 4.120732923893584e-05, \"step\": 9, \"test_objective\": [{\"annualised_returns\": -0.8313675633384072, \"annualised_returns_over_hodl\": -0.06130615871652745, \"annualised_returns_over_uniform_hodl\": -0.020796222078430016, \"calmar\": -1.4263315894310478, \"daily_log_sharpe\": -2.7843546937168826, \"daily_returns\": 0.007267905027056986, \"fee_revenue_over_value\": 2.6793880138222024e-05, \"jax_sharpe\": -2.336123076444815, \"return\": -0.5117301056880721, \"returns_over_hodl\": -0.02515770627185976, \"returns_over_uniform_hodl\": -0.008428023763124348, \"sharpe\": -2.474224365710967, \"sterling\": -3.5123499689900815, \"ulcer\": -0.11466630103778763}], \"train_objective\": [{\"annualised_returns\": 0.5233352528426132, \"annualised_returns_over_hodl\": 0.03441097055463338, \"annualised_returns_over_uniform_hodl\": 0.03441097055463338, \"calmar\": 0.8717387606857198, \"daily_log_sharpe\": 0.5470416706728221, \"daily_returns\": 0.008896608412888241, \"fee_revenue_over_value\": 7.840623754027143e-06, \"jax_sharpe\": 0.9334678617406748, \"return\": 0.29115699566277753, \"returns_over_hodl\": 0.02075262645702658, \"returns_over_uniform_hodl\": 0.02075262645702658, \"sharpe\": 0.918656529354088, \"sterling\": 2.2301241990097074, \"ulcer\": -0.1216028446691469}], \"train_return\": 0.29115699566277753, \"train_returns_over_hodl\": 0.02075262645702658, \"train_sharpe\": 0.9334678617406748, \"validation_return\": 0.06459789759978452, \"validation_returns_over_hodl\": 0.015310709909341247, \"validation_sharpe\": 1.3055843904287245}, {\"centeredness_margin\": 0.7229025600980851, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8305726682491644, \"annualised_returns_over_hodl\": -0.00833144363377658, \"annualised_returns_over_uniform_hodl\": -0.01618047738632289, \"calmar\": -1.4310374333948934, \"daily_log_sharpe\": -2.93342186153626, \"daily_returns\": 0.006825928174585989, \"fee_revenue_over_value\": 3.860423296892394e-05, \"jax_sharpe\": -2.5388268368621496, \"return\": -0.5108044684488182, \"returns_over_hodl\": -0.003363773006633486, \"returns_over_uniform_hodl\": -0.006548252027660517, \"sharpe\": -2.645669231716398, \"sterling\": -3.587478153291932, \"ulcer\": -0.11576585418044211}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.109363146691791, \"optuna_trial_number\": 10, \"price_ratio\": 4.150720577555207, \"shift_exponent\": 0.018197431627551088, \"step\": 10, \"test_objective\": [{\"annualised_returns\": -0.8305726682491644, \"annualised_returns_over_hodl\": -0.00833144363377658, \"annualised_returns_over_uniform_hodl\": -0.01618047738632289, \"calmar\": -1.4310374333948934, \"daily_log_sharpe\": -2.93342186153626, \"daily_returns\": 0.006825928174585989, \"fee_revenue_over_value\": 3.860423296892394e-05, \"jax_sharpe\": -2.5388268368621496, \"return\": -0.5108044684488182, \"returns_over_hodl\": -0.003363773006633486, \"returns_over_uniform_hodl\": -0.006548252027660517, \"sharpe\": -2.645669231716398, \"sterling\": -3.587478153291932, \"ulcer\": -0.11576585418044211}], \"train_objective\": [{\"annualised_returns\": 0.35622993106156775, \"annualised_returns_over_hodl\": -0.07906081956236077, \"annualised_returns_over_uniform_hodl\": -0.07906081956236088, \"calmar\": 0.7203778707493282, \"daily_log_sharpe\": 0.506843808922815, \"daily_returns\": 0.008885038907995583, \"fee_revenue_over_value\": 5.76480075207445e-05, \"jax_sharpe\": 0.8085345671916063, \"return\": 0.20321265600066285, \"returns_over_hodl\": -0.04877370999445341, \"returns_over_uniform_hodl\": -0.04877370999445352, \"sharpe\": 0.7964091268035137, \"sterling\": 1.9023322963516551, \"ulcer\": -0.0911858686724364}], \"train_return\": 0.20321265600066285, \"train_returns_over_hodl\": -0.04877370999445341, \"train_sharpe\": 0.8085345671916063, \"validation_return\": 0.04405967395432664, \"validation_returns_over_hodl\": -0.0019720410781097764, \"validation_sharpe\": 0.8953664533184503}, {\"centeredness_margin\": 0.9114420352685174, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8324789238155123, \"annualised_returns_over_hodl\": -0.017957795166049895, \"annualised_returns_over_uniform_hodl\": -0.02724959723778808, \"calmar\": -1.4314934026989827, \"daily_log_sharpe\": -2.9927582884990453, \"daily_returns\": 0.0068467334942168175, \"fee_revenue_over_value\": 6.447886302139568e-05, \"jax_sharpe\": -2.608169256037157, \"return\": -0.513028639023525, \"returns_over_hodl\": -0.007271443516858045, \"returns_over_uniform_hodl\": -0.011065067907447212, \"sharpe\": -2.710603559079583, \"sterling\": -3.617254339084041, \"ulcer\": -0.11633660778511494}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.11720113603959184, \"optuna_trial_number\": 11, \"price_ratio\": 2.154842091081405, \"shift_exponent\": 0.6420231506041836, \"step\": 11, \"test_objective\": [{\"annualised_returns\": -0.8324789238155123, \"annualised_returns_over_hodl\": -0.017957795166049895, \"annualised_returns_over_uniform_hodl\": -0.02724959723778808, \"calmar\": -1.4314934026989827, \"daily_log_sharpe\": -2.9927582884990453, \"daily_returns\": 0.0068467334942168175, \"fee_revenue_over_value\": 6.447886302139568e-05, \"jax_sharpe\": -2.608169256037157, \"return\": -0.513028639023525, \"returns_over_hodl\": -0.007271443516858045, \"returns_over_uniform_hodl\": -0.011065067907447212, \"sharpe\": -2.710603559079583, \"sterling\": -3.617254339084041, \"ulcer\": -0.11633660778511494}], \"train_objective\": [{\"annualised_returns\": 0.3743402028842202, \"annualised_returns_over_hodl\": -0.0667631563801292, \"annualised_returns_over_uniform_hodl\": -0.0667631563801292, \"calmar\": 0.7641891702213396, \"daily_log_sharpe\": 0.5425108168744578, \"daily_returns\": 0.008886323919519866, \"fee_revenue_over_value\": 0.00011654369347895392, \"jax_sharpe\": 0.831976766898351, \"return\": 0.2129418280465536, \"returns_over_hodl\": -0.04108209855412859, \"returns_over_uniform_hodl\": -0.04108209855412859, \"sharpe\": 0.8294999676552521, \"sterling\": 1.989409224443268, \"ulcer\": -0.09135470256454852}], \"train_return\": 0.2129418280465536, \"train_returns_over_hodl\": -0.04108209855412859, \"train_sharpe\": 0.8319767668983509, \"validation_return\": 0.041488854114796636, \"validation_returns_over_hodl\": -0.007055358052271976, \"validation_sharpe\": 0.8449491216657721}, {\"centeredness_margin\": 0.47563356501562465, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8295566650303866, \"annualised_returns_over_hodl\": -0.05235765407348891, \"annualised_returns_over_uniform_hodl\": -0.010280816503143275, \"calmar\": -1.4282236673345743, \"daily_log_sharpe\": -2.8093273616259338, \"daily_returns\": 0.007271046195127116, \"fee_revenue_over_value\": 8.149692577839692e-05, \"jax_sharpe\": -2.373295944381749, \"return\": -0.5096251242923733, \"returns_over_hodl\": -0.021425622277323808, \"returns_over_uniform_hodl\": -0.004153255675248713, \"sharpe\": -2.5047500718244535, \"sterling\": -3.53380948633333, \"ulcer\": -0.1143083062121382}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.19736351006304206, \"optuna_trial_number\": 12, \"price_ratio\": 1.5454116532695201, \"shift_exponent\": 8.993755337959179e-05, \"step\": 12, \"test_objective\": [{\"annualised_returns\": -0.8295566650303866, \"annualised_returns_over_hodl\": -0.05235765407348891, \"annualised_returns_over_uniform_hodl\": -0.010280816503143275, \"calmar\": -1.4282236673345743, \"daily_log_sharpe\": -2.8093273616259338, \"daily_returns\": 0.007271046195127116, \"fee_revenue_over_value\": 8.149692577839692e-05, \"jax_sharpe\": -2.373295944381749, \"return\": -0.5096251242923733, \"returns_over_hodl\": -0.021425622277323808, \"returns_over_uniform_hodl\": -0.004153255675248713, \"sharpe\": -2.5047500718244535, \"sterling\": -3.53380948633333, \"ulcer\": -0.1143083062121382}], \"train_objective\": [{\"annualised_returns\": 0.513220188233853, \"annualised_returns_over_hodl\": 0.027542401223197066, \"annualised_returns_over_uniform_hodl\": 0.027542401223197288, \"calmar\": 0.8594057294451084, \"daily_log_sharpe\": 0.540247867802316, \"daily_returns\": 0.008895566520347753, \"fee_revenue_over_value\": 1.886959426813164e-05, \"jax_sharpe\": 0.9254775560568712, \"return\": 0.2859451011701135, \"returns_over_hodl\": 0.016632248369718106, \"returns_over_uniform_hodl\": 0.016632248369718328, \"sharpe\": 0.9100386485863743, \"sterling\": 2.200642890282185, \"ulcer\": -0.12065291576672835}], \"train_return\": 0.2859451011701135, \"train_returns_over_hodl\": 0.016632248369718106, \"train_sharpe\": 0.9254775560568713, \"validation_return\": 0.06168325164783228, \"validation_returns_over_hodl\": 0.013092387303135444, \"validation_sharpe\": 1.2496497023257285}, {\"centeredness_margin\": 0.7944991216991398, \"continuous_test_metrics\": [{\"annualised_returns\": -0.830379942839416, \"annualised_returns_over_hodl\": -0.037185652384764456, \"annualised_returns_over_uniform_hodl\": -0.015061372111778937, \"calmar\": -1.428955435365574, \"daily_log_sharpe\": -2.855951347168461, \"daily_returns\": 0.007080654204007596, \"fee_revenue_over_value\": 2.4195548061486552e-05, \"jax_sharpe\": -2.435679552302757, \"return\": -0.510580434959788, \"returns_over_hodl\": -0.015145749117225105, \"returns_over_uniform_hodl\": -0.006093287812072301, \"sharpe\": -2.5570773373925206, \"sterling\": -3.55167878394649, \"ulcer\": -0.11510940243879925}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.18835137403600594, \"optuna_trial_number\": 13, \"price_ratio\": 1.8752691251044875, \"shift_exponent\": 1.6337270838164026e-05, \"step\": 13, \"test_objective\": [{\"annualised_returns\": -0.830379942839416, \"annualised_returns_over_hodl\": -0.037185652384764456, \"annualised_returns_over_uniform_hodl\": -0.015061372111778937, \"calmar\": -1.428955435365574, \"daily_log_sharpe\": -2.855951347168461, \"daily_returns\": 0.007080654204007596, \"fee_revenue_over_value\": 2.4195548061486552e-05, \"jax_sharpe\": -2.435679552302757, \"return\": -0.510580434959788, \"returns_over_hodl\": -0.015145749117225105, \"returns_over_uniform_hodl\": -0.006093287812072301, \"sharpe\": -2.5570773373925206, \"sterling\": -3.55167878394649, \"ulcer\": -0.11510940243879925}], \"train_objective\": [{\"annualised_returns\": 0.5339209879792657, \"annualised_returns_over_hodl\": 0.04159914566993117, \"annualised_returns_over_uniform_hodl\": 0.04159914566993117, \"calmar\": 0.9138946827347675, \"daily_log_sharpe\": 0.5693669913702121, \"daily_returns\": 0.008887533155324047, \"fee_revenue_over_value\": 8.080391229929259e-06, \"jax_sharpe\": 0.9465165666524029, \"return\": 0.29659686884783776, \"returns_over_hodl\": 0.025053238125395394, \"returns_over_uniform_hodl\": 0.025053238125395394, \"sharpe\": 0.9327078526139801, \"sterling\": 2.32768317096358, \"ulcer\": -0.11771272914665726}], \"train_return\": 0.29659686884783776, \"train_returns_over_hodl\": 0.025053238125395394, \"train_sharpe\": 0.946516566652403, \"validation_return\": 0.058730518359496386, \"validation_returns_over_hodl\": 0.009551668152397275, \"validation_sharpe\": 1.1590768230765836}, {\"centeredness_margin\": 0.30899321083127573, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8408276362356205, \"annualised_returns_over_hodl\": -0.24256277861733788, \"annualised_returns_over_uniform_hodl\": -0.07572835319003912, \"calmar\": -1.4297141807861329, \"daily_log_sharpe\": -2.6602481800602953, \"daily_returns\": 0.008967689676384007, \"fee_revenue_over_value\": 5.983282888168915e-05, \"jax_sharpe\": -2.1827855711584343, \"return\": -0.522952174152203, \"returns_over_hodl\": -0.10585422321399374, \"returns_over_uniform_hodl\": -0.031217650430819144, \"sharpe\": -2.3231069409951344, \"sterling\": -3.339834071909653, \"ulcer\": -0.11709584143105936}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2675741016367101, \"optuna_trial_number\": 14, \"price_ratio\": 1.0784919293760131, \"shift_exponent\": 1.5639783899759538e-05, \"step\": 14, \"test_objective\": [{\"annualised_returns\": -0.8408276362356205, \"annualised_returns_over_hodl\": -0.24256277861733788, \"annualised_returns_over_uniform_hodl\": -0.07572835319003912, \"calmar\": -1.4297141807861329, \"daily_log_sharpe\": -2.6602481800602953, \"daily_returns\": 0.008967689676384007, \"fee_revenue_over_value\": 5.983282888168915e-05, \"jax_sharpe\": -2.1827855711584343, \"return\": -0.522952174152203, \"returns_over_hodl\": -0.10585422321399374, \"returns_over_uniform_hodl\": -0.031217650430819144, \"sharpe\": -2.3231069409951344, \"sterling\": -3.339834071909653, \"ulcer\": -0.11709584143105936}], \"train_objective\": [{\"annualised_returns\": 0.5068015756517763, \"annualised_returns_over_hodl\": 0.02318388378046743, \"annualised_returns_over_uniform_hodl\": 0.02318388378046743, \"calmar\": 0.8217641846806102, \"daily_log_sharpe\": 0.5102613464905734, \"daily_returns\": 0.008964724855355757, \"fee_revenue_over_value\": 1.905222349313506e-05, \"jax_sharpe\": 0.9168076824427702, \"return\": 0.28263074308627023, \"returns_over_hodl\": 0.014012009521563895, \"returns_over_uniform_hodl\": 0.014012009521563895, \"sharpe\": 0.8898544662170093, \"sterling\": 2.11383171580962, \"ulcer\": -0.1256045229553684}], \"train_return\": 0.28263074308627023, \"train_returns_over_hodl\": 0.014012009521563895, \"train_sharpe\": 0.9168076824427701, \"validation_return\": 0.06264546569370699, \"validation_returns_over_hodl\": 0.010742059099857482, \"validation_sharpe\": 1.4175765115053374}, {\"centeredness_margin\": 0.031175537700796896, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8303859637642476, \"annualised_returns_over_hodl\": -0.04040450018609354, \"annualised_returns_over_uniform_hodl\": -0.015096334023368185, \"calmar\": -1.4285912880687461, \"daily_log_sharpe\": -2.8461736855035196, \"daily_returns\": 0.007117763299864432, \"fee_revenue_over_value\": 8.355183772048136e-05, \"jax_sharpe\": -2.4225629293079325, \"return\": -0.510587431677449, \"returns_over_hodl\": -0.01647310205549335, \"returns_over_uniform_hodl\": -0.00610749665279553, \"sharpe\": -2.5459097953535488, \"sterling\": -3.546052398243033, \"ulcer\": -0.11504128141588546}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.19056544524686986, \"optuna_trial_number\": 15, \"price_ratio\": 1.7861766594548845, \"shift_exponent\": 3.505996170227486e-05, \"step\": 15, \"test_objective\": [{\"annualised_returns\": -0.8303859637642476, \"annualised_returns_over_hodl\": -0.04040450018609354, \"annualised_returns_over_uniform_hodl\": -0.015096334023368185, \"calmar\": -1.4285912880687461, \"daily_log_sharpe\": -2.8461736855035196, \"daily_returns\": 0.007117763299864432, \"fee_revenue_over_value\": 8.355183772048136e-05, \"jax_sharpe\": -2.4225629293079325, \"return\": -0.510587431677449, \"returns_over_hodl\": -0.01647310205549335, \"returns_over_uniform_hodl\": -0.00610749665279553, \"sharpe\": -2.5459097953535488, \"sterling\": -3.546052398243033, \"ulcer\": -0.11504128141588546}], \"train_objective\": [{\"annualised_returns\": 0.5334724941426519, \"annualised_returns_over_hodl\": 0.0412945988251352, \"annualised_returns_over_uniform_hodl\": 0.0412945988251352, \"calmar\": 0.9089427943621707, \"daily_log_sharpe\": 0.566366312001129, \"daily_returns\": 0.008889753273739162, \"fee_revenue_over_value\": 6.486395866143006e-05, \"jax_sharpe\": 0.9453184362772594, \"return\": 0.29636669343264965, \"returns_over_hodl\": 0.024871267876704906, \"returns_over_uniform_hodl\": 0.024871267876704906, \"sharpe\": 0.9314073834915214, \"sterling\": 2.31684795965029, \"ulcer\": -0.11835220709545792}], \"train_return\": 0.29636669343264965, \"train_returns_over_hodl\": 0.024871267876704906, \"train_sharpe\": 0.9453184362772595, \"validation_return\": 0.05948825266196245, \"validation_returns_over_hodl\": 0.01035205327694988, \"validation_sharpe\": 1.1777550780186088}, {\"centeredness_margin\": 0.890696589057117, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8281961951378582, \"annualised_returns_over_hodl\": -0.038494364722348906, \"annualised_returns_over_uniform_hodl\": -0.002380929121537423, \"calmar\": -1.4307992942873047, \"daily_log_sharpe\": -2.8517315615792853, \"daily_returns\": 0.007186847518235992, \"fee_revenue_over_value\": 1.603157002112571e-05, \"jax_sharpe\": -2.4295455411715023, \"return\": -0.508052490933995, \"returns_over_hodl\": -0.015685101255447198, \"returns_over_uniform_hodl\": -0.0009595728673814641, \"sharpe\": -2.554789780051911, \"sterling\": -3.5655692515307047, \"ulcer\": -0.1142528145290365}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.18331815309755178, \"optuna_trial_number\": 16, \"price_ratio\": 1.7645234228930982, \"shift_exponent\": 0.0001484542426769082, \"step\": 16, \"test_objective\": [{\"annualised_returns\": -0.8281961951378582, \"annualised_returns_over_hodl\": -0.038494364722348906, \"annualised_returns_over_uniform_hodl\": -0.002380929121537423, \"calmar\": -1.4307992942873047, \"daily_log_sharpe\": -2.8517315615792853, \"daily_returns\": 0.007186847518235992, \"fee_revenue_over_value\": 1.603157002112571e-05, \"jax_sharpe\": -2.4295455411715023, \"return\": -0.508052490933995, \"returns_over_hodl\": -0.015685101255447198, \"returns_over_uniform_hodl\": -0.0009595728673814641, \"sharpe\": -2.554789780051911, \"sterling\": -3.5655692515307047, \"ulcer\": -0.1142528145290365}], \"train_objective\": [{\"annualised_returns\": 0.5147266487111959, \"annualised_returns_over_hodl\": 0.02856535348637257, \"annualised_returns_over_uniform_hodl\": 0.02856535348637257, \"calmar\": 0.8770663334839929, \"daily_log_sharpe\": 0.5482196532804594, \"daily_returns\": 0.008890390843547435, \"fee_revenue_over_value\": 3.57804504468688e-06, \"jax_sharpe\": 0.9288465105367153, \"return\": 0.2867221864351952, \"returns_over_hodl\": 0.01724659025686104, \"returns_over_uniform_hodl\": 0.01724659025686104, \"sharpe\": 0.9128754326458223, \"sterling\": 2.2379019707596086, \"ulcer\": -0.11832993421867889}], \"train_return\": 0.2867221864351952, \"train_returns_over_hodl\": 0.01724659025686104, \"train_sharpe\": 0.9288465105367154, \"validation_return\": 0.05784097426711576, \"validation_returns_over_hodl\": 0.009900191147423243, \"validation_sharpe\": 1.1680028800254914}, {\"centeredness_margin\": 0.2656799080981239, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8367163192308819, \"annualised_returns_over_hodl\": -0.27008278673686836, \"annualised_returns_over_uniform_hodl\": -0.051855027138585585, \"calmar\": -1.434864307160716, \"daily_log_sharpe\": -2.590924862607728, \"daily_returns\": 0.008690524952244583, \"fee_revenue_over_value\": 3.4479039071954005e-05, \"jax_sharpe\": -2.129862816789058, \"return\": -0.5180274670576843, \"returns_over_hodl\": -0.11908275991004891, \"returns_over_uniform_hodl\": -0.02121662107597655, \"sharpe\": -2.248640676411852, \"sterling\": -3.3052734891835978, \"ulcer\": -0.11535250418560554}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2737221518025831, \"optuna_trial_number\": 17, \"price_ratio\": 1.0259768121803423, \"shift_exponent\": 1.2137654048546867e-05, \"step\": 17, \"test_objective\": [{\"annualised_returns\": -0.8367163192308819, \"annualised_returns_over_hodl\": -0.27008278673686836, \"annualised_returns_over_uniform_hodl\": -0.051855027138585585, \"calmar\": -1.434864307160716, \"daily_log_sharpe\": -2.590924862607728, \"daily_returns\": 0.008690524952244583, \"fee_revenue_over_value\": 3.4479039071954005e-05, \"jax_sharpe\": -2.129862816789058, \"return\": -0.5180274670576843, \"returns_over_hodl\": -0.11908275991004891, \"returns_over_uniform_hodl\": -0.02121662107597655, \"sharpe\": -2.248640676411852, \"sterling\": -3.3052734891835978, \"ulcer\": -0.11535250418560554}], \"train_objective\": [{\"annualised_returns\": 0.5071254690794667, \"annualised_returns_over_hodl\": 0.02340382152185949, \"annualised_returns_over_uniform_hodl\": 0.02340382152185949, \"calmar\": 0.8195659130211637, \"daily_log_sharpe\": 0.49556785678511706, \"daily_returns\": 0.009116787861735958, \"fee_revenue_over_value\": 1.6538638924599666e-05, \"jax_sharpe\": 0.9169477686721593, \"return\": 0.2827981236047856, \"returns_over_hodl\": 0.014144335880377001, \"returns_over_uniform_hodl\": 0.014144335880377001, \"sharpe\": 0.8755625253474213, \"sterling\": 2.1099625633322128, \"ulcer\": -0.12610065819758498}], \"train_return\": 0.2827981236047856, \"train_returns_over_hodl\": 0.014144335880377001, \"train_sharpe\": 0.9169477686721592, \"validation_return\": 0.057912090811726546, \"validation_returns_over_hodl\": 0.0010309555075063148, \"validation_sharpe\": 1.3400769090219544}, {\"centeredness_margin\": 0.04138523717244236, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8443216671210937, \"annualised_returns_over_hodl\": -0.022979868832079253, \"annualised_returns_over_uniform_hodl\": -0.09601726267247679, \"calmar\": -1.411658792890672, \"daily_log_sharpe\": -2.8105941694881467, \"daily_returns\": 0.006522547972040203, \"fee_revenue_over_value\": 0.00016410692208874614, \"jax_sharpe\": -2.371119474378295, \"return\": -0.5271975348692828, \"returns_over_hodl\": -0.009319162729896302, \"returns_over_uniform_hodl\": -0.039839072240995366, \"sharpe\": -2.492843678815763, \"sterling\": -3.426890430803211, \"ulcer\": -0.11916946912692794}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.13579331132261385, \"optuna_trial_number\": 18, \"price_ratio\": 1.2615918090830154, \"shift_exponent\": 1.6692119333326882, \"step\": 18, \"test_objective\": [{\"annualised_returns\": -0.8443216671210937, \"annualised_returns_over_hodl\": -0.022979868832079253, \"annualised_returns_over_uniform_hodl\": -0.09601726267247679, \"calmar\": -1.411658792890672, \"daily_log_sharpe\": -2.8105941694881467, \"daily_returns\": 0.006522547972040203, \"fee_revenue_over_value\": 0.00016410692208874614, \"jax_sharpe\": -2.371119474378295, \"return\": -0.5271975348692828, \"returns_over_hodl\": -0.009319162729896302, \"returns_over_uniform_hodl\": -0.039839072240995366, \"sharpe\": -2.492843678815763, \"sterling\": -3.426890430803211, \"ulcer\": -0.11916946912692794}], \"train_objective\": [{\"annualised_returns\": -0.20013697595074642, \"annualised_returns_over_hodl\": -0.456858176508677, \"annualised_returns_over_uniform_hodl\": -0.456858176508677, \"calmar\": -0.34615394214651635, \"daily_log_sharpe\": -0.4008832897824629, \"daily_returns\": 0.008903748995746539, \"fee_revenue_over_value\": 0.000270564118000289, \"jax_sharpe\": -0.017613284125693636, \"return\": -0.12678998244063078, \"returns_over_hodl\": -0.3096645707172321, \"returns_over_uniform_hodl\": -0.3096645707172321, \"sharpe\": -0.09067095860957429, \"sterling\": -0.9891009071533153, \"ulcer\": -0.10418020313166501}], \"train_return\": -0.12678998244063078, \"train_returns_over_hodl\": -0.3096645707172321, \"train_sharpe\": -0.017613284125693636, \"validation_return\": 0.009927500739681072, \"validation_returns_over_hodl\": -0.02770686568836289, \"validation_sharpe\": 0.36131362505245407}, {\"centeredness_margin\": 0.04982162126121692, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8349377480760989, \"annualised_returns_over_hodl\": -0.26213214711353006, \"annualised_returns_over_uniform_hodl\": -0.041527336757399946, \"calmar\": -1.43630401660645, \"daily_log_sharpe\": -2.6204315009690378, \"daily_returns\": 0.008690524952162796, \"fee_revenue_over_value\": 9.220923580248768e-05, \"jax_sharpe\": -2.1456276594345716, \"return\": -0.5159199731076636, \"returns_over_hodl\": -0.11523082306948496, \"returns_over_uniform_hodl\": -0.016936750526361344, \"sharpe\": -2.284876375017495, \"sterling\": -3.359366874381369, \"ulcer\": -0.1137727080723779}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2661031207517777, \"optuna_trial_number\": 19, \"price_ratio\": 1.0445074986230658, \"shift_exponent\": 8.686066175453515e-05, \"step\": 19, \"test_objective\": [{\"annualised_returns\": -0.8349377480760989, \"annualised_returns_over_hodl\": -0.26213214711353006, \"annualised_returns_over_uniform_hodl\": -0.041527336757399946, \"calmar\": -1.43630401660645, \"daily_log_sharpe\": -2.6204315009690378, \"daily_returns\": 0.008690524952162796, \"fee_revenue_over_value\": 9.220923580248768e-05, \"jax_sharpe\": -2.1456276594345716, \"return\": -0.5159199731076636, \"returns_over_hodl\": -0.11523082306948496, \"returns_over_uniform_hodl\": -0.016936750526361344, \"sharpe\": -2.284876375017495, \"sterling\": -3.359366874381369, \"ulcer\": -0.1137727080723779}], \"train_objective\": [{\"annualised_returns\": 0.4935157280617548, \"annualised_returns_over_hodl\": 0.014162214732508671, \"annualised_returns_over_uniform_hodl\": 0.014162214732508227, \"calmar\": 0.7992113551505089, \"daily_log_sharpe\": 0.4816918922239037, \"daily_returns\": 0.009009579338884598, \"fee_revenue_over_value\": 2.4466804928058268e-05, \"jax_sharpe\": 0.9053457709739269, \"return\": 0.2757526976927125, \"returns_over_hodl\": 0.008574419109283582, \"returns_over_uniform_hodl\": 0.00857441910928336, \"sharpe\": 0.8612970165606822, \"sterling\": 2.056610577679534, \"ulcer\": -0.12575149728211638}], \"train_return\": 0.2757526976927125, \"train_returns_over_hodl\": 0.008574419109283582, \"train_sharpe\": 0.9053457709739269, \"validation_return\": 0.047506325392280946, \"validation_returns_over_hodl\": 0.002909526801248452, \"validation_sharpe\": 1.143781917666264}, {\"centeredness_margin\": 0.25302723256534865, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8258814087500562, \"annualised_returns_over_hodl\": -0.15923509177241246, \"annualised_returns_over_uniform_hodl\": 0.011060420721291209, \"calmar\": -1.440809625730514, \"daily_log_sharpe\": -2.64492019701093, \"daily_returns\": 0.008573992086204987, \"fee_revenue_over_value\": 0.00012796292804829733, \"jax_sharpe\": -2.167553787328244, \"return\": -0.5053937203963805, \"returns_over_hodl\": -0.06746830189241348, \"returns_over_uniform_hodl\": 0.004439822809179539, \"sharpe\": -2.323477453391594, \"sterling\": -3.4578508689896483, \"ulcer\": -0.10956511184455138}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.24587007952664924, \"optuna_trial_number\": 20, \"price_ratio\": 1.1342391766718742, \"shift_exponent\": 0.0001650739827753108, \"step\": 20, \"test_objective\": [{\"annualised_returns\": -0.8258814087500562, \"annualised_returns_over_hodl\": -0.15923509177241246, \"annualised_returns_over_uniform_hodl\": 0.011060420721291209, \"calmar\": -1.440809625730514, \"daily_log_sharpe\": -2.64492019701093, \"daily_returns\": 0.008573992086204987, \"fee_revenue_over_value\": 0.00012796292804829733, \"jax_sharpe\": -2.167553787328244, \"return\": -0.5053937203963805, \"returns_over_hodl\": -0.06746830189241348, \"returns_over_uniform_hodl\": 0.004439822809179539, \"sharpe\": -2.323477453391594, \"sterling\": -3.4578508689896483, \"ulcer\": -0.10956511184455138}], \"train_objective\": [{\"annualised_returns\": 0.47260937588760976, \"annualised_returns_over_hodl\": -3.410876414777775e-05, \"annualised_returns_over_uniform_hodl\": -3.410876414799979e-05, \"calmar\": 0.7682424998667975, \"daily_log_sharpe\": 0.47689089618966357, \"daily_returns\": 0.008930978716784548, \"fee_revenue_over_value\": 2.9999757302441055e-05, \"jax_sharpe\": 0.8874382568571829, \"return\": 0.2648806621044326, \"returns_over_hodl\": -2.0708298886007448e-05, \"returns_over_uniform_hodl\": -2.070829888611847e-05, \"sharpe\": 0.8555801946416176, \"sterling\": 1.973820996287053, \"ulcer\": -0.12533358037953504}], \"train_return\": 0.2648806621044326, \"train_returns_over_hodl\": -2.0708298886007448e-05, \"train_sharpe\": 0.8874382568571829, \"validation_return\": 0.05345050839514376, \"validation_returns_over_hodl\": 0.009044744835050578, \"validation_sharpe\": 1.2480158126680745}, {\"centeredness_margin\": 0.26645620813345267, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8362615665603798, \"annualised_returns_over_hodl\": -0.26804993324089965, \"annualised_returns_over_uniform_hodl\": -0.04921439914440395, \"calmar\": -1.435234329440402, \"daily_log_sharpe\": -2.5814788491358023, \"daily_returns\": 0.008690524952298781, \"fee_revenue_over_value\": 1.7750312087977807e-05, \"jax_sharpe\": -2.1175053120038037, \"return\": -0.5174873145889012, \"returns_over_hodl\": -0.11809550526685553, \"returns_over_uniform_hodl\": -0.02011968666085051, \"sharpe\": -2.238892651864599, \"sterling\": -3.303181195386332, \"ulcer\": -0.11523901061613925}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.27671681674209847, \"optuna_trial_number\": 21, \"price_ratio\": 1.0135446536102861, \"shift_exponent\": 1.0439828724530492e-05, \"step\": 21, \"test_objective\": [{\"annualised_returns\": -0.8362615665603798, \"annualised_returns_over_hodl\": -0.26804993324089965, \"annualised_returns_over_uniform_hodl\": -0.04921439914440395, \"calmar\": -1.435234329440402, \"daily_log_sharpe\": -2.5814788491358023, \"daily_returns\": 0.008690524952298781, \"fee_revenue_over_value\": 1.7750312087977807e-05, \"jax_sharpe\": -2.1175053120038037, \"return\": -0.5174873145889012, \"returns_over_hodl\": -0.11809550526685553, \"returns_over_uniform_hodl\": -0.02011968666085051, \"sharpe\": -2.238892651864599, \"sterling\": -3.303181195386332, \"ulcer\": -0.11523901061613925}], \"train_objective\": [{\"annualised_returns\": 0.5157354159454561, \"annualised_returns_over_hodl\": 0.029250350365364053, \"annualised_returns_over_uniform_hodl\": 0.029250350365364053, \"calmar\": 0.83285098947154, \"daily_log_sharpe\": 0.49676897742369075, \"daily_returns\": 0.009334658504192888, \"fee_revenue_over_value\": 1.3030734387329742e-05, \"jax_sharpe\": 0.9242459607196962, \"return\": 0.28724237430346933, \"returns_over_hodl\": 0.017657836243658576, \"returns_over_uniform_hodl\": 0.017657836243658576, \"sharpe\": 0.8768599921001071, \"sterling\": 2.1445284105786593, \"ulcer\": -0.12622726497005154}], \"train_return\": 0.28724237430346933, \"train_returns_over_hodl\": 0.017657836243658576, \"train_sharpe\": 0.9242459607196962, \"validation_return\": 0.06290366197822661, \"validation_returns_over_hodl\": -0.0020408622660387232, \"validation_sharpe\": 1.4365391815603652}, {\"centeredness_margin\": 0.26394281025053246, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8367693596755121, \"annualised_returns_over_hodl\": -0.27031989023344816, \"annualised_returns_over_uniform_hodl\": -0.05216301891521402, \"calmar\": -1.434821063473087, \"daily_log_sharpe\": -2.620518637002718, \"daily_returns\": 0.008690524952084498, \"fee_revenue_over_value\": 3.436900870613593e-05, \"jax_sharpe\": -2.144307202148997, \"return\": -0.518090526658382, \"returns_over_hodl\": -0.11919801603544744, \"returns_over_uniform_hodl\": -0.02134468167035819, \"sharpe\": -2.2837281909880445, \"sterling\": -3.348440959717062, \"ulcer\": -0.11526058024442395}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.26439624794570793, \"optuna_trial_number\": 22, \"price_ratio\": 1.0216868747837775, \"shift_exponent\": 8.219846550233926e-05, \"step\": 22, \"test_objective\": [{\"annualised_returns\": -0.8367693596755121, \"annualised_returns_over_hodl\": -0.27031989023344816, \"annualised_returns_over_uniform_hodl\": -0.05216301891521402, \"calmar\": -1.434821063473087, \"daily_log_sharpe\": -2.620518637002718, \"daily_returns\": 0.008690524952084498, \"fee_revenue_over_value\": 3.436900870613593e-05, \"jax_sharpe\": -2.144307202148997, \"return\": -0.518090526658382, \"returns_over_hodl\": -0.11919801603544744, \"returns_over_uniform_hodl\": -0.02134468167035819, \"sharpe\": -2.2837281909880445, \"sterling\": -3.348440959717062, \"ulcer\": -0.11526058024442395}], \"train_objective\": [{\"annualised_returns\": 0.48533309450509443, \"annualised_returns_over_hodl\": 0.008605850233463386, \"annualised_returns_over_uniform_hodl\": 0.008605850233463386, \"calmar\": 0.7840975971261747, \"daily_log_sharpe\": 0.4698898558181, \"daily_returns\": 0.009150407875525625, \"fee_revenue_over_value\": 1.3017964430462804e-05, \"jax_sharpe\": 0.8982298566930247, \"return\": 0.27150460801491394, \"returns_over_hodl\": 0.005215998165428948, \"returns_over_uniform_hodl\": 0.005215998165428948, \"sharpe\": 0.8497129325011753, \"sterling\": 2.018794065807877, \"ulcer\": -0.12615354325985875}], \"train_return\": 0.27150460801491394, \"train_returns_over_hodl\": 0.005215998165428948, \"train_sharpe\": 0.8982298566930247, \"validation_return\": 0.04335162795090497, \"validation_returns_over_hodl\": 0.0013850731310534048, \"validation_sharpe\": 1.0635537536970692}, {\"centeredness_margin\": 0.33756654538808095, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8393183852303532, \"annualised_returns_over_hodl\": -0.2817146458052202, \"annualised_returns_over_uniform_hodl\": -0.06696453339684283, \"calmar\": -1.4325858017612998, \"daily_log_sharpe\": -2.638021378920803, \"daily_returns\": 0.008746719131845663, \"fee_revenue_over_value\": 5.266623520448375e-05, \"jax_sharpe\": -2.1663932575488483, \"return\": -0.5211356011671522, \"returns_over_hodl\": -0.12476360007027087, \"returns_over_uniform_hodl\": -0.027528578289060857, \"sharpe\": -2.2995519515343608, \"sterling\": -3.3347852880334203, \"ulcer\": -0.11610009494845351}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.26997417282750924, \"optuna_trial_number\": 23, \"price_ratio\": 1.0570312512335722, \"shift_exponent\": 3.009051056689297e-05, \"step\": 23, \"test_objective\": [{\"annualised_returns\": -0.8393183852303532, \"annualised_returns_over_hodl\": -0.2817146458052202, \"annualised_returns_over_uniform_hodl\": -0.06696453339684283, \"calmar\": -1.4325858017612998, \"daily_log_sharpe\": -2.638021378920803, \"daily_returns\": 0.008746719131845663, \"fee_revenue_over_value\": 5.266623520448375e-05, \"jax_sharpe\": -2.1663932575488483, \"return\": -0.5211356011671522, \"returns_over_hodl\": -0.12476360007027087, \"returns_over_uniform_hodl\": -0.027528578289060857, \"sharpe\": -2.2995519515343608, \"sterling\": -3.3347852880334203, \"ulcer\": -0.11610009494845351}], \"train_objective\": [{\"annualised_returns\": 0.5067358357121368, \"annualised_returns_over_hodl\": 0.023139243498796036, \"annualised_returns_over_uniform_hodl\": 0.023139243498795592, \"calmar\": 0.8215820825501414, \"daily_log_sharpe\": 0.5035618145791341, \"daily_returns\": 0.008971980121942676, \"fee_revenue_over_value\": 1.7715326330700508e-05, \"jax_sharpe\": 0.9166846505828016, \"return\": 0.28259676850340076, \"returns_over_hodl\": 0.01398515016610724, \"returns_over_uniform_hodl\": 0.013985150166107019, \"sharpe\": 0.8831668441209247, \"sterling\": 2.1136201828938135, \"ulcer\": -0.12557262001274}], \"train_return\": 0.28259676850340076, \"train_returns_over_hodl\": 0.01398515016610724, \"train_sharpe\": 0.9166846505828014, \"validation_return\": 0.05554659496591241, \"validation_returns_over_hodl\": 0.00468000998825735, \"validation_sharpe\": 1.2897098171621402}, {\"centeredness_margin\": 0.041099021184069495, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8348590893940556, \"annualised_returns_over_hodl\": -0.261780523939275, \"annualised_returns_over_uniform_hodl\": -0.041070586679253895, \"calmar\": -1.436367226456557, \"daily_log_sharpe\": -2.601074845420591, \"daily_returns\": 0.008690524952105549, \"fee_revenue_over_value\": 5.306512462841181e-05, \"jax_sharpe\": -2.133550838912773, \"return\": -0.5158270814550774, \"returns_over_hodl\": -0.1150610419039001, \"returns_over_uniform_hodl\": -0.016748107399672696, \"sharpe\": -2.263057556863442, \"sterling\": -3.332379297694621, \"ulcer\": -0.11446641275011771}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.26718816296236353, \"optuna_trial_number\": 24, \"price_ratio\": 1.0218340687138447, \"shift_exponent\": 6.695955834663245e-05, \"step\": 24, \"test_objective\": [{\"annualised_returns\": -0.8348590893940556, \"annualised_returns_over_hodl\": -0.261780523939275, \"annualised_returns_over_uniform_hodl\": -0.041070586679253895, \"calmar\": -1.436367226456557, \"daily_log_sharpe\": -2.601074845420591, \"daily_returns\": 0.008690524952105549, \"fee_revenue_over_value\": 5.306512462841181e-05, \"jax_sharpe\": -2.133550838912773, \"return\": -0.5158270814550774, \"returns_over_hodl\": -0.1150610419039001, \"returns_over_uniform_hodl\": -0.016748107399672696, \"sharpe\": -2.263057556863442, \"sterling\": -3.332379297694621, \"ulcer\": -0.11446641275011771}], \"train_objective\": [{\"annualised_returns\": 0.4915692210098297, \"annualised_returns_over_hodl\": 0.012840451683293619, \"annualised_returns_over_uniform_hodl\": 0.012840451683293397, \"calmar\": 0.7942196407244396, \"daily_log_sharpe\": 0.47560150967863557, \"daily_returns\": 0.009190325394765192, \"fee_revenue_over_value\": 1.5406109809694515e-05, \"jax_sharpe\": 0.9036181226407076, \"return\": 0.2747429812041944, \"returns_over_hodl\": 0.007776165480101183, \"returns_over_uniform_hodl\": 0.0077761654801009605, \"sharpe\": 0.8554405403877194, \"sterling\": 2.044827819837126, \"ulcer\": -0.1261429069928364}], \"train_return\": 0.2747429812041944, \"train_returns_over_hodl\": 0.007776165480101183, \"train_sharpe\": 0.9036181226407076, \"validation_return\": 0.04780932594607279, \"validation_returns_over_hodl\": 0.0025765036003297936, \"validation_sharpe\": 1.1523624884437755}, {\"centeredness_margin\": 0.27674216922827, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8350006258933965, \"annualised_returns_over_hodl\": -0.12045924971793887, \"annualised_returns_over_uniform_hodl\": -0.041892451544709775, \"calmar\": -1.4299787958238046, \"daily_log_sharpe\": -2.699816135460676, \"daily_returns\": 0.007752934896805338, \"fee_revenue_over_value\": 0.00010053430025066562, \"jax_sharpe\": -2.2287100227836056, \"return\": -0.5159942474279848, \"returns_over_hodl\": -0.05038018321117055, \"returns_over_uniform_hodl\": -0.01708758582348391, \"sharpe\": -2.3755870242809594, \"sterling\": -3.4309673273908037, \"ulcer\": -0.11431348993935464}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2464128424783917, \"optuna_trial_number\": 25, \"price_ratio\": 1.1951309882002557, \"shift_exponent\": 1.656932351464279e-05, \"step\": 25, \"test_objective\": [{\"annualised_returns\": -0.8350006258933965, \"annualised_returns_over_hodl\": -0.12045924971793887, \"annualised_returns_over_uniform_hodl\": -0.041892451544709775, \"calmar\": -1.4299787958238046, \"daily_log_sharpe\": -2.699816135460676, \"daily_returns\": 0.007752934896805338, \"fee_revenue_over_value\": 0.00010053430025066562, \"jax_sharpe\": -2.2287100227836056, \"return\": -0.5159942474279848, \"returns_over_hodl\": -0.05038018321117055, \"returns_over_uniform_hodl\": -0.01708758582348391, \"sharpe\": -2.3755870242809594, \"sterling\": -3.4309673273908037, \"ulcer\": -0.11431348993935464}], \"train_objective\": [{\"annualised_returns\": 0.5010971137688645, \"annualised_returns_over_hodl\": 0.01931030576027526, \"annualised_returns_over_uniform_hodl\": 0.01931030576027526, \"calmar\": 0.8183875226088874, \"daily_log_sharpe\": 0.5164636662273501, \"daily_returns\": 0.008907751709353658, \"fee_revenue_over_value\": 2.8049545334460462e-05, \"jax_sharpe\": 0.9121857663529829, \"return\": 0.2796804903758925, \"returns_over_hodl\": 0.011679622203099704, \"returns_over_uniform_hodl\": 0.011679622203099704, \"sharpe\": 0.894358495892406, \"sterling\": 2.0980907811333025, \"ulcer\": -0.12479985866670507}], \"train_return\": 0.2796804903758925, \"train_returns_over_hodl\": 0.011679622203099704, \"train_sharpe\": 0.9121857663529829, \"validation_return\": 0.06943133757473197, \"validation_returns_over_hodl\": 0.019072876174397013, \"validation_sharpe\": 1.4860864751821796}, {\"centeredness_margin\": 0.5766589290569646, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8351752984866373, \"annualised_returns_over_hodl\": -0.14394483902323485, \"annualised_returns_over_uniform_hodl\": -0.042906728907872616, \"calmar\": -1.4308786542438452, \"daily_log_sharpe\": -2.67757493196577, \"daily_returns\": 0.008010582865414817, \"fee_revenue_over_value\": 4.1638472781808464e-05, \"jax_sharpe\": -2.2030468379647234, \"return\": -0.5162006678996902, \"returns_over_hodl\": -0.060674955691102817, \"returns_over_uniform_hodl\": -0.017506781758865442, \"sharpe\": -2.35007317318637, \"sterling\": -3.4124820163064955, \"ulcer\": -0.11431607748042086}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2525279206784929, \"optuna_trial_number\": 26, \"price_ratio\": 1.1565539070217534, \"shift_exponent\": 3.532791434709187e-05, \"step\": 26, \"test_objective\": [{\"annualised_returns\": -0.8351752984866373, \"annualised_returns_over_hodl\": -0.14394483902323485, \"annualised_returns_over_uniform_hodl\": -0.042906728907872616, \"calmar\": -1.4308786542438452, \"daily_log_sharpe\": -2.67757493196577, \"daily_returns\": 0.008010582865414817, \"fee_revenue_over_value\": 4.1638472781808464e-05, \"jax_sharpe\": -2.2030468379647234, \"return\": -0.5162006678996902, \"returns_over_hodl\": -0.060674955691102817, \"returns_over_uniform_hodl\": -0.017506781758865442, \"sharpe\": -2.35007317318637, \"sterling\": -3.4124820163064955, \"ulcer\": -0.11431607748042086}], \"train_objective\": [{\"annualised_returns\": 0.4981972320805068, \"annualised_returns_over_hodl\": 0.017341159818073493, \"annualised_returns_over_uniform_hodl\": 0.017341159818073493, \"calmar\": 0.8115640547791028, \"daily_log_sharpe\": 0.5103515854192056, \"daily_returns\": 0.008923437904402077, \"fee_revenue_over_value\": 1.2665268155448411e-05, \"jax_sharpe\": 0.909603771440941, \"return\": 0.27817903131176025, \"returns_over_hodl\": 0.010492610640308264, \"returns_over_uniform_hodl\": 0.010492610640308264, \"sharpe\": 0.888879557799301, \"sterling\": 2.0825400888039995, \"ulcer\": -0.12509295255164846}], \"train_return\": 0.27817903131176025, \"train_returns_over_hodl\": 0.010492610640308264, \"train_sharpe\": 0.909603771440941, \"validation_return\": 0.06412743329555948, \"validation_returns_over_hodl\": 0.014365214930549985, \"validation_sharpe\": 1.4132401062725959}, {\"centeredness_margin\": 0.5264588216508843, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8366834534574517, \"annualised_returns_over_hodl\": -0.15966807307200837, \"annualised_returns_over_uniform_hodl\": -0.0516641843200607, \"calmar\": -1.4301284691533327, \"daily_log_sharpe\": -2.667389736830322, \"daily_returns\": 0.008113818075082272, \"fee_revenue_over_value\": 4.39721763454675e-05, \"jax_sharpe\": -2.1834680226768057, \"return\": -0.5179883991925279, \"returns_over_hodl\": -0.06766174238436684, \"returns_over_uniform_hodl\": -0.02113728257751102, \"sharpe\": -2.336249871769828, \"sterling\": -3.380308827510557, \"ulcer\": -0.11516151723994186}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2589282033340461, \"optuna_trial_number\": 27, \"price_ratio\": 1.13426578679329, \"shift_exponent\": 1.6946863122309983e-05, \"step\": 27, \"test_objective\": [{\"annualised_returns\": -0.8366834534574517, \"annualised_returns_over_hodl\": -0.15966807307200837, \"annualised_returns_over_uniform_hodl\": -0.0516641843200607, \"calmar\": -1.4301284691533327, \"daily_log_sharpe\": -2.667389736830322, \"daily_returns\": 0.008113818075082272, \"fee_revenue_over_value\": 4.39721763454675e-05, \"jax_sharpe\": -2.1834680226768057, \"return\": -0.5179883991925279, \"returns_over_hodl\": -0.06766174238436684, \"returns_over_uniform_hodl\": -0.02113728257751102, \"sharpe\": -2.336249871769828, \"sterling\": -3.380308827510557, \"ulcer\": -0.11516151723994186}], \"train_objective\": [{\"annualised_returns\": 0.5010690622569229, \"annualised_returns_over_hodl\": 0.019291257562159103, \"annualised_returns_over_uniform_hodl\": 0.019291257562159103, \"calmar\": 0.8147305824902266, \"daily_log_sharpe\": 0.5122729860432876, \"daily_returns\": 0.008923674377136525, \"fee_revenue_over_value\": 1.4039637887266487e-05, \"jax_sharpe\": 0.9120090562738159, \"return\": 0.2796659717284955, \"returns_over_hodl\": 0.011668144166336436, \"returns_over_uniform_hodl\": 0.011668144166336436, \"sharpe\": 0.8911261047767407, \"sterling\": 2.09307275963574, \"ulcer\": -0.12529255139122086}], \"train_return\": 0.2796659717284955, \"train_returns_over_hodl\": 0.011668144166336436, \"train_sharpe\": 0.9120090562738161, \"validation_return\": 0.06584793553635704, \"validation_returns_over_hodl\": 0.015260133793472352, \"validation_sharpe\": 1.4507001735317573}, {\"centeredness_margin\": 0.5783715474177578, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8352357998349492, \"annualised_returns_over_hodl\": -0.14783798379922197, \"annualised_returns_over_uniform_hodl\": -0.043258044170873866, \"calmar\": -1.4310179704588453, \"daily_log_sharpe\": -2.6706866976458876, \"daily_returns\": 0.00804662412865197, \"fee_revenue_over_value\": 4.0873440922118606e-05, \"jax_sharpe\": -2.193812694252443, \"return\": -0.5162721962240374, \"returns_over_hodl\": -0.06239772835962465, \"returns_over_uniform_hodl\": -0.017652040524060664, \"sharpe\": -2.341846515768623, \"sterling\": -3.3996412905528066, \"ulcer\": -0.11439317456361317}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2552016851132779, \"optuna_trial_number\": 28, \"price_ratio\": 1.1476713750825196, \"shift_exponent\": 2.9908358166649875e-05, \"step\": 28, \"test_objective\": [{\"annualised_returns\": -0.8352357998349492, \"annualised_returns_over_hodl\": -0.14783798379922197, \"annualised_returns_over_uniform_hodl\": -0.043258044170873866, \"calmar\": -1.4310179704588453, \"daily_log_sharpe\": -2.6706866976458876, \"daily_returns\": 0.00804662412865197, \"fee_revenue_over_value\": 4.0873440922118606e-05, \"jax_sharpe\": -2.193812694252443, \"return\": -0.5162721962240374, \"returns_over_hodl\": -0.06239772835962465, \"returns_over_uniform_hodl\": -0.017652040524060664, \"sharpe\": -2.341846515768623, \"sterling\": -3.3996412905528066, \"ulcer\": -0.11439317456361317}], \"train_objective\": [{\"annualised_returns\": 0.49936760813158165, \"annualised_returns_over_hodl\": 0.018135896120963668, \"annualised_returns_over_uniform_hodl\": 0.018135896120963224, \"calmar\": 0.812920292575683, \"daily_log_sharpe\": 0.5111380300900709, \"daily_returns\": 0.008931383912843862, \"fee_revenue_over_value\": 1.2962697531382203e-05, \"jax_sharpe\": 0.9105821133537252, \"return\": 0.2787851490990052, \"returns_over_hodl\": 0.010971790418871974, \"returns_over_uniform_hodl\": 0.010971790418871752, \"sharpe\": 0.889791333487875, \"sterling\": 2.086923801933755, \"ulcer\": -0.12516146219890004}], \"train_return\": 0.2787851490990052, \"train_returns_over_hodl\": 0.010971790418871974, \"train_sharpe\": 0.9105821133537251, \"validation_return\": 0.06483211025336799, \"validation_returns_over_hodl\": 0.014620033747324568, \"validation_sharpe\": 1.4289774882403106}, {\"centeredness_margin\": 0.36574733028065565, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8355589475187567, \"annualised_returns_over_hodl\": -0.014830237235032406, \"annualised_returns_over_uniform_hodl\": -0.04513447695613848, \"calmar\": -1.4237954856732375, \"daily_log_sharpe\": -2.9134522106330825, \"daily_returns\": 0.006731888243908098, \"fee_revenue_over_value\": 8.587458044916075e-05, \"jax_sharpe\": -2.5178951397165443, \"return\": -0.5166545067175949, \"returns_over_hodl\": -0.005999360061407999, \"returns_over_uniform_hodl\": -0.018428431565263748, \"sharpe\": -2.618765071304071, \"sterling\": -3.55038169671286, \"ulcer\": -0.11671153948139021}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.017813645279616763, \"optuna_trial_number\": 29, \"price_ratio\": 1.7227595513532332, \"shift_exponent\": 27.497006657394774, \"step\": 29, \"test_objective\": [{\"annualised_returns\": -0.8355589475187567, \"annualised_returns_over_hodl\": -0.014830237235032406, \"annualised_returns_over_uniform_hodl\": -0.04513447695613848, \"calmar\": -1.4237954856732375, \"daily_log_sharpe\": -2.9134522106330825, \"daily_returns\": 0.006731888243908098, \"fee_revenue_over_value\": 8.587458044916075e-05, \"jax_sharpe\": -2.5178951397165443, \"return\": -0.5166545067175949, \"returns_over_hodl\": -0.005999360061407999, \"returns_over_uniform_hodl\": -0.018428431565263748, \"sharpe\": -2.618765071304071, \"sterling\": -3.55038169671286, \"ulcer\": -0.11671153948139021}], \"train_objective\": [{\"annualised_returns\": 0.09825861633562916, \"annualised_returns_over_hodl\": -0.2542345756629709, \"annualised_returns_over_uniform_hodl\": -0.2542345756629709, \"calmar\": 0.18556672006114275, \"daily_log_sharpe\": 0.13617596248998173, \"daily_returns\": 0.008890769852780995, \"fee_revenue_over_value\": 0.00014943808440530311, \"jax_sharpe\": 0.45714692564945697, \"return\": 0.05855309133769926, \"returns_over_hodl\": -0.16313751785660457, \"returns_over_uniform_hodl\": -0.16313751785660457, \"sharpe\": 0.42871162284596265, \"sterling\": 0.5091671571429425, \"ulcer\": -0.09747715036734546}], \"train_return\": 0.05855309133769926, \"train_returns_over_hodl\": -0.16313751785660457, \"train_sharpe\": 0.457146925649457, \"validation_return\": 0.03244130606319118, \"validation_returns_over_hodl\": -0.010955883499512709, \"validation_sharpe\": 0.7250152730039906}, {\"centeredness_margin\": 0.3546898076134619, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8077105574656179, \"annualised_returns_over_hodl\": -0.14042007509467247, \"annualised_returns_over_uniform_hodl\": 0.11657372870651339, \"calmar\": -1.4559723785061727, \"daily_log_sharpe\": -2.7735563220001627, \"daily_returns\": 0.008690524951294248, \"fee_revenue_over_value\": 0.00013124243626909884, \"jax_sharpe\": -2.3057886566287285, \"return\": -0.4852198731422589, \"returns_over_hodl\": -0.05911922851443796, \"returns_over_uniform_hodl\": 0.045408602213979465, \"sharpe\": -2.489374974932905, \"sterling\": -3.767307794161374, \"ulcer\": -0.09907246854663462}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.0498210856295795, \"optuna_trial_number\": 30, \"price_ratio\": 1.0989523650436248, \"shift_exponent\": 0.0018664512347820682, \"step\": 30, \"test_objective\": [{\"annualised_returns\": -0.8077105574656179, \"annualised_returns_over_hodl\": -0.14042007509467247, \"annualised_returns_over_uniform_hodl\": 0.11657372870651339, \"calmar\": -1.4559723785061727, \"daily_log_sharpe\": -2.7735563220001627, \"daily_returns\": 0.008690524951294248, \"fee_revenue_over_value\": 0.00013124243626909884, \"jax_sharpe\": -2.3057886566287285, \"return\": -0.4852198731422589, \"returns_over_hodl\": -0.05911922851443796, \"returns_over_uniform_hodl\": 0.045408602213979465, \"sharpe\": -2.489374974932905, \"sterling\": -3.767307794161374, \"ulcer\": -0.09907246854663462}], \"train_objective\": [{\"annualised_returns\": 0.038045037108545454, \"annualised_returns_over_hodl\": -0.29512221796799065, \"annualised_returns_over_uniform_hodl\": -0.29512221796799076, \"calmar\": 0.06168095193185533, \"daily_log_sharpe\": 0.007502303808807395, \"daily_returns\": 0.008948423336017211, \"fee_revenue_over_value\": 1.519631337547053e-05, \"jax_sharpe\": 0.42680241255461393, \"return\": 0.022928313685890878, \"returns_over_hodl\": -0.19130147117690888, \"returns_over_uniform_hodl\": -0.191301471176909, \"sharpe\": 0.37933198845525207, \"sterling\": 0.1614901207366151, \"ulcer\": -0.12495324149207149}], \"train_return\": 0.022928313685890878, \"train_returns_over_hodl\": -0.19130147117690888, \"train_sharpe\": 0.4268024125546139, \"validation_return\": 0.030486270473979227, \"validation_returns_over_hodl\": -4.0983438864827804e-11, \"validation_sharpe\": 0.8017265750341069}, {\"centeredness_margin\": 0.2424189025288776, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8229176798681792, \"annualised_returns_over_hodl\": -0.025185474465655444, \"annualised_returns_over_uniform_hodl\": 0.028270007295036814, \"calmar\": -1.4378898457671208, \"daily_log_sharpe\": -2.9186479179399267, \"daily_returns\": 0.0072593456733821005, \"fee_revenue_over_value\": 5.373434952411137e-05, \"jax_sharpe\": -2.522522969659071, \"return\": -0.5020202098665205, \"returns_over_hodl\": -0.010220471436380718, \"returns_over_uniform_hodl\": 0.011290702910364603, \"sharpe\": -2.6363767328031047, \"sterling\": -3.639202592863523, \"ulcer\": -0.11253873534245971}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.04642914973219929, \"optuna_trial_number\": 31, \"price_ratio\": 2.7595208347475806, \"shift_exponent\": 49.236204519494265, \"step\": 31, \"test_objective\": [{\"annualised_returns\": -0.8229176798681792, \"annualised_returns_over_hodl\": -0.025185474465655444, \"annualised_returns_over_uniform_hodl\": 0.028270007295036814, \"calmar\": -1.4378898457671208, \"daily_log_sharpe\": -2.9186479179399267, \"daily_returns\": 0.0072593456733821005, \"fee_revenue_over_value\": 5.373434952411137e-05, \"jax_sharpe\": -2.522522969659071, \"return\": -0.5020202098665205, \"returns_over_hodl\": -0.010220471436380718, \"returns_over_uniform_hodl\": 0.011290702910364603, \"sharpe\": -2.6363767328031047, \"sterling\": -3.639202592863523, \"ulcer\": -0.11253873534245971}], \"train_objective\": [{\"annualised_returns\": 0.18588399971970682, \"annualised_returns_over_hodl\": -0.19473312468400483, \"annualised_returns_over_uniform_hodl\": -0.19473312468400483, \"calmar\": 0.3473047003232584, \"daily_log_sharpe\": 0.2515339697993838, \"daily_returns\": 0.00888660232400238, \"fee_revenue_over_value\": 9.250240754939277e-05, \"jax_sharpe\": 0.5853825127216776, \"return\": 0.10905378612394334, \"returns_over_hodl\": -0.12321308030631095, \"returns_over_uniform_hodl\": -0.12321308030631095, \"sharpe\": 0.5547856797162176, \"sterling\": 0.9495264328989064, \"ulcer\": -0.09879932388809971}], \"train_return\": 0.10905378612394334, \"train_returns_over_hodl\": -0.12321308030631095, \"train_sharpe\": 0.5853825127216776, \"validation_return\": 0.033583603278537044, \"validation_returns_over_hodl\": -0.007183000691897701, \"validation_sharpe\": 0.7702219236281578}, {\"centeredness_margin\": 0.14419405874049318, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8218626967777318, \"annualised_returns_over_hodl\": -0.041216444755332815, \"annualised_returns_over_uniform_hodl\": 0.03439601394156644, \"calmar\": -1.4371348215477078, \"daily_log_sharpe\": -2.866300909229338, \"daily_returns\": 0.007459450831085553, \"fee_revenue_over_value\": 8.814638707957089e-05, \"jax_sharpe\": -2.454153571531999, \"return\": -0.5008275035282774, \"returns_over_hodl\": -0.01680834265820419, \"returns_over_uniform_hodl\": 0.013712835003002688, \"sharpe\": -2.57790393980204, \"sterling\": -3.619147479999706, \"ulcer\": -0.11162832483044058}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -0.0524043468371231, \"optuna_trial_number\": 32, \"price_ratio\": 1.7320909332763945, \"shift_exponent\": 0.027671836158132887, \"step\": 32, \"test_objective\": [{\"annualised_returns\": -0.8218626967777318, \"annualised_returns_over_hodl\": -0.041216444755332815, \"annualised_returns_over_uniform_hodl\": 0.03439601394156644, \"calmar\": -1.4371348215477078, \"daily_log_sharpe\": -2.866300909229338, \"daily_returns\": 0.007459450831085553, \"fee_revenue_over_value\": 8.814638707957089e-05, \"jax_sharpe\": -2.454153571531999, \"return\": -0.5008275035282774, \"returns_over_hodl\": -0.01680834265820419, \"returns_over_uniform_hodl\": 0.013712835003002688, \"sharpe\": -2.57790393980204, \"sterling\": -3.619147479999706, \"ulcer\": -0.11162832483044058}], \"train_objective\": [{\"annualised_returns\": -0.025050276677797467, \"annualised_returns_over_hodl\": -0.3379666835243351, \"annualised_returns_over_uniform_hodl\": -0.337966683524335, \"calmar\": -0.044187600077762644, \"daily_log_sharpe\": -0.08482508620103081, \"daily_returns\": 0.00889352342180917, \"fee_revenue_over_value\": 0.00012648358402192345, \"jax_sharpe\": 0.28978520013231335, \"return\": -0.015284281545304812, \"returns_over_hodl\": -0.2215112807329963, \"returns_over_uniform_hodl\": -0.2215112807329962, \"sharpe\": 0.23206651360065217, \"sterling\": -0.12512359678345883, \"ulcer\": -0.10239800709048946}], \"train_return\": -0.015284281545304812, \"train_returns_over_hodl\": -0.2215112807329963, \"train_sharpe\": 0.28978520013231335, \"validation_return\": 0.025240008018478255, \"validation_returns_over_hodl\": -0.009332515763111893, \"validation_sharpe\": 0.646864832284423}, {\"centeredness_margin\": 0.2334887956992607, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8147764284709256, \"annualised_returns_over_hodl\": -0.045539209045002416, \"annualised_returns_over_uniform_hodl\": 0.07554409218060298, \"calmar\": -1.4434996962912425, \"daily_log_sharpe\": -2.866645570847533, \"daily_returns\": 0.007779029907874285, \"fee_revenue_over_value\": 9.493500468523606e-05, \"jax_sharpe\": -2.4567230703018987, \"return\": -0.49292338410661607, \"returns_over_hodl\": -0.018596012572387965, \"returns_over_uniform_hodl\": 0.02976441509559402, \"sharpe\": -2.5855320180841552, \"sterling\": -3.6712632210462486, \"ulcer\": -0.10878917528277374}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.12629552323766205, \"optuna_trial_number\": 33, \"price_ratio\": 1.6599637743830482, \"shift_exponent\": 0.001126540626839893, \"step\": 33, \"test_objective\": [{\"annualised_returns\": -0.8147764284709256, \"annualised_returns_over_hodl\": -0.045539209045002416, \"annualised_returns_over_uniform_hodl\": 0.07554409218060298, \"calmar\": -1.4434996962912425, \"daily_log_sharpe\": -2.866645570847533, \"daily_returns\": 0.007779029907874285, \"fee_revenue_over_value\": 9.493500468523606e-05, \"jax_sharpe\": -2.4567230703018987, \"return\": -0.49292338410661607, \"returns_over_hodl\": -0.018596012572387965, \"returns_over_uniform_hodl\": 0.02976441509559402, \"sharpe\": -2.5855320180841552, \"sterling\": -3.6712632210462486, \"ulcer\": -0.10878917528277374}], \"train_objective\": [{\"annualised_returns\": 0.36108031269531016, \"annualised_returns_over_hodl\": -0.07576719922241437, \"annualised_returns_over_uniform_hodl\": -0.07576719922241437, \"calmar\": 0.6111672676735843, \"daily_log_sharpe\": 0.39206859214101264, \"daily_returns\": 0.00889449115044861, \"fee_revenue_over_value\": 4.7702107926604345e-05, \"jax_sharpe\": 0.7868460494958401, \"return\": 0.20582334895749366, \"returns_over_hodl\": -0.04670976912474534, \"returns_over_uniform_hodl\": -0.04670976912474534, \"sharpe\": 0.7557080866112884, \"sterling\": 1.5736663553199273, \"ulcer\": -0.11861053398721383}], \"train_return\": 0.20582334895749366, \"train_returns_over_hodl\": -0.04670976912474534, \"train_sharpe\": 0.7868460494958401, \"validation_return\": 0.04691820605849606, \"validation_returns_over_hodl\": 0.0062444118698492534, \"validation_sharpe\": 1.076222548343492}, {\"centeredness_margin\": 0.2801610104496445, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8331205390652725, \"annualised_returns_over_hodl\": -0.0693202706045507, \"annualised_returns_over_uniform_hodl\": -0.030975286606780505, \"calmar\": -1.4258124588230068, \"daily_log_sharpe\": -2.786379394691744, \"daily_returns\": 0.007267504995874629, \"fee_revenue_over_value\": 9.772175017283864e-05, \"jax_sharpe\": -2.3348864572008337, \"return\": -0.5137806575294952, \"returns_over_hodl\": -0.02851816926090467, \"returns_over_uniform_hodl\": -0.012592257039561239, \"sharpe\": -2.4747109180264335, \"sterling\": -3.506753830835507, \"ulcer\": -0.11501821830654217}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.21387140180994596, \"optuna_trial_number\": 34, \"price_ratio\": 1.4469589444743778, \"shift_exponent\": 1.518276193497397e-05, \"step\": 34, \"test_objective\": [{\"annualised_returns\": -0.8331205390652725, \"annualised_returns_over_hodl\": -0.0693202706045507, \"annualised_returns_over_uniform_hodl\": -0.030975286606780505, \"calmar\": -1.4258124588230068, \"daily_log_sharpe\": -2.786379394691744, \"daily_returns\": 0.007267504995874629, \"fee_revenue_over_value\": 9.772175017283864e-05, \"jax_sharpe\": -2.3348864572008337, \"return\": -0.5137806575294952, \"returns_over_hodl\": -0.02851816926090467, \"returns_over_uniform_hodl\": -0.012592257039561239, \"sharpe\": -2.4747109180264335, \"sterling\": -3.506753830835507, \"ulcer\": -0.11501821830654217}], \"train_objective\": [{\"annualised_returns\": 0.5277319562020988, \"annualised_returns_over_hodl\": 0.03739652359086709, \"annualised_returns_over_uniform_hodl\": 0.037396523590867536, \"calmar\": 0.8785114537647406, \"daily_log_sharpe\": 0.5510349613048624, \"daily_returns\": 0.008894998307345652, \"fee_revenue_over_value\": 2.7526801243209327e-05, \"jax_sharpe\": 0.9370815784634957, \"return\": 0.2934182024676346, \"returns_over_hodl\": 0.02254027334487496, \"returns_over_uniform_hodl\": 0.022540273344875184, \"sharpe\": 0.9229760725924991, \"sterling\": 2.2462788891197607, \"ulcer\": -0.1217193636942749}], \"train_return\": 0.2934182024676346, \"train_returns_over_hodl\": 0.02254027334487496, \"train_sharpe\": 0.9370815784634958, \"validation_return\": 0.06569250662968762, \"validation_returns_over_hodl\": 0.015991431295024228, \"validation_sharpe\": 1.3203732475572376}, {\"centeredness_margin\": 0.36424502137433135, \"continuous_test_metrics\": [{\"annualised_returns\": -0.837846399419047, \"annualised_returns_over_hodl\": -0.2751345162043617, \"annualised_returns_over_uniform_hodl\": -0.058417102688891664, \"calmar\": -1.4339390643416807, \"daily_log_sharpe\": -2.6030260416255215, \"daily_returns\": 0.008690524952252464, \"fee_revenue_over_value\": 3.358622969255911e-05, \"jax_sharpe\": -2.1414083472479137, \"return\": -0.5193736720058442, \"returns_over_hodl\": -0.12154326349990774, \"returns_over_uniform_hodl\": -0.023950476094311668, \"sharpe\": -2.2611391415056867, \"sterling\": -3.309910434268811, \"ulcer\": -0.11584146441965554}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.27326218850888606, \"optuna_trial_number\": 35, \"price_ratio\": 1.0337817169078793, \"shift_exponent\": 1.1654255830763121e-05, \"step\": 35, \"test_objective\": [{\"annualised_returns\": -0.837846399419047, \"annualised_returns_over_hodl\": -0.2751345162043617, \"annualised_returns_over_uniform_hodl\": -0.058417102688891664, \"calmar\": -1.4339390643416807, \"daily_log_sharpe\": -2.6030260416255215, \"daily_returns\": 0.008690524952252464, \"fee_revenue_over_value\": 3.358622969255911e-05, \"jax_sharpe\": -2.1414083472479137, \"return\": -0.5193736720058442, \"returns_over_hodl\": -0.12154326349990774, \"returns_over_uniform_hodl\": -0.023950476094311668, \"sharpe\": -2.2611391415056867, \"sterling\": -3.309910434268811, \"ulcer\": -0.11584146441965554}], \"train_objective\": [{\"annualised_returns\": 0.5089286418686425, \"annualised_returns_over_hodl\": 0.024628254365153923, \"annualised_returns_over_uniform_hodl\": 0.024628254365153923, \"calmar\": 0.82354620790388, \"daily_log_sharpe\": 0.5004906116625754, \"daily_returns\": 0.00905248694613863, \"fee_revenue_over_value\": 1.5328531305747444e-05, \"jax_sharpe\": 0.9184757798600741, \"return\": 0.2837297027478205, \"returns_over_hodl\": 0.01488081630855076, \"returns_over_uniform_hodl\": 0.01488081630855076, \"sharpe\": 0.8803272539064485, \"sterling\": 2.1195929688117783, \"ulcer\": -0.12587831648673486}], \"train_return\": 0.2837297027478205, \"train_returns_over_hodl\": 0.01488081630855076, \"train_sharpe\": 0.918475779860074, \"validation_return\": 0.05748391468604841, \"validation_returns_over_hodl\": 0.0016326388526684, \"validation_sharpe\": 1.3278335139345334}, {\"centeredness_margin\": 0.3508386103838136, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8409515442358118, \"annualised_returns_over_hodl\": -0.23707571380067116, \"annualised_returns_over_uniform_hodl\": -0.07644785404233068, \"calmar\": -1.4294699211967248, \"daily_log_sharpe\": -2.661022116559123, \"daily_returns\": 0.008879568731008399, \"fee_revenue_over_value\": 5.5651070689894164e-05, \"jax_sharpe\": -2.18385389827413, \"return\": -0.5231017691294477, \"returns_over_hodl\": -0.10325114475761599, \"returns_over_uniform_hodl\": -0.03152144591145334, \"sharpe\": -2.3237334799801075, \"sterling\": -3.3387170855853117, \"ulcer\": -0.11711234324062875}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2678373522429586, \"optuna_trial_number\": 36, \"price_ratio\": 1.080568941655359, \"shift_exponent\": 1.1388162428130138e-05, \"step\": 36, \"test_objective\": [{\"annualised_returns\": -0.8409515442358118, \"annualised_returns_over_hodl\": -0.23707571380067116, \"annualised_returns_over_uniform_hodl\": -0.07644785404233068, \"calmar\": -1.4294699211967248, \"daily_log_sharpe\": -2.661022116559123, \"daily_returns\": 0.008879568731008399, \"fee_revenue_over_value\": 5.5651070689894164e-05, \"jax_sharpe\": -2.18385389827413, \"return\": -0.5231017691294477, \"returns_over_hodl\": -0.10325114475761599, \"returns_over_uniform_hodl\": -0.03152144591145334, \"sharpe\": -2.3237334799801075, \"sterling\": -3.3387170855853117, \"ulcer\": -0.11711234324062875}], \"train_objective\": [{\"annualised_returns\": 0.5085093732020689, \"annualised_returns_over_hodl\": 0.02434355268343036, \"annualised_returns_over_uniform_hodl\": 0.02434355268343036, \"calmar\": 0.8247049409526682, \"daily_log_sharpe\": 0.5124027980118229, \"daily_returns\": 0.008960779035999903, \"fee_revenue_over_value\": 1.8400730884437214e-05, \"jax_sharpe\": 0.9182570640386873, \"return\": 0.2835131336273229, \"returns_over_hodl\": 0.014709602816078693, \"returns_over_uniform_hodl\": 0.014709602816078693, \"sharpe\": 0.8919528378061428, \"sterling\": 2.1212528703113542, \"ulcer\": -0.12557152396326576}], \"train_return\": 0.2835131336273229, \"train_returns_over_hodl\": 0.014709602816078693, \"train_sharpe\": 0.9182570640386871, \"validation_return\": 0.06393170085493716, \"validation_returns_over_hodl\": 0.011770160801770846, \"validation_sharpe\": 1.4400088119228285}, {\"centeredness_margin\": 0.32718550439888994, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8336799986151874, \"annualised_returns_over_hodl\": -0.07193047821087539, \"annualised_returns_over_uniform_hodl\": -0.034223919643912804, \"calmar\": -1.4259757480973994, \"daily_log_sharpe\": -2.7869917211505197, \"daily_returns\": 0.00725381026311497, \"fee_revenue_over_value\": 9.679725395392679e-05, \"jax_sharpe\": -2.336676458122122, \"return\": -0.5144377943600846, \"returns_over_hodl\": -0.029616406159897135, \"returns_over_uniform_hodl\": -0.013926761733289661, \"sharpe\": -2.474844043163772, \"sterling\": -3.50443027620235, \"ulcer\": -0.11499382633960124}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.21574798549352767, \"optuna_trial_number\": 37, \"price_ratio\": 1.4373516492194178, \"shift_exponent\": 1.0089672192731767e-05, \"step\": 37, \"test_objective\": [{\"annualised_returns\": -0.8336799986151874, \"annualised_returns_over_hodl\": -0.07193047821087539, \"annualised_returns_over_uniform_hodl\": -0.034223919643912804, \"calmar\": -1.4259757480973994, \"daily_log_sharpe\": -2.7869917211505197, \"daily_returns\": 0.00725381026311497, \"fee_revenue_over_value\": 9.679725395392679e-05, \"jax_sharpe\": -2.336676458122122, \"return\": -0.5144377943600846, \"returns_over_hodl\": -0.029616406159897135, \"returns_over_uniform_hodl\": -0.013926761733289661, \"sharpe\": -2.474844043163772, \"sterling\": -3.50443027620235, \"ulcer\": -0.11499382633960124}], \"train_objective\": [{\"annualised_returns\": 0.5293394727333098, \"annualised_returns_over_hodl\": 0.03848809731512115, \"annualised_returns_over_uniform_hodl\": 0.03848809731512115, \"calmar\": 0.8807761110294758, \"daily_log_sharpe\": 0.5522095318105846, \"daily_returns\": 0.008894602737109928, \"fee_revenue_over_value\": 2.339954118602609e-05, \"jax_sharpe\": 0.938358522447105, \"return\": 0.2942443032825328, \"returns_over_hodl\": 0.023193365555472356, \"returns_over_uniform_hodl\": 0.023193365555472356, \"sharpe\": 0.9243729463915784, \"sterling\": 2.2514775093059476, \"ulcer\": -0.12182450034553229}], \"train_return\": 0.2942443032825328, \"train_returns_over_hodl\": 0.023193365555472356, \"train_sharpe\": 0.9383585224471049, \"validation_return\": 0.06626621977470304, \"validation_returns_over_hodl\": 0.016492791826545394, \"validation_sharpe\": 1.331201281763685}, {\"centeredness_margin\": 0.3075523740347971, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8353890945632636, \"annualised_returns_over_hodl\": -0.13300489540046967, \"annualised_returns_over_uniform_hodl\": -0.044148185949482, \"calmar\": -1.4301819220472456, \"daily_log_sharpe\": -2.6809401051876, \"daily_returns\": 0.007891888145408612, \"fee_revenue_over_value\": 8.492416820679642e-05, \"jax_sharpe\": -2.2067970510581394, \"return\": -0.5164535007073439, \"returns_over_hodl\": -0.055858780342318726, \"returns_over_uniform_hodl\": -0.018020231245846174, \"sharpe\": -2.353281639387032, \"sterling\": -3.4096806859446547, \"ulcer\": -0.11450101363032586}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.25158891386231624, \"optuna_trial_number\": 38, \"price_ratio\": 1.1668255010026958, \"shift_exponent\": 1.2619562767880982e-05, \"step\": 38, \"test_objective\": [{\"annualised_returns\": -0.8353890945632636, \"annualised_returns_over_hodl\": -0.13300489540046967, \"annualised_returns_over_uniform_hodl\": -0.044148185949482, \"calmar\": -1.4301819220472456, \"daily_log_sharpe\": -2.6809401051876, \"daily_returns\": 0.007891888145408612, \"fee_revenue_over_value\": 8.492416820679642e-05, \"jax_sharpe\": -2.2067970510581394, \"return\": -0.5164535007073439, \"returns_over_hodl\": -0.055858780342318726, \"returns_over_uniform_hodl\": -0.018020231245846174, \"sharpe\": -2.353281639387032, \"sterling\": -3.4096806859446547, \"ulcer\": -0.11450101363032586}], \"train_objective\": [{\"annualised_returns\": 0.5016991959075687, \"annualised_returns_over_hodl\": 0.019719145750217715, \"annualised_returns_over_uniform_hodl\": 0.019719145750217715, \"calmar\": 0.8179085540289237, \"daily_log_sharpe\": 0.5155565569353623, \"daily_returns\": 0.008916170694524704, \"fee_revenue_over_value\": 2.9344293401981403e-05, \"jax_sharpe\": 0.9126346638537802, \"return\": 0.27999208490215133, \"returns_over_hodl\": 0.01192596012493019, \"returns_over_uniform_hodl\": 0.01192596012493019, \"sharpe\": 0.8939172808798403, \"sterling\": 2.099840058179112, \"ulcer\": -0.12499648262657598}], \"train_return\": 0.27999208490215133, \"train_returns_over_hodl\": 0.01192596012493019, \"train_sharpe\": 0.9126346638537802, \"validation_return\": 0.06647896155578703, \"validation_returns_over_hodl\": 0.01597877030006245, \"validation_sharpe\": 1.4446243890908872}, {\"centeredness_margin\": 0.34825637738524456, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8404229905240047, \"annualised_returns_over_hodl\": -0.24903972265181507, \"annualised_returns_over_uniform_hodl\": -0.0733786829997829, \"calmar\": -1.430273972949315, \"daily_log_sharpe\": -2.6570660267760213, \"daily_returns\": 0.009057000809718337, \"fee_revenue_over_value\": 5.6527352381074186e-05, \"jax_sharpe\": -2.1799475050877066, \"return\": -0.522464126496788, \"returns_over_hodl\": -0.10894143859487182, \"returns_over_uniform_hodl\": -0.03022652977436424, \"sharpe\": -2.3199268167356943, \"sterling\": -3.3397756890451116, \"ulcer\": -0.11683558267525805}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2679214206351387, \"optuna_trial_number\": 39, \"price_ratio\": 1.0754497926740432, \"shift_exponent\": 2.0432739938245116e-05, \"step\": 39, \"test_objective\": [{\"annualised_returns\": -0.8404229905240047, \"annualised_returns_over_hodl\": -0.24903972265181507, \"annualised_returns_over_uniform_hodl\": -0.0733786829997829, \"calmar\": -1.430273972949315, \"daily_log_sharpe\": -2.6570660267760213, \"daily_returns\": 0.009057000809718337, \"fee_revenue_over_value\": 5.6527352381074186e-05, \"jax_sharpe\": -2.1799475050877066, \"return\": -0.522464126496788, \"returns_over_hodl\": -0.10894143859487182, \"returns_over_uniform_hodl\": -0.03022652977436424, \"sharpe\": -2.3199268167356943, \"sterling\": -3.3397756890451116, \"ulcer\": -0.11683558267525805}], \"train_objective\": [{\"annualised_returns\": 0.5070037312430065, \"annualised_returns_over_hodl\": 0.02332115622980968, \"annualised_returns_over_uniform_hodl\": 0.02332115622980968, \"calmar\": 0.822191455545669, \"daily_log_sharpe\": 0.5093651328076374, \"daily_returns\": 0.008967896735494462, \"fee_revenue_over_value\": 1.8561219268102855e-05, \"jax_sharpe\": 0.916965316185233, \"return\": 0.28273521400707735, \"returns_over_hodl\": 0.014094601311067656, \"returns_over_uniform_hodl\": 0.014094601311067656, \"sharpe\": 0.8889230197485417, \"sterling\": 2.11496491524274, \"ulcer\": -0.12557392900272024}], \"train_return\": 0.28273521400707735, \"train_returns_over_hodl\": 0.014094601311067656, \"train_sharpe\": 0.9169653161852329, \"validation_return\": 0.06139158698774194, \"validation_returns_over_hodl\": 0.010228330922488205, \"validation_sharpe\": 1.3966020058894195}, {\"centeredness_margin\": 0.32277724762761556, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8379169904276723, \"annualised_returns_over_hodl\": -0.2754500749381291, \"annualised_returns_over_uniform_hodl\": -0.0588270059299385, \"calmar\": -1.4338809968758643, \"daily_log_sharpe\": -2.624558637339173, \"daily_returns\": 0.00869052495219643, \"fee_revenue_over_value\": 4.578836381493638e-05, \"jax_sharpe\": -2.1494763509947403, \"return\": -0.5194579490385254, \"returns_over_hodl\": -0.12169729943531715, \"returns_over_uniform_hodl\": -0.024121624766020977, \"sharpe\": -2.286266090819585, \"sterling\": -3.344304117151131, \"ulcer\": -0.11523867747786726}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.26925765359260956, \"optuna_trial_number\": 40, \"price_ratio\": 1.0415385584555679, \"shift_exponent\": 5.7254842338588133e-05, \"step\": 40, \"test_objective\": [{\"annualised_returns\": -0.8379169904276723, \"annualised_returns_over_hodl\": -0.2754500749381291, \"annualised_returns_over_uniform_hodl\": -0.0588270059299385, \"calmar\": -1.4338809968758643, \"daily_log_sharpe\": -2.624558637339173, \"daily_returns\": 0.00869052495219643, \"fee_revenue_over_value\": 4.578836381493638e-05, \"jax_sharpe\": -2.1494763509947403, \"return\": -0.5194579490385254, \"returns_over_hodl\": -0.12169729943531715, \"returns_over_uniform_hodl\": -0.024121624766020977, \"sharpe\": -2.286266090819585, \"sterling\": -3.344304117151131, \"ulcer\": -0.11523867747786726}], \"train_objective\": [{\"annualised_returns\": 0.5002734463005729, \"annualised_returns_over_hodl\": 0.018750999682574987, \"annualised_returns_over_uniform_hodl\": 0.018750999682574987, \"calmar\": 0.8099864218123098, \"daily_log_sharpe\": 0.489996951639544, \"daily_returns\": 0.009021545641419512, \"fee_revenue_over_value\": 1.3614178960922425e-05, \"jax_sharpe\": 0.9111276347547683, \"return\": 0.2792541396328685, \"returns_over_hodl\": 0.011342561224306547, \"returns_over_uniform_hodl\": 0.011342561224306547, \"sharpe\": 0.8696761541545905, \"sterling\": 2.084435557779764, \"ulcer\": -0.12578582325305482}], \"train_return\": 0.2792541396328685, \"train_returns_over_hodl\": 0.011342561224306547, \"train_sharpe\": 0.9111276347547683, \"validation_return\": 0.051262819020999384, \"validation_returns_over_hodl\": 0.0032307400342324044, \"validation_sharpe\": 1.214930287423765}, {\"centeredness_margin\": 0.3543019819510827, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8330349159357333, \"annualised_returns_over_hodl\": -0.13797315579024605, \"annualised_returns_over_uniform_hodl\": -0.030478095831493524, \"calmar\": -1.4327794673946008, \"daily_log_sharpe\": -2.6734996012276664, \"daily_returns\": 0.008030077580106917, \"fee_revenue_over_value\": 8.449002977159206e-05, \"jax_sharpe\": -2.1963191625903256, \"return\": -0.51368020142599, \"returns_over_hodl\": -0.05804146964801005, \"returns_over_uniform_hodl\": -0.012388252127036203, \"sharpe\": -2.3477020370180988, \"sterling\": -3.4237749748852524, \"ulcer\": -0.11326818956999965}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.2502107702364681, \"optuna_trial_number\": 41, \"price_ratio\": 1.1589816466030654, \"shift_exponent\": 5.496440847786953e-05, \"step\": 41, \"test_objective\": [{\"annualised_returns\": -0.8330349159357333, \"annualised_returns_over_hodl\": -0.13797315579024605, \"annualised_returns_over_uniform_hodl\": -0.030478095831493524, \"calmar\": -1.4327794673946008, \"daily_log_sharpe\": -2.6734996012276664, \"daily_returns\": 0.008030077580106917, \"fee_revenue_over_value\": 8.449002977159206e-05, \"jax_sharpe\": -2.1963191625903256, \"return\": -0.51368020142599, \"returns_over_hodl\": -0.05804146964801005, \"returns_over_uniform_hodl\": -0.012388252127036203, \"sharpe\": -2.3477020370180988, \"sterling\": -3.4237749748852524, \"ulcer\": -0.11326818956999965}], \"train_objective\": [{\"annualised_returns\": 0.494707261038678, \"annualised_returns_over_hodl\": 0.014971317509330673, \"annualised_returns_over_uniform_hodl\": 0.014971317509330673, \"calmar\": 0.8060070634668675, \"daily_log_sharpe\": 0.5063307277854058, \"daily_returns\": 0.00892619782202271, \"fee_revenue_over_value\": 2.515546309804574e-05, \"jax_sharpe\": 0.9066179908132338, \"return\": 0.2763705294803207, \"returns_over_hodl\": 0.009062859649068944, \"returns_over_uniform_hodl\": 0.009062859649068944, \"sharpe\": 0.8847914636147434, \"sterling\": 2.0685883332353923, \"ulcer\": -0.12507460903270706}], \"train_return\": 0.2763705294803207, \"train_returns_over_hodl\": 0.009062859649068944, \"train_sharpe\": 0.9066179908132338, \"validation_return\": 0.06266166340105328, \"validation_returns_over_hodl\": 0.013652367399159315, \"validation_sharpe\": 1.3899948721177398}, {\"centeredness_margin\": 0.3896953399973982, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8227234851284226, \"annualised_returns_over_hodl\": -0.01992733154209736, \"annualised_returns_over_uniform_hodl\": 0.02939764457874472, \"calmar\": -1.4385292625120358, \"daily_log_sharpe\": -2.9341925793312957, \"daily_returns\": 0.007216255347303855, \"fee_revenue_over_value\": 4.298836004599449e-05, \"jax_sharpe\": -2.542271758684213, \"return\": -0.5018003458144731, \"returns_over_hodl\": -0.008073762549881458, \"returns_over_uniform_hodl\": 0.011737199889047423, \"sharpe\": -2.654236650051916, \"sterling\": -3.64848477407704, \"ulcer\": -0.11263857541031942}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.14989925125908363, \"optuna_trial_number\": 42, \"price_ratio\": 3.8156983518900685, \"shift_exponent\": 0.0009946084822200464, \"step\": 42, \"test_objective\": [{\"annualised_returns\": -0.8227234851284226, \"annualised_returns_over_hodl\": -0.01992733154209736, \"annualised_returns_over_uniform_hodl\": 0.02939764457874472, \"calmar\": -1.4385292625120358, \"daily_log_sharpe\": -2.9341925793312957, \"daily_returns\": 0.007216255347303855, \"fee_revenue_over_value\": 4.298836004599449e-05, \"jax_sharpe\": -2.542271758684213, \"return\": -0.5018003458144731, \"returns_over_hodl\": -0.008073762549881458, \"returns_over_uniform_hodl\": 0.011737199889047423, \"sharpe\": -2.654236650051916, \"sterling\": -3.64848477407704, \"ulcer\": -0.11263857541031942}], \"train_objective\": [{\"annualised_returns\": 0.45857871146785256, \"annualised_returns_over_hodl\": -0.009561540872664476, \"annualised_returns_over_uniform_hodl\": -0.009561540872664476, \"calmar\": 0.8520736870449754, \"daily_log_sharpe\": 0.5588489376683345, \"daily_returns\": 0.008885244385443978, \"fee_revenue_over_value\": 3.398509313574425e-05, \"jax_sharpe\": 0.9016432238744928, \"return\": 0.2575501984204771, \"returns_over_hodl\": -0.0058159679641948125, \"returns_over_uniform_hodl\": -0.0058159679641948125, \"sharpe\": 0.8820061902917039, \"sterling\": 2.2319258887727793, \"ulcer\": -0.10226185591858819}], \"train_return\": 0.2575501984204771, \"train_returns_over_hodl\": -0.0058159679641948125, \"train_sharpe\": 0.9016432238744927, \"validation_return\": 0.05069039310034107, \"validation_returns_over_hodl\": 0.004624077212784794, \"validation_sharpe\": 1.0360919005540623}, {\"centeredness_margin\": 0.17965021963531358, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8296941043267907, \"annualised_returns_over_hodl\": -0.020843746248927153, \"annualised_returns_over_uniform_hodl\": -0.011078889999191599, \"calmar\": -1.4314711197825227, \"daily_log_sharpe\": -2.9140823784257113, \"daily_returns\": 0.006951029160017879, \"fee_revenue_over_value\": 4.416671465191439e-05, \"jax_sharpe\": -2.514347850676355, \"return\": -0.509784413423154, \"returns_over_hodl\": -0.008447405577164835, \"returns_over_uniform_hodl\": -0.004476737913333029, \"sharpe\": -2.624101340460753, \"sterling\": -3.587057603349221, \"ulcer\": -0.11537973350084013}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.1848539247156815, \"optuna_trial_number\": 43, \"price_ratio\": 3.5919672095887107, \"shift_exponent\": 1.2866204121814264e-05, \"step\": 43, \"test_objective\": [{\"annualised_returns\": -0.8296941043267907, \"annualised_returns_over_hodl\": -0.020843746248927153, \"annualised_returns_over_uniform_hodl\": -0.011078889999191599, \"calmar\": -1.4314711197825227, \"daily_log_sharpe\": -2.9140823784257113, \"daily_returns\": 0.006951029160017879, \"fee_revenue_over_value\": 4.416671465191439e-05, \"jax_sharpe\": -2.514347850676355, \"return\": -0.509784413423154, \"returns_over_hodl\": -0.008447405577164835, \"returns_over_uniform_hodl\": -0.004476737913333029, \"sharpe\": -2.624101340460753, \"sterling\": -3.587057603349221, \"ulcer\": -0.11537973350084013}], \"train_objective\": [{\"annualised_returns\": 0.5477772571825077, \"annualised_returns_over_hodl\": 0.05100815583236673, \"annualised_returns_over_uniform_hodl\": 0.051008155832366286, \"calmar\": 1.0034630045359936, \"daily_log_sharpe\": 0.6342182659178693, \"daily_returns\": 0.008885109064618714, \"fee_revenue_over_value\": 5.6699788109140306e-05, \"jax_sharpe\": 0.9828472494074714, \"return\": 0.30369519106672405, \"returns_over_hodl\": 0.03066497323793871, \"returns_over_uniform_hodl\": 0.030664973237938487, \"sharpe\": 0.9661403155922857, \"sterling\": 2.598126094691133, \"ulcer\": -0.10507146024006378}], \"train_return\": 0.30369519106672405, \"train_returns_over_hodl\": 0.03066497323793871, \"train_sharpe\": 0.9828472494074713, \"validation_return\": 0.05386109543149131, \"validation_returns_over_hodl\": 0.005061953133945085, \"validation_sharpe\": 1.051768258672148}, {\"centeredness_margin\": 0.4956390571108398, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8031490201735354, \"annualised_returns_over_hodl\": -0.12002890951278655, \"annualised_returns_over_uniform_hodl\": 0.14306136440676087, \"calmar\": -1.4588578197072049, \"daily_log_sharpe\": -2.5692331050911736, \"daily_returns\": 0.008690524952359922, \"fee_revenue_over_value\": 4.557027842062713e-05, \"jax_sharpe\": -2.057962838717037, \"return\": -0.48033615103904215, \"returns_over_hodl\": -0.050193087126162994, \"returns_over_uniform_hodl\": 0.0553263997962008, \"sharpe\": -2.2619792711878293, \"sterling\": -3.5737717088417837, \"ulcer\": -0.09892803751595598}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.1999675668536461, \"optuna_trial_number\": 44, \"price_ratio\": 1.0246671970129546, \"shift_exponent\": 0.0005244297020637258, \"step\": 44, \"test_objective\": [{\"annualised_returns\": -0.8031490201735354, \"annualised_returns_over_hodl\": -0.12002890951278655, \"annualised_returns_over_uniform_hodl\": 0.14306136440676087, \"calmar\": -1.4588578197072049, \"daily_log_sharpe\": -2.5692331050911736, \"daily_returns\": 0.008690524952359922, \"fee_revenue_over_value\": 4.557027842062713e-05, \"jax_sharpe\": -2.057962838717037, \"return\": -0.48033615103904215, \"returns_over_hodl\": -0.050193087126162994, \"returns_over_uniform_hodl\": 0.0553263997962008, \"sharpe\": -2.2619792711878293, \"sterling\": -3.5737717088417837, \"ulcer\": -0.09892803751595598}], \"train_objective\": [{\"annualised_returns\": 0.33621209420463694, \"annualised_returns_over_hodl\": -0.09265380246808863, \"annualised_returns_over_uniform_hodl\": -0.09265380246808863, \"calmar\": 0.5427742486643841, \"daily_log_sharpe\": 0.33641141836383875, \"daily_returns\": 0.009098270242687384, \"fee_revenue_over_value\": 1.0421662109074028e-05, \"jax_sharpe\": 0.762239003010411, \"return\": 0.19239911809385646, \"returns_over_hodl\": -0.057322590770953696, \"returns_over_uniform_hodl\": -0.057322590770953696, \"sharpe\": 0.7165142800843302, \"sterling\": 1.3976998386042823, \"ulcer\": -0.1262809314986471}], \"train_return\": 0.19239911809385646, \"train_returns_over_hodl\": -0.057322590770953696, \"train_sharpe\": 0.762239003010411, \"validation_return\": 0.030486270507650515, \"validation_returns_over_hodl\": 1.134869975771835e-12, \"validation_sharpe\": 0.8017265757971355}, {\"centeredness_margin\": 0.7587167236717354, \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": -Infinity, \"optuna_trial_number\": 45, \"price_ratio\": 1.7240148426959157, \"shift_exponent\": 88.52173587636129, \"step\": 45, \"test_objective\": -Infinity, \"train_objective\": -Infinity, \"train_return\": -Infinity, \"train_returns_over_hodl\": -Infinity, \"train_sharpe\": -Infinity, \"validation_return\": -Infinity, \"validation_returns_over_hodl\": -Infinity, \"validation_sharpe\": -Infinity}, {\"centeredness_margin\": 0.28585824911907054, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8392577063544111, \"annualised_returns_over_hodl\": -0.2814433966753487, \"annualised_returns_over_uniform_hodl\": -0.06661218727803697, \"calmar\": -1.4327720375308524, \"daily_log_sharpe\": -2.6287469193149335, \"daily_returns\": 0.008690524952248424, \"fee_revenue_over_value\": 5.160322171636549e-05, \"jax_sharpe\": -2.155288589319053, \"return\": -0.5210627800496945, \"returns_over_hodl\": -0.12463050248926821, \"returns_over_uniform_hodl\": -0.02738069413689348, \"sharpe\": -2.2893963573792218, \"sterling\": -3.334498488947675, \"ulcer\": -0.11600885254160236}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.27059999997384493, \"optuna_trial_number\": 46, \"price_ratio\": 1.044897403115403, \"shift_exponent\": 3.681037916507494e-05, \"step\": 46, \"test_objective\": [{\"annualised_returns\": -0.8392577063544111, \"annualised_returns_over_hodl\": -0.2814433966753487, \"annualised_returns_over_uniform_hodl\": -0.06661218727803697, \"calmar\": -1.4327720375308524, \"daily_log_sharpe\": -2.6287469193149335, \"daily_returns\": 0.008690524952248424, \"fee_revenue_over_value\": 5.160322171636549e-05, \"jax_sharpe\": -2.155288589319053, \"return\": -0.5210627800496945, \"returns_over_hodl\": -0.12463050248926821, \"returns_over_uniform_hodl\": -0.02738069413689348, \"sharpe\": -2.2893963573792218, \"sterling\": -3.334498488947675, \"ulcer\": -0.11600885254160236}], \"train_objective\": [{\"annualised_returns\": 0.5031893321864263, \"annualised_returns_over_hodl\": 0.02073101317178172, \"annualised_returns_over_uniform_hodl\": 0.02073101317178172, \"calmar\": 0.8146816223420116, \"daily_log_sharpe\": 0.49577199741766637, \"daily_returns\": 0.008999278938317524, \"fee_revenue_over_value\": 1.769087290033203e-05, \"jax_sharpe\": 0.9136356431080943, \"return\": 0.2807630615605117, \"returns_over_hodl\": 0.012535472718365304, \"returns_over_uniform_hodl\": 0.012535472718365304, \"sharpe\": 0.8755246579118057, \"sterling\": 2.096533249110199, \"ulcer\": -0.12579447110196254}], \"train_return\": 0.2807630615605117, \"train_returns_over_hodl\": 0.012535472718365304, \"train_sharpe\": 0.9136356431080943, \"validation_return\": 0.053421079207566224, \"validation_returns_over_hodl\": 0.0030190659439219836, \"validation_sharpe\": 1.2524560975803911}, {\"centeredness_margin\": 0.2545511269367291, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8417232942490541, \"annualised_returns_over_hodl\": -0.23398920136220447, \"annualised_returns_over_uniform_hodl\": -0.0809292014244739, \"calmar\": -1.4285263128791987, \"daily_log_sharpe\": -2.668430016111086, \"daily_returns\": 0.008833732404336159, \"fee_revenue_over_value\": 6.718096335962129e-05, \"jax_sharpe\": -2.1836295067284106, \"return\": -0.5240350788833026, \"returns_over_hodl\": -0.1017918065314034, \"returns_over_uniform_hodl\": -0.03341679888703497, \"sharpe\": -2.3315484252884553, \"sterling\": -3.3426090822896337, \"ulcer\": -0.11747438601626717}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.26687079302170613, \"optuna_trial_number\": 47, \"price_ratio\": 1.0844866114611427, \"shift_exponent\": 1.0443433908728175e-05, \"step\": 47, \"test_objective\": [{\"annualised_returns\": -0.8417232942490541, \"annualised_returns_over_hodl\": -0.23398920136220447, \"annualised_returns_over_uniform_hodl\": -0.0809292014244739, \"calmar\": -1.4285263128791987, \"daily_log_sharpe\": -2.668430016111086, \"daily_returns\": 0.008833732404336159, \"fee_revenue_over_value\": 6.718096335962129e-05, \"jax_sharpe\": -2.1836295067284106, \"return\": -0.5240350788833026, \"returns_over_hodl\": -0.1017918065314034, \"returns_over_uniform_hodl\": -0.03341679888703497, \"sharpe\": -2.3315484252884553, \"sterling\": -3.3426090822896337, \"ulcer\": -0.11747438601626717}], \"train_objective\": [{\"annualised_returns\": 0.5069474800255247, \"annualised_returns_over_hodl\": 0.02328295920367096, \"annualised_returns_over_uniform_hodl\": 0.023282959203671183, \"calmar\": 0.8221007520235338, \"daily_log_sharpe\": 0.5118190383605998, \"daily_returns\": 0.00895913836231686, \"fee_revenue_over_value\": 2.2064981521593356e-05, \"jax_sharpe\": 0.9169376665256113, \"return\": 0.28270614479024725, \"returns_over_hodl\": 0.01407162000087192, \"returns_over_uniform_hodl\": 0.014071620000872143, \"sharpe\": 0.891363783431784, \"sterling\": 2.1145217582318954, \"ulcer\": -0.12559649317467336}], \"train_return\": 0.28270614479024725, \"train_returns_over_hodl\": 0.01407162000087192, \"train_sharpe\": 0.9169376665256113, \"validation_return\": 0.06435200557791765, \"validation_returns_over_hodl\": 0.012666775740830749, \"validation_sharpe\": 1.4468791182517067}, {\"centeredness_margin\": 0.36972517591482373, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8329829470501537, \"annualised_returns_over_hodl\": -0.2533937176911457, \"annualised_returns_over_uniform_hodl\": -0.030176326313710478, \"calmar\": -1.4378633700679249, \"daily_log_sharpe\": -2.645228856862493, \"daily_returns\": 0.008690524951785865, \"fee_revenue_over_value\": 3.932969393059401e-05, \"jax_sharpe\": -2.156850905708826, \"return\": -0.5136192446836376, \"returns_over_hodl\": -0.11102570514463272, \"returns_over_uniform_hodl\": -0.012264461989279218, \"sharpe\": -2.3161062412406537, \"sterling\": -3.4139507503025057, \"ulcer\": -0.11277314537459374}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.25008768011783566, \"optuna_trial_number\": 48, \"price_ratio\": 1.0252293092076576, \"shift_exponent\": 0.00018125751636554547, \"step\": 48, \"test_objective\": [{\"annualised_returns\": -0.8329829470501537, \"annualised_returns_over_hodl\": -0.2533937176911457, \"annualised_returns_over_uniform_hodl\": -0.030176326313710478, \"calmar\": -1.4378633700679249, \"daily_log_sharpe\": -2.645228856862493, \"daily_returns\": 0.008690524951785865, \"fee_revenue_over_value\": 3.932969393059401e-05, \"jax_sharpe\": -2.156850905708826, \"return\": -0.5136192446836376, \"returns_over_hodl\": -0.11102570514463272, \"returns_over_uniform_hodl\": -0.012264461989279218, \"sharpe\": -2.3161062412406537, \"sterling\": -3.4139507503025057, \"ulcer\": -0.11277314537459374}], \"train_objective\": [{\"annualised_returns\": 0.45049873489464765, \"annualised_returns_over_hodl\": -0.015048196809728354, \"annualised_returns_over_uniform_hodl\": -0.015048196809728354, \"calmar\": 0.7277615548386465, \"daily_log_sharpe\": 0.44268487221278635, \"daily_returns\": 0.00913835924770349, \"fee_revenue_over_value\": 1.3330942260260735e-05, \"jax_sharpe\": 0.867784270391922, \"return\": 0.25331616138840296, \"returns_over_hodl\": -0.009163279279181369, \"returns_over_uniform_hodl\": -0.009163279279181369, \"sharpe\": 0.8225517536115937, \"sterling\": 1.873780840001201, \"ulcer\": -0.12616959448143428}], \"train_return\": 0.25331616138840296, \"train_returns_over_hodl\": -0.009163279279181369, \"train_sharpe\": 0.867784270391922, \"validation_return\": 0.032822781551564706, \"validation_returns_over_hodl\": 0.0022673868780500595, \"validation_sharpe\": 0.8502456995444336}, {\"centeredness_margin\": 0.2076436598526294, \"continuous_test_metrics\": [{\"annualised_returns\": -0.8322528064020767, \"annualised_returns_over_hodl\": -0.2501298139017025, \"annualised_returns_over_uniform_hodl\": -0.02593659346536281, \"calmar\": -1.4383236377249324, \"daily_log_sharpe\": -2.628173816566865, \"daily_returns\": 0.008739595404478757, \"fee_revenue_over_value\": 9.217363822356774e-05, \"jax_sharpe\": -2.1441414010495166, \"return\": -0.512764023434475, \"returns_over_hodl\": -0.10946258881095516, \"returns_over_uniform_hodl\": -0.010527690105469167, \"sharpe\": -2.2967344386589157, \"sterling\": -3.375077709666533, \"ulcer\": -0.11253255561144242}], \"hessian_trace\": 0, \"iterations_since_improvement\": 0, \"local_learning_rate\": 0, \"objective\": 0.25921843195344, \"optuna_trial_number\": 49, \"price_ratio\": 1.0743548090215316, \"shift_exponent\": 0.00010728908627115939, \"step\": 49, \"test_objective\": [{\"annualised_returns\": -0.8322528064020767, \"annualised_returns_over_hodl\": -0.2501298139017025, \"annualised_returns_over_uniform_hodl\": -0.02593659346536281, \"calmar\": -1.4383236377249324, \"daily_log_sharpe\": -2.628173816566865, \"daily_returns\": 0.008739595404478757, \"fee_revenue_over_value\": 9.217363822356774e-05, \"jax_sharpe\": -2.1441414010495166, \"return\": -0.512764023434475, \"returns_over_hodl\": -0.10946258881095516, \"returns_over_uniform_hodl\": -0.010527690105469167, \"sharpe\": -2.2967344386589157, \"sterling\": -3.375077709666533, \"ulcer\": -0.11253255561144242}], \"train_objective\": [{\"annualised_returns\": 0.48967066833898953, \"annualised_returns_over_hodl\": 0.011551251747016433, \"annualised_returns_over_uniform_hodl\": 0.011551251747016877, \"calmar\": 0.7940886215538232, \"daily_log_sharpe\": 0.4844367417622134, \"daily_returns\": 0.008963441381538626, \"fee_revenue_over_value\": 2.2429116213383827e-05, \"jax_sharpe\": 0.9021272331260283, \"return\": 0.27375764137452996, \"returns_over_hodl\": 0.006997183355958558, \"returns_over_uniform_hodl\": 0.00699718335595878, \"sharpe\": 0.8638450421978052, \"sterling\": 2.0426738078130176, \"ulcer\": -0.12557286718575436}], \"train_return\": 0.27375764137452996, \"train_returns_over_hodl\": 0.006997183355958558, \"train_sharpe\": 0.9021272331260283, \"validation_return\": 0.04779028123095386, \"validation_returns_over_hodl\": 0.0030590026923063007, \"validation_sharpe\": 1.1453009943774282}]" \ No newline at end of file diff --git a/scripts/prepare_bold_usdc_data.py b/scripts/prepare_bold_usdc_data.py new file mode 100644 index 00000000..bb547dce --- /dev/null +++ b/scripts/prepare_bold_usdc_data.py @@ -0,0 +1,154 @@ +"""Prepare price data for BOLD/USDC reCLAMM simulations. + +1. Fetches BOLD/USD daily price + traded volume from CoinGecko (free tier = + last 365 days), expands to a minute grid by forward-fill, and writes + quantammsim/data/BOLD_USD.parquet. The real daily traded volume is spread + evenly across the day so that market_features' daily resample recovers the + true daily USD volume (used by the organic-volume noise model). + +2. Rebuilds quantammsim/data/USDC_USD.parquet pegged at $1.00 over the union + of its existing coverage and the BOLD grid, so both the BOLD/USDC window + and the earlier Monad window stay covered. + +BOLD (Liquity V2, CoinGecko id "liquity-bold-2") is not on Binance/Coinbase, +so CoinGecko is the only source. Free-tier history is capped at 365 days — +this bounds the earliest possible simulation start date. + +Usage (from repo root): + python scripts/prepare_bold_usdc_data.py +""" + +from __future__ import annotations + +import time +from pathlib import Path + +import numpy as np +import pandas as pd +import requests + +REPO_ROOT = Path(__file__).resolve().parent.parent +DATA_DIR = REPO_ROOT / "quantammsim" / "data" + +CG_ID = "liquity-bold-2" +_CG_URL = f"https://api.coingecko.com/api/v3/coins/{CG_ID}/market_chart" + + +def _fetch_coingecko_daily() -> tuple[list, list]: + """Return (prices, total_volumes) as [[unix_ms, value], ...] for last 365d.""" + params = {"vs_currency": "usd", "days": "365", "interval": "daily"} + for attempt in range(4): + try: + resp = requests.get(_CG_URL, params=params, timeout=40) + if resp.status_code == 429: + print(" rate-limited, sleeping 60s …") + time.sleep(60) + continue + resp.raise_for_status() + j = resp.json() + return j.get("prices", []), j.get("total_volumes", []) + except Exception as exc: + print(f" CoinGecko attempt {attempt+1} failed: {exc}") + time.sleep(10) + raise RuntimeError("CoinGecko fetch failed for BOLD") + + +def build_bold_parquet() -> pd.DataFrame: + prices, volumes = _fetch_coingecko_daily() + if len(prices) < 100: + raise RuntimeError(f"BOLD price series too short: {len(prices)}") + + # Daily frame: price (close) + daily USD volume, indexed by midnight-UTC day. + pdf = pd.DataFrame(prices, columns=["unix", "close"]) + vdf = pd.DataFrame(volumes, columns=["unix", "vol_usd"]) + pdf["day"] = pd.to_datetime(pdf["unix"], unit="ms", utc=True).dt.normalize() + vdf["day"] = pd.to_datetime(vdf["unix"], unit="ms", utc=True).dt.normalize() + daily = (pdf.groupby("day")["close"].last() + .to_frame() + .join(vdf.groupby("day")["vol_usd"].last())) + daily["vol_usd"] = daily["vol_usd"].fillna(0.0) + daily = daily.dropna(subset=["close"]).sort_index() + + start = daily.index.min() + end = daily.index.max() + print(f" BOLD daily: {start.date()} → {end.date()} ({len(daily)} days)") + print(f" price ${daily['close'].min():.4f}–${daily['close'].max():.4f}, " + f"last ${daily['close'].iloc[-1]:.4f}") + print(f" daily vol (USD): median ${daily['vol_usd'].median():,.0f}, " + f"mean ${daily['vol_usd'].mean():,.0f}") + + # Expand to a complete minute grid covering full days [start, end+1day). + epoch = pd.Timestamp("1970-01-01", tz="UTC") + grid_end = end + pd.Timedelta(days=1) - pd.Timedelta(minutes=1) + min_range = pd.date_range(start=start, end=grid_end, freq="1min") + min_idx = ((min_range - epoch).total_seconds() * 1000).astype(np.int64) + + out = pd.DataFrame(index=min_idx) + out.index.name = "unix" + out["date"] = min_range + out["day"] = min_range.normalize() + + # Map close + per-minute volume (daily vol spread across 1440 minutes so the + # daily resample-sum in market_features recovers the real daily volume). + close_map = daily["close"].to_dict() + vol_map = (daily["vol_usd"] / 1440.0).to_dict() + out["close"] = out["day"].map(close_map).astype(float) + out["Volume USD"] = out["day"].map(vol_map).astype(float) + out["close"] = out["close"].ffill().bfill() + out["Volume USD"] = out["Volume USD"].fillna(0.0) + out["open"] = out["high"] = out["low"] = out["close"] + out["Volume BOLD"] = out["Volume USD"] / out["close"].clip(lower=1e-9) + out["symbol"] = "BOLD" + out = out.drop(columns=["day"]) + out = out[["date", "symbol", "open", "high", "low", "close", + "Volume USD", "Volume BOLD"]] + + path = DATA_DIR / "BOLD_USD.parquet" + out.to_parquet(path, engine="pyarrow") + print(f" Saved {len(out):,} rows → {path}") + return out + + +def rebuild_usdc_parquet(bold_df: pd.DataFrame) -> None: + """USDC pegged at $1.00 over the union of its current grid and BOLD's.""" + path = DATA_DIR / "USDC_USD.parquet" + start_ms = int(bold_df.index.min()) + end_ms = int(bold_df.index.max()) + if path.exists(): + existing = pd.read_parquet(path) + start_ms = min(start_ms, int(existing.index.min())) + end_ms = max(end_ms, int(existing.index.max())) + + idx = np.arange(start_ms, end_ms + 60_000, 60_000, dtype=np.int64) + n = len(idx) + out = pd.DataFrame({ + "date": pd.to_datetime(idx, unit="ms", utc=True), + "symbol": "USDC", + "open": np.ones(n), "high": np.ones(n), + "low": np.ones(n), "close": np.ones(n), + "Volume USD": np.zeros(n), "Volume USDC": np.zeros(n), + }, index=idx) + out.index.name = "unix" + out.to_parquet(path, engine="pyarrow") + print(f" Saved {n:,} rows → {path} (peg $1.00, " + f"{out['date'].iloc[0].date()} → {out['date'].iloc[-1].date()})") + + +def main() -> None: + DATA_DIR.mkdir(parents=True, exist_ok=True) + print("Fetching BOLD/USD from CoinGecko …") + bold = build_bold_parquet() + print("Rebuilding USDC peg parquet …") + rebuild_usdc_parquet(bold) + + print("\n--- Data summary ---") + for token in ["BOLD", "USDC"]: + df = pd.read_parquet(DATA_DIR / f"{token}_USD.parquet") + dates = pd.to_datetime(df.index, unit="ms", utc=True) + print(f" {token:5s}: {dates.min().date()} → {dates.max().date()} " + f"({len(df):,} rows) close=[{df['close'].min():.4f}, " + f"{df['close'].max():.4f}]") + + +if __name__ == "__main__": + main() diff --git a/scripts/run_final_sims.py b/scripts/run_final_sims.py index 3bd09e4f..31162f3f 100644 --- a/scripts/run_final_sims.py +++ b/scripts/run_final_sims.py @@ -38,7 +38,9 @@ BG = "#162536" TC = "#E6CE97" -# Train/test split around the Oct 10 flash crash +# Train/test split around the Oct 10 flash crash. Pairs whose price data +# starts later (e.g. CoinGecko free tier = last 365 days) can override these +# with per-pair "train"/"test" (start, end) tuples in PAIR_CONFIGS. TRAIN_START = "2025-01-01 00:00:00" TRAIN_END = "2025-10-05 00:00:00" TEST_START = "2025-10-25 00:00:00" @@ -67,6 +69,29 @@ "20m": 20_000_000, }, }, + "btceth": { + # WBTC/WETH pool from the MM artifact (results/mm_noise/meta.json) + "tokens": ["BTC", "ETH"], + "pool_id": "0xa6f548df93de92", + "gas_cost": 1.0, + "fees": 0.0025, + "tvls": { + "5m": 5_000_000, + }, + }, + "boldusdc": { + # Not in the MM artifact — noise model uses the median-pool fallback. + # BOLD price data (CoinGecko) only starts 2025-07-09, hence the + # shortened train window. + "tokens": ["BOLD", "USDC"], + "pool_id": "boldusdc", + "gas_cost": 1.0, + "fees": 0.0005, + "tvls": { + "1m": 1_000_000, + }, + "train": ("2025-07-15 00:00:00", "2025-10-05 00:00:00"), + }, } @@ -98,8 +123,8 @@ def _get_val_roh(t): for entry in to: if isinstance(entry, dict) and "returns_over_hodl" in entry: return float(entry["returns_over_hodl"]) - # If the trial's own objective IS returns_over_hodl, use test_value - if t.get("return_val") == "returns_over_hodl": + # If the trial's own objective matches the requested metric, use test_value + if t.get("return_val") == metric_key: return float(t.get("test_value", float("-inf"))) return float("-inf") @@ -302,7 +327,18 @@ def make_plot_data(all_results, pair_cfg, start, end): return configs, time_series, hodl_values, ref_config -def run_pair(pair_name, pair_cfg, trials, output_dir): +def filter_trials_to_window(trials, train_start, train_end): + """Keep only trials swept over the given train window.""" + start_day = train_start.split(" ")[0] + end_day = train_end.split(" ")[0] + return [ + t for t in trials + if t.get("start_date", "").startswith(start_day) + and end_day in t.get("end_date", "") + ] + + +def run_pair(pair_name, pair_cfg, trials, output_dir, args=None): """Run all TVL variants for a pair, for train and test periods.""" tokens = pair_cfg["tokens"] tokens_set = tuple(sorted(tokens)) @@ -310,9 +346,15 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): print(f" {'/'.join(tokens)} — {pair_name}") print(f"{'='*60}") + # Per-pair window overrides (defaults: the global flash-crash split) + train_start, train_end = pair_cfg.get("train", (TRAIN_START, TRAIN_END)) + test_start, test_end = pair_cfg.get("test", (TEST_START, TEST_END)) + trials = filter_trials_to_window(trials, train_start, train_end) + print(f" {len(trials)} trials swept over {train_start[:10]} → {train_end[:10]}") + # Build noise arrays for both periods - train_noise = build_noise_arrays(pair_cfg, TRAIN_START, TRAIN_END) - test_noise = build_noise_arrays(pair_cfg, TEST_START, TEST_END) + train_noise = build_noise_arrays(pair_cfg, train_start, train_end) + test_noise = build_noise_arrays(pair_cfg, test_start, test_end) train_results = {} test_results = {} @@ -322,7 +364,8 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): print(f"\n --- {full_label} (TVL=${tvl:,.0f}) ---") # Select best params for this TVL - best = select_best_params(trials, tokens_set, tvl) + metric = args.metric if args is not None else "returns_over_hodl" + best = select_best_params(trials, tokens_set, tvl, metric_key=metric) if best is None: print(f" No results found for {tokens_set} TVL={tvl}") continue @@ -341,9 +384,9 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): # Train period forward pass print(f" Running train period...") - train_result = run_forward(pair_cfg, tvl, params, TRAIN_START, TRAIN_END, train_noise) + train_result = run_forward(pair_cfg, tvl, params, train_start, train_end, train_noise) source_hash = _trial_hash(best) - train_fp = build_fingerprint(pair_cfg, tvl, TRAIN_START, TRAIN_END, train_noise) + train_fp = build_fingerprint(pair_cfg, tvl, train_start, train_end, train_noise) export_forward_csvs( train_result, train_fp, output_dir, identifier=f"{pair_name}_{tvl_label}_train", @@ -356,8 +399,8 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): # Test period forward pass print(f" Running test period...") - test_result = run_forward(pair_cfg, tvl, params, TEST_START, TEST_END, test_noise) - test_fp = build_fingerprint(pair_cfg, tvl, TEST_START, TEST_END, test_noise) + test_result = run_forward(pair_cfg, tvl, params, test_start, test_end, test_noise) + test_fp = build_fingerprint(pair_cfg, tvl, test_start, test_end, test_noise) export_forward_csvs( test_result, test_fp, output_dir, identifier=f"{pair_name}_{tvl_label}_test", @@ -381,8 +424,8 @@ def run_pair(pair_name, pair_cfg, trials, output_dir): # Generate plots using existing plotting functions for period_name, results_dict, start, end in [ - ("train", train_results, TRAIN_START, TRAIN_END), - ("test", test_results, TEST_START, TEST_END), + ("train", train_results, train_start, train_end), + ("test", test_results, test_start, test_end), ]: if not results_dict: continue @@ -406,7 +449,7 @@ class PlotArgs: def main(): p = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) - p.add_argument("--pair", choices=["aave", "cow"], default=None) + p.add_argument("--pair", choices=sorted(PAIR_CONFIGS), default=None) p.add_argument("--all", action="store_true") p.add_argument("--method", default="optuna", choices=["optuna", "cma_es"], help="Which sweep method results to use") @@ -421,21 +464,14 @@ def main(): if not args.all and not args.pair: p.error("Specify --pair or --all") - # Load all trials + # Load all trials; each pair filters to its own train window in run_pair print("Loading sweep results...") trials = load_reclamm_results("./results/", metric_key=args.metric) - - # Filter to our training window - trials = [ - t for t in trials - if t.get("start_date", "").startswith("2025-01-01") - and "2025-10-05" in t.get("end_date", "") - ] - print(f" {len(trials)} trials from Jan-Oct 2025 sweep") + print(f" {len(trials)} reCLAMM trials loaded") for pair_name in pairs: pair_cfg = PAIR_CONFIGS[pair_name] - run_pair(pair_name, pair_cfg, trials, args.output_dir) + run_pair(pair_name, pair_cfg, trials, args.output_dir, args=args) print("\nDone.") diff --git a/scripts/run_full_sweep.sh b/scripts/run_full_sweep.sh index 5b11353a..ab4000b0 100755 --- a/scripts/run_full_sweep.sh +++ b/scripts/run_full_sweep.sh @@ -49,6 +49,7 @@ CONFIGS=( "COW ETH 0xd321300ef77067 3.0 0.003 cow_500k 500000" "COW ETH 0xd321300ef77067 3.0 0.003 cow_2m 2000000" "COW ETH 0xd321300ef77067 3.0 0.003 cow_20m 20000000" + "BTC ETH 0xa6f548df93de92 1.0 0.0025 btceth_5m 5000000" ) OUTDIR="results/full_sweep"