Files

170 lines
6.5 KiB
Python
Raw Permalink Normal View History

"""
A/B tests of individual ideas, mostly salvaged from the 2024 model's branches.
Each experiment states its hypothesis and pairs a control with variants that
change one component. Experiments are written down before they are run, and
every result is reported, including the failures; with a few dozen
comparisons over a handful of independent windows, some "wins" will be luck.
Verdicts use one rule, fixed in advance (see `verdict`), and anything that
passes still has to hold up on the holdout.
"""
from dataclasses import dataclass
import pandas as pd
from .evaluate import backtest, summarize
from .models import MODELS
from .models.base import Composite
from .models.drift import (
CycleDrift,
PowerLawDrift,
PowerLawScaledCycleDrift,
ShrunkDrift,
TrailingMeanDrift,
)
from .models.shape import Empirical, StudentT
from .models.volatility import CycleVol, EwmaVol, ReversionVol, TrailingVol, TrendReversionVol
@dataclass(frozen=True)
class Experiment:
name: str
hypothesis: str
source: str # where in the old history the idea came from
control: Composite
variants: tuple[Composite, ...]
@property
def models(self) -> list[Composite]:
return [self.control, *self.variants]
def run(experiment: Experiment, prices: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
scores = backtest(experiment.models, prices)
return scores, summarize(scores, baseline=experiment.control.name)
def verdict(variant_summary: pd.DataFrame) -> str:
"""
Decide from skill vs the control, per horizon, with 90% intervals:
- "worse" if the interval is entirely below zero at any horizon;
- "better" if the interval is entirely above zero at two or more horizons
and the point estimate is non-negative at every horizon;
- "inconclusive" otherwise.
"""
if (variant_summary["skill_hi"] < 0).any():
return "worse"
if (variant_summary["skill_lo"] > 0).sum() >= 2 and (variant_summary["skill"] >= 0).all():
return "better"
return "inconclusive"
# Idea labels (D1, V2, ...) refer to docs/2024-ideas.md.
CYCLE = MODELS["cycle"]
DRIFT_RW = MODELS["drift_rw"]
POWERLAW = MODELS["powerlaw"]
# Volatility and shape experiments use drift_rw as the control: the zero-drift
# random walk is biased low at long horizons, so anything that merely widened
# its intervals would look like an improvement.
_VOL = dict(drift=TrailingMeanDrift())
EXPERIMENTS: dict[str, Experiment] = {
e.name: e
for e in (
Experiment(
"shrink-cycle",
"the cycle drift is overfit; pulling it toward zero improves it",
"D2: damping constants throughout the 2024 model",
CYCLE,
tuple(
Composite(f"cycle_x{f}", ShrunkDrift(CycleDrift(), f), TrailingVol())
for f in (0.25, 0.5, 0.75)
),
),
Experiment(
"diminishing-returns",
"each cycle grows less than the last; a power-law trend captures that",
"D3: initial commit, old/backtests-trend-enhancement-1",
DRIFT_RW,
(
Composite("powerlaw", PowerLawDrift(), TrailingVol()),
Composite("powerlaw_revert", PowerLawDrift(revert=True), TrailingVol()),
Composite("cycle_on_powerlaw", PowerLawScaledCycleDrift(), TrailingVol()),
),
),
Experiment(
"vol-window",
"recent volatility predicts future volatility better than a flat 365-day window",
"V1: 'Add improved vol calculation'",
DRIFT_RW,
(
Composite("ewma_blend", volatility=EwmaVol(), **_VOL),
Composite("ewma_90", volatility=EwmaVol(spans=(90,), weights=(1.0,)), **_VOL),
Composite("trailing_730", volatility=TrailingVol(730), **_VOL),
),
),
Experiment(
"vol-reversion",
"volatility shocks fade toward a long-run level, which itself is falling",
"V2, V4, C2: adaptive windows, market maturity, horizon multipliers",
DRIFT_RW,
(
Composite("revert_level", volatility=ReversionVol(), **_VOL),
Composite("revert_trend", volatility=ReversionVol(long_run="trend"), **_VOL),
),
),
Experiment(
"cycle-vol",
"volatility depends on position in the halving cycle",
"V3: old/backtest-vol-cycle",
DRIFT_RW,
(Composite("cycle_vol", volatility=CycleVol(), **_VOL),),
),
Experiment(
"tails",
"log returns are heavier-tailed than normal, even at long horizons",
"S1: skewed innovations in tuning-1",
DRIFT_RW,
(
Composite("student_t4", volatility=TrailingVol(), shape=StudentT(4), **_VOL),
Composite("empirical", volatility=TrailingVol(), shape=Empirical(), **_VOL),
),
),
Experiment(
"cycle-phase",
"cycles align better by fraction elapsed than by days since the halving",
"D1: the 2024 model stretched every cycle to 1460 days",
CYCLE,
(Composite("cycle_fraction", CycleDrift(phase="fraction"), TrailingVol()),),
),
# Round 2. Before running these, the holdout candidates were fixed as:
# the three original models, powerlaw, and any variant below that is
# "better" than its control.
Experiment(
"powerlaw-ou",
"deviations from the power-law trend fade, so long-horizon uncertainty is bounded",
"follow-up to diminishing-returns: powerlaw_revert beat drift_rw, but its bands"
" were too wide",
POWERLAW,
(
Composite("powerlaw_ou", PowerLawDrift(revert=True), TrendReversionVol()),
Composite(
"powerlaw_ou_param",
PowerLawDrift(revert=True),
TrendReversionVol(parameter_uncertainty=True),
),
),
),
Experiment(
"cycle-on-powerlaw",
"with the level set by the power law, the cycle's timing adds information",
"D1 on D3. This pair was already compared informally on the same data after"
" round 1, so only the holdout can really settle it",
POWERLAW,
(Composite("cycle_on_powerlaw", PowerLawScaledCycleDrift(), TrailingVol()),),
),
)
}