Add TrendReversionVol: deviations from the power-law trend follow a daily AR(1), so uncertainty levels off, optionally plus trend-parameter uncertainty with an autocorrelation-adjusted effective sample size. Two experiments, run under the unchanged verdict rule: - powerlaw-ou: +21% to +45% vs powerlaw at 2-4 years, but slightly negative point estimates at 1 month make it inconclusive. - cycle-on-powerlaw: inconclusive (+18% at 2 years, negative elsewhere). The holdout (outcomes after 2024-11-26) was scored once, for the four candidates fixed beforehand. powerlaw is the best long-horizon forecast (+45% and +58% vs the random walk at 2 and 3 years); nothing beats the random walk inside a year; cycle fails badly. Results are in the README.
170 lines
6.5 KiB
Python
170 lines
6.5 KiB
Python
"""
|
|
A/B tests of individual ideas, mostly salvaged from the 2024 model's branches.
|
|
|
|
Each experiment states its hypothesis and pairs a control with variants that
|
|
change one component. Experiments are written down before they are run, and
|
|
every result is reported, including the failures; with a few dozen
|
|
comparisons over a handful of independent windows, some "wins" will be luck.
|
|
Verdicts use one rule, fixed in advance (see `verdict`), and anything that
|
|
passes still has to hold up on the holdout.
|
|
"""
|
|
|
|
from dataclasses import dataclass
|
|
|
|
import pandas as pd
|
|
|
|
from .evaluate import backtest, summarize
|
|
from .models import MODELS
|
|
from .models.base import Composite
|
|
from .models.drift import (
|
|
CycleDrift,
|
|
PowerLawDrift,
|
|
PowerLawScaledCycleDrift,
|
|
ShrunkDrift,
|
|
TrailingMeanDrift,
|
|
)
|
|
from .models.shape import Empirical, StudentT
|
|
from .models.volatility import CycleVol, EwmaVol, ReversionVol, TrailingVol, TrendReversionVol
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Experiment:
|
|
name: str
|
|
hypothesis: str
|
|
source: str # where in the old history the idea came from
|
|
control: Composite
|
|
variants: tuple[Composite, ...]
|
|
|
|
@property
|
|
def models(self) -> list[Composite]:
|
|
return [self.control, *self.variants]
|
|
|
|
|
|
def run(experiment: Experiment, prices: pd.DataFrame) -> tuple[pd.DataFrame, pd.DataFrame]:
|
|
scores = backtest(experiment.models, prices)
|
|
return scores, summarize(scores, baseline=experiment.control.name)
|
|
|
|
|
|
def verdict(variant_summary: pd.DataFrame) -> str:
|
|
"""
|
|
Decide from skill vs the control, per horizon, with 90% intervals:
|
|
|
|
- "worse" if the interval is entirely below zero at any horizon;
|
|
- "better" if the interval is entirely above zero at two or more horizons
|
|
and the point estimate is non-negative at every horizon;
|
|
- "inconclusive" otherwise.
|
|
"""
|
|
if (variant_summary["skill_hi"] < 0).any():
|
|
return "worse"
|
|
if (variant_summary["skill_lo"] > 0).sum() >= 2 and (variant_summary["skill"] >= 0).all():
|
|
return "better"
|
|
return "inconclusive"
|
|
|
|
|
|
# Idea labels (D1, V2, ...) refer to docs/2024-ideas.md.
|
|
CYCLE = MODELS["cycle"]
|
|
DRIFT_RW = MODELS["drift_rw"]
|
|
POWERLAW = MODELS["powerlaw"]
|
|
# Volatility and shape experiments use drift_rw as the control: the zero-drift
|
|
# random walk is biased low at long horizons, so anything that merely widened
|
|
# its intervals would look like an improvement.
|
|
_VOL = dict(drift=TrailingMeanDrift())
|
|
|
|
EXPERIMENTS: dict[str, Experiment] = {
|
|
e.name: e
|
|
for e in (
|
|
Experiment(
|
|
"shrink-cycle",
|
|
"the cycle drift is overfit; pulling it toward zero improves it",
|
|
"D2: damping constants throughout the 2024 model",
|
|
CYCLE,
|
|
tuple(
|
|
Composite(f"cycle_x{f}", ShrunkDrift(CycleDrift(), f), TrailingVol())
|
|
for f in (0.25, 0.5, 0.75)
|
|
),
|
|
),
|
|
Experiment(
|
|
"diminishing-returns",
|
|
"each cycle grows less than the last; a power-law trend captures that",
|
|
"D3: initial commit, old/backtests-trend-enhancement-1",
|
|
DRIFT_RW,
|
|
(
|
|
Composite("powerlaw", PowerLawDrift(), TrailingVol()),
|
|
Composite("powerlaw_revert", PowerLawDrift(revert=True), TrailingVol()),
|
|
Composite("cycle_on_powerlaw", PowerLawScaledCycleDrift(), TrailingVol()),
|
|
),
|
|
),
|
|
Experiment(
|
|
"vol-window",
|
|
"recent volatility predicts future volatility better than a flat 365-day window",
|
|
"V1: 'Add improved vol calculation'",
|
|
DRIFT_RW,
|
|
(
|
|
Composite("ewma_blend", volatility=EwmaVol(), **_VOL),
|
|
Composite("ewma_90", volatility=EwmaVol(spans=(90,), weights=(1.0,)), **_VOL),
|
|
Composite("trailing_730", volatility=TrailingVol(730), **_VOL),
|
|
),
|
|
),
|
|
Experiment(
|
|
"vol-reversion",
|
|
"volatility shocks fade toward a long-run level, which itself is falling",
|
|
"V2, V4, C2: adaptive windows, market maturity, horizon multipliers",
|
|
DRIFT_RW,
|
|
(
|
|
Composite("revert_level", volatility=ReversionVol(), **_VOL),
|
|
Composite("revert_trend", volatility=ReversionVol(long_run="trend"), **_VOL),
|
|
),
|
|
),
|
|
Experiment(
|
|
"cycle-vol",
|
|
"volatility depends on position in the halving cycle",
|
|
"V3: old/backtest-vol-cycle",
|
|
DRIFT_RW,
|
|
(Composite("cycle_vol", volatility=CycleVol(), **_VOL),),
|
|
),
|
|
Experiment(
|
|
"tails",
|
|
"log returns are heavier-tailed than normal, even at long horizons",
|
|
"S1: skewed innovations in tuning-1",
|
|
DRIFT_RW,
|
|
(
|
|
Composite("student_t4", volatility=TrailingVol(), shape=StudentT(4), **_VOL),
|
|
Composite("empirical", volatility=TrailingVol(), shape=Empirical(), **_VOL),
|
|
),
|
|
),
|
|
Experiment(
|
|
"cycle-phase",
|
|
"cycles align better by fraction elapsed than by days since the halving",
|
|
"D1: the 2024 model stretched every cycle to 1460 days",
|
|
CYCLE,
|
|
(Composite("cycle_fraction", CycleDrift(phase="fraction"), TrailingVol()),),
|
|
),
|
|
# Round 2. Before running these, the holdout candidates were fixed as:
|
|
# the three original models, powerlaw, and any variant below that is
|
|
# "better" than its control.
|
|
Experiment(
|
|
"powerlaw-ou",
|
|
"deviations from the power-law trend fade, so long-horizon uncertainty is bounded",
|
|
"follow-up to diminishing-returns: powerlaw_revert beat drift_rw, but its bands"
|
|
" were too wide",
|
|
POWERLAW,
|
|
(
|
|
Composite("powerlaw_ou", PowerLawDrift(revert=True), TrendReversionVol()),
|
|
Composite(
|
|
"powerlaw_ou_param",
|
|
PowerLawDrift(revert=True),
|
|
TrendReversionVol(parameter_uncertainty=True),
|
|
),
|
|
),
|
|
),
|
|
Experiment(
|
|
"cycle-on-powerlaw",
|
|
"with the level set by the power law, the cycle's timing adds information",
|
|
"D1 on D3. This pair was already compared informally on the same data after"
|
|
" round 1, so only the holdout can really settle it",
|
|
POWERLAW,
|
|
(Composite("cycle_on_powerlaw", PowerLawScaledCycleDrift(), TrailingVol()),),
|
|
),
|
|
)
|
|
}
|