`snapshot` writes each tracked model's forecast quantiles at the seven backtest horizons from the latest price to data/forecasts/<origin>.csv. It refuses stale data (older than two days) and duplicate dates, so snapshots can't be reconstructed after the fact; committing them dates them. `forward` scores every recorded forecast whose target date has passed, reusing the backtest's scoring (now factored out as evaluate.score). Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou, powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't settle. `just weekly` runs update, snapshot and forward. First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due 2026-10-23.
134 lines
5.1 KiB
Python
134 lines
5.1 KiB
Python
"""
|
|
Walk-forward evaluation.
|
|
|
|
From each origin (every ORIGIN_STEP_DAYS from FIRST_ORIGIN), each model sees the
|
|
data up to that day only and forecasts every horizon. Each forecast whose target
|
|
date has been observed is scored against what happened.
|
|
|
|
Development runs load data only up to DEV_CUTOFF, so outcomes after it cannot
|
|
influence model design. The holdout run scores only targets after DEV_CUTOFF.
|
|
"""
|
|
|
|
import numpy as np
|
|
import pandas as pd
|
|
|
|
from .forecast import Forecast, crps, pit
|
|
from .models import BASELINE
|
|
|
|
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
|
|
FIRST_ORIGIN = pd.Timestamp("2014-01-01")
|
|
ORIGIN_STEP_DAYS = 30
|
|
COVERAGES = (0.5, 0.8, 0.95)
|
|
|
|
|
|
def backtest(
|
|
models, data: pd.DataFrame, horizons=HORIZONS, score_after: pd.Timestamp | None = None
|
|
) -> pd.DataFrame:
|
|
"""One row per (model, origin, horizon) with an observed outcome."""
|
|
log_close = np.log(data["close"])
|
|
last = data.index[-1]
|
|
rows = []
|
|
for origin in pd.date_range(FIRST_ORIGIN, last, freq=f"{ORIGIN_STEP_DAYS}D"):
|
|
targets = origin + pd.to_timedelta(horizons, unit="D")
|
|
scored = targets <= last
|
|
if score_after is not None:
|
|
scored &= targets > score_after
|
|
if not scored.any():
|
|
continue
|
|
history = data.loc[:origin]
|
|
outcome = log_close.loc[targets[scored]].to_numpy()
|
|
for model in models:
|
|
rows.append(score(model.name, model.forecast(history, horizons[scored]), outcome))
|
|
return pd.concat(rows, ignore_index=True)
|
|
|
|
|
|
def score(model: str, f: Forecast, outcome: np.ndarray) -> pd.DataFrame:
|
|
"""Score one forecast against the observed log prices at its horizons."""
|
|
row = {
|
|
"model": model,
|
|
"origin": f.origin,
|
|
"horizon": f.horizons,
|
|
"outcome": outcome,
|
|
"median": f.quantile(0.5),
|
|
"crps": crps(f.log_quantiles, outcome),
|
|
"pit": pit(f.log_quantiles, outcome),
|
|
}
|
|
for c in COVERAGES:
|
|
lo, hi = f.interval(c)
|
|
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
|
|
return pd.DataFrame(row)
|
|
|
|
|
|
def summarize(
|
|
scores: pd.DataFrame,
|
|
baseline: str = BASELINE,
|
|
step_days: int = ORIGIN_STEP_DAYS,
|
|
n_boot: int = 2000,
|
|
seed: int = 0,
|
|
) -> pd.DataFrame:
|
|
"""
|
|
Per model and horizon: mean CRPS, skill relative to `baseline`, and coverage.
|
|
|
|
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
|
|
with a 90% block bootstrap interval over origins spaced `step_days` apart. Forecasts from
|
|
nearby origins overlap heavily, so `windows` (the span covered divided by
|
|
the horizon) is the honest count of independent outcomes. Treat intervals
|
|
with fewer than ~5 windows as optimistic.
|
|
"""
|
|
rng = np.random.default_rng(seed)
|
|
rows = []
|
|
for horizon, at_h in scores.groupby("horizon"):
|
|
base = at_h[at_h.model == baseline].set_index("origin")["crps"].sort_index()
|
|
span = (base.index[-1] - base.index[0]).days + horizon
|
|
block = max(1, min(int(np.ceil(horizon / step_days)), len(base) // 2))
|
|
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
|
|
for model, g in at_h.groupby("model", sort=False):
|
|
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()
|
|
boot = 1 - m[boot_index].mean(axis=1) / base.to_numpy()[boot_index].mean(axis=1)
|
|
rows.append(
|
|
{
|
|
"model": model,
|
|
"horizon": horizon,
|
|
"forecasts": len(g),
|
|
"windows": span / horizon,
|
|
"crps": g["crps"].mean(),
|
|
"skill": 1 - m.mean() / base.mean(),
|
|
"skill_lo": np.quantile(boot, 0.05),
|
|
"skill_hi": np.quantile(boot, 0.95),
|
|
**{f"cov{c:.0%}": g[f"in{c:.0%}"].mean() for c in COVERAGES},
|
|
"mean_pit": g["pit"].mean(),
|
|
}
|
|
)
|
|
return pd.DataFrame(rows)
|
|
|
|
|
|
def _block_bootstrap_indices(n, block, n_boot, rng) -> np.ndarray:
|
|
"""Circular block bootstrap, so the first and last origins aren't under-sampled."""
|
|
n_blocks = int(np.ceil(n / block))
|
|
starts = rng.integers(0, n, size=(n_boot, n_blocks))
|
|
return ((starts[:, :, None] + np.arange(block)) % n).reshape(n_boot, -1)[:, :n]
|
|
|
|
|
|
def format_summary(summary: pd.DataFrame, baseline: str = BASELINE) -> str:
|
|
table = pd.DataFrame(
|
|
{
|
|
"model": summary["model"],
|
|
"horizon": summary["horizon"].map(horizon_label),
|
|
"windows": summary["windows"].map("{:.1f}".format),
|
|
"crps": summary["crps"].map("{:.3f}".format),
|
|
f"skill vs {baseline} [90%]": [
|
|
f"{s:+.0%} [{lo:+.0%}, {hi:+.0%}]"
|
|
for s, lo, hi in zip(summary.skill, summary.skill_lo, summary.skill_hi, strict=True)
|
|
],
|
|
**{f"in {c:.0%}": summary[f"cov{c:.0%}"].map("{:.0%}".format) for c in COVERAGES},
|
|
"mean pit": summary["mean_pit"].map("{:.2f}".format),
|
|
}
|
|
)
|
|
return table.to_string(index=False)
|
|
|
|
|
|
def horizon_label(days: int) -> str:
|
|
if days < 365:
|
|
return f"{round(days / 30.4)}mo"
|
|
return f"{round(days / 365)}y"
|