Forward test: record forecasts before their outcomes exist.

`snapshot` writes each tracked model's forecast quantiles at the seven
backtest horizons from the latest price to data/forecasts/<origin>.csv. It
refuses stale data (older than two days) and duplicate dates, so snapshots
can't be reconstructed after the fact; committing them dates them.
`forward` scores every recorded forecast whose target date has passed,
reusing the backtest's scoring (now factored out as evaluate.score).

Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou,
powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't
settle. `just weekly` runs update, snapshot and forward.

First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due
2026-10-23.
This commit is contained in:
sam
2026-09-24 03:08:30 -07:00
parent 082bcfbbbc
commit 67d016fe25
7 changed files with 316 additions and 19 deletions
+26 -18
View File
@@ -12,7 +12,7 @@ influence model design. The holdout run scores only targets after DEV_CUTOFF.
import numpy as np
import pandas as pd
from .forecast import crps, pit
from .forecast import Forecast, crps, pit
from .models import BASELINE
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
@@ -38,31 +38,39 @@ def backtest(
history = data.loc[:origin]
outcome = log_close.loc[targets[scored]].to_numpy()
for model in models:
f = model.forecast(history, horizons[scored])
row = {
"model": model.name,
"origin": origin,
"horizon": f.horizons,
"outcome": outcome,
"median": f.quantile(0.5),
"crps": crps(f.log_quantiles, outcome),
"pit": pit(f.log_quantiles, outcome),
}
for c in COVERAGES:
lo, hi = f.interval(c)
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
rows.append(pd.DataFrame(row))
rows.append(score(model.name, model.forecast(history, horizons[scored]), outcome))
return pd.concat(rows, ignore_index=True)
def score(model: str, f: Forecast, outcome: np.ndarray) -> pd.DataFrame:
"""Score one forecast against the observed log prices at its horizons."""
row = {
"model": model,
"origin": f.origin,
"horizon": f.horizons,
"outcome": outcome,
"median": f.quantile(0.5),
"crps": crps(f.log_quantiles, outcome),
"pit": pit(f.log_quantiles, outcome),
}
for c in COVERAGES:
lo, hi = f.interval(c)
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
return pd.DataFrame(row)
def summarize(
scores: pd.DataFrame, baseline: str = BASELINE, n_boot: int = 2000, seed: int = 0
scores: pd.DataFrame,
baseline: str = BASELINE,
step_days: int = ORIGIN_STEP_DAYS,
n_boot: int = 2000,
seed: int = 0,
) -> pd.DataFrame:
"""
Per model and horizon: mean CRPS, skill relative to `baseline`, and coverage.
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
with a 90% moving-block bootstrap interval over origins. Forecasts from
with a 90% block bootstrap interval over origins spaced `step_days` apart. Forecasts from
nearby origins overlap heavily, so `windows` (the span covered divided by
the horizon) is the honest count of independent outcomes. Treat intervals
with fewer than ~5 windows as optimistic.
@@ -72,7 +80,7 @@ def summarize(
for horizon, at_h in scores.groupby("horizon"):
base = at_h[at_h.model == baseline].set_index("origin")["crps"].sort_index()
span = (base.index[-1] - base.index[0]).days + horizon
block = max(1, min(int(np.ceil(horizon / ORIGIN_STEP_DAYS)), len(base) // 2))
block = max(1, min(int(np.ceil(horizon / step_days)), len(base) // 2))
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
for model, g in at_h.groupby("model", sort=False):
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()