Forward test: record forecasts before their outcomes exist.
`snapshot` writes each tracked model's forecast quantiles at the seven backtest horizons from the latest price to data/forecasts/<origin>.csv. It refuses stale data (older than two days) and duplicate dates, so snapshots can't be reconstructed after the fact; committing them dates them. `forward` scores every recorded forecast whose target date has passed, reusing the backtest's scoring (now factored out as evaluate.score). Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou, powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't settle. `just weekly` runs update, snapshot and forward. First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due 2026-10-23.
This commit is contained in:
+26
-18
@@ -12,7 +12,7 @@ influence model design. The holdout run scores only targets after DEV_CUTOFF.
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .forecast import crps, pit
|
||||
from .forecast import Forecast, crps, pit
|
||||
from .models import BASELINE
|
||||
|
||||
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
|
||||
@@ -38,31 +38,39 @@ def backtest(
|
||||
history = data.loc[:origin]
|
||||
outcome = log_close.loc[targets[scored]].to_numpy()
|
||||
for model in models:
|
||||
f = model.forecast(history, horizons[scored])
|
||||
row = {
|
||||
"model": model.name,
|
||||
"origin": origin,
|
||||
"horizon": f.horizons,
|
||||
"outcome": outcome,
|
||||
"median": f.quantile(0.5),
|
||||
"crps": crps(f.log_quantiles, outcome),
|
||||
"pit": pit(f.log_quantiles, outcome),
|
||||
}
|
||||
for c in COVERAGES:
|
||||
lo, hi = f.interval(c)
|
||||
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
|
||||
rows.append(pd.DataFrame(row))
|
||||
rows.append(score(model.name, model.forecast(history, horizons[scored]), outcome))
|
||||
return pd.concat(rows, ignore_index=True)
|
||||
|
||||
|
||||
def score(model: str, f: Forecast, outcome: np.ndarray) -> pd.DataFrame:
|
||||
"""Score one forecast against the observed log prices at its horizons."""
|
||||
row = {
|
||||
"model": model,
|
||||
"origin": f.origin,
|
||||
"horizon": f.horizons,
|
||||
"outcome": outcome,
|
||||
"median": f.quantile(0.5),
|
||||
"crps": crps(f.log_quantiles, outcome),
|
||||
"pit": pit(f.log_quantiles, outcome),
|
||||
}
|
||||
for c in COVERAGES:
|
||||
lo, hi = f.interval(c)
|
||||
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
|
||||
return pd.DataFrame(row)
|
||||
|
||||
|
||||
def summarize(
|
||||
scores: pd.DataFrame, baseline: str = BASELINE, n_boot: int = 2000, seed: int = 0
|
||||
scores: pd.DataFrame,
|
||||
baseline: str = BASELINE,
|
||||
step_days: int = ORIGIN_STEP_DAYS,
|
||||
n_boot: int = 2000,
|
||||
seed: int = 0,
|
||||
) -> pd.DataFrame:
|
||||
"""
|
||||
Per model and horizon: mean CRPS, skill relative to `baseline`, and coverage.
|
||||
|
||||
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
|
||||
with a 90% moving-block bootstrap interval over origins. Forecasts from
|
||||
with a 90% block bootstrap interval over origins spaced `step_days` apart. Forecasts from
|
||||
nearby origins overlap heavily, so `windows` (the span covered divided by
|
||||
the horizon) is the honest count of independent outcomes. Treat intervals
|
||||
with fewer than ~5 windows as optimistic.
|
||||
@@ -72,7 +80,7 @@ def summarize(
|
||||
for horizon, at_h in scores.groupby("horizon"):
|
||||
base = at_h[at_h.model == baseline].set_index("origin")["crps"].sort_index()
|
||||
span = (base.index[-1] - base.index[0]).days + horizon
|
||||
block = max(1, min(int(np.ceil(horizon / ORIGIN_STEP_DAYS)), len(base) // 2))
|
||||
block = max(1, min(int(np.ceil(horizon / step_days)), len(base) // 2))
|
||||
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
|
||||
for model, g in at_h.groupby("model", sort=False):
|
||||
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()
|
||||
|
||||
Reference in New Issue
Block a user