Forward test: record forecasts before their outcomes exist.

`snapshot` writes each tracked model's forecast quantiles at the seven
backtest horizons from the latest price to data/forecasts/<origin>.csv. It
refuses stale data (older than two days) and duplicate dates, so snapshots
can't be reconstructed after the fact; committing them dates them.
`forward` scores every recorded forecast whose target date has passed,
reusing the backtest's scoring (now factored out as evaluate.score).

Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou,
powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't
settle. `just weekly` runs update, snapshot and forward.

First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due
2026-10-23.
This commit is contained in:
sam
2026-09-24 03:08:30 -07:00
parent 082bcfbbbc
commit 67d016fe25
7 changed files with 316 additions and 19 deletions
+49 -1
View File
@@ -6,6 +6,8 @@ Command line entry point.
python -m btcmodel backtest --holdout score models on outcomes after DEV_CUTOFF
python -m btcmodel forecast forecast from the latest price
python -m btcmodel ab [NAME ...] run A/B experiments on development data
python -m btcmodel snapshot record tracked models' forecasts for the forward test
python -m btcmodel forward score recorded forecasts whose targets have passed
"""
import argparse
@@ -14,7 +16,7 @@ from pathlib import Path
import numpy as np
import pandas as pd
from . import data, evaluate, experiments, plots
from . import data, evaluate, experiments, forward, plots
from .models import MODELS
FORECAST_REPORT_HORIZONS = (182, 365, 730, 1095, 1460)
@@ -36,6 +38,8 @@ def main() -> None:
commands.add_parser("forecast", help="forecast from the latest price")
ab = commands.add_parser("ab", help="run A/B experiments on development data")
ab.add_argument("names", nargs="*", help="experiments to run (default: all)")
commands.add_parser("snapshot", help="record forecasts for the forward test")
commands.add_parser("forward", help="score recorded forecasts")
args = parser.parse_args()
if args.command == "ab" and (unknown := set(args.names) - set(experiments.EXPERIMENTS)):
parser.error(f"unknown experiments: {', '.join(sorted(unknown))}")
@@ -50,6 +54,10 @@ def main() -> None:
run_forecast(models, args.output)
elif args.command == "ab":
run_ab(args.names or list(experiments.EXPERIMENTS), args.output)
elif args.command == "snapshot":
run_snapshot()
elif args.command == "forward":
run_forward(args.output)
def run_backtest(models, output: Path, holdout: bool) -> None:
@@ -112,6 +120,46 @@ def run_ab(names: list[str], output: Path) -> None:
(output / "ab" / "verdicts.txt").write_text(table + "\n")
def run_snapshot() -> None:
prices = data.load_prices()
path = forward.snapshot(prices)
recorded = [(m, f) for m, f in forward.load_snapshots() if f.origin == prices.index[-1]]
medians = pd.DataFrame(
{m: [plots.price_formatter(p) for p in np.exp(f.quantile(0.5))] for m, f in recorded},
index=[evaluate.horizon_label(h) for h in evaluate.HORIZONS],
).T
print(
f"recorded {len(recorded)} models' forecasts from {prices.index[-1]:%Y-%m-%d} in {path}\n"
)
print("medians:\n" + medians.to_string())
def run_forward(output: Path) -> None:
prices = data.load_prices()
scores = forward.score_snapshots(prices)
due = forward.next_due(prices)
if scores.empty:
when = f"; the first is due {due:%Y-%m-%d}" if due is not None else ""
print(f"no recorded forecast has reached its target date yet{when}")
return
out = output / "forward"
out.mkdir(parents=True, exist_ok=True)
summary = evaluate.summarize(scores, step_days=forward.SNAPSHOT_STEP_DAYS)
report = (
f"forward test: {scores.origin.nunique()} snapshots from {scores.origin.min():%Y-%m-%d}, "
f"outcomes through {prices.index[-1]:%Y-%m-%d}\n\n" + evaluate.format_summary(summary)
)
if due is not None:
report += f"\n\nnext outcome due {due:%Y-%m-%d}"
print(report)
(out / "report.txt").write_text(report + "\n")
scores.to_csv(out / "scores.csv", index=False)
summary.to_csv(out / "summary.csv", index=False)
plots.skill_chart(summary, out / "skill.png", "forward test: skill by horizon")
plots.calibration_chart(summary, out / "calibration.png", "forward test: interval coverage")
print(f"\nwrote {out}/")
def run_forecast(models, output: Path) -> None:
prices = data.load_prices()
horizons = np.arange(1, max(FORECAST_REPORT_HORIZONS) + 1)
+26 -18
View File
@@ -12,7 +12,7 @@ influence model design. The holdout run scores only targets after DEV_CUTOFF.
import numpy as np
import pandas as pd
from .forecast import crps, pit
from .forecast import Forecast, crps, pit
from .models import BASELINE
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
@@ -38,31 +38,39 @@ def backtest(
history = data.loc[:origin]
outcome = log_close.loc[targets[scored]].to_numpy()
for model in models:
f = model.forecast(history, horizons[scored])
row = {
"model": model.name,
"origin": origin,
"horizon": f.horizons,
"outcome": outcome,
"median": f.quantile(0.5),
"crps": crps(f.log_quantiles, outcome),
"pit": pit(f.log_quantiles, outcome),
}
for c in COVERAGES:
lo, hi = f.interval(c)
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
rows.append(pd.DataFrame(row))
rows.append(score(model.name, model.forecast(history, horizons[scored]), outcome))
return pd.concat(rows, ignore_index=True)
def score(model: str, f: Forecast, outcome: np.ndarray) -> pd.DataFrame:
"""Score one forecast against the observed log prices at its horizons."""
row = {
"model": model,
"origin": f.origin,
"horizon": f.horizons,
"outcome": outcome,
"median": f.quantile(0.5),
"crps": crps(f.log_quantiles, outcome),
"pit": pit(f.log_quantiles, outcome),
}
for c in COVERAGES:
lo, hi = f.interval(c)
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
return pd.DataFrame(row)
def summarize(
scores: pd.DataFrame, baseline: str = BASELINE, n_boot: int = 2000, seed: int = 0
scores: pd.DataFrame,
baseline: str = BASELINE,
step_days: int = ORIGIN_STEP_DAYS,
n_boot: int = 2000,
seed: int = 0,
) -> pd.DataFrame:
"""
Per model and horizon: mean CRPS, skill relative to `baseline`, and coverage.
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
with a 90% moving-block bootstrap interval over origins. Forecasts from
with a 90% block bootstrap interval over origins spaced `step_days` apart. Forecasts from
nearby origins overlap heavily, so `windows` (the span covered divided by
the horizon) is the honest count of independent outcomes. Treat intervals
with fewer than ~5 windows as optimistic.
@@ -72,7 +80,7 @@ def summarize(
for horizon, at_h in scores.groupby("horizon"):
base = at_h[at_h.model == baseline].set_index("origin")["crps"].sort_index()
span = (base.index[-1] - base.index[0]).days + horizon
block = max(1, min(int(np.ceil(horizon / ORIGIN_STEP_DAYS)), len(base) // 2))
block = max(1, min(int(np.ceil(horizon / step_days)), len(base) // 2))
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
for model, g in at_h.groupby("model", sort=False):
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()
+104
View File
@@ -0,0 +1,104 @@
"""
Forward test: forecasts recorded before their outcomes exist.
`snapshot` stores every tracked model's forecast from the latest price in
data/forecasts/<origin>.csv. The files are committed, so version control shows
each one was written before its outcome was known. `score_snapshots` grades
every recorded forecast whose target date has passed, exactly as the backtest
does.
Snapshots store the forecast quantiles themselves, so model code can change
without rewriting the record. If a model's definition changes, give it a new
name rather than reusing the old one.
"""
import datetime as dt
from pathlib import Path
import numpy as np
import pandas as pd
from .data import DATA_DIR
from .evaluate import HORIZONS, score
from .experiments import EXPERIMENTS
from .forecast import LEVELS, Forecast
from .models import MODELS
FORECASTS_DIR = DATA_DIR / "forecasts"
# A snapshot must be made from fresh data, not reconstructed after the fact.
MAX_DATA_AGE_DAYS = 2
# Intended cadence (weekly), used to size bootstrap blocks when scoring.
SNAPSHOT_STEP_DAYS = 7
_VARIANTS = {m.name: m for e in EXPERIMENTS.values() for m in e.models}
# The registered models, plus candidates that could only be settled on new data.
TRACKED = [
*MODELS.values(),
*(_VARIANTS[n] for n in ("powerlaw_ou", "powerlaw_ou_param", "cycle_on_powerlaw")),
]
# Named by percentile: p0.5, p1.5, ..., p99.5 (the grid has no exact median).
_QUANTILE_COLUMNS = [f"p{100 * level:.1f}" for level in LEVELS]
def snapshot(
prices: pd.DataFrame,
models=TRACKED,
directory: Path = FORECASTS_DIR,
today: pd.Timestamp | None = None,
) -> Path:
"""Record each model's forecast from the last price in `prices`."""
origin = prices.index[-1]
if today is None:
today = pd.Timestamp(dt.datetime.now(dt.UTC).date())
if (today - origin).days > MAX_DATA_AGE_DAYS:
raise ValueError(
f"latest price is from {origin:%Y-%m-%d}; update the data first "
"(forecasts must be recorded in real time)"
)
path = directory / f"{origin:%Y-%m-%d}.csv"
if path.exists():
raise FileExistsError(f"{path} already exists")
frames = []
for model in models:
f = model.forecast(prices, HORIZONS)
frame = pd.DataFrame(f.log_quantiles, columns=_QUANTILE_COLUMNS)
frame.insert(0, "model", model.name)
frame.insert(1, "horizon", f.horizons)
frame.insert(2, "target", f.dates.strftime("%Y-%m-%d"))
frames.append(frame)
directory.mkdir(parents=True, exist_ok=True)
pd.concat(frames).to_csv(path, index=False, float_format="%.6f")
return path
def load_snapshots(directory: Path = FORECASTS_DIR) -> list[tuple[str, Forecast]]:
"""Every recorded forecast, as (model name, Forecast)."""
forecasts = []
for path in sorted(directory.glob("*.csv")):
origin = pd.Timestamp(path.stem)
for model, g in pd.read_csv(path).groupby("model", sort=False):
quantiles = g[_QUANTILE_COLUMNS].to_numpy()
forecasts.append((model, Forecast(origin, g["horizon"].to_numpy(), quantiles)))
return forecasts
def score_snapshots(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.DataFrame:
"""Score every recorded forecast horizon whose target date is in `prices`."""
log_close = np.log(prices["close"])
rows = []
for model, f in load_snapshots(directory):
due = f.dates <= prices.index[-1]
if not due.any():
continue
due_forecast = Forecast(f.origin, f.horizons[due], f.log_quantiles[due])
outcome = log_close.loc[due_forecast.dates].to_numpy()
rows.append(score(model, due_forecast, outcome))
return pd.concat(rows, ignore_index=True) if rows else pd.DataFrame()
def next_due(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.Timestamp | None:
"""The earliest target date not yet observed, if any."""
pending = [d for _, f in load_snapshots(directory) for d in f.dates if d > prices.index[-1]]
return min(pending) if pending else None