Forward test: record forecasts before their outcomes exist.

`snapshot` writes each tracked model's forecast quantiles at the seven
backtest horizons from the latest price to data/forecasts/<origin>.csv. It
refuses stale data (older than two days) and duplicate dates, so snapshots
can't be reconstructed after the fact; committing them dates them.
`forward` scores every recorded forecast whose target date has passed,
reusing the backtest's scoring (now factored out as evaluate.score).

Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou,
powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't
settle. `just weekly` runs update, snapshot and forward.

First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due
2026-10-23.
This commit is contained in:
sam
2026-09-24 03:08:30 -07:00
parent 082bcfbbbc
commit 67d016fe25
7 changed files with 316 additions and 19 deletions
+49 -1
View File
@@ -6,6 +6,8 @@ Command line entry point.
python -m btcmodel backtest --holdout score models on outcomes after DEV_CUTOFF
python -m btcmodel forecast forecast from the latest price
python -m btcmodel ab [NAME ...] run A/B experiments on development data
python -m btcmodel snapshot record tracked models' forecasts for the forward test
python -m btcmodel forward score recorded forecasts whose targets have passed
"""
import argparse
@@ -14,7 +16,7 @@ from pathlib import Path
import numpy as np
import pandas as pd
from . import data, evaluate, experiments, plots
from . import data, evaluate, experiments, forward, plots
from .models import MODELS
FORECAST_REPORT_HORIZONS = (182, 365, 730, 1095, 1460)
@@ -36,6 +38,8 @@ def main() -> None:
commands.add_parser("forecast", help="forecast from the latest price")
ab = commands.add_parser("ab", help="run A/B experiments on development data")
ab.add_argument("names", nargs="*", help="experiments to run (default: all)")
commands.add_parser("snapshot", help="record forecasts for the forward test")
commands.add_parser("forward", help="score recorded forecasts")
args = parser.parse_args()
if args.command == "ab" and (unknown := set(args.names) - set(experiments.EXPERIMENTS)):
parser.error(f"unknown experiments: {', '.join(sorted(unknown))}")
@@ -50,6 +54,10 @@ def main() -> None:
run_forecast(models, args.output)
elif args.command == "ab":
run_ab(args.names or list(experiments.EXPERIMENTS), args.output)
elif args.command == "snapshot":
run_snapshot()
elif args.command == "forward":
run_forward(args.output)
def run_backtest(models, output: Path, holdout: bool) -> None:
@@ -112,6 +120,46 @@ def run_ab(names: list[str], output: Path) -> None:
(output / "ab" / "verdicts.txt").write_text(table + "\n")
def run_snapshot() -> None:
prices = data.load_prices()
path = forward.snapshot(prices)
recorded = [(m, f) for m, f in forward.load_snapshots() if f.origin == prices.index[-1]]
medians = pd.DataFrame(
{m: [plots.price_formatter(p) for p in np.exp(f.quantile(0.5))] for m, f in recorded},
index=[evaluate.horizon_label(h) for h in evaluate.HORIZONS],
).T
print(
f"recorded {len(recorded)} models' forecasts from {prices.index[-1]:%Y-%m-%d} in {path}\n"
)
print("medians:\n" + medians.to_string())
def run_forward(output: Path) -> None:
prices = data.load_prices()
scores = forward.score_snapshots(prices)
due = forward.next_due(prices)
if scores.empty:
when = f"; the first is due {due:%Y-%m-%d}" if due is not None else ""
print(f"no recorded forecast has reached its target date yet{when}")
return
out = output / "forward"
out.mkdir(parents=True, exist_ok=True)
summary = evaluate.summarize(scores, step_days=forward.SNAPSHOT_STEP_DAYS)
report = (
f"forward test: {scores.origin.nunique()} snapshots from {scores.origin.min():%Y-%m-%d}, "
f"outcomes through {prices.index[-1]:%Y-%m-%d}\n\n" + evaluate.format_summary(summary)
)
if due is not None:
report += f"\n\nnext outcome due {due:%Y-%m-%d}"
print(report)
(out / "report.txt").write_text(report + "\n")
scores.to_csv(out / "scores.csv", index=False)
summary.to_csv(out / "summary.csv", index=False)
plots.skill_chart(summary, out / "skill.png", "forward test: skill by horizon")
plots.calibration_chart(summary, out / "calibration.png", "forward test: interval coverage")
print(f"\nwrote {out}/")
def run_forecast(models, output: Path) -> None:
prices = data.load_prices()
horizons = np.arange(1, max(FORECAST_REPORT_HORIZONS) + 1)