Forward test: record forecasts before their outcomes exist.
`snapshot` writes each tracked model's forecast quantiles at the seven backtest horizons from the latest price to data/forecasts/<origin>.csv. It refuses stale data (older than two days) and duplicate dates, so snapshots can't be reconstructed after the fact; committing them dates them. `forward` scores every recorded forecast whose target date has passed, reusing the backtest's scoring (now factored out as evaluate.score). Tracked: random_walk, drift_rw, cycle, powerlaw, plus powerlaw_ou, powerlaw_ou_param and cycle_on_powerlaw, which development data couldn't settle. `just weekly` runs update, snapshot and forward. First snapshot: 2026-09-23 (BTC $84.4K). The first outcomes are due 2026-10-23.
This commit is contained in:
+49
-1
@@ -6,6 +6,8 @@ Command line entry point.
|
||||
python -m btcmodel backtest --holdout score models on outcomes after DEV_CUTOFF
|
||||
python -m btcmodel forecast forecast from the latest price
|
||||
python -m btcmodel ab [NAME ...] run A/B experiments on development data
|
||||
python -m btcmodel snapshot record tracked models' forecasts for the forward test
|
||||
python -m btcmodel forward score recorded forecasts whose targets have passed
|
||||
"""
|
||||
|
||||
import argparse
|
||||
@@ -14,7 +16,7 @@ from pathlib import Path
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from . import data, evaluate, experiments, plots
|
||||
from . import data, evaluate, experiments, forward, plots
|
||||
from .models import MODELS
|
||||
|
||||
FORECAST_REPORT_HORIZONS = (182, 365, 730, 1095, 1460)
|
||||
@@ -36,6 +38,8 @@ def main() -> None:
|
||||
commands.add_parser("forecast", help="forecast from the latest price")
|
||||
ab = commands.add_parser("ab", help="run A/B experiments on development data")
|
||||
ab.add_argument("names", nargs="*", help="experiments to run (default: all)")
|
||||
commands.add_parser("snapshot", help="record forecasts for the forward test")
|
||||
commands.add_parser("forward", help="score recorded forecasts")
|
||||
args = parser.parse_args()
|
||||
if args.command == "ab" and (unknown := set(args.names) - set(experiments.EXPERIMENTS)):
|
||||
parser.error(f"unknown experiments: {', '.join(sorted(unknown))}")
|
||||
@@ -50,6 +54,10 @@ def main() -> None:
|
||||
run_forecast(models, args.output)
|
||||
elif args.command == "ab":
|
||||
run_ab(args.names or list(experiments.EXPERIMENTS), args.output)
|
||||
elif args.command == "snapshot":
|
||||
run_snapshot()
|
||||
elif args.command == "forward":
|
||||
run_forward(args.output)
|
||||
|
||||
|
||||
def run_backtest(models, output: Path, holdout: bool) -> None:
|
||||
@@ -112,6 +120,46 @@ def run_ab(names: list[str], output: Path) -> None:
|
||||
(output / "ab" / "verdicts.txt").write_text(table + "\n")
|
||||
|
||||
|
||||
def run_snapshot() -> None:
|
||||
prices = data.load_prices()
|
||||
path = forward.snapshot(prices)
|
||||
recorded = [(m, f) for m, f in forward.load_snapshots() if f.origin == prices.index[-1]]
|
||||
medians = pd.DataFrame(
|
||||
{m: [plots.price_formatter(p) for p in np.exp(f.quantile(0.5))] for m, f in recorded},
|
||||
index=[evaluate.horizon_label(h) for h in evaluate.HORIZONS],
|
||||
).T
|
||||
print(
|
||||
f"recorded {len(recorded)} models' forecasts from {prices.index[-1]:%Y-%m-%d} in {path}\n"
|
||||
)
|
||||
print("medians:\n" + medians.to_string())
|
||||
|
||||
|
||||
def run_forward(output: Path) -> None:
|
||||
prices = data.load_prices()
|
||||
scores = forward.score_snapshots(prices)
|
||||
due = forward.next_due(prices)
|
||||
if scores.empty:
|
||||
when = f"; the first is due {due:%Y-%m-%d}" if due is not None else ""
|
||||
print(f"no recorded forecast has reached its target date yet{when}")
|
||||
return
|
||||
out = output / "forward"
|
||||
out.mkdir(parents=True, exist_ok=True)
|
||||
summary = evaluate.summarize(scores, step_days=forward.SNAPSHOT_STEP_DAYS)
|
||||
report = (
|
||||
f"forward test: {scores.origin.nunique()} snapshots from {scores.origin.min():%Y-%m-%d}, "
|
||||
f"outcomes through {prices.index[-1]:%Y-%m-%d}\n\n" + evaluate.format_summary(summary)
|
||||
)
|
||||
if due is not None:
|
||||
report += f"\n\nnext outcome due {due:%Y-%m-%d}"
|
||||
print(report)
|
||||
(out / "report.txt").write_text(report + "\n")
|
||||
scores.to_csv(out / "scores.csv", index=False)
|
||||
summary.to_csv(out / "summary.csv", index=False)
|
||||
plots.skill_chart(summary, out / "skill.png", "forward test: skill by horizon")
|
||||
plots.calibration_chart(summary, out / "calibration.png", "forward test: interval coverage")
|
||||
print(f"\nwrote {out}/")
|
||||
|
||||
|
||||
def run_forecast(models, output: Path) -> None:
|
||||
prices = data.load_prices()
|
||||
horizons = np.arange(1, max(FORECAST_REPORT_HORIZONS) + 1)
|
||||
|
||||
+26
-18
@@ -12,7 +12,7 @@ influence model design. The holdout run scores only targets after DEV_CUTOFF.
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .forecast import crps, pit
|
||||
from .forecast import Forecast, crps, pit
|
||||
from .models import BASELINE
|
||||
|
||||
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
|
||||
@@ -38,31 +38,39 @@ def backtest(
|
||||
history = data.loc[:origin]
|
||||
outcome = log_close.loc[targets[scored]].to_numpy()
|
||||
for model in models:
|
||||
f = model.forecast(history, horizons[scored])
|
||||
row = {
|
||||
"model": model.name,
|
||||
"origin": origin,
|
||||
"horizon": f.horizons,
|
||||
"outcome": outcome,
|
||||
"median": f.quantile(0.5),
|
||||
"crps": crps(f.log_quantiles, outcome),
|
||||
"pit": pit(f.log_quantiles, outcome),
|
||||
}
|
||||
for c in COVERAGES:
|
||||
lo, hi = f.interval(c)
|
||||
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
|
||||
rows.append(pd.DataFrame(row))
|
||||
rows.append(score(model.name, model.forecast(history, horizons[scored]), outcome))
|
||||
return pd.concat(rows, ignore_index=True)
|
||||
|
||||
|
||||
def score(model: str, f: Forecast, outcome: np.ndarray) -> pd.DataFrame:
|
||||
"""Score one forecast against the observed log prices at its horizons."""
|
||||
row = {
|
||||
"model": model,
|
||||
"origin": f.origin,
|
||||
"horizon": f.horizons,
|
||||
"outcome": outcome,
|
||||
"median": f.quantile(0.5),
|
||||
"crps": crps(f.log_quantiles, outcome),
|
||||
"pit": pit(f.log_quantiles, outcome),
|
||||
}
|
||||
for c in COVERAGES:
|
||||
lo, hi = f.interval(c)
|
||||
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
|
||||
return pd.DataFrame(row)
|
||||
|
||||
|
||||
def summarize(
|
||||
scores: pd.DataFrame, baseline: str = BASELINE, n_boot: int = 2000, seed: int = 0
|
||||
scores: pd.DataFrame,
|
||||
baseline: str = BASELINE,
|
||||
step_days: int = ORIGIN_STEP_DAYS,
|
||||
n_boot: int = 2000,
|
||||
seed: int = 0,
|
||||
) -> pd.DataFrame:
|
||||
"""
|
||||
Per model and horizon: mean CRPS, skill relative to `baseline`, and coverage.
|
||||
|
||||
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
|
||||
with a 90% moving-block bootstrap interval over origins. Forecasts from
|
||||
with a 90% block bootstrap interval over origins spaced `step_days` apart. Forecasts from
|
||||
nearby origins overlap heavily, so `windows` (the span covered divided by
|
||||
the horizon) is the honest count of independent outcomes. Treat intervals
|
||||
with fewer than ~5 windows as optimistic.
|
||||
@@ -72,7 +80,7 @@ def summarize(
|
||||
for horizon, at_h in scores.groupby("horizon"):
|
||||
base = at_h[at_h.model == baseline].set_index("origin")["crps"].sort_index()
|
||||
span = (base.index[-1] - base.index[0]).days + horizon
|
||||
block = max(1, min(int(np.ceil(horizon / ORIGIN_STEP_DAYS)), len(base) // 2))
|
||||
block = max(1, min(int(np.ceil(horizon / step_days)), len(base) // 2))
|
||||
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
|
||||
for model, g in at_h.groupby("model", sort=False):
|
||||
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
"""
|
||||
Forward test: forecasts recorded before their outcomes exist.
|
||||
|
||||
`snapshot` stores every tracked model's forecast from the latest price in
|
||||
data/forecasts/<origin>.csv. The files are committed, so version control shows
|
||||
each one was written before its outcome was known. `score_snapshots` grades
|
||||
every recorded forecast whose target date has passed, exactly as the backtest
|
||||
does.
|
||||
|
||||
Snapshots store the forecast quantiles themselves, so model code can change
|
||||
without rewriting the record. If a model's definition changes, give it a new
|
||||
name rather than reusing the old one.
|
||||
"""
|
||||
|
||||
import datetime as dt
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from .data import DATA_DIR
|
||||
from .evaluate import HORIZONS, score
|
||||
from .experiments import EXPERIMENTS
|
||||
from .forecast import LEVELS, Forecast
|
||||
from .models import MODELS
|
||||
|
||||
FORECASTS_DIR = DATA_DIR / "forecasts"
|
||||
# A snapshot must be made from fresh data, not reconstructed after the fact.
|
||||
MAX_DATA_AGE_DAYS = 2
|
||||
# Intended cadence (weekly), used to size bootstrap blocks when scoring.
|
||||
SNAPSHOT_STEP_DAYS = 7
|
||||
|
||||
_VARIANTS = {m.name: m for e in EXPERIMENTS.values() for m in e.models}
|
||||
# The registered models, plus candidates that could only be settled on new data.
|
||||
TRACKED = [
|
||||
*MODELS.values(),
|
||||
*(_VARIANTS[n] for n in ("powerlaw_ou", "powerlaw_ou_param", "cycle_on_powerlaw")),
|
||||
]
|
||||
|
||||
# Named by percentile: p0.5, p1.5, ..., p99.5 (the grid has no exact median).
|
||||
_QUANTILE_COLUMNS = [f"p{100 * level:.1f}" for level in LEVELS]
|
||||
|
||||
|
||||
def snapshot(
|
||||
prices: pd.DataFrame,
|
||||
models=TRACKED,
|
||||
directory: Path = FORECASTS_DIR,
|
||||
today: pd.Timestamp | None = None,
|
||||
) -> Path:
|
||||
"""Record each model's forecast from the last price in `prices`."""
|
||||
origin = prices.index[-1]
|
||||
if today is None:
|
||||
today = pd.Timestamp(dt.datetime.now(dt.UTC).date())
|
||||
if (today - origin).days > MAX_DATA_AGE_DAYS:
|
||||
raise ValueError(
|
||||
f"latest price is from {origin:%Y-%m-%d}; update the data first "
|
||||
"(forecasts must be recorded in real time)"
|
||||
)
|
||||
path = directory / f"{origin:%Y-%m-%d}.csv"
|
||||
if path.exists():
|
||||
raise FileExistsError(f"{path} already exists")
|
||||
|
||||
frames = []
|
||||
for model in models:
|
||||
f = model.forecast(prices, HORIZONS)
|
||||
frame = pd.DataFrame(f.log_quantiles, columns=_QUANTILE_COLUMNS)
|
||||
frame.insert(0, "model", model.name)
|
||||
frame.insert(1, "horizon", f.horizons)
|
||||
frame.insert(2, "target", f.dates.strftime("%Y-%m-%d"))
|
||||
frames.append(frame)
|
||||
directory.mkdir(parents=True, exist_ok=True)
|
||||
pd.concat(frames).to_csv(path, index=False, float_format="%.6f")
|
||||
return path
|
||||
|
||||
|
||||
def load_snapshots(directory: Path = FORECASTS_DIR) -> list[tuple[str, Forecast]]:
|
||||
"""Every recorded forecast, as (model name, Forecast)."""
|
||||
forecasts = []
|
||||
for path in sorted(directory.glob("*.csv")):
|
||||
origin = pd.Timestamp(path.stem)
|
||||
for model, g in pd.read_csv(path).groupby("model", sort=False):
|
||||
quantiles = g[_QUANTILE_COLUMNS].to_numpy()
|
||||
forecasts.append((model, Forecast(origin, g["horizon"].to_numpy(), quantiles)))
|
||||
return forecasts
|
||||
|
||||
|
||||
def score_snapshots(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.DataFrame:
|
||||
"""Score every recorded forecast horizon whose target date is in `prices`."""
|
||||
log_close = np.log(prices["close"])
|
||||
rows = []
|
||||
for model, f in load_snapshots(directory):
|
||||
due = f.dates <= prices.index[-1]
|
||||
if not due.any():
|
||||
continue
|
||||
due_forecast = Forecast(f.origin, f.horizons[due], f.log_quantiles[due])
|
||||
outcome = log_close.loc[due_forecast.dates].to_numpy()
|
||||
rows.append(score(model, due_forecast, outcome))
|
||||
return pd.concat(rows, ignore_index=True) if rows else pd.DataFrame()
|
||||
|
||||
|
||||
def next_due(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.Timestamp | None:
|
||||
"""The earliest target date not yet observed, if any."""
|
||||
pending = [d for _, f in load_snapshots(directory) for d in f.dates if d > prices.index[-1]]
|
||||
return min(pending) if pending else None
|
||||
Reference in New Issue
Block a user