Rewrite as a probabilistic model with walk-forward evaluation.

Replace the 2024 model (model.py, ~2000 lines) with the btcmodel package, the
baseline for future work:

- Forecasts are quantiles of log price at each horizon, scored with CRPS in a
  walk-forward backtest (origins every 30 days from 2014, horizons 1 month to
  4 years). Skill is relative to a zero-drift random walk, with circular
  block-bootstrap intervals and a count of independent windows.
- Development data stops at 2024-11-26, the last day the 2024 model saw.
  Later outcomes are a holdout, scored only by `backtest --holdout`.
- Models: random_walk, drift_rw, and cycle (the 2024 model's cycle-position
  drift, now kernel-smoothed and recency-weighted). On development data
  nothing beats the random walk with confidence; cycle loses at every horizon.
- Prices: the Investing.com archive moves to data/ (cut at 2024-11-26; its
  last row was intraday) and is extended with Coinbase daily closes by
  `update`.

Also: Nix flake dev shell (Python 3.13, pandas 3), ruff in place of black,
pytest suite, and a rewritten README. NOTES.md is removed as inaccurate, and
poetry is dropped.
This commit is contained in:
sam
2026-09-24 02:19:02 -07:00
parent eefff47070
commit cfc27a38de
24 changed files with 1767 additions and 3124 deletions
+123
View File
@@ -0,0 +1,123 @@
"""
Walk-forward evaluation.
From each origin (every ORIGIN_STEP_DAYS from FIRST_ORIGIN), each model sees the
data up to that day only and forecasts every horizon. Each forecast whose target
date has been observed is scored against what happened.
Development runs load data only up to DEV_CUTOFF, so outcomes after it cannot
influence model design. The holdout run scores only targets after DEV_CUTOFF.
"""
import numpy as np
import pandas as pd
from .forecast import crps, pit
from .models import BASELINE
HORIZONS = np.array([30, 91, 182, 365, 730, 1095, 1460])
FIRST_ORIGIN = pd.Timestamp("2014-01-01")
ORIGIN_STEP_DAYS = 30
COVERAGES = (0.5, 0.8, 0.95)
def backtest(
models, data: pd.DataFrame, horizons=HORIZONS, score_after: pd.Timestamp | None = None
) -> pd.DataFrame:
"""One row per (model, origin, horizon) with an observed outcome."""
log_close = np.log(data["close"])
last = data.index[-1]
rows = []
for origin in pd.date_range(FIRST_ORIGIN, last, freq=f"{ORIGIN_STEP_DAYS}D"):
targets = origin + pd.to_timedelta(horizons, unit="D")
scored = targets <= last
if score_after is not None:
scored &= targets > score_after
if not scored.any():
continue
history = data.loc[:origin]
outcome = log_close.loc[targets[scored]].to_numpy()
for model in models:
f = model.forecast(history, horizons[scored])
row = {
"model": model.name,
"origin": origin,
"horizon": f.horizons,
"outcome": outcome,
"median": f.quantile(0.5),
"crps": crps(f.log_quantiles, outcome),
"pit": pit(f.log_quantiles, outcome),
}
for c in COVERAGES:
lo, hi = f.interval(c)
row[f"in{c:.0%}"] = (lo <= outcome) & (outcome <= hi)
rows.append(pd.DataFrame(row))
return pd.concat(rows, ignore_index=True)
def summarize(scores: pd.DataFrame, n_boot: int = 2000, seed: int = 0) -> pd.DataFrame:
"""
Per model and horizon: mean CRPS, skill relative to the baseline, and coverage.
Skill is 1 - CRPS / baseline CRPS (positive = better than the baseline),
with a 90% moving-block bootstrap interval over origins. Forecasts from
nearby origins overlap heavily, so `windows` (the span covered divided by
the horizon) is the honest count of independent outcomes. Treat intervals
with fewer than ~5 windows as optimistic.
"""
rng = np.random.default_rng(seed)
rows = []
for horizon, at_h in scores.groupby("horizon"):
base = at_h[at_h.model == BASELINE].set_index("origin")["crps"].sort_index()
span = (base.index[-1] - base.index[0]).days + horizon
block = max(1, min(int(np.ceil(horizon / ORIGIN_STEP_DAYS)), len(base) // 2))
boot_index = _block_bootstrap_indices(len(base), block, n_boot, rng)
for model, g in at_h.groupby("model", sort=False):
m = g.set_index("origin")["crps"].reindex(base.index).to_numpy()
boot = 1 - m[boot_index].mean(axis=1) / base.to_numpy()[boot_index].mean(axis=1)
rows.append(
{
"model": model,
"horizon": horizon,
"forecasts": len(g),
"windows": span / horizon,
"crps": g["crps"].mean(),
"skill": 1 - m.mean() / base.mean(),
"skill_lo": np.quantile(boot, 0.05),
"skill_hi": np.quantile(boot, 0.95),
**{f"cov{c:.0%}": g[f"in{c:.0%}"].mean() for c in COVERAGES},
"mean_pit": g["pit"].mean(),
}
)
return pd.DataFrame(rows)
def _block_bootstrap_indices(n, block, n_boot, rng) -> np.ndarray:
"""Circular block bootstrap, so the first and last origins aren't under-sampled."""
n_blocks = int(np.ceil(n / block))
starts = rng.integers(0, n, size=(n_boot, n_blocks))
return ((starts[:, :, None] + np.arange(block)) % n).reshape(n_boot, -1)[:, :n]
def format_summary(summary: pd.DataFrame) -> str:
table = pd.DataFrame(
{
"model": summary["model"],
"horizon": summary["horizon"].map(horizon_label),
"windows": summary["windows"].map("{:.1f}".format),
"crps": summary["crps"].map("{:.3f}".format),
"skill vs rw [90%]": [
f"{s:+.0%} [{lo:+.0%}, {hi:+.0%}]"
for s, lo, hi in zip(summary.skill, summary.skill_lo, summary.skill_hi, strict=True)
],
**{f"in {c:.0%}": summary[f"cov{c:.0%}"].map("{:.0%}".format) for c in COVERAGES},
"mean pit": summary["mean_pit"].map("{:.2f}".format),
}
)
return table.to_string(index=False)
def horizon_label(days: int) -> str:
if days < 365:
return f"{round(days / 30.4)}mo"
return f"{round(days / 365)}y"