Files
bitcoin-model/btcmodel/forward.py
T

105 lines
4.0 KiB
Python
Raw Normal View History

"""
Forward test: forecasts recorded before their outcomes exist.
`snapshot` stores every tracked model's forecast from the latest price in
data/forecasts/<origin>.csv. The files are committed, so version control shows
each one was written before its outcome was known. `score_snapshots` grades
every recorded forecast whose target date has passed, exactly as the backtest
does.
Snapshots store the forecast quantiles themselves, so model code can change
without rewriting the record. If a model's definition changes, give it a new
name rather than reusing the old one.
"""
import datetime as dt
from pathlib import Path
import numpy as np
import pandas as pd
from .data import DATA_DIR
from .evaluate import HORIZONS, score
from .experiments import EXPERIMENTS
from .forecast import LEVELS, Forecast
from .models import MODELS
FORECASTS_DIR = DATA_DIR / "forecasts"
# A snapshot must be made from fresh data, not reconstructed after the fact.
MAX_DATA_AGE_DAYS = 2
# Intended cadence (weekly), used to size bootstrap blocks when scoring.
SNAPSHOT_STEP_DAYS = 7
_VARIANTS = {m.name: m for e in EXPERIMENTS.values() for m in e.models}
# The registered models, plus candidates that could only be settled on new data.
TRACKED = [
*MODELS.values(),
*(_VARIANTS[n] for n in ("powerlaw_ou", "powerlaw_ou_param", "cycle_on_powerlaw")),
]
# Named by percentile: p0.5, p1.5, ..., p99.5 (the grid has no exact median).
_QUANTILE_COLUMNS = [f"p{100 * level:.1f}" for level in LEVELS]
def snapshot(
prices: pd.DataFrame,
models=TRACKED,
directory: Path = FORECASTS_DIR,
today: pd.Timestamp | None = None,
) -> Path:
"""Record each model's forecast from the last price in `prices`."""
origin = prices.index[-1]
if today is None:
today = pd.Timestamp(dt.datetime.now(dt.UTC).date())
if (today - origin).days > MAX_DATA_AGE_DAYS:
raise ValueError(
f"latest price is from {origin:%Y-%m-%d}; update the data first "
"(forecasts must be recorded in real time)"
)
path = directory / f"{origin:%Y-%m-%d}.csv"
if path.exists():
raise FileExistsError(f"{path} already exists")
frames = []
for model in models:
f = model.forecast(prices, HORIZONS)
frame = pd.DataFrame(f.log_quantiles, columns=_QUANTILE_COLUMNS)
frame.insert(0, "model", model.name)
frame.insert(1, "horizon", f.horizons)
frame.insert(2, "target", f.dates.strftime("%Y-%m-%d"))
frames.append(frame)
directory.mkdir(parents=True, exist_ok=True)
pd.concat(frames).to_csv(path, index=False, float_format="%.6f")
return path
def load_snapshots(directory: Path = FORECASTS_DIR) -> list[tuple[str, Forecast]]:
"""Every recorded forecast, as (model name, Forecast)."""
forecasts = []
for path in sorted(directory.glob("*.csv")):
origin = pd.Timestamp(path.stem)
for model, g in pd.read_csv(path).groupby("model", sort=False):
quantiles = g[_QUANTILE_COLUMNS].to_numpy()
forecasts.append((model, Forecast(origin, g["horizon"].to_numpy(), quantiles)))
return forecasts
def score_snapshots(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.DataFrame:
"""Score every recorded forecast horizon whose target date is in `prices`."""
log_close = np.log(prices["close"])
rows = []
for model, f in load_snapshots(directory):
due = f.dates <= prices.index[-1]
if not due.any():
continue
due_forecast = Forecast(f.origin, f.horizons[due], f.log_quantiles[due])
outcome = log_close.loc[due_forecast.dates].to_numpy()
rows.append(score(model, due_forecast, outcome))
return pd.concat(rows, ignore_index=True) if rows else pd.DataFrame()
def next_due(prices: pd.DataFrame, directory: Path = FORECASTS_DIR) -> pd.Timestamp | None:
"""The earliest target date not yet observed, if any."""
pending = [d for _, f in load_snapshots(directory) for d in f.dates if d > prices.index[-1]]
return min(pending) if pending else None