Rewrite as a probabilistic model with walk-forward evaluation.

Replace the 2024 model (model.py, ~2000 lines) with the btcmodel package, the
baseline for future work:

- Forecasts are quantiles of log price at each horizon, scored with CRPS in a
  walk-forward backtest (origins every 30 days from 2014, horizons 1 month to
  4 years). Skill is relative to a zero-drift random walk, with circular
  block-bootstrap intervals and a count of independent windows.
- Development data stops at 2024-11-26, the last day the 2024 model saw.
  Later outcomes are a holdout, scored only by `backtest --holdout`.
- Models: random_walk, drift_rw, and cycle (the 2024 model's cycle-position
  drift, now kernel-smoothed and recency-weighted). On development data
  nothing beats the random walk with confidence; cycle loses at every horizon.
- Prices: the Investing.com archive moves to data/ (cut at 2024-11-26; its
  last row was intraday) and is extended with Coinbase daily closes by
  `update`.

Also: Nix flake dev shell (Python 3.13, pandas 3), ruff in place of black,
pytest suite, and a rewritten README. NOTES.md is removed as inaccurate, and
poetry is dropped.
This commit is contained in:
sam
2026-09-24 02:19:02 -07:00
parent eefff47070
commit cfc27a38de
24 changed files with 1767 additions and 3124 deletions
+117
View File
@@ -0,0 +1,117 @@
"""Daily BTC-USD closing prices.
Two sources, stitched at ARCHIVE_END:
- data/investing.csv: the original Investing.com download (2010-07-18 onward).
Its last row (2024-11-27) was an intraday snapshot, so it is cut a day early.
- data/coinbase.csv: Coinbase Exchange daily candles (UTC days), appended by
`python -m btcmodel update`.
Everything up to ARCHIVE_END is development data. Everything after it is the
holdout: outcomes nobody had seen while the 2024 model was being built, and which
model development here must not look at (see evaluate.py).
"""
import datetime as dt
import json
import urllib.request
from pathlib import Path
import numpy as np
import pandas as pd
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
ARCHIVE_CSV = DATA_DIR / "investing.csv"
COINBASE_CSV = DATA_DIR / "coinbase.csv"
ARCHIVE_END = pd.Timestamp("2024-11-26")
DEV_CUTOFF = ARCHIVE_END
# 2010 has four distinct prices and no change on 89% of days; it is noise.
DATA_START = pd.Timestamp("2011-01-01")
COINBASE_URL = "https://api.exchange.coinbase.com/products/BTC-USD/candles"
COINBASE_MAX_CANDLES = 300
def load_prices(
until: pd.Timestamp | str | None = None, start: pd.Timestamp | str = DATA_START
) -> pd.DataFrame:
"""
Load daily closes as a frame indexed by date, with a single `close` column.
Models receive a prefix of this frame, so additional data sources can be
joined in as extra columns later without changing the model interface.
"""
archive = _read_archive()
parts = [archive[archive.index <= ARCHIVE_END]]
if COINBASE_CSV.exists():
coinbase = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"])
parts.append(coinbase.loc[coinbase.index > ARCHIVE_END, ["close"]])
df = pd.concat(parts).sort_index()
df = df[df.index >= pd.Timestamp(start)]
if until is not None:
df = df[df.index <= pd.Timestamp(until)]
expected = pd.date_range(df.index[0], df.index[-1], freq="D")
missing = expected.difference(df.index)
if len(missing):
raise ValueError(f"{len(missing)} missing days, first {missing[0].date()}")
if not (df["close"] > 0).all():
raise ValueError("non-positive closing price")
return df
def log_returns(df: pd.DataFrame) -> pd.Series:
"""Daily log returns of the close, without the leading NaN."""
return np.log(df["close"]).diff().iloc[1:]
def _read_archive() -> pd.DataFrame:
raw = pd.read_csv(ARCHIVE_CSV, encoding="utf-8-sig", thousands=",")
dates = pd.to_datetime(raw["Date"], format="%m/%d/%Y")
return pd.DataFrame({"close": raw["Price"].astype(float).values}, index=dates.rename("date"))
def update_coinbase() -> int:
"""Append completed daily candles since the last stored day. Returns rows added."""
if COINBASE_CSV.exists():
existing = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"])
first = existing.index.max() + pd.Timedelta(days=1)
else:
existing = None
first = ARCHIVE_END + pd.Timedelta(days=1)
# Today's candle is still forming; only take finished UTC days.
last = pd.Timestamp(dt.datetime.now(dt.UTC).date()) - pd.Timedelta(days=1)
rows = []
chunk_start = first
while chunk_start <= last:
chunk_end = min(chunk_start + pd.Timedelta(days=COINBASE_MAX_CANDLES - 1), last)
rows.extend(_fetch_candles(chunk_start, chunk_end))
chunk_start = chunk_end + pd.Timedelta(days=1)
if not rows:
return 0
new = pd.DataFrame(rows, columns=["date", "close"]).set_index("date").sort_index()
new = new[(new.index >= first) & (new.index <= last)]
combined = new if existing is None else pd.concat([existing, new])
combined = combined[~combined.index.duplicated(keep="last")].sort_index()
combined.to_csv(COINBASE_CSV, date_format="%Y-%m-%d")
return len(new)
def _fetch_candles(start: pd.Timestamp, end: pd.Timestamp) -> list[tuple[pd.Timestamp, float]]:
url = (
f"{COINBASE_URL}?granularity=86400"
f"&start={start:%Y-%m-%d}T00:00:00Z&end={end:%Y-%m-%d}T00:00:00Z"
)
# Coinbase rejects requests without a User-Agent.
request = urllib.request.Request(url, headers={"User-Agent": "btcmodel"})
with urllib.request.urlopen(request, timeout=30) as response:
candles = json.load(response)
# Each candle is [time, low, high, open, close, volume].
return [
(pd.Timestamp(dt.datetime.fromtimestamp(c[0], dt.UTC).date()), float(c[4])) for c in candles
]