Rewrite as a probabilistic model with walk-forward evaluation.

Replace the 2024 model (model.py, ~2000 lines) with the btcmodel package, the
baseline for future work:

- Forecasts are quantiles of log price at each horizon, scored with CRPS in a
  walk-forward backtest (origins every 30 days from 2014, horizons 1 month to
  4 years). Skill is relative to a zero-drift random walk, with circular
  block-bootstrap intervals and a count of independent windows.
- Development data stops at 2024-11-26, the last day the 2024 model saw.
  Later outcomes are a holdout, scored only by `backtest --holdout`.
- Models: random_walk, drift_rw, and cycle (the 2024 model's cycle-position
  drift, now kernel-smoothed and recency-weighted). On development data
  nothing beats the random walk with confidence; cycle loses at every horizon.
- Prices: the Investing.com archive moves to data/ (cut at 2024-11-26; its
  last row was intraday) and is extended with Coinbase daily closes by
  `update`.

Also: Nix flake dev shell (Python 3.13, pandas 3), ruff in place of black,
pytest suite, and a rewritten README. NOTES.md is removed as inaccurate, and
poetry is dropped.
This commit is contained in:
sam
2026-09-24 02:19:02 -07:00
parent eefff47070
commit cfc27a38de
24 changed files with 1767 additions and 3124 deletions
+44
View File
@@ -0,0 +1,44 @@
import numpy as np
import pandas as pd
from scipy.stats import norm
from btcmodel.forecast import Forecast, crps, pit
def normal_crps(mu, sigma, y):
"""Closed-form CRPS of N(mu, sigma^2) at y."""
z = (y - mu) / sigma
return sigma * (z * (2 * norm.cdf(z) - 1) + 2 * norm.pdf(z) - 1 / np.sqrt(np.pi))
def test_crps_matches_closed_form_for_normal():
f = Forecast.normal("2020-01-01", np.array([1, 2, 3]), mean=[0.0, 1.0, 2.0], sd=[1.0, 0.5, 2.0])
y = np.array([0.3, -0.2, 5.0])
expected = normal_crps(np.array([0.0, 1.0, 2.0]), np.array([1.0, 0.5, 2.0]), y)
np.testing.assert_allclose(crps(f.log_quantiles, y), expected, rtol=0.02)
def test_crps_prefers_the_right_forecast():
y = np.array([0.0])
good = Forecast.normal("2020-01-01", np.array([1]), 0.0, 0.1)
biased = Forecast.normal("2020-01-01", np.array([1]), 0.5, 0.1)
vague = Forecast.normal("2020-01-01", np.array([1]), 0.0, 2.0)
assert crps(good.log_quantiles, y) < crps(biased.log_quantiles, y)
assert crps(good.log_quantiles, y) < crps(vague.log_quantiles, y)
def test_pit_and_intervals():
f = Forecast.normal("2020-01-01", np.array([1, 1, 1]), 0.0, 1.0)
np.testing.assert_allclose(
pit(f.log_quantiles, [0.0, 1.0, -10.0]), [0.5, 0.841, 0.0], atol=0.01
)
lo, hi = f.interval(0.95)
np.testing.assert_allclose(hi, 1.96, atol=0.01)
np.testing.assert_allclose(lo, -1.96, atol=0.01)
def test_from_samples_recovers_quantiles():
rng = np.random.default_rng(0)
f = Forecast.from_samples("2020-01-01", np.array([10]), rng.normal(0, 1, size=(200_000, 1)))
np.testing.assert_allclose(f.quantile(0.5), 0.0, atol=0.02)
assert f.dates[0] == pd.Timestamp("2020-01-11")
+83
View File
@@ -0,0 +1,83 @@
import numpy as np
import pandas as pd
import pytest
from btcmodel import data
from btcmodel.evaluate import backtest
from btcmodel.halving import HALVINGS, cycle_position
from btcmodel.models import MODELS, CycleModel
def synthetic_prices(daily_return, start="2011-01-01", end="2024-11-26", noise=0.0, seed=0):
"""Prices whose daily log return is `daily_return(cycle_day)` plus optional noise."""
dates = pd.date_range(start, end, freq="D", name="date")
_, day = cycle_position(dates)
r = daily_return(day) + noise * np.random.default_rng(seed).standard_normal(len(dates))
return pd.DataFrame({"close": 100 * np.exp(np.cumsum(r))}, index=dates)
def test_cycle_position():
index, day = cycle_position(pd.DatetimeIndex(["2009-01-03", "2012-11-27", *HALVINGS]))
assert list(index) == [0, 0, 1, 2, 3, 4]
assert list(day) == [0, 1424, 0, 0, 0, 0]
# A projected halving about four years after the last one starts cycle 5.
index, _ = cycle_position(pd.DatetimeIndex(["2028-06-01"]))
assert index[0] == 5
def test_cycle_model_recovers_a_cycle_shaped_drift():
# Up for the first half of each cycle, down in the second half.
def shape(day):
return np.where(day < 700, 0.002, -0.001)
prices = synthetic_prices(shape)
drift = CycleModel(prior_days=0).drift_by_cycle_day(prices)
assert drift[300] == pytest.approx(0.002, abs=2e-4)
assert drift[1100] == pytest.approx(-0.001, abs=2e-4)
def test_cycle_model_weights_recent_cycles_more():
# The same day of the cycle returns less in each later cycle.
prices = synthetic_prices(lambda day: np.zeros_like(day, dtype=float))
cycle, _ = cycle_position(prices.index)
r = 0.004 / 2.0**cycle
prices["close"] = 100 * np.exp(np.cumsum(r))
drift = CycleModel(recency_half_life=0.25).drift_by_cycle_day(prices)
equal = CycleModel(recency_half_life=1e9).drift_by_cycle_day(prices)
latest_complete = 0.004 / 2.0**3
assert abs(drift[900] - latest_complete) < abs(equal[900] - latest_complete)
@pytest.mark.parametrize("name", list(MODELS))
def test_models_produce_valid_forecasts(name):
prices = synthetic_prices(lambda day: 0.001 + 0 * day, noise=0.03)
horizons = np.array([1, 30, 365, 1460])
f = MODELS[name].forecast(prices, horizons)
assert f.log_quantiles.shape == (4, 100)
assert np.all(np.diff(f.log_quantiles, axis=1) >= 0)
lo, hi = f.interval(0.8)
assert np.all(np.diff(hi - lo) > 0), "uncertainty should grow with horizon"
def test_backtest_never_shows_models_the_future():
prices = synthetic_prices(lambda day: 0.001 + 0 * day, noise=0.03)
class Spy:
name = "random_walk"
def forecast(self, history, horizons):
origin = history.index[-1]
assert history.index.max() == origin
assert (origin + pd.Timedelta(days=int(horizons.min()))) > history.index.max()
return MODELS["random_walk"].forecast(history, horizons)
scores = backtest([Spy()], prices)
assert len(scores) > 0
targets = scores.origin + pd.to_timedelta(scores.horizon, unit="D")
assert targets.max() <= prices.index[-1]
def test_development_data_stops_at_cutoff():
prices = data.load_prices(until=data.DEV_CUTOFF)
assert prices.index[0] == data.DATA_START
assert prices.index[-1] == data.DEV_CUTOFF