Rewrite as a probabilistic model with walk-forward evaluation.
Replace the 2024 model (model.py, ~2000 lines) with the btcmodel package, the baseline for future work: - Forecasts are quantiles of log price at each horizon, scored with CRPS in a walk-forward backtest (origins every 30 days from 2014, horizons 1 month to 4 years). Skill is relative to a zero-drift random walk, with circular block-bootstrap intervals and a count of independent windows. - Development data stops at 2024-11-26, the last day the 2024 model saw. Later outcomes are a holdout, scored only by `backtest --holdout`. - Models: random_walk, drift_rw, and cycle (the 2024 model's cycle-position drift, now kernel-smoothed and recency-weighted). On development data nothing beats the random walk with confidence; cycle loses at every horizon. - Prices: the Investing.com archive moves to data/ (cut at 2024-11-26; its last row was intraday) and is extended with Coinbase daily closes by `update`. Also: Nix flake dev shell (Python 3.13, pandas 3), ruff in place of black, pytest suite, and a rewritten README. NOTES.md is removed as inaccurate, and poetry is dropped.
This commit is contained in:
@@ -0,0 +1,15 @@
|
||||
"""
|
||||
Candidate models.
|
||||
|
||||
A model is any object with a `name` and a
|
||||
`forecast(history: pd.DataFrame, horizons: np.ndarray) -> Forecast` method.
|
||||
`history` holds every row up to and including the forecast origin and nothing
|
||||
after it; the harness guarantees that, so models can use all of it freely.
|
||||
"""
|
||||
|
||||
from .baselines import DriftRandomWalk, RandomWalk
|
||||
from .cycle import CycleModel
|
||||
|
||||
# Order is fixed: it sets each model's colour in every chart.
|
||||
MODELS = {m.name: m for m in (RandomWalk(), DriftRandomWalk(), CycleModel())}
|
||||
BASELINE = RandomWalk.name
|
||||
@@ -0,0 +1,47 @@
|
||||
"""Reference forecasts every other model has to beat."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import ClassVar
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from ..data import log_returns
|
||||
from ..forecast import Forecast
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RandomWalk:
|
||||
"""
|
||||
Zero-drift random walk in log price: "it stays about here, give or take".
|
||||
|
||||
Volatility is the trailing standard deviation of daily log returns.
|
||||
"""
|
||||
|
||||
name: ClassVar[str] = "random_walk"
|
||||
vol_window: int = 365
|
||||
|
||||
def forecast(self, history: pd.DataFrame, horizons: np.ndarray) -> Forecast:
|
||||
sigma = log_returns(history).iloc[-self.vol_window :].std()
|
||||
mean = np.log(history["close"].iloc[-1])
|
||||
return Forecast.normal(history.index[-1], horizons, mean, sigma * np.sqrt(horizons))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DriftRandomWalk:
|
||||
"""
|
||||
Random walk whose drift is the mean daily log return over the trailing
|
||||
`drift_window` days (one halving cycle by default): "it keeps doing what it
|
||||
did last cycle".
|
||||
"""
|
||||
|
||||
name: ClassVar[str] = "drift_rw"
|
||||
drift_window: int = 1460
|
||||
vol_window: int = 365
|
||||
|
||||
def forecast(self, history: pd.DataFrame, horizons: np.ndarray) -> Forecast:
|
||||
returns = log_returns(history)
|
||||
mu = returns.iloc[-self.drift_window :].mean()
|
||||
sigma = returns.iloc[-self.vol_window :].std()
|
||||
mean = np.log(history["close"].iloc[-1]) + mu * horizons
|
||||
return Forecast.normal(history.index[-1], horizons, mean, sigma * np.sqrt(horizons))
|
||||
@@ -0,0 +1,73 @@
|
||||
"""The 2024 model, distilled."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import ClassVar
|
||||
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
|
||||
from ..data import log_returns
|
||||
from ..forecast import Forecast
|
||||
from ..halving import cycle_position
|
||||
|
||||
# Longer than any cycle so far (the longest, cycle 0, is 1425 days).
|
||||
MAX_CYCLE_DAYS = 1500
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CycleModel:
|
||||
"""
|
||||
Expected return depends on how many days it has been since the last halving.
|
||||
|
||||
The drift for day d of the cycle is a weighted mean of the daily log returns
|
||||
observed around day d of every past cycle. Of the 2024 model's ~2000 lines,
|
||||
this idea did all the work. It differs from that model in two ways:
|
||||
|
||||
- Neighbouring cycle days are pooled with a Gaussian kernel. The 2024 model
|
||||
averaged each day separately and then took a rolling mean.
|
||||
- Past cycles are down-weighted, halving each `recency_half_life` cycles, so
|
||||
the 10-100x cycles of 2011-2017 don't set the level. The 2024 model
|
||||
averaged all cycles equally, then scaled by ~0.7; it overshot the
|
||||
2025 peak by ~60%. (This is the idea on the old `tuning-b` branch.)
|
||||
|
||||
Where the data is thin, the drift shrinks toward the overall weighted mean,
|
||||
as if `prior_days` extra observations sat at that value. Noise is a
|
||||
constant-volatility random walk, so the distribution is closed-form.
|
||||
"""
|
||||
|
||||
name: ClassVar[str] = "cycle"
|
||||
bandwidth_days: float = 30.0
|
||||
recency_half_life: float = 1.0
|
||||
prior_days: float = 10.0
|
||||
vol_window: int = 365
|
||||
|
||||
def drift_by_cycle_day(self, history: pd.DataFrame) -> np.ndarray:
|
||||
"""Expected daily log return for each day of the cycle, shape (MAX_CYCLE_DAYS,)."""
|
||||
returns = log_returns(history)
|
||||
cycle, day = cycle_position(returns.index)
|
||||
current_cycle = cycle_position(history.index[-1:])[0][0]
|
||||
weight = 0.5 ** ((current_cycle - cycle) / self.recency_half_life)
|
||||
|
||||
sum_wr = np.bincount(day, weights=weight * returns.values, minlength=MAX_CYCLE_DAYS)
|
||||
sum_w = np.bincount(day, weights=weight, minlength=MAX_CYCLE_DAYS)
|
||||
|
||||
# Peak-1 kernel, so smoothed weights count (recency-weighted) days of data.
|
||||
half_width = int(np.ceil(4 * self.bandwidth_days))
|
||||
offsets = np.arange(-half_width, half_width + 1)
|
||||
kernel = np.exp(-0.5 * (offsets / self.bandwidth_days) ** 2)
|
||||
smooth_wr = np.convolve(sum_wr, kernel, mode="same")
|
||||
smooth_w = np.convolve(sum_w, kernel, mode="same")
|
||||
|
||||
overall = sum_wr.sum() / sum_w.sum()
|
||||
return (smooth_wr + self.prior_days * overall) / (smooth_w + self.prior_days)
|
||||
|
||||
def forecast(self, history: pd.DataFrame, horizons: np.ndarray) -> Forecast:
|
||||
drift = self.drift_by_cycle_day(history)
|
||||
origin = history.index[-1]
|
||||
future = origin + pd.to_timedelta(np.arange(1, horizons.max() + 1), unit="D")
|
||||
_, future_day = cycle_position(future)
|
||||
cumulative = np.cumsum(drift[future_day])
|
||||
|
||||
sigma = log_returns(history).iloc[-self.vol_window :].std()
|
||||
mean = np.log(history["close"].iloc[-1]) + cumulative[horizons - 1]
|
||||
return Forecast.normal(origin, horizons, mean, sigma * np.sqrt(horizons))
|
||||
Reference in New Issue
Block a user