"""The 2024 model, distilled.""" from dataclasses import dataclass from typing import ClassVar import numpy as np import pandas as pd from ..data import log_returns from ..forecast import Forecast from ..halving import cycle_position # Longer than any cycle so far (the longest, cycle 0, is 1425 days). MAX_CYCLE_DAYS = 1500 @dataclass(frozen=True) class CycleModel: """ Expected return depends on how many days it has been since the last halving. The drift for day d of the cycle is a weighted mean of the daily log returns observed around day d of every past cycle. Of the 2024 model's ~2000 lines, this idea did all the work. It differs from that model in two ways: - Neighbouring cycle days are pooled with a Gaussian kernel. The 2024 model averaged each day separately and then took a rolling mean. - Past cycles are down-weighted, halving each `recency_half_life` cycles, so the 10-100x cycles of 2011-2017 don't set the level. The 2024 model averaged all cycles equally, then scaled by ~0.7; it overshot the 2025 peak by ~60%. (This is the idea on the old `tuning-b` branch.) Where the data is thin, the drift shrinks toward the overall weighted mean, as if `prior_days` extra observations sat at that value. Noise is a constant-volatility random walk, so the distribution is closed-form. """ name: ClassVar[str] = "cycle" bandwidth_days: float = 30.0 recency_half_life: float = 1.0 prior_days: float = 10.0 vol_window: int = 365 def drift_by_cycle_day(self, history: pd.DataFrame) -> np.ndarray: """Expected daily log return for each day of the cycle, shape (MAX_CYCLE_DAYS,).""" returns = log_returns(history) cycle, day = cycle_position(returns.index) current_cycle = cycle_position(history.index[-1:])[0][0] weight = 0.5 ** ((current_cycle - cycle) / self.recency_half_life) sum_wr = np.bincount(day, weights=weight * returns.values, minlength=MAX_CYCLE_DAYS) sum_w = np.bincount(day, weights=weight, minlength=MAX_CYCLE_DAYS) # Peak-1 kernel, so smoothed weights count (recency-weighted) days of data. half_width = int(np.ceil(4 * self.bandwidth_days)) offsets = np.arange(-half_width, half_width + 1) kernel = np.exp(-0.5 * (offsets / self.bandwidth_days) ** 2) smooth_wr = np.convolve(sum_wr, kernel, mode="same") smooth_w = np.convolve(sum_w, kernel, mode="same") overall = sum_wr.sum() / sum_w.sum() return (smooth_wr + self.prior_days * overall) / (smooth_w + self.prior_days) def forecast(self, history: pd.DataFrame, horizons: np.ndarray) -> Forecast: drift = self.drift_by_cycle_day(history) origin = history.index[-1] future = origin + pd.to_timedelta(np.arange(1, horizons.max() + 1), unit="D") _, future_day = cycle_position(future) cumulative = np.cumsum(drift[future_day]) sigma = log_returns(history).iloc[-self.vol_window :].std() mean = np.log(history["close"].iloc[-1]) + cumulative[horizons - 1] return Forecast.normal(origin, horizons, mean, sigma * np.sqrt(horizons))