74 lines
3.1 KiB
Python
74 lines
3.1 KiB
Python
"""The 2024 model, distilled."""
|
|||
|
|
|
||
|
|
from dataclasses import dataclass
|
||
|
|
from typing import ClassVar
|
||
|
|
|
||
|
|
import numpy as np
|
||
|
|
import pandas as pd
|
||
|
|
|
||
|
|
from ..data import log_returns
|
||
|
|
from ..forecast import Forecast
|
||
|
|
from ..halving import cycle_position
|
||
|
|
|
||
|
|
# Longer than any cycle so far (the longest, cycle 0, is 1425 days).
|
||
|
|
MAX_CYCLE_DAYS = 1500
|
||
|
|
|
||
|
|
|
||
|
|
@dataclass(frozen=True)
|
||
|
|
class CycleModel:
|
||
|
|
"""
|
||
|
|
Expected return depends on how many days it has been since the last halving.
|
||
|
|
|
||
|
|
The drift for day d of the cycle is a weighted mean of the daily log returns
|
||
|
|
observed around day d of every past cycle. Of the 2024 model's ~2000 lines,
|
||
|
|
this idea did all the work. It differs from that model in two ways:
|
||
|
|
|
||
|
|
- Neighbouring cycle days are pooled with a Gaussian kernel. The 2024 model
|
||
|
|
averaged each day separately and then took a rolling mean.
|
||
|
|
- Past cycles are down-weighted, halving each `recency_half_life` cycles, so
|
||
|
|
the 10-100x cycles of 2011-2017 don't set the level. The 2024 model
|
||
|
|
averaged all cycles equally, then scaled by ~0.7; it overshot the
|
||
|
|
2025 peak by ~60%. (This is the idea on the old `tuning-b` branch.)
|
||
|
|
|
||
|
|
Where the data is thin, the drift shrinks toward the overall weighted mean,
|
||
|
|
as if `prior_days` extra observations sat at that value. Noise is a
|
||
|
|
constant-volatility random walk, so the distribution is closed-form.
|
||
|
|
"""
|
||
|
|
|
||
|
|
name: ClassVar[str] = "cycle"
|
||
|
|
bandwidth_days: float = 30.0
|
||
|
|
recency_half_life: float = 1.0
|
||
|
|
prior_days: float = 10.0
|
||
|
|
vol_window: int = 365
|
||
|
|
|
||
|
|
def drift_by_cycle_day(self, history: pd.DataFrame) -> np.ndarray:
|
||
|
|
"""Expected daily log return for each day of the cycle, shape (MAX_CYCLE_DAYS,)."""
|
||
|
|
returns = log_returns(history)
|
||
|
|
cycle, day = cycle_position(returns.index)
|
||
|
|
current_cycle = cycle_position(history.index[-1:])[0][0]
|
||
|
|
weight = 0.5 ** ((current_cycle - cycle) / self.recency_half_life)
|
||
|
|
|
||
|
|
sum_wr = np.bincount(day, weights=weight * returns.values, minlength=MAX_CYCLE_DAYS)
|
||
|
|
sum_w = np.bincount(day, weights=weight, minlength=MAX_CYCLE_DAYS)
|
||
|
|
|
||
|
|
# Peak-1 kernel, so smoothed weights count (recency-weighted) days of data.
|
||
|
|
half_width = int(np.ceil(4 * self.bandwidth_days))
|
||
|
|
offsets = np.arange(-half_width, half_width + 1)
|
||
|
|
kernel = np.exp(-0.5 * (offsets / self.bandwidth_days) ** 2)
|
||
|
|
smooth_wr = np.convolve(sum_wr, kernel, mode="same")
|
||
|
|
smooth_w = np.convolve(sum_w, kernel, mode="same")
|
||
|
|
|
||
|
|
overall = sum_wr.sum() / sum_w.sum()
|
||
|
|
return (smooth_wr + self.prior_days * overall) / (smooth_w + self.prior_days)
|
||
|
|
|
||
|
|
def forecast(self, history: pd.DataFrame, horizons: np.ndarray) -> Forecast:
|
||
|
|
drift = self.drift_by_cycle_day(history)
|
||
|
|
origin = history.index[-1]
|
||
|
|
future = origin + pd.to_timedelta(np.arange(1, horizons.max() + 1), unit="D")
|
||
|
|
_, future_day = cycle_position(future)
|
||
|
|
cumulative = np.cumsum(drift[future_day])
|
||
|
|
|
||
|
|
sigma = log_returns(history).iloc[-self.vol_window :].std()
|
||
|
|
mean = np.log(history["close"].iloc[-1]) + cumulative[horizons - 1]
|
||
|
|
return Forecast.normal(origin, horizons, mean, sigma * np.sqrt(horizons))
|