"""Daily BTC-USD closing prices. Two sources, stitched at ARCHIVE_END: - data/investing.csv: the original Investing.com download (2010-07-18 onward). Its last row (2024-11-27) was an intraday snapshot, so it is cut a day early. - data/coinbase.csv: Coinbase Exchange daily candles (UTC days), appended by `python -m btcmodel update`. Everything up to ARCHIVE_END is development data. Everything after it is the holdout: outcomes nobody had seen while the 2024 model was being built, and which model development here must not look at (see evaluate.py). """ import datetime as dt import json import urllib.request from pathlib import Path import numpy as np import pandas as pd DATA_DIR = Path(__file__).resolve().parent.parent / "data" ARCHIVE_CSV = DATA_DIR / "investing.csv" COINBASE_CSV = DATA_DIR / "coinbase.csv" ARCHIVE_END = pd.Timestamp("2024-11-26") DEV_CUTOFF = ARCHIVE_END # 2010 has four distinct prices and no change on 89% of days; it is noise. DATA_START = pd.Timestamp("2011-01-01") COINBASE_URL = "https://api.exchange.coinbase.com/products/BTC-USD/candles" COINBASE_MAX_CANDLES = 300 def load_prices( until: pd.Timestamp | str | None = None, start: pd.Timestamp | str = DATA_START ) -> pd.DataFrame: """ Load daily closes as a frame indexed by date, with a single `close` column. Models receive a prefix of this frame, so additional data sources can be joined in as extra columns later without changing the model interface. """ archive = _read_archive() parts = [archive[archive.index <= ARCHIVE_END]] if COINBASE_CSV.exists(): coinbase = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"]) parts.append(coinbase.loc[coinbase.index > ARCHIVE_END, ["close"]]) df = pd.concat(parts).sort_index() df = df[df.index >= pd.Timestamp(start)] if until is not None: df = df[df.index <= pd.Timestamp(until)] expected = pd.date_range(df.index[0], df.index[-1], freq="D") missing = expected.difference(df.index) if len(missing): raise ValueError(f"{len(missing)} missing days, first {missing[0].date()}") if not (df["close"] > 0).all(): raise ValueError("non-positive closing price") return df def log_returns(df: pd.DataFrame) -> pd.Series: """Daily log returns of the close, without the leading NaN.""" return np.log(df["close"]).diff().iloc[1:] def _read_archive() -> pd.DataFrame: raw = pd.read_csv(ARCHIVE_CSV, encoding="utf-8-sig", thousands=",") dates = pd.to_datetime(raw["Date"], format="%m/%d/%Y") return pd.DataFrame({"close": raw["Price"].astype(float).values}, index=dates.rename("date")) def update_coinbase() -> int: """Append completed daily candles since the last stored day. Returns rows added.""" if COINBASE_CSV.exists(): existing = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"]) first = existing.index.max() + pd.Timedelta(days=1) else: existing = None first = ARCHIVE_END + pd.Timedelta(days=1) # Today's candle is still forming; only take finished UTC days. last = pd.Timestamp(dt.datetime.now(dt.UTC).date()) - pd.Timedelta(days=1) rows = [] chunk_start = first while chunk_start <= last: chunk_end = min(chunk_start + pd.Timedelta(days=COINBASE_MAX_CANDLES - 1), last) rows.extend(_fetch_candles(chunk_start, chunk_end)) chunk_start = chunk_end + pd.Timedelta(days=1) if not rows: return 0 new = pd.DataFrame(rows, columns=["date", "close"]).set_index("date").sort_index() new = new[(new.index >= first) & (new.index <= last)] combined = new if existing is None else pd.concat([existing, new]) combined = combined[~combined.index.duplicated(keep="last")].sort_index() combined.to_csv(COINBASE_CSV, date_format="%Y-%m-%d") return len(new) def _fetch_candles(start: pd.Timestamp, end: pd.Timestamp) -> list[tuple[pd.Timestamp, float]]: url = ( f"{COINBASE_URL}?granularity=86400" f"&start={start:%Y-%m-%d}T00:00:00Z&end={end:%Y-%m-%d}T00:00:00Z" ) # Coinbase rejects requests without a User-Agent. request = urllib.request.Request(url, headers={"User-Agent": "btcmodel"}) with urllib.request.urlopen(request, timeout=30) as response: candles = json.load(response) # Each candle is [time, low, high, open, close, volume]. return [ (pd.Timestamp(dt.datetime.fromtimestamp(c[0], dt.UTC).date()), float(c[4])) for c in candles ]