118 lines
4.4 KiB
Python
118 lines
4.4 KiB
Python
"""Daily BTC-USD closing prices.
|
|||
|
|
|
||
|
|
Two sources, stitched at ARCHIVE_END:
|
||
|
|
|
||
|
|
- data/investing.csv: the original Investing.com download (2010-07-18 onward).
|
||
|
|
Its last row (2024-11-27) was an intraday snapshot, so it is cut a day early.
|
||
|
|
- data/coinbase.csv: Coinbase Exchange daily candles (UTC days), appended by
|
||
|
|
`python -m btcmodel update`.
|
||
|
|
|
||
|
|
Everything up to ARCHIVE_END is development data. Everything after it is the
|
||
|
|
holdout: outcomes nobody had seen while the 2024 model was being built, and which
|
||
|
|
model development here must not look at (see evaluate.py).
|
||
|
|
"""
|
||
|
|
|
||
|
|
import datetime as dt
|
||
|
|
import json
|
||
|
|
import urllib.request
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
import numpy as np
|
||
|
|
import pandas as pd
|
||
|
|
|
||
|
|
DATA_DIR = Path(__file__).resolve().parent.parent / "data"
|
||
|
|
ARCHIVE_CSV = DATA_DIR / "investing.csv"
|
||
|
|
COINBASE_CSV = DATA_DIR / "coinbase.csv"
|
||
|
|
|
||
|
|
ARCHIVE_END = pd.Timestamp("2024-11-26")
|
||
|
|
DEV_CUTOFF = ARCHIVE_END
|
||
|
|
|
||
|
|
# 2010 has four distinct prices and no change on 89% of days; it is noise.
|
||
|
|
DATA_START = pd.Timestamp("2011-01-01")
|
||
|
|
|
||
|
|
COINBASE_URL = "https://api.exchange.coinbase.com/products/BTC-USD/candles"
|
||
|
|
COINBASE_MAX_CANDLES = 300
|
||
|
|
|
||
|
|
|
||
|
|
def load_prices(
|
||
|
|
until: pd.Timestamp | str | None = None, start: pd.Timestamp | str = DATA_START
|
||
|
|
) -> pd.DataFrame:
|
||
|
|
"""
|
||
|
|
Load daily closes as a frame indexed by date, with a single `close` column.
|
||
|
|
|
||
|
|
Models receive a prefix of this frame, so additional data sources can be
|
||
|
|
joined in as extra columns later without changing the model interface.
|
||
|
|
"""
|
||
|
|
archive = _read_archive()
|
||
|
|
parts = [archive[archive.index <= ARCHIVE_END]]
|
||
|
|
if COINBASE_CSV.exists():
|
||
|
|
coinbase = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"])
|
||
|
|
parts.append(coinbase.loc[coinbase.index > ARCHIVE_END, ["close"]])
|
||
|
|
df = pd.concat(parts).sort_index()
|
||
|
|
|
||
|
|
df = df[df.index >= pd.Timestamp(start)]
|
||
|
|
if until is not None:
|
||
|
|
df = df[df.index <= pd.Timestamp(until)]
|
||
|
|
|
||
|
|
expected = pd.date_range(df.index[0], df.index[-1], freq="D")
|
||
|
|
missing = expected.difference(df.index)
|
||
|
|
if len(missing):
|
||
|
|
raise ValueError(f"{len(missing)} missing days, first {missing[0].date()}")
|
||
|
|
if not (df["close"] > 0).all():
|
||
|
|
raise ValueError("non-positive closing price")
|
||
|
|
return df
|
||
|
|
|
||
|
|
|
||
|
|
def log_returns(df: pd.DataFrame) -> pd.Series:
|
||
|
|
"""Daily log returns of the close, without the leading NaN."""
|
||
|
|
return np.log(df["close"]).diff().iloc[1:]
|
||
|
|
|
||
|
|
|
||
|
|
def _read_archive() -> pd.DataFrame:
|
||
|
|
raw = pd.read_csv(ARCHIVE_CSV, encoding="utf-8-sig", thousands=",")
|
||
|
|
dates = pd.to_datetime(raw["Date"], format="%m/%d/%Y")
|
||
|
|
return pd.DataFrame({"close": raw["Price"].astype(float).values}, index=dates.rename("date"))
|
||
|
|
|
||
|
|
|
||
|
|
def update_coinbase() -> int:
|
||
|
|
"""Append completed daily candles since the last stored day. Returns rows added."""
|
||
|
|
if COINBASE_CSV.exists():
|
||
|
|
existing = pd.read_csv(COINBASE_CSV, index_col="date", parse_dates=["date"])
|
||
|
|
first = existing.index.max() + pd.Timedelta(days=1)
|
||
|
|
else:
|
||
|
|
existing = None
|
||
|
|
first = ARCHIVE_END + pd.Timedelta(days=1)
|
||
|
|
# Today's candle is still forming; only take finished UTC days.
|
||
|
|
last = pd.Timestamp(dt.datetime.now(dt.UTC).date()) - pd.Timedelta(days=1)
|
||
|
|
|
||
|
|
rows = []
|
||
|
|
chunk_start = first
|
||
|
|
while chunk_start <= last:
|
||
|
|
chunk_end = min(chunk_start + pd.Timedelta(days=COINBASE_MAX_CANDLES - 1), last)
|
||
|
|
rows.extend(_fetch_candles(chunk_start, chunk_end))
|
||
|
|
chunk_start = chunk_end + pd.Timedelta(days=1)
|
||
|
|
if not rows:
|
||
|
|
return 0
|
||
|
|
|
||
|
|
new = pd.DataFrame(rows, columns=["date", "close"]).set_index("date").sort_index()
|
||
|
|
new = new[(new.index >= first) & (new.index <= last)]
|
||
|
|
combined = new if existing is None else pd.concat([existing, new])
|
||
|
|
combined = combined[~combined.index.duplicated(keep="last")].sort_index()
|
||
|
|
combined.to_csv(COINBASE_CSV, date_format="%Y-%m-%d")
|
||
|
|
return len(new)
|
||
|
|
|
||
|
|
|
||
|
|
def _fetch_candles(start: pd.Timestamp, end: pd.Timestamp) -> list[tuple[pd.Timestamp, float]]:
|
||
|
|
url = (
|
||
|
|
f"{COINBASE_URL}?granularity=86400"
|
||
|
|
f"&start={start:%Y-%m-%d}T00:00:00Z&end={end:%Y-%m-%d}T00:00:00Z"
|
||
|
|
)
|
||
|
|
# Coinbase rejects requests without a User-Agent.
|
||
|
|
request = urllib.request.Request(url, headers={"User-Agent": "btcmodel"})
|
||
|
|
with urllib.request.urlopen(request, timeout=30) as response:
|
||
|
|
candles = json.load(response)
|
||
|
|
# Each candle is [time, low, high, open, close, volume].
|
||
|
|
return [
|
||
|
|
(pd.Timestamp(dt.datetime.fromtimestamp(c[0], dt.UTC).date()), float(c[4])) for c in candles
|
||
|
|
]
|