""" Command line entry point. python -m btcmodel update fetch new daily prices from Coinbase python -m btcmodel backtest score models on development data python -m btcmodel backtest --holdout score models on outcomes after DEV_CUTOFF python -m btcmodel forecast forecast from the latest price python -m btcmodel ab [NAME ...] run A/B experiments on development data """ import argparse from pathlib import Path import numpy as np import pandas as pd from . import data, evaluate, experiments, plots from .models import MODELS FORECAST_REPORT_HORIZONS = (182, 365, 730, 1095, 1460) FORECAST_REPORT_LEVELS = (0.05, 0.25, 0.5, 0.75, 0.95) def main() -> None: parser = argparse.ArgumentParser(prog="btcmodel", description="Bitcoin price model") parser.add_argument("-o", "--output", type=Path, default=Path("output")) parser.add_argument("-m", "--models", nargs="+", choices=list(MODELS), default=list(MODELS)) commands = parser.add_subparsers(dest="command", required=True) commands.add_parser("update", help="fetch new daily prices") backtest = commands.add_parser("backtest", help="walk-forward evaluation") backtest.add_argument( "--holdout", action="store_true", help=f"score outcomes after {data.DEV_CUTOFF:%Y-%m-%d} (don't use while developing)", ) commands.add_parser("forecast", help="forecast from the latest price") ab = commands.add_parser("ab", help="run A/B experiments on development data") ab.add_argument("names", nargs="*", help="experiments to run (default: all)") args = parser.parse_args() if args.command == "ab" and (unknown := set(args.names) - set(experiments.EXPERIMENTS)): parser.error(f"unknown experiments: {', '.join(sorted(unknown))}") models = [MODELS[name] for name in args.models] if args.command == "update": added = data.update_coinbase() print(f"added {added} days; latest {data.load_prices().index[-1]:%Y-%m-%d}") elif args.command == "backtest": run_backtest(models, args.output, args.holdout) elif args.command == "forecast": run_forecast(models, args.output) elif args.command == "ab": run_ab(args.names or list(experiments.EXPERIMENTS), args.output) def run_backtest(models, output: Path, holdout: bool) -> None: if holdout: name, prices = "holdout", data.load_prices() scores = evaluate.backtest(models, prices, score_after=data.DEV_CUTOFF) else: name, prices = "backtest", data.load_prices(until=data.DEV_CUTOFF) scores = evaluate.backtest(models, prices) out = output / name out.mkdir(parents=True, exist_ok=True) summary = evaluate.summarize(scores) report = ( f"{name}: origins every {evaluate.ORIGIN_STEP_DAYS} days from " f"{evaluate.FIRST_ORIGIN:%Y-%m-%d}, outcomes through {prices.index[-1]:%Y-%m-%d}\n\n" + evaluate.format_summary(summary) ) print(report) (out / "report.txt").write_text(report + "\n") scores.to_csv(out / "scores.csv", index=False) summary.to_csv(out / "summary.csv", index=False) plots.skill_chart(summary, out / "skill.png", f"{name}: skill by horizon") plots.calibration_chart(summary, out / "calibration.png", f"{name}: interval coverage") print(f"\nwrote {out}/") def run_ab(names: list[str], output: Path) -> None: prices = data.load_prices(until=data.DEV_CUTOFF) verdicts = [] for name in names: experiment = experiments.EXPERIMENTS[name] scores, summary = experiments.run(experiment, prices) out = output / "ab" / name out.mkdir(parents=True, exist_ok=True) control = experiment.control.name report = ( f"{name}: {experiment.hypothesis}\n(from {experiment.source})\n\n" + evaluate.format_summary(summary, baseline=control) ) print(report + "\n") (out / "report.txt").write_text(report + "\n") summary.to_csv(out / "summary.csv", index=False) plots.skill_chart(summary, out / "skill.png", f"{name}: skill vs {control}", control) plots.calibration_chart(summary, out / "calibration.png", f"{name}: interval coverage") for variant in experiment.variants: v = summary[summary.model == variant.name].set_index("horizon") verdicts.append( { "experiment": name, "variant": variant.name, "control": control, **{evaluate.horizon_label(h): f"{s:+.0%}" for h, s in v["skill"].items()}, "verdict": experiments.verdict(v), } ) table = pd.DataFrame(verdicts).to_string(index=False) print(table) (output / "ab").mkdir(parents=True, exist_ok=True) (output / "ab" / "verdicts.txt").write_text(table + "\n") def run_forecast(models, output: Path) -> None: prices = data.load_prices() horizons = np.arange(1, max(FORECAST_REPORT_HORIZONS) + 1) forecasts = {m.name: m.forecast(prices, horizons) for m in models} out = output / "forecast" out.mkdir(parents=True, exist_ok=True) rows = [] for name, f in forecasts.items(): for h in FORECAST_REPORT_HORIZONS: row = {"model": name, "date": f.dates[h - 1].date(), "horizon": h} for level in FORECAST_REPORT_LEVELS: row[f"p{level * 100:02.0f}"] = np.exp(f.quantile(level)[h - 1]) rows.append(row) table = pd.DataFrame(rows) table.to_csv(out / "forecast.csv", index=False) shown = table.copy() for column in shown.columns[3:]: shown[column] = shown[column].map(plots.price_formatter) print(f"from {prices.index[-1]:%Y-%m-%d} at {plots.price_formatter(prices.close.iloc[-1])}\n") print(shown.to_string(index=False)) plots.fan_chart(prices, forecasts, out / "fan.png") print(f"\nwrote {out}/") if __name__ == "__main__": main()