From bbd7d493b78a3a0560d9cc8747c3cc231278aba5 Mon Sep 17 00:00:00 2001 From: Sam Fredrickson Date: Fri, 15 Nov 2024 23:59:47 -0800 Subject: [PATCH] New systematic backtest framework. --- model.py | 473 ++++++++++++++++++++++++++++++++++++++++++++----------- 1 file changed, 377 insertions(+), 96 deletions(-) diff --git a/model.py b/model.py index 560c069..bfd4abc 100644 --- a/model.py +++ b/model.py @@ -5,7 +5,7 @@ import matplotlib.pyplot as plt import seaborn as sns from scipy.stats import norm from scipy.signal import savgol_filter -from multiprocessing import Process +from multiprocessing import Process, Pool # Utility functions @@ -815,12 +815,16 @@ def create_backtest_plot( ): """ Create a plot comparing actual price history against model projections from a historical date. + Returns both the projections and performance metrics. Args: df: DataFrame with historical price data - backtest_date: Date to start the backtest from (default: third halving) - start_date: Date to start considering historical data (default: first halving) + backtest_date: Date to start the backtest from + start_date: Date to start considering historical data project_days: Number of days to project forward from backtest date + + Returns: + tuple: (projections DataFrame, metrics dictionary) """ # Convert dates to datetime backtest_date = pd.to_datetime(backtest_date) @@ -932,7 +936,8 @@ def create_backtest_plot( alpha=0.7, ) - # Calculate and add model performance metrics + # Calculate model performance metrics + metrics = {} if len(validation_df) > 0: # Create a common date range for comparison actual_prices = validation_df.set_index("Date")["Close"] @@ -943,51 +948,43 @@ def create_backtest_plot( projections_aligned = historical_projections.loc[common_dates] # Calculate metrics using aligned data - mape = ( - np.mean( + metrics = { + "mape": np.mean( np.abs( (actual_aligned - projections_aligned["Expected_Trend"]) / actual_aligned ) ) - * 100 - ) - coverage_95 = ( - np.mean( + * 100, + "rmse": np.sqrt( + np.mean( + (actual_aligned - projections_aligned["Expected_Trend"]) ** 2 + ) + ), + "max_error": np.max( + np.abs(actual_aligned - projections_aligned["Expected_Trend"]) + ), + "coverage_95": np.mean( (actual_aligned >= projections_aligned["Lower_95"]) & (actual_aligned <= projections_aligned["Upper_95"]) ) - * 100 - ) - coverage_68 = ( - np.mean( + * 100, + "coverage_68": np.mean( (actual_aligned >= projections_aligned["Lower_68"]) & (actual_aligned <= projections_aligned["Upper_68"]) ) - * 100 - ) - rmse = np.sqrt( - np.mean((actual_aligned - projections_aligned["Expected_Trend"]) ** 2) - ) - max_error = np.max( - np.abs(actual_aligned - projections_aligned["Expected_Trend"]) - ) + * 100, + } # Add metrics to plot metrics_text = ( f"Model Performance Metrics:\n" - f"MAPE: {mape:.1f}%\n" - f"RMSE: ${rmse:,.0f}\n" - f"Max Error: ${max_error:,.0f}\n" - f"95% CI Coverage: {coverage_95:.1f}%\n" - f"68% CI Coverage: {coverage_68:.1f}%" + f"MAPE: {metrics['mape']:.1f}%\n" + f"RMSE: ${metrics['rmse']:,.0f}\n" + f"Max Error: ${metrics['max_error']:,.0f}\n" + f"95% CI Coverage: {metrics['coverage_95']:.1f}%\n" + f"68% CI Coverage: {metrics['coverage_68']:.1f}%" ) - with open( - f'bitcoin_backtest_{start_date.strftime("%Y%m%d")}_to_{backtest_date.strftime("%Y%m%d")}.txt', - "w", - ) as f: - f.write(f"{heading_label}\n") - f.write(metrics_text) ax.text( 0.02, 0.98, @@ -1013,46 +1010,15 @@ def create_backtest_plot( plt.savefig(filename, dpi=300, bbox_inches="tight") plt.close() - return historical_projections + return historical_projections, metrics -def run_projection(df, start): +def run_projection(args): + df, start = args projections = create_plots(df, start=start, project_days=365 * 4) - # print("\nProjected Prices at Key Points:") - # print(projections.iloc[[29, 89, 179, 364]].round(2)) # 30, 90, 180, 365 days -def run_backtest(params, df): - print( - f"\nRunning backtest from {params['start_date']} to {params['backtest_date']}" - ) - backtest_projections = create_backtest_plot(df, **params) - - # Print some key projection points vs actual prices - print("\nBacktest Results - Projected vs Actual Prices:") - validation_df = df[df["Date"] > params["backtest_date"]] - actual_prices = validation_df.set_index("Date")["Close"] - - for days in [30, 90, 180, 365]: - target_date = pd.to_datetime(params["backtest_date"]) + pd.Timedelta(days=days) - if ( - target_date in actual_prices.index - and target_date in backtest_projections.index - ): - projected = backtest_projections.loc[target_date] - actual = actual_prices.loc[target_date] - print(f"\n{days} days out ({target_date.strftime('%Y-%m-%d')}):") - print(f"Actual Price: ${actual:,.2f}") - print(f"Projected (Expected): ${projected['Expected_Trend']:,.2f}") - print( - f"Projected Range: ${projected['Lower_95']:,.2f} - ${projected['Upper_95']:,.2f}" - ) - - -if __name__ == "__main__": - analysis, df = analyze_bitcoin_prices("prices.csv") - procs = [] - +def run_projections(df): # Create main projection projection_starts = [ "2013-01-01", @@ -1060,51 +1026,366 @@ if __name__ == "__main__": "2015-01-01", "2016-07-09", ] - for start in projection_starts: - proc = Process(target=run_projection, args=(df, start)) - proc.start() - procs.append(proc) + args = [(df, start) for start in projection_starts] + with Pool() as pool: + pool.map(run_projection, args) - # Create multiple backtests for different periods - # Create multiple backtests for different periods - backtests = [ - # Base case: Second until fourth halving + +def run_single_backtest(args): + """ + Run a single backtest with the given parameters. + Must be defined at module level for multiprocessing. + + Args: + args: tuple of (params dict, DataFrame) + """ + params, df = args + try: + # Create a copy of params without the description + backtest_params = params.copy() + backtest_params.pop("description", None) + + projections, metrics = create_backtest_plot(df, **backtest_params) + + # Ensure metrics has all required keys with default values + if metrics is None: + metrics = {} + + default_metrics = { + "mape": 0.0, + "rmse": 0.0, + "max_error": 0.0, + "coverage_95": 0.0, + "coverage_68": 0.0, + } + + # Update metrics with defaults for any missing keys + metrics = {**default_metrics, **metrics} + + return { + "params": params, + "projections": projections, + "metrics": metrics, + "success": True, + } + except Exception as e: + print( + f"Error in backtest for period {params['description']}: {str(e)}" + ) # Debug print + return {"params": params, "error": str(e), "success": False} + + +def run_systematic_backtests(df, validation_years=1, min_training_years=8): + """ + Run a comprehensive suite of backtests with consistent validation periods. + Uses sliding windows for both start and end dates. + """ + # Convert years to days + validation_days = validation_years * 365 + min_training_days = min_training_years * 365 + + # Define start date for reliable data + mature_start = pd.Timestamp("2013-01-01") + last_possible_start = df["Date"].max() - pd.Timedelta( + days=min_training_days + validation_days + ) + end_date = df["Date"].max() - pd.Timedelta(days=validation_days) + + if mature_start >= last_possible_start: + raise ValueError( + f"Insufficient data for backtesting with current parameters:\n" + f"- Data range: {mature_start} to {df['Date'].max()}\n" + f"- Minimum training period: {min_training_years} years\n" + f"- Validation period: {validation_years} years" + ) + + old_backtests = [ { "start_date": "2016-07-09", "backtest_date": "2024-04-19", - "project_days": 1460, + "project_days": validation_days, + "description": "Second until fourth halving", }, - # Post-Futures Window with two cycles of training { "start_date": "2013-01-01", # Includes pre-futures for cycle learning "backtest_date": "2020-05-11", - "project_days": 1460, + "project_days": validation_days, + "description": "Post-Futures Window with two cycles of training", }, - # Cross-Regime Test with two cycles of training { "start_date": "2014-01-01", "backtest_date": "2021-12-31", - "project_days": 1460, + "project_days": validation_days, + "description": "Cross-Regime Test with two cycles of training", }, - # Recent Window focusing on post-2022 behavior { - "start_date": "2015-01-01", # About two cycles before 2022 + "start_date": "2015-01-01", "backtest_date": "2022-01-01", - "project_days": 1460, + "project_days": validation_days, + "description": "Recent Window focusing on post-2022 behavior", }, ] + backtest_periods = [] + backtest_periods.extend(old_backtests) - # Run all backtests - for params in backtests: - proc = Process( - target=run_backtest, - args=( - params, - df, - ), + # Generate backtest periods with sliding windows + window_start = mature_start + step = pd.Timedelta(days=180) # 6 month steps + + while window_start <= last_possible_start: + backtest_date = window_start + pd.Timedelta(days=min_training_days) + + backtest_periods.append( + { + "start_date": window_start.strftime("%Y-%m-%d"), + "backtest_date": backtest_date.strftime("%Y-%m-%d"), + "project_days": validation_days, + "description": f"Training {window_start.strftime('%Y-%m-%d')} to {backtest_date.strftime('%Y-%m-%d')}", + } ) - procs.append(proc) - proc.start() + window_start += step - for proc in procs: - proc.join() + # Add specific periods of interest + special_periods = [] + + # Halving-based periods + halving_dates = get_halving_dates() + relevant_halvings = [ + h + for h in halving_dates + if h < end_date and h > (mature_start + pd.Timedelta(days=min_training_days)) + ] + + for halving in relevant_halvings: + earliest_start = halving - pd.Timedelta(days=min_training_days) + if earliest_start >= mature_start: + special_periods.append( + { + "start_date": earliest_start.strftime("%Y-%m-%d"), + "backtest_date": halving.strftime("%Y-%m-%d"), + "project_days": validation_days, + "description": f"Pre-halving {halving.strftime('%Y')}", + } + ) + + # Market structure change periods + important_dates = [ + ("2017-12-01", "Post-futures introduction"), + ("2020-03-01", "Post-COVID crash"), + ("2021-11-01", "Post-2021 peak"), + ] + + for date, description in important_dates: + test_date = pd.Timestamp(date) + if test_date < end_date: + earliest_start = test_date - pd.Timedelta(days=min_training_days) + if earliest_start >= mature_start: + special_periods.append( + { + "start_date": earliest_start.strftime("%Y-%m-%d"), + "backtest_date": date, + "project_days": validation_days, + "description": description, + } + ) + + # Combine and remove any duplicates + all_periods = backtest_periods + special_periods + unique_periods = [] + seen_dates = set() + for period in all_periods: + key = f"{period['start_date']}_{period['backtest_date']}" + if key not in seen_dates: + unique_periods.append(period) + seen_dates.add(key) + + if not unique_periods: + raise ValueError("No valid backtest periods found with current parameters") + + # Sort periods by backtest date for clearer analysis + unique_periods.sort(key=lambda x: pd.Timestamp(x["backtest_date"])) + + print(f"\nRunning backtests with:") + print( + f"- Start dates range: {unique_periods[0]['start_date']} to {unique_periods[-1]['start_date']}" + ) + print( + f"- Backtest dates range: {unique_periods[0]['backtest_date']} to {unique_periods[-1]['backtest_date']}" + ) + print(f"- Minimum training period: {min_training_years} years") + print(f"- Validation period: {validation_years} years") + print(f"- Number of test periods: {len(unique_periods)}") + + print("\nTest periods:") + for period in unique_periods: + print(f"- {period['description']}") + + # Create args tuples with params and DataFrame + args = [(params, df) for params in unique_periods] + + # Use multiprocessing + with Pool() as pool: + results = pool.map(run_single_backtest, args) + + # Analyze results + successful_tests = [r for r in results if r["success"]] + failed_tests = [r for r in results if not r["success"]] + + # Define stress periods + stress_periods = { + # COVID crash and recovery + ("2020-03-01", "2020-09-01"): "COVID crash period", + # 2021 peak and subsequent crash + ("2021-11-01", "2022-06-01"): "2021 peak aftermath", + # Add more stress periods as needed + } + + def is_stress_period(test_date): + """Check if a test date falls in any stress period""" + test_date = pd.Timestamp(test_date) + for (start, end), _ in stress_periods.items(): + if pd.Timestamp(start) <= test_date <= pd.Timestamp(end): + return True + return False + + # Categorize results + normal_periods = [] + stress_periods_results = [] + + for result in successful_tests: + if is_stress_period(result["params"]["backtest_date"]): + stress_periods_results.append(result) + else: + normal_periods.append(result) + + # Calculate metrics for each category + def calculate_category_metrics(results): + if not results: + return None + return { + "count": len(results), + "mape": np.mean([r["metrics"]["mape"] for r in results]), + "rmse": np.mean([r["metrics"]["rmse"] for r in results]), + "max_error": np.mean([r["metrics"]["max_error"] for r in results]), + "coverage_95": np.mean([r["metrics"]["coverage_95"] for r in results]), + "coverage_68": np.mean([r["metrics"]["coverage_68"] for r in results]), + } + + normal_metrics = calculate_category_metrics(normal_periods) + stress_metrics = calculate_category_metrics(stress_periods_results) + + # Write detailed results + with open("bitcoin_backtest_results_summary.txt", "w") as f: + f.write("Systematic Backtest Results\n") + f.write("==========================\n\n") + + f.write("Configuration:\n") + f.write(f"- Minimum training period: {min_training_years} years\n") + f.write(f"- Validation period: {validation_years} years\n") + f.write( + f"- Start dates range: {unique_periods[0]['start_date']} to {unique_periods[-1]['start_date']}\n" + ) + f.write( + f"- Backtest dates range: {unique_periods[0]['backtest_date']} to {unique_periods[-1]['backtest_date']}\n" + ) + f.write(f"- Number of test periods: {len(unique_periods)}\n\n") + + # Normal Periods + f.write("Normal Market Periods\n") + f.write("====================\n") + f.write(f"Number of periods: {len(normal_periods)}\n\n") + + for result in normal_periods: + f.write("\n" + "=" * 50 + "\n") + f.write(f"Period: {result['params']['description']}\n") + f.write( + f"Training: {result['params']['start_date']} to {result['params']['backtest_date']}\n" + ) + f.write( + f"Validation: {result['params']['backtest_date']} to {pd.Timestamp(result['params']['backtest_date']) + pd.Timedelta(days=validation_years*365):%Y-%m-%d}\n" + ) + f.write("\nMetrics:\n") + for metric, value in result["metrics"].items(): + if metric in ["mape", "coverage_95", "coverage_68"]: + f.write(f"- {metric}: {value:.1f}%\n") + else: + f.write(f"- {metric}: ${value:,.0f}\n") + f.write("\n") + + if normal_metrics: + f.write("\nNormal Periods Aggregate Metrics:\n") + f.write(f"MAPE: {normal_metrics['mape']:.1f}%\n") + f.write(f"RMSE: ${normal_metrics['rmse']:,.0f}\n") + f.write(f"Average Max Error: ${normal_metrics['max_error']:,.0f}\n") + f.write(f"95% CI Coverage: {normal_metrics['coverage_95']:.1f}%\n") + f.write(f"68% CI Coverage: {normal_metrics['coverage_68']:.1f}%\n") + + # Stress Periods + f.write("\n\nStress Periods\n") + f.write("=============\n") + f.write(f"Number of periods: {len(stress_periods_results)}\n\n") + + for result in stress_periods_results: + f.write("\n" + "=" * 50 + "\n") + f.write(f"Period: {result['params']['description']}\n") + f.write( + f"Training: {result['params']['start_date']} to {result['params']['backtest_date']}\n" + ) + f.write( + f"Validation: {result['params']['backtest_date']} to {pd.Timestamp(result['params']['backtest_date']) + pd.Timedelta(days=validation_years*365):%Y-%m-%d}\n" + ) + f.write("\nMetrics:\n") + for metric, value in result["metrics"].items(): + if metric in ["mape", "coverage_95", "coverage_68"]: + f.write(f"- {metric}: {value:.1f}%\n") + else: + f.write(f"- {metric}: ${value:,.0f}\n") + f.write("\n") + + if stress_metrics: + f.write("\nStress Periods Aggregate Metrics:\n") + f.write(f"MAPE: {stress_metrics['mape']:.1f}%\n") + f.write(f"RMSE: ${stress_metrics['rmse']:,.0f}\n") + f.write(f"Average Max Error: ${stress_metrics['max_error']:,.0f}\n") + f.write(f"95% CI Coverage: {stress_metrics['coverage_95']:.1f}%\n") + f.write(f"68% CI Coverage: {stress_metrics['coverage_68']:.1f}%\n") + + return ( + normal_metrics, + stress_metrics, + normal_periods, + stress_periods_results, + failed_tests, + ) + + +# if __name__ == "__main__": +# analysis, df = analyze_bitcoin_prices("prices.csv") +# procs = [] +# +# for proc in procs: +# proc.join() +# + + +if __name__ == "__main__": + analysis, df = analyze_bitcoin_prices("prices.csv") + run_projections(df) + normal_metrics, stress_metrics, normal_results, stress_results, failed_tests = ( + run_systematic_backtests(df) + ) + + print("\nAggregate Metrics:") + print(f"Total backtests run: {normal_metrics['count']}") + print(f"Successful tests: {len(normal_results)}") + print(f"Failed tests: {len(failed_tests)}") + print("\nAverage Performance:") + print(f"MAPE: {normal_metrics['mape']:.1f}%") + print(f"RMSE: ${normal_metrics['rmse']:,.0f}") + print(f"95% CI Coverage: {normal_metrics['coverage_95']:.1f}%") + print(f"68% CI Coverage: {normal_metrics['coverage_68']:.1f}%") + + print("\nFailed Tests:") + for test in failed_tests: + print(f"Period: {test['params']['description']}") + print(f"Error: {test['error']}\n")