New systematic backtest framework.

This commit is contained in:
sam
2024-11-16 03:20:21 -08:00
parent f5bed410ae
commit bbd7d493b7
+377 -96
View File
@@ -5,7 +5,7 @@ import matplotlib.pyplot as plt
import seaborn as sns import seaborn as sns
from scipy.stats import norm from scipy.stats import norm
from scipy.signal import savgol_filter from scipy.signal import savgol_filter
from multiprocessing import Process from multiprocessing import Process, Pool
# Utility functions # Utility functions
@@ -815,12 +815,16 @@ def create_backtest_plot(
): ):
""" """
Create a plot comparing actual price history against model projections from a historical date. Create a plot comparing actual price history against model projections from a historical date.
Returns both the projections and performance metrics.
Args: Args:
df: DataFrame with historical price data df: DataFrame with historical price data
backtest_date: Date to start the backtest from (default: third halving) backtest_date: Date to start the backtest from
start_date: Date to start considering historical data (default: first halving) start_date: Date to start considering historical data
project_days: Number of days to project forward from backtest date project_days: Number of days to project forward from backtest date
Returns:
tuple: (projections DataFrame, metrics dictionary)
""" """
# Convert dates to datetime # Convert dates to datetime
backtest_date = pd.to_datetime(backtest_date) backtest_date = pd.to_datetime(backtest_date)
@@ -932,7 +936,8 @@ def create_backtest_plot(
alpha=0.7, alpha=0.7,
) )
# Calculate and add model performance metrics # Calculate model performance metrics
metrics = {}
if len(validation_df) > 0: if len(validation_df) > 0:
# Create a common date range for comparison # Create a common date range for comparison
actual_prices = validation_df.set_index("Date")["Close"] actual_prices = validation_df.set_index("Date")["Close"]
@@ -943,51 +948,43 @@ def create_backtest_plot(
projections_aligned = historical_projections.loc[common_dates] projections_aligned = historical_projections.loc[common_dates]
# Calculate metrics using aligned data # Calculate metrics using aligned data
mape = ( metrics = {
np.mean( "mape": np.mean(
np.abs( np.abs(
(actual_aligned - projections_aligned["Expected_Trend"]) (actual_aligned - projections_aligned["Expected_Trend"])
/ actual_aligned / actual_aligned
) )
) )
* 100 * 100,
) "rmse": np.sqrt(
coverage_95 = ( np.mean(
np.mean( (actual_aligned - projections_aligned["Expected_Trend"]) ** 2
)
),
"max_error": np.max(
np.abs(actual_aligned - projections_aligned["Expected_Trend"])
),
"coverage_95": np.mean(
(actual_aligned >= projections_aligned["Lower_95"]) (actual_aligned >= projections_aligned["Lower_95"])
& (actual_aligned <= projections_aligned["Upper_95"]) & (actual_aligned <= projections_aligned["Upper_95"])
) )
* 100 * 100,
) "coverage_68": np.mean(
coverage_68 = (
np.mean(
(actual_aligned >= projections_aligned["Lower_68"]) (actual_aligned >= projections_aligned["Lower_68"])
& (actual_aligned <= projections_aligned["Upper_68"]) & (actual_aligned <= projections_aligned["Upper_68"])
) )
* 100 * 100,
) }
rmse = np.sqrt(
np.mean((actual_aligned - projections_aligned["Expected_Trend"]) ** 2)
)
max_error = np.max(
np.abs(actual_aligned - projections_aligned["Expected_Trend"])
)
# Add metrics to plot # Add metrics to plot
metrics_text = ( metrics_text = (
f"Model Performance Metrics:\n" f"Model Performance Metrics:\n"
f"MAPE: {mape:.1f}%\n" f"MAPE: {metrics['mape']:.1f}%\n"
f"RMSE: ${rmse:,.0f}\n" f"RMSE: ${metrics['rmse']:,.0f}\n"
f"Max Error: ${max_error:,.0f}\n" f"Max Error: ${metrics['max_error']:,.0f}\n"
f"95% CI Coverage: {coverage_95:.1f}%\n" f"95% CI Coverage: {metrics['coverage_95']:.1f}%\n"
f"68% CI Coverage: {coverage_68:.1f}%" f"68% CI Coverage: {metrics['coverage_68']:.1f}%"
) )
with open(
f'bitcoin_backtest_{start_date.strftime("%Y%m%d")}_to_{backtest_date.strftime("%Y%m%d")}.txt',
"w",
) as f:
f.write(f"{heading_label}\n")
f.write(metrics_text)
ax.text( ax.text(
0.02, 0.02,
0.98, 0.98,
@@ -1013,46 +1010,15 @@ def create_backtest_plot(
plt.savefig(filename, dpi=300, bbox_inches="tight") plt.savefig(filename, dpi=300, bbox_inches="tight")
plt.close() plt.close()
return historical_projections return historical_projections, metrics
def run_projection(df, start): def run_projection(args):
df, start = args
projections = create_plots(df, start=start, project_days=365 * 4) projections = create_plots(df, start=start, project_days=365 * 4)
# print("\nProjected Prices at Key Points:")
# print(projections.iloc[[29, 89, 179, 364]].round(2)) # 30, 90, 180, 365 days
def run_backtest(params, df): def run_projections(df):
print(
f"\nRunning backtest from {params['start_date']} to {params['backtest_date']}"
)
backtest_projections = create_backtest_plot(df, **params)
# Print some key projection points vs actual prices
print("\nBacktest Results - Projected vs Actual Prices:")
validation_df = df[df["Date"] > params["backtest_date"]]
actual_prices = validation_df.set_index("Date")["Close"]
for days in [30, 90, 180, 365]:
target_date = pd.to_datetime(params["backtest_date"]) + pd.Timedelta(days=days)
if (
target_date in actual_prices.index
and target_date in backtest_projections.index
):
projected = backtest_projections.loc[target_date]
actual = actual_prices.loc[target_date]
print(f"\n{days} days out ({target_date.strftime('%Y-%m-%d')}):")
print(f"Actual Price: ${actual:,.2f}")
print(f"Projected (Expected): ${projected['Expected_Trend']:,.2f}")
print(
f"Projected Range: ${projected['Lower_95']:,.2f} - ${projected['Upper_95']:,.2f}"
)
if __name__ == "__main__":
analysis, df = analyze_bitcoin_prices("prices.csv")
procs = []
# Create main projection # Create main projection
projection_starts = [ projection_starts = [
"2013-01-01", "2013-01-01",
@@ -1060,51 +1026,366 @@ if __name__ == "__main__":
"2015-01-01", "2015-01-01",
"2016-07-09", "2016-07-09",
] ]
for start in projection_starts: args = [(df, start) for start in projection_starts]
proc = Process(target=run_projection, args=(df, start)) with Pool() as pool:
proc.start() pool.map(run_projection, args)
procs.append(proc)
# Create multiple backtests for different periods
# Create multiple backtests for different periods def run_single_backtest(args):
backtests = [ """
# Base case: Second until fourth halving Run a single backtest with the given parameters.
Must be defined at module level for multiprocessing.
Args:
args: tuple of (params dict, DataFrame)
"""
params, df = args
try:
# Create a copy of params without the description
backtest_params = params.copy()
backtest_params.pop("description", None)
projections, metrics = create_backtest_plot(df, **backtest_params)
# Ensure metrics has all required keys with default values
if metrics is None:
metrics = {}
default_metrics = {
"mape": 0.0,
"rmse": 0.0,
"max_error": 0.0,
"coverage_95": 0.0,
"coverage_68": 0.0,
}
# Update metrics with defaults for any missing keys
metrics = {**default_metrics, **metrics}
return {
"params": params,
"projections": projections,
"metrics": metrics,
"success": True,
}
except Exception as e:
print(
f"Error in backtest for period {params['description']}: {str(e)}"
) # Debug print
return {"params": params, "error": str(e), "success": False}
def run_systematic_backtests(df, validation_years=1, min_training_years=8):
"""
Run a comprehensive suite of backtests with consistent validation periods.
Uses sliding windows for both start and end dates.
"""
# Convert years to days
validation_days = validation_years * 365
min_training_days = min_training_years * 365
# Define start date for reliable data
mature_start = pd.Timestamp("2013-01-01")
last_possible_start = df["Date"].max() - pd.Timedelta(
days=min_training_days + validation_days
)
end_date = df["Date"].max() - pd.Timedelta(days=validation_days)
if mature_start >= last_possible_start:
raise ValueError(
f"Insufficient data for backtesting with current parameters:\n"
f"- Data range: {mature_start} to {df['Date'].max()}\n"
f"- Minimum training period: {min_training_years} years\n"
f"- Validation period: {validation_years} years"
)
old_backtests = [
{ {
"start_date": "2016-07-09", "start_date": "2016-07-09",
"backtest_date": "2024-04-19", "backtest_date": "2024-04-19",
"project_days": 1460, "project_days": validation_days,
"description": "Second until fourth halving",
}, },
# Post-Futures Window with two cycles of training
{ {
"start_date": "2013-01-01", # Includes pre-futures for cycle learning "start_date": "2013-01-01", # Includes pre-futures for cycle learning
"backtest_date": "2020-05-11", "backtest_date": "2020-05-11",
"project_days": 1460, "project_days": validation_days,
"description": "Post-Futures Window with two cycles of training",
}, },
# Cross-Regime Test with two cycles of training
{ {
"start_date": "2014-01-01", "start_date": "2014-01-01",
"backtest_date": "2021-12-31", "backtest_date": "2021-12-31",
"project_days": 1460, "project_days": validation_days,
"description": "Cross-Regime Test with two cycles of training",
}, },
# Recent Window focusing on post-2022 behavior
{ {
"start_date": "2015-01-01", # About two cycles before 2022 "start_date": "2015-01-01",
"backtest_date": "2022-01-01", "backtest_date": "2022-01-01",
"project_days": 1460, "project_days": validation_days,
"description": "Recent Window focusing on post-2022 behavior",
}, },
] ]
backtest_periods = []
backtest_periods.extend(old_backtests)
# Run all backtests # Generate backtest periods with sliding windows
for params in backtests: window_start = mature_start
proc = Process( step = pd.Timedelta(days=180) # 6 month steps
target=run_backtest,
args=( while window_start <= last_possible_start:
params, backtest_date = window_start + pd.Timedelta(days=min_training_days)
df,
), backtest_periods.append(
{
"start_date": window_start.strftime("%Y-%m-%d"),
"backtest_date": backtest_date.strftime("%Y-%m-%d"),
"project_days": validation_days,
"description": f"Training {window_start.strftime('%Y-%m-%d')} to {backtest_date.strftime('%Y-%m-%d')}",
}
) )
procs.append(proc) window_start += step
proc.start()
for proc in procs: # Add specific periods of interest
proc.join() special_periods = []
# Halving-based periods
halving_dates = get_halving_dates()
relevant_halvings = [
h
for h in halving_dates
if h < end_date and h > (mature_start + pd.Timedelta(days=min_training_days))
]
for halving in relevant_halvings:
earliest_start = halving - pd.Timedelta(days=min_training_days)
if earliest_start >= mature_start:
special_periods.append(
{
"start_date": earliest_start.strftime("%Y-%m-%d"),
"backtest_date": halving.strftime("%Y-%m-%d"),
"project_days": validation_days,
"description": f"Pre-halving {halving.strftime('%Y')}",
}
)
# Market structure change periods
important_dates = [
("2017-12-01", "Post-futures introduction"),
("2020-03-01", "Post-COVID crash"),
("2021-11-01", "Post-2021 peak"),
]
for date, description in important_dates:
test_date = pd.Timestamp(date)
if test_date < end_date:
earliest_start = test_date - pd.Timedelta(days=min_training_days)
if earliest_start >= mature_start:
special_periods.append(
{
"start_date": earliest_start.strftime("%Y-%m-%d"),
"backtest_date": date,
"project_days": validation_days,
"description": description,
}
)
# Combine and remove any duplicates
all_periods = backtest_periods + special_periods
unique_periods = []
seen_dates = set()
for period in all_periods:
key = f"{period['start_date']}_{period['backtest_date']}"
if key not in seen_dates:
unique_periods.append(period)
seen_dates.add(key)
if not unique_periods:
raise ValueError("No valid backtest periods found with current parameters")
# Sort periods by backtest date for clearer analysis
unique_periods.sort(key=lambda x: pd.Timestamp(x["backtest_date"]))
print(f"\nRunning backtests with:")
print(
f"- Start dates range: {unique_periods[0]['start_date']} to {unique_periods[-1]['start_date']}"
)
print(
f"- Backtest dates range: {unique_periods[0]['backtest_date']} to {unique_periods[-1]['backtest_date']}"
)
print(f"- Minimum training period: {min_training_years} years")
print(f"- Validation period: {validation_years} years")
print(f"- Number of test periods: {len(unique_periods)}")
print("\nTest periods:")
for period in unique_periods:
print(f"- {period['description']}")
# Create args tuples with params and DataFrame
args = [(params, df) for params in unique_periods]
# Use multiprocessing
with Pool() as pool:
results = pool.map(run_single_backtest, args)
# Analyze results
successful_tests = [r for r in results if r["success"]]
failed_tests = [r for r in results if not r["success"]]
# Define stress periods
stress_periods = {
# COVID crash and recovery
("2020-03-01", "2020-09-01"): "COVID crash period",
# 2021 peak and subsequent crash
("2021-11-01", "2022-06-01"): "2021 peak aftermath",
# Add more stress periods as needed
}
def is_stress_period(test_date):
"""Check if a test date falls in any stress period"""
test_date = pd.Timestamp(test_date)
for (start, end), _ in stress_periods.items():
if pd.Timestamp(start) <= test_date <= pd.Timestamp(end):
return True
return False
# Categorize results
normal_periods = []
stress_periods_results = []
for result in successful_tests:
if is_stress_period(result["params"]["backtest_date"]):
stress_periods_results.append(result)
else:
normal_periods.append(result)
# Calculate metrics for each category
def calculate_category_metrics(results):
if not results:
return None
return {
"count": len(results),
"mape": np.mean([r["metrics"]["mape"] for r in results]),
"rmse": np.mean([r["metrics"]["rmse"] for r in results]),
"max_error": np.mean([r["metrics"]["max_error"] for r in results]),
"coverage_95": np.mean([r["metrics"]["coverage_95"] for r in results]),
"coverage_68": np.mean([r["metrics"]["coverage_68"] for r in results]),
}
normal_metrics = calculate_category_metrics(normal_periods)
stress_metrics = calculate_category_metrics(stress_periods_results)
# Write detailed results
with open("bitcoin_backtest_results_summary.txt", "w") as f:
f.write("Systematic Backtest Results\n")
f.write("==========================\n\n")
f.write("Configuration:\n")
f.write(f"- Minimum training period: {min_training_years} years\n")
f.write(f"- Validation period: {validation_years} years\n")
f.write(
f"- Start dates range: {unique_periods[0]['start_date']} to {unique_periods[-1]['start_date']}\n"
)
f.write(
f"- Backtest dates range: {unique_periods[0]['backtest_date']} to {unique_periods[-1]['backtest_date']}\n"
)
f.write(f"- Number of test periods: {len(unique_periods)}\n\n")
# Normal Periods
f.write("Normal Market Periods\n")
f.write("====================\n")
f.write(f"Number of periods: {len(normal_periods)}\n\n")
for result in normal_periods:
f.write("\n" + "=" * 50 + "\n")
f.write(f"Period: {result['params']['description']}\n")
f.write(
f"Training: {result['params']['start_date']} to {result['params']['backtest_date']}\n"
)
f.write(
f"Validation: {result['params']['backtest_date']} to {pd.Timestamp(result['params']['backtest_date']) + pd.Timedelta(days=validation_years*365):%Y-%m-%d}\n"
)
f.write("\nMetrics:\n")
for metric, value in result["metrics"].items():
if metric in ["mape", "coverage_95", "coverage_68"]:
f.write(f"- {metric}: {value:.1f}%\n")
else:
f.write(f"- {metric}: ${value:,.0f}\n")
f.write("\n")
if normal_metrics:
f.write("\nNormal Periods Aggregate Metrics:\n")
f.write(f"MAPE: {normal_metrics['mape']:.1f}%\n")
f.write(f"RMSE: ${normal_metrics['rmse']:,.0f}\n")
f.write(f"Average Max Error: ${normal_metrics['max_error']:,.0f}\n")
f.write(f"95% CI Coverage: {normal_metrics['coverage_95']:.1f}%\n")
f.write(f"68% CI Coverage: {normal_metrics['coverage_68']:.1f}%\n")
# Stress Periods
f.write("\n\nStress Periods\n")
f.write("=============\n")
f.write(f"Number of periods: {len(stress_periods_results)}\n\n")
for result in stress_periods_results:
f.write("\n" + "=" * 50 + "\n")
f.write(f"Period: {result['params']['description']}\n")
f.write(
f"Training: {result['params']['start_date']} to {result['params']['backtest_date']}\n"
)
f.write(
f"Validation: {result['params']['backtest_date']} to {pd.Timestamp(result['params']['backtest_date']) + pd.Timedelta(days=validation_years*365):%Y-%m-%d}\n"
)
f.write("\nMetrics:\n")
for metric, value in result["metrics"].items():
if metric in ["mape", "coverage_95", "coverage_68"]:
f.write(f"- {metric}: {value:.1f}%\n")
else:
f.write(f"- {metric}: ${value:,.0f}\n")
f.write("\n")
if stress_metrics:
f.write("\nStress Periods Aggregate Metrics:\n")
f.write(f"MAPE: {stress_metrics['mape']:.1f}%\n")
f.write(f"RMSE: ${stress_metrics['rmse']:,.0f}\n")
f.write(f"Average Max Error: ${stress_metrics['max_error']:,.0f}\n")
f.write(f"95% CI Coverage: {stress_metrics['coverage_95']:.1f}%\n")
f.write(f"68% CI Coverage: {stress_metrics['coverage_68']:.1f}%\n")
return (
normal_metrics,
stress_metrics,
normal_periods,
stress_periods_results,
failed_tests,
)
# if __name__ == "__main__":
# analysis, df = analyze_bitcoin_prices("prices.csv")
# procs = []
#
# for proc in procs:
# proc.join()
#
if __name__ == "__main__":
analysis, df = analyze_bitcoin_prices("prices.csv")
run_projections(df)
normal_metrics, stress_metrics, normal_results, stress_results, failed_tests = (
run_systematic_backtests(df)
)
print("\nAggregate Metrics:")
print(f"Total backtests run: {normal_metrics['count']}")
print(f"Successful tests: {len(normal_results)}")
print(f"Failed tests: {len(failed_tests)}")
print("\nAverage Performance:")
print(f"MAPE: {normal_metrics['mape']:.1f}%")
print(f"RMSE: ${normal_metrics['rmse']:,.0f}")
print(f"95% CI Coverage: {normal_metrics['coverage_95']:.1f}%")
print(f"68% CI Coverage: {normal_metrics['coverage_68']:.1f}%")
print("\nFailed Tests:")
for test in failed_tests:
print(f"Period: {test['params']['description']}")
print(f"Error: {test['error']}\n")