Improve volume handling in market maturity calculations.
This commit is contained in:
@@ -170,44 +170,108 @@ def calculate_market_maturity_score(df):
|
||||
"""
|
||||
df = df.copy()
|
||||
|
||||
# 1. Volume-based metrics
|
||||
df["log_volume"] = np.log(df["Volume"])
|
||||
df["volume_ma"] = df["log_volume"].rolling(window=365).mean()
|
||||
volume_growth = (df["volume_ma"] - df["volume_ma"].shift(365)) / df[
|
||||
"volume_ma"
|
||||
].shift(365)
|
||||
# 1. Enhanced volume-based metrics
|
||||
# Use rolling median instead of mean to reduce impact of outliers
|
||||
df["volume_ma90"] = df["Volume"].rolling(window=90).median()
|
||||
df["volume_ma365"] = df["Volume"].rolling(window=365).median()
|
||||
|
||||
# Calculate relative volume growth using log differences
|
||||
# This better handles exponential growth in volume over time
|
||||
volume_growth_90d = np.log(df["volume_ma90"] / df["volume_ma90"].shift(90)).fillna(
|
||||
0
|
||||
)
|
||||
volume_growth_365d = np.log(
|
||||
df["volume_ma365"] / df["volume_ma365"].shift(365)
|
||||
).fillna(0)
|
||||
|
||||
# Normalize volume growth to rolling volatility of volume
|
||||
# This adapts to different market epochs
|
||||
volume_growth_std_90 = volume_growth_90d.rolling(window=90).std()
|
||||
volume_growth_std_365 = volume_growth_365d.rolling(window=365).std()
|
||||
|
||||
normalized_volume_growth = (
|
||||
(volume_growth_90d / volume_growth_std_90).clip(-2, 2)
|
||||
* 0.4 # Short-term component
|
||||
+ (volume_growth_365d / volume_growth_std_365).clip(-2, 2)
|
||||
* 0.6 # Long-term component
|
||||
).fillna(0)
|
||||
|
||||
# Transform to 0-1 scale using sigmoid function
|
||||
volume_score = 1 / (1 + np.exp(-normalized_volume_growth))
|
||||
|
||||
# 2. Volatility maturity (lower volatility = more mature)
|
||||
df["rolling_vol"] = df["Daily_Return"].rolling(window=365).std() * np.sqrt(365)
|
||||
vol_maturity = 1 / (1 + df["rolling_vol"])
|
||||
df["rolling_vol_90"] = df["Daily_Return"].rolling(window=90).std() * np.sqrt(365)
|
||||
df["rolling_vol_365"] = df["Daily_Return"].rolling(window=365).std() * np.sqrt(365)
|
||||
|
||||
# 3. Market efficiency score
|
||||
df["autocorr"] = (
|
||||
df["Daily_Return"]
|
||||
.rolling(window=30)
|
||||
.apply(lambda x: abs(pd.Series(x).autocorr(1)))
|
||||
# Normalize volatility relative to its historical range
|
||||
vol_score_90 = 1 / (
|
||||
1 + df["rolling_vol_90"] / df["rolling_vol_90"].rolling(window=365).median()
|
||||
)
|
||||
efficiency = 1 - df["autocorr"] # Lower autocorrelation = more efficient
|
||||
vol_score_365 = 1 / (
|
||||
1 + df["rolling_vol_365"] / df["rolling_vol_365"].rolling(window=730).median()
|
||||
)
|
||||
|
||||
vol_maturity = vol_score_90 * 0.4 + vol_score_365 * 0.6
|
||||
|
||||
# 3. Market efficiency score using multiple timeframes
|
||||
efficiency_scores = []
|
||||
for window in [30, 90]:
|
||||
# Calculate absolute autocorrelation at multiple lags
|
||||
for lag in [1, 2, 3, 5]:
|
||||
autocorr = (
|
||||
df["Daily_Return"]
|
||||
.rolling(window=window)
|
||||
.apply(lambda x: abs(pd.Series(x).autocorr(lag)))
|
||||
)
|
||||
efficiency_scores.append(1 - autocorr)
|
||||
|
||||
efficiency = pd.concat(efficiency_scores, axis=1).mean(axis=1)
|
||||
|
||||
# 4. Futures market impact (post-2017)
|
||||
futures_date = pd.Timestamp("2017-12-10")
|
||||
futures_impact = (df["Date"] > futures_date).astype(float) * 0.2
|
||||
futures_impact = (df["Date"] > futures_date).astype(float)
|
||||
|
||||
# Combine scores with time-varying weights
|
||||
weights = {"volume": 0.3, "volatility": 0.3, "efficiency": 0.2, "futures": 0.2}
|
||||
# Progressive futures market maturation
|
||||
days_since_futures = (df["Date"] - futures_date).dt.total_seconds() / (24 * 60 * 60)
|
||||
futures_maturity = futures_impact * (1 - np.exp(-days_since_futures / 365))
|
||||
|
||||
# Combine scores with dynamic weights
|
||||
base_weights = {
|
||||
"volume": 0.25,
|
||||
"volatility": 0.30,
|
||||
"efficiency": 0.25,
|
||||
"futures": 0.20,
|
||||
}
|
||||
|
||||
# Adjust weights based on data availability
|
||||
lookback = pd.Timestamp("2016-01-01")
|
||||
historical_period = (df["Date"] < lookback).astype(float)
|
||||
|
||||
# Reduce weight of futures impact for historical data
|
||||
weights = base_weights.copy()
|
||||
weights["futures"] = weights["futures"] * (1 - historical_period)
|
||||
|
||||
# Redistribute futures weight to other components in historical period
|
||||
historical_adjustment = (weights["futures"] * historical_period) / 3
|
||||
weights["volume"] += historical_adjustment
|
||||
weights["volatility"] += historical_adjustment
|
||||
weights["efficiency"] += historical_adjustment
|
||||
|
||||
# Calculate final score
|
||||
maturity_score = (
|
||||
weights["volume"] * volume_growth.clip(-1, 1).map(lambda x: (x + 1) / 2)
|
||||
weights["volume"] * volume_score
|
||||
+ weights["volatility"] * vol_maturity
|
||||
+ weights["efficiency"] * efficiency
|
||||
+ weights["futures"] * futures_impact
|
||||
+ weights["futures"] * futures_maturity
|
||||
)
|
||||
|
||||
# Normalize to 0-1 range and smooth
|
||||
maturity_score = (maturity_score - maturity_score.min()) / (
|
||||
maturity_score.max() - maturity_score.min()
|
||||
)
|
||||
maturity_score = maturity_score.rolling(window=30, min_periods=1).mean()
|
||||
# Apply non-linear transformation to better distinguish maturity levels
|
||||
maturity_score = 1 / (1 + np.exp(-4 * (maturity_score - 0.5)))
|
||||
|
||||
# Final smoothing
|
||||
maturity_score = maturity_score.rolling(
|
||||
window=30, min_periods=1, center=True
|
||||
).mean()
|
||||
|
||||
return maturity_score
|
||||
|
||||
|
||||
Reference in New Issue
Block a user