ParallelLLC/algorithmic_trading
2732
1"""Style attribution: is this alpha, or is it beta you could have bought cheaply?2 3The most common way a strategy is oversold is not fraud or overfitting — it is4that the "edge" is a well-known risk premium wearing a new name. A book that5loads on market beta, or on momentum, or on low-volatility, will produce a6respectable Sharpe and an exciting story, and you can buy the same exposure in7an ETF for a few basis points.8 9So we regress the strategy's returns on style factors built from the panel10itself and ask what is left over. If the intercept is not distinguishable from11zero, the strategy has no alpha — however good its Sharpe looked.12 13Factors are constructed from the universe under test rather than downloaded,14which keeps this offline and self-consistent. The tradeoff is that they are15proxies: without fundamentals there is no true value or size factor, so those16are labelled honestly as what they are.17"""18 19from __future__ import annotations20 21from typing import Dict, Optional22 23import numpy as np24import pandas as pd25 26from .panel import Panel27from .portfolio import run_portfolio_backtest28from .types import CostModel29 30__all__ = ["build_style_factors", "factor_attribution"]31 32_FREE = CostModel(0.0, 0.0, 0.0) # factors are theoretical portfolios33 34 35def _long_short(panel: Panel, scores: pd.DataFrame, rebalance: int = 21) -> pd.Series:36 from .cross_sectional import _rebalance_hold, scores_to_weights37 38 weights = _rebalance_hold(scores_to_weights(scores), rebalance)39 return run_portfolio_backtest(panel, weights, costs=_FREE).returns40 41 42def build_style_factors(panel: Panel, rebalance: int = 21) -> pd.DataFrame:43 """Construct market and style factor returns from the panel itself."""44 close = panel.close45 listed = close.notna()46 counts = listed.sum(axis=1).replace(0, np.nan)47 48 equal = listed.astype(float).div(counts, axis=0).fillna(0.0)49 market = run_portfolio_backtest(panel, equal, costs=_FREE).returns50 51 factors = {52 "market": market,53 "momentum": _long_short(panel, close.shift(20) / close.shift(250) - 1.0, rebalance),54 "low_vol": _long_short(panel, -close.pct_change().rolling(60, min_periods=60).std(), rebalance),55 "reversal": _long_short(panel, -close.pct_change(5), rebalance),56 # Dollar volume is a liquidity proxy, not market cap. Named accordingly.57 "liquidity": _long_short(panel, -np.log(panel.dollar_volume().replace(0, np.nan)), rebalance),58 }59 return pd.DataFrame(factors).reindex(panel.index).fillna(0.0)60 61 62def _white_standard_errors(x: np.ndarray, residuals: np.ndarray, xtx_inv: np.ndarray) -> np.ndarray:63 """Heteroskedasticity-robust (White) standard errors.64 65 Return series are famously heteroskedastic — volatility clusters — and66 classical standard errors would overstate the significance of alpha.67 """68 meat = x.T @ (x * (residuals**2)[:, None])69 covariance = xtx_inv @ meat @ xtx_inv70 return np.sqrt(np.maximum(np.diag(covariance), 0.0))71 72 73def factor_attribution(74 returns: pd.Series,75 factors: pd.DataFrame,76 periods_per_year: int = 252,77 alpha_t_threshold: float = 2.0,78) -> Dict[str, object]:79 """Regress strategy returns on style factors; report annualised alpha and betas."""80 aligned = pd.concat([returns.rename("strategy"), factors], axis=1).dropna()81 if len(aligned) < 60 or factors.shape[1] == 0:82 return {"available": False, "note": "Not enough overlapping observations for attribution."}83 84 y = aligned["strategy"].to_numpy(dtype=float)85 names = list(factors.columns)86 x = np.column_stack([np.ones(len(aligned))] + [aligned[c].to_numpy(dtype=float) for c in names])87 88 # Drop factors that are constant or collinear; a singular fit is worse than89 # a smaller one.90 keep = [0] + [i + 1 for i, c in enumerate(names) if aligned[c].std() > 1e-12]91 x = x[:, keep]92 names = [names[i - 1] for i in keep[1:]]93 94 try:95 xtx_inv = np.linalg.pinv(x.T @ x)96 except np.linalg.LinAlgError: # pragma: no cover - pinv rarely fails97 return {"available": False, "note": "Factor matrix is singular."}98 99 beta = xtx_inv @ x.T @ y100 fitted = x @ beta101 residuals = y - fitted102 errors = _white_standard_errors(x, residuals, xtx_inv)103 t_stats = np.divide(beta, errors, out=np.zeros_like(beta), where=errors > 0)104 105 ss_res = float(residuals @ residuals)106 ss_tot = float(((y - y.mean()) ** 2).sum())107 r_squared = 1.0 - ss_res / ss_tot if ss_tot > 0 else 0.0108 109 alpha_period = float(beta[0])110 alpha_annual = alpha_period * periods_per_year111 alpha_t = float(t_stats[0])112 significant = bool(abs(alpha_t) >= alpha_t_threshold and alpha_annual > 0)113 114 betas = {name: float(b) for name, b in zip(names, beta[1:])}115 beta_ts = {name: float(t) for name, t in zip(names, t_stats[1:])}116 dominant = max(betas, key=lambda k: abs(betas[k])) if betas else None117 118 if significant:119 note = (120 f"Alpha of {alpha_annual:.1%} a year survives the style regression "121 f"(t = {alpha_t:.1f}). Something here is not explained by market, momentum, "122 "low-volatility, reversal or liquidity exposure."123 )124 else:125 explained = (126 f" Most of the variation is {dominant} exposure (beta {betas[dominant]:.2f})."127 if dominant128 else ""129 )130 note = (131 f"Alpha is {alpha_annual:.1%} a year with t = {alpha_t:.1f}, which is not "132 f"distinguishable from zero. The style factors explain {r_squared:.0%} of the "133 f"returns.{explained} You can buy that exposure far more cheaply than by "134 "running this strategy."135 )136 137 return {138 "available": True,139 "alpha_annual": alpha_annual,140 "alpha_t_stat": alpha_t,141 "alpha_significant": significant,142 "betas": betas,143 "beta_t_stats": beta_ts,144 "r_squared": float(r_squared),145 "dominant_factor": dominant,146 "n_obs": int(len(aligned)),147 "note": note,148 }149 