Team Ai
Modelpublic

ParallelLLC/algorithmic_trading

sourceHugging Faceapache-2.0updated 2mo agoView on Hugging Face
27likes32downloads
cross_permutation.py147 linesDownload Raw Back to validation
1"""The cross-sectional null.2 3For a timing rule, shuffling the price path is the right null. For a rule that4*ranks names*, it is the wrong test entirely: shuffling time destroys the5market's whole correlation structure, and the resulting null is so weak that6almost any long-short book clears it.7 8The question a cross-sectional strategy has to answer is narrower. Not "does9this market have structure?" but: **given these dates, these assets and this10book's shape, does the strategy put its weight on the right names?**11 12So we permute the *weights across assets within each date*. Every calendar13effect survives. Every correlation between names survives. The gross and net14exposure of the book on each date survives exactly. The one thing destroyed is15the link between the strategy's choice and the asset it chose.16 17A momentum book that beats this null is picking names. One that does not was18being paid for its market exposure, its sector tilt, or the calendar -- all of19which are available far more cheaply.20"""21 22from __future__ import annotations23 24from dataclasses import dataclass25from typing import Callable, Optional26 27import numpy as np28import pandas as pd29 30from ..panel import Panel31from ..portfolio import run_portfolio_backtest32from ..types import CostModel33 34__all__ = ["cross_sectional_permutation_test", "permute_within_dates", "CrossPermutationResult"]35 36 37@dataclass38class CrossPermutationResult:39    observed: float40    null: np.ndarray41    p_value: float42    n_permutations: int43 44    @property45    def null_mean(self) -> float:46        return float(np.mean(self.null)) if self.null.size else 0.047 48    @property49    def percentile(self) -> float:50        if not self.null.size:51            return 50.052        return float((self.null < self.observed).mean() * 100.0)53 54 55def permute_within_dates(56    weights: pd.DataFrame,57    investable: pd.DataFrame,58    rng: np.random.Generator,59) -> pd.DataFrame:60    """Reassign each date's weights among that date's investable assets.61 62    The multiset of weights on every row is preserved exactly -- so gross63    exposure, net exposure, leg sizes and position counts are all identical to64    the real book -- but which asset receives which weight is randomised.65 66    Vectorised across all dates at once: sorting each row puts the investable67    weights first and pushes non-investable slots to NaN, then a random rank per68    investable slot picks each weight exactly once.69    """70    values = weights.to_numpy(dtype=float, copy=True)71    mask = investable.to_numpy(dtype=bool)72 73    masked = np.where(mask, values, np.nan)74    # NaNs sort last, so the first k entries of each row are that row's real weights.75    ordered = np.sort(masked, axis=1)76 77    noise = np.where(mask, rng.random(values.shape), np.inf)78    # Double argsort turns random values into ranks 0..k-1 for investable slots,79    # and k..n-1 for the rest, which then index into the NaN tail.80    random_rank = np.argsort(np.argsort(noise, axis=1), axis=1)81 82    shuffled = np.take_along_axis(ordered, random_rank, axis=1)83    return pd.DataFrame(84        np.nan_to_num(shuffled, nan=0.0), index=weights.index, columns=weights.columns85    )86 87 88def cross_sectional_permutation_test(89    panel: Panel,90    weights: pd.DataFrame,91    n_permutations: int = 200,92    costs: Optional[CostModel] = None,93    lag: int = 1,94    gross_leverage: float = 1.0,95    max_weight: Optional[float] = None,96    allow_short: bool = True,97    rebalance_on: Optional[pd.Series] = None,98    seed: int = 0,99    observed: Optional[float] = None,100    neutralise_costs: bool = True,101    progress: Optional[Callable[[float, str], None]] = None,102) -> CrossPermutationResult:103    """Test whether a book's Sharpe survives randomising which names it picked.104 105    ``neutralise_costs`` defaults to True, and it matters more than it looks.106    A real momentum book holds many of the same names from one rebalance to the107    next, so it churns slowly. A shuffled book reassigns names at random every108    date, so it churns furiously and pays for it. Charging costs would penalise109    the null for turnover the strategy never had, and the strategy would look110    good by comparison for reasons that have nothing to do with skill.111 112    So this test asks only "did it pick the right names?" and leaves "can you113    afford to trade it?" to the cost stress test, which measures that directly.114    """115    costs = CostModel(0.0, 0.0, 0.0) if neutralise_costs else (costs or CostModel())116    rng = np.random.default_rng(seed)117    investable = panel.close.notna()118 119    def sharpe_of(w: pd.DataFrame) -> float:120        return run_portfolio_backtest(121            panel,122            w,123            costs=costs,124            lag=lag,125            gross_leverage=gross_leverage,126            max_weight=max_weight,127            allow_short=allow_short,128            rebalance_on=rebalance_on,129        ).sharpe130 131    if observed is None:132        observed = sharpe_of(weights)133 134    null = np.empty(n_permutations, dtype=float)135    for i in range(n_permutations):136        null[i] = sharpe_of(permute_within_dates(weights, investable, rng))137        if progress is not None and (i % 10 == 0 or i == n_permutations - 1):138            progress((i + 1) / n_permutations, f"Cross-sectional shuffle {i + 1}/{n_permutations}")139 140    p_value = float((1 + np.sum(null >= observed)) / (n_permutations + 1))141    return CrossPermutationResult(142        observed=float(observed),143        null=null,144        p_value=p_value,145        n_permutations=n_permutations,146    )147