Team Ai
Modelpublic

ParallelLLC/algorithmic_trading

sourceHugging Faceapache-2.0updated 2mo agoView on Hugging Face
27likes32downloads
app.py850 linesDownload Raw Back to root
1"""Backtest Reality Check — the Hugging Face Space entrypoint.2 3Point it at a ticker and a trading rule. It runs the backtest, then spends the4rest of its time trying to prove the result was luck: shuffled markets,5selection-bias deflation, walk-forward, and a cost stress test.6 7Run locally with ``python app.py``.8"""9 10from __future__ import annotations11 12import logging13import os14from typing import Dict, List15 16import gradio as gr17import pandas as pd18 19from algotrader import __version__20from algotrader.charts import (21    arena_chart,22    attribution_chart,23    cross_permutation_chart,24    drawdown_chart,25    empty_figure,26    equity_chart,27    exposure_chart,28    permutation_chart,29    score_chart,30    walkforward_chart,31    weights_chart,32)33from algotrader.cross_sectional import XS_REGISTRY, get_xs_strategy34from algotrader.data import DEFAULT_UNIVERSE35from algotrader.lab import LabConfig, run_arena, run_lab36from algotrader.portfolio_lab import DEFAULT_UNIVERSE as PORTFOLIO_UNIVERSE37from algotrader.portfolio_lab import PortfolioLabConfig, run_portfolio_lab38from algotrader.strategies import REGISTRY, get_strategy39 40logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")41logger = logging.getLogger("app")42 43MAX_PARAMS = 344MAX_XS_PARAMS = 445STRATEGY_CHOICES = [(s.name, key) for key, s in REGISTRY.items()]46XS_STRATEGY_CHOICES = [(s.name, key) for key, s in XS_REGISTRY.items()]47 48GRADE_COLORS = {49    "A": "#0ca30c",50    "B": "#3987e5",51    "C": "#fab219",52    "D": "#ec835a",53    "F": "#d03b3b",54}55 56CSS = """57.gradio-container { max-width: 1280px !important; }58#hero h1 { font-size: 2.4rem; line-height: 1.1; margin: 0 0 .4rem 0; letter-spacing: -.02em; }59#hero p { color: #c3c2b7; margin: 0; font-size: 1.05rem; max-width: 60ch; }60.score-card {61  display: flex; gap: 24px; align-items: center; padding: 22px 24px;62  border: 1px solid rgba(255,255,255,.10); border-radius: 14px; background: #1a1a19;63}64.score-badge {65  min-width: 132px; text-align: center; padding: 14px 10px; border-radius: 12px;66  background: #0d0d0d; border: 1px solid rgba(255,255,255,.10);67}68.score-badge .grade { font-size: 3.2rem; font-weight: 700; line-height: 1; }69.score-badge .num { font-size: .95rem; color: #898781; margin-top: 6px; }70.score-body h3 { margin: 0 0 6px 0; font-size: 1.15rem; color: #fff; }71.score-body p { margin: 0; color: #c3c2b7; line-height: 1.55; }72.tiles { display: grid; grid-template-columns: repeat(auto-fit, minmax(132px,1fr)); gap: 10px; margin-top: 14px; }73.tile { padding: 12px 14px; border: 1px solid rgba(255,255,255,.10); border-radius: 10px; background: #1a1a19; }74.tile .label { font-size: .72rem; text-transform: uppercase; letter-spacing: .06em; color: #898781; }75.tile .value { font-size: 1.5rem; font-weight: 600; color: #fff; margin-top: 4px; }76.tile .sub { font-size: .75rem; color: #898781; margin-top: 2px; }77.flags { margin-top: 14px; padding: 0; list-style: none; }78.flags li {79  padding: 9px 12px; margin-bottom: 7px; border-radius: 8px; background: #1a1a19;80  border-left: 3px solid #ec835a; color: #c3c2b7; font-size: .9rem; line-height: 1.5;81}82.provenance { font-size: .82rem; color: #898781; margin-top: 10px; }83.provenance.sim { color: #fab219; }84"""85 86 87def _fmt_pct(value: float) -> str:88    return f"{value * 100:,.1f}%"89 90 91def param_controls(strategy_key: str):92    """Re-label the shared sliders to match the selected strategy."""93    strategy = get_strategy(strategy_key)94    updates = []95    for i in range(MAX_PARAMS):96        if i < len(strategy.params):97            spec = strategy.params[i]98            lo = spec.minimum if spec.minimum is not None else min(spec.grid)99            hi = spec.maximum if spec.maximum is not None else max(spec.grid)100            updates.append(101                gr.update(102                    visible=True,103                    label=spec.label,104                    value=spec.cast(spec.default),105                    minimum=lo,106                    maximum=hi,107                    step=spec.step or (1 if spec.kind == "int" else 0.1),108                )109            )110        else:111            updates.append(gr.update(visible=False))112    return tuple(updates)113 114 115def collect_params(strategy_key: str, *values) -> Dict[str, float]:116    strategy = get_strategy(strategy_key)117    return {spec.name: spec.cast(values[i]) for i, spec in enumerate(strategy.params[:MAX_PARAMS])}118 119 120def _score_card(report) -> str:121    v = report.verdict122    color = GRADE_COLORS.get(v["grade"], "#898781")123    market = report.market124    provenance = (125        f'<div class="provenance">Data: {market.source} · {market.symbol} · '126        f"{market.start.date()} to {market.end.date()} · {len(market.df):,} bars</div>"127        if market.is_real128        else f'<div class="provenance sim">⚠ {market.note}</div>'129    )130    flags = "".join(f"<li>{f}</li>" for f in v["flags"])131    flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""132 133    return f"""134<div class="score-card">135  <div class="score-badge">136    <div class="grade" style="color:{color}">{v['grade']}</div>137    <div class="num">{v['score']} / 100</div>138  </div>139  <div class="score-body">140    <h3>{report.strategy.name} on {market.symbol}</h3>141    <p>{v['verdict']}</p>142  </div>143</div>144{flags_html}145{provenance}146"""147 148 149def _tiles(report) -> str:150    m = report.backtest.metrics151    b = report.backtest.benchmark_metrics152    perm = report.permutation153    p_text = f"{perm.p_value:.3f}" if perm else "—"154    p_sub = "vs shuffled markets" if perm else "test skipped"155    pbo = report.pbo.get("pbo")156    pbo_text = f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a"157 158    cells = [159        ("Total return", _fmt_pct(m.get("total_return", 0)), f"buy & hold {_fmt_pct(b.get('total_return', 0))}"),160        ("CAGR", _fmt_pct(m.get("cagr", 0)), f"over {m.get('years', 0):.1f} years"),161        ("Sharpe", f"{m.get('sharpe', 0):.2f}", f"buy & hold {b.get('sharpe', 0):.2f}"),162        ("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),163        ("p-value", p_text, p_sub),164        ("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),165        ("Overfit prob.", pbo_text, "in-sample winner fails OOS"),166        ("Trades", f"{int(m.get('n_trades', 0)):,}", f"{m.get('turnover_ann', 0):.0f}x turnover/yr"),167    ]168    tiles = "".join(169        f'<div class="tile"><div class="label">{label}</div>'170        f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'171        for label, value, sub in cells172    )173    return f'<div class="tiles">{tiles}</div>'174 175 176def xs_param_controls(strategy_key: str):177    """Re-label the shared portfolio sliders for the selected strategy."""178    strategy = get_xs_strategy(strategy_key)179    updates = []180    for i in range(MAX_XS_PARAMS):181        if i < len(strategy.params):182            spec = strategy.params[i]183            updates.append(184                gr.update(185                    visible=True,186                    label=spec.label,187                    value=spec.cast(spec.default),188                    minimum=spec.minimum if spec.minimum is not None else min(spec.grid),189                    maximum=spec.maximum if spec.maximum is not None else max(spec.grid),190                    step=spec.step or (1 if spec.kind == "int" else 0.05),191                )192            )193        else:194            updates.append(gr.update(visible=False))195    return tuple(updates)196 197 198def _portfolio_card(report) -> str:199    v = report.verdict200    color = GRADE_COLORS.get(v["grade"], "#898781")201    panel = report.panel202    survivorship = report.survivorship203 204    provenance = (205        f'<div class="provenance">{len(panel.symbols)} symbols · '206        f"{panel.index[0].date()} to {panel.index[-1].date()} · {len(panel):,} bars · "207        f"rebalance {report.config.rebalance}</div>"208    )209    if panel.note:210        provenance += f'<div class="provenance sim">⚠ {panel.note}</div>'211    # The survivorship note is already in the flag list when it is a problem;212    # repeating it under the tiles just makes the card noisy.213 214    flags = "".join(f"<li>{f}</li>" for f in v["flags"])215    flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""216 217    return f"""218<div class="score-card">219  <div class="score-badge">220    <div class="grade" style="color:{color}">{v['grade']}</div>221    <div class="num">{v['score']} / 100</div>222  </div>223  <div class="score-body">224    <h3>{report.strategy.name} across {len(panel.symbols)} names</h3>225    <p>{v['verdict']}</p>226  </div>227</div>228{flags_html}229{provenance}230"""231 232 233def _portfolio_tiles(report) -> str:234    m = report.backtest.metrics235    b = report.backtest.benchmark_metrics236    perm = report.permutation237    attribution = report.attribution238    pbo = report.pbo.get("pbo")239 240    cells = [241        ("Total return", _fmt_pct(m.get("total_return", 0)), f"equal weight {_fmt_pct(b.get('total_return', 0))}"),242        ("Sharpe", f"{m.get('sharpe', 0):.2f}", f"equal weight {b.get('sharpe', 0):.2f}"),243        ("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),244        ("Name-shuffle p", f"{perm.p_value:.3f}" if perm else "—", "vs same book, random names"),245        ("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),246        (247            "Style alpha",248            f"{attribution.get('alpha_annual', 0) * 100:,.1f}%" if attribution.get("available") else "n/a",249            f"t = {attribution.get('alpha_t_stat', 0):.1f}" if attribution.get("available") else "not available",250        ),251        ("Overfit prob.", f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a", "in-sample winner fails OOS"),252        ("Gross / net", f"{m.get('gross_exposure', 0):.2f}", f"net {m.get('net_exposure', 0):+.2f}"),253        ("Avg positions", f"{m.get('avg_positions', 0):.1f}", f"{m.get('turnover_ann', 0):.1f}x turnover/yr"),254        (255            "Survivorship",256            f"{report.survivorship.survival_rate:.0%}",257            f"{report.survivorship.n_delisted} of {report.survivorship.n_symbols} delisted",258        ),259    ]260    tiles = "".join(261        f'<div class="tile"><div class="label">{label}</div>'262        f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'263        for label, value, sub in cells264    )265    return f'<div class="tiles">{tiles}</div>'266 267 268def _portfolio_detail(report) -> str:269    attribution, survivorship, wf = report.attribution, report.survivorship, report.walkforward270    lines = ["### Reading the evidence", ""]271 272    if report.permutation:273        lines += [274            f"**Did it pick the right names?** We rebuilt the book "275            f"{report.permutation.n_permutations} times, keeping every date's gross exposure, net "276            f"exposure and position count exactly as they were, and only randomising *which* asset "277            f"got which weight. The real book's Sharpe of {report.permutation.observed:.2f} sits at "278            f"the {report.permutation.percentile:.0f}th percentile of that null "279            f"(p = {report.permutation.p_value:.3f}). This test deliberately ignores trading costs: "280            "a shuffled book churns far more than a real one, and charging it for that would "281            "flatter the strategy for reasons unrelated to skill.",282            "",283        ]284    if attribution.get("available"):285        lines += [f"**Is it alpha or is it beta?** {attribution['note']}", ""]286    lines += [f"**Survivorship.** {survivorship.note}", ""]287    if survivorship.delisted_symbols:288        lines += [f"Stopped trading during the sample: `{'`, `'.join(survivorship.delisted_symbols)}`.", ""]289    if wf.get("folds"):290        lines += [291            f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "292            f"{wf.get('mean_is_sharpe', 0):.2f} against {wf.get('mean_oos_sharpe', 0):.2f} blind out "293            f"of sample — {wf.get('efficiency', 0):.0%} efficiency, with "294            f"{wf.get('oos_win_rate', 0):.0%} of folds profitable.",295            "",296        ]297    lines += [298        f"**Costs.** Sharpe is {report.cost_stress.get('sharpe_1x', 0):.2f} at the modelled friction "299        f"and {report.cost_stress.get('sharpe_3x', 0):.2f} at triple it "300        f"({report.cost_stress.get('ratio', 0):.0%} retained), on turnover of "301        f"{report.backtest.metrics.get('turnover_ann', 0):.1f}x a year.",302        "",303        "> Past performance, simulated or otherwise, does not predict future returns. "304        "This is research tooling, not investment advice.",305    ]306    return "\n".join(lines)307 308 309def analyse_portfolio(310    symbols: str,311    start: str,312    end: str,313    strategy_key: str,314    p1: float,315    p2: float,316    p3: float,317    p4: float,318    rebalance: str,319    commission: float,320    slippage: float,321    allow_short: bool,322    gross_leverage: float,323    max_weight: float,324    n_permutations: int,325    progress=gr.Progress(),326):327    """Portfolio Lab handler. Like the single-asset one, it never raises into the UI."""328    try:329        universe = [s.strip().upper() for s in (symbols or "").replace("\n", ",").split(",") if s.strip()]330        if len(universe) < 4:331            raise ValueError(332                "A cross-sectional strategy needs at least 4 symbols — it ranks names against "333                "each other, and there is nothing to rank in a list this short."334            )335        strategy = get_xs_strategy(strategy_key)336        values = (p1, p2, p3, p4)337        params = {338            spec.name: spec.cast(values[i])339            for i, spec in enumerate(strategy.params[:MAX_XS_PARAMS])340        }341        cfg = PortfolioLabConfig(342            symbols=universe,343            start=start or "2015-01-01",344            end=end or None,345            source="yahoo",346            strategy=strategy_key,347            params=params,348            commission_bps=float(commission),349            slippage_bps=float(slippage),350            allow_short=bool(allow_short),351            gross_leverage=float(gross_leverage),352            max_weight=float(max_weight) if max_weight else None,353            rebalance=rebalance,354            n_permutations=int(n_permutations),355        )356        report = run_portfolio_lab(cfg, progress=lambda f, m: progress(f, desc=m))357    except Exception as exc:  # noqa: BLE001358        logger.exception("Portfolio lab run failed")359        message = (360            f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3>'361            f"<p>{exc}</p></div></div>"362        )363        blank = empty_figure("No results.")364        return message, "", blank, blank, blank, blank, blank, ""365 366    return (367        _portfolio_card(report),368        _portfolio_tiles(report),369        cross_permutation_chart(report),370        equity_chart(report, benchmark_label="Equal weight"),371        attribution_chart(report.attribution),372        weights_chart(report),373        score_chart(report.verdict, significance_label="Picks the right names"),374        _portfolio_detail(report),375    )376 377 378def _detail_markdown(report) -> str:379    dsr, wf, pbo, stress = report.dsr, report.walkforward, report.pbo, report.cost_stress380    mtr = dsr.get("min_track_record_years", float("inf"))381    mtr_text = f"{mtr:.1f} years" if mtr == mtr and mtr != float("inf") else "never, at this effect size"382 383    lines = [384        "### Reading the evidence",385        "",386        f"**Selection bias.** {report.trials.get('n', 1)} parameter variants of "387        f"*{report.strategy.name}* were backtested. The luckiest skill-free variant of that many "388        f"would be expected to show an annualised Sharpe of about "389        f"**{dsr.get('threshold_sr_annual', 0):.2f}** on its own. Yours was "390        f"**{report.backtest.sharpe:.2f}**, which puts the deflated probability of a real edge at "391        f"**{dsr.get('dsr', 0):.0%}**.",392        "",393        f"**Track record needed.** To call this Sharpe significant at 95% confidence given its "394        f"skew ({dsr.get('skew', 0):.2f}) and kurtosis ({dsr.get('kurtosis', 0):.1f}), you would need "395        f"about **{mtr_text}** of live returns.",396        "",397        f"**Costs.** At the modelled friction the Sharpe is {stress.get('sharpe_1x', 0):.2f}. "398        f"Triple the costs and it becomes {stress.get('sharpe_3x', 0):.2f} "399        f"({stress.get('ratio', 0):.0%} retained).",400        "",401    ]402 403    if wf.get("folds"):404        lines += [405            f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "406            f"{wf.get('mean_is_sharpe', 0):.2f} and the blind out-of-sample Sharpe averaged "407            f"{wf.get('mean_oos_sharpe', 0):.2f} — an efficiency of {wf.get('efficiency', 0):.0%}. "408            f"{wf.get('oos_win_rate', 0):.0%} of folds were profitable out of sample, and the tuner "409            f"picked a different parameter set in {wf.get('param_instability', 0):.0%} of them.",410            "",411        ]412    if pbo.get("note"):413        lines += [f"**Overfitting.** {pbo['note']}", ""]414    elif pbo.get("pbo") == pbo.get("pbo"):415        lines += [416            f"**Overfitting (CSCV).** Over {pbo.get('n_combinations', 0)} in/out splits of "417            f"{pbo.get('n_strategies', 0)} variants, the in-sample winner landed in the bottom half "418            f"out of sample **{pbo.get('pbo', 0):.0%}** of the time, and lost money outright "419            f"{pbo.get('prob_oos_loss', 0):.0%} of the time. The most frequently selected variant was "420            f"`{pbo.get('most_selected_label', 'n/a')}` "421            f"({pbo.get('selection_stability', 0):.0%} of splits).",422            "",423        ]424 425    lines += [426        "> Past performance, simulated or otherwise, does not predict future returns. "427        "This is research tooling, not investment advice.",428    ]429    return "\n".join(lines)430 431 432def analyse(433    symbol: str,434    start: str,435    end: str,436    strategy_key: str,437    p1: float,438    p2: float,439    p3: float,440    commission: float,441    slippage: float,442    allow_short: bool,443    n_permutations: int,444    perm_method: str,445    wf_folds: int,446    progress=gr.Progress(),447):448    """Main Lab handler. Never raises into the UI — it returns a readable message."""449    try:450        cfg = LabConfig(451            symbol=symbol or "SPY",452            start=start or "2015-01-01",453            end=end or None,454            source="yahoo",455            strategy=strategy_key,456            params=collect_params(strategy_key, p1, p2, p3),457            commission_bps=float(commission),458            slippage_bps=float(slippage),459            allow_short=bool(allow_short),460            n_permutations=int(n_permutations),461            permutation_method="block" if perm_method.startswith("Block") else "permute",462            wf_folds=int(wf_folds),463        )464        report = run_lab(cfg, progress=lambda f, m: progress(f, desc=m))465    except Exception as exc:  # noqa: BLE001 - the UI must always say something useful466        logger.exception("Lab run failed")467        message = f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3><p>{exc}</p></div></div>'468        blank = empty_figure("No results.")469        return message, "", blank, blank, blank, blank, blank, blank, ""470 471    return (472        _score_card(report),473        _tiles(report),474        permutation_chart(report),475        equity_chart(report),476        drawdown_chart(report),477        exposure_chart(report),478        walkforward_chart(report),479        score_chart(report.verdict),480        _detail_markdown(report),481    )482 483 484def race(symbol: str, start: str, allow_short: bool, n_permutations: int, progress=gr.Progress()):485    try:486        cfg = LabConfig(487            symbol=symbol or "SPY",488            start=start or "2015-01-01",489            source="yahoo",490            allow_short=bool(allow_short),491        )492        table, market, _ = run_arena(493            cfg,494            n_permutations=int(n_permutations),495            progress=lambda f, m: progress(f, desc=m),496        )497    except Exception as exc:  # noqa: BLE001498        logger.exception("Arena run failed")499        return pd.DataFrame({"Error": [str(exc)]}), empty_figure("No results."), ""500 501    display = table.copy()502    for col in ("Return", "CAGR", "MaxDD"):503        display[col] = display[col].map(lambda v: f"{v * 100:,.1f}%")504    for col in ("Sharpe", "DSR", "Evidence"):505        display[col] = display[col].map(lambda v: f"{v:.2f}")506    display["p-value"] = display["p-value"].map(lambda v: "—" if v != v else f"{v:.3f}")507    display = display.drop(columns=["key"])508 509    provenance = (510        f"Data: {market.source} · {market.symbol} · {market.start.date()} to {market.end.date()}"511        if market.is_real512        else f"⚠ {market.note}"513    )514    note = (515        f"{provenance}\n\nRanked by **evidence** — `(1 − p) × deflated Sharpe` — not by return. "516        "*Buy & Hold* and *Coin Flip* are in the field on purpose: a leaderboard without a "517        "control group is marketing, not measurement."518    )519    return display, arena_chart(table), note520 521 522HOW_IT_WORKS = """523## Why most backtests are wrong524 525A backtest is a measurement taken with a ruler you built after seeing the thing you526are measuring. Four failure modes do almost all the damage, and this Space tests for527each one.528 529### 1. The market had no structure to find — permutation test530 531We take the real price series and shuffle it: each bar's gap, high, low, body and532volume are kept intact, but their **order** is destroyed. The result is a market with533the same volatility and the same fat tails, and no exploitable structure whatsoever.534Then we re-run *your exact rule* on hundreds of these shuffled markets.535 536If your Sharpe sits inside that cloud of results, your rule found nothing that a537coin-flip market would not also have handed it. The **p-value** is the share of538shuffled markets that did as well or better.539 540*Block mode* resamples contiguous chunks instead of single bars, preserving541short-horizon momentum and volatility clustering. It is a harder null, and trend542strategies should be held to it.543 544### 2. You tried 200 things and reported the best — Deflated Sharpe Ratio545 546If you test 200 worthless strategies, the best of them will show a Sharpe near 1.0547purely by chance. The **Deflated Sharpe Ratio** (Bailey & López de Prado, 2014) works548out what the luckiest of *N* skill-free variants would have scored, and asks whether549yours beats that bar — with an extra penalty for negative skew and fat tails, the550return shapes that flatter naive Sharpe ratios.551 552This Space counts the whole parameter grid as trials, because that is what a553researcher would really have run.554 555### 3. The parameters were fitted to the past — PBO and walk-forward556 557**Probability of Backtest Overfitting** (CSCV) cuts the timeline into chunks, and for558every way of splitting them half in-sample and half out-of-sample, checks whether the559in-sample winner stayed a winner. If the winner lands in the bottom half about half560the time, PBO ≈ 50% and your selection process has no skill at all.561 562**Walk-forward** re-tunes on a training window and trades the next window blind,563rolling forward. Efficiency is out-of-sample Sharpe over in-sample Sharpe: 100% means564the edge survived intact, 0% means it was entirely curve-fit.565 566### 4. The edge is smaller than the costs — stress test567 568Every result here is net of commission and slippage charged on exposure changes, plus569a borrow fee on short positions. We then re-run at **triple** the friction. A real edge570degrades; a fake one disappears.571 572---573 574## The Reality Score575 576| Weight | Component | What it measures |577|---:|---|---|578| 30% | Significance | How far outside the shuffled-market null the result sits |579| 25% | Selection | Deflated Sharpe — does it clear the best-of-N bar |580| 20% | Walk-forward | How much of the tuned Sharpe survived trading forward |581| 15% | Overfitting | 1 − PBO, from combinatorially symmetric cross-validation |582| 10% | Robustness | Sharpe retained when costs triple |583 584Grades: **A** ≥ 85 · **B** ≥ 70 · **C** ≥ 55 · **D** ≥ 40 · **F** below 40.585 586The scale is deliberately harsh. Most strategies people post online score below 40,587and the honest response to that is not to soften the scale.588 589## No look-ahead, by construction590 591A strategy emits a target exposure at each bar's close using only data up to that592bar. The engine holds `position[t] = target[t - lag]` with `lag ≥ 1`, so a signal593computed on Tuesday's close cannot earn Tuesday's move. That is the single line where594look-ahead could enter, and the test suite asserts it directly.595 596## Use it from Python597 598```python599from algotrader import LabConfig, run_lab600 601report = run_lab(LabConfig(symbol="SPY", strategy="sma_cross", params={"fast": 20, "slow": 100}))602print(report.verdict["grade"], report.verdict["score"])603print(report.permutation.p_value, report.dsr["dsr"], report.pbo["pbo"])604```605 606Or from the command line:607 608```bash609python -m algotrader.cli lab --symbol SPY --strategy donchian_breakout --permutations 500610python -m algotrader.cli arena --symbol BTC-USD611```612 613---614 615*Research tooling, not investment advice. Nothing here is a recommendation to trade.*616"""617 618 619# Gradio 6 moved `css` and `theme` from the Blocks constructor to launch().620# Spaces pin their own version, so pass them wherever the installed one wants.621_GRADIO_MAJOR = int(gr.__version__.split(".")[0])622_STYLE_KWARGS = {"css": CSS, "theme": gr.themes.Base()}623_BLOCKS_KWARGS = {} if _GRADIO_MAJOR >= 6 else _STYLE_KWARGS624# Gradio 6 also dropped launch(show_api=...).625_LAUNCH_KWARGS = dict(_STYLE_KWARGS) if _GRADIO_MAJOR >= 6 else {"show_api": False}626 627 628def build_app() -> gr.Blocks:629    with gr.Blocks(title="Backtest Reality Check", **_BLOCKS_KWARGS) as demo:630        with gr.Column(elem_id="hero"):631            gr.HTML(632                "<h1>Backtest Reality Check</h1>"633                "<p>Your backtest is probably lying to you. Pick a market and a trading rule — "634                "this runs it, then spends the rest of its effort trying to prove the result "635                "was luck.</p>"636            )637 638        with gr.Tabs():639            with gr.Tab("The Lab"):640                with gr.Row():641                    with gr.Column(scale=1):642                        symbol = gr.Dropdown(643                            choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker",644                            allow_custom_value=True,645                            info="Any Yahoo Finance symbol. Falls back to a simulated market if offline.",646                        )647                        with gr.Row():648                            start = gr.Textbox(value="2015-01-01", label="Start", scale=1)649                            end = gr.Textbox(value="", label="End (blank = today)", scale=1)650 651                        strategy = gr.Dropdown(652                            choices=STRATEGY_CHOICES, value="sma_cross", label="Strategy"653                        )654                        strategy_note = gr.Markdown(get_strategy("sma_cross").description)655 656                        param_sliders = [657                            gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)658                            for i in range(MAX_PARAMS)659                        ]660 661                        with gr.Accordion("Costs and testing", open=False):662                            commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps per trade)")663                            slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps per trade)")664                            allow_short = gr.Checkbox(value=True, label="Allow short positions")665                            n_perms = gr.Slider(666                                0, 1000, value=250, step=50, label="Shuffled markets to test against",667                                info="More is stricter and slower. 250 is plenty for a first look.",668                            )669                            perm_method = gr.Radio(670                                ["Shuffle bars (standard)", "Block bootstrap (harder)"],671                                value="Shuffle bars (standard)", label="Null market",672                            )673                            wf_folds = gr.Slider(2, 8, value=5, step=1, label="Walk-forward folds")674 675                        run_button = gr.Button("Run reality check", variant="primary", size="lg")676 677                        gr.Examples(678                            label="Or try one of these",679                            examples=[680                                ["SPY", "sma_cross"],681                                ["BTC-USD", "donchian_breakout"],682                                ["NVDA", "rsi_reversion"],683                                ["QQQ", "momentum"],684                                ["SPY", "coin_flip"],685                            ],686                            inputs=[symbol, strategy],687                        )688 689                    with gr.Column(scale=2):690                        verdict_html = gr.HTML(691                            '<div class="score-card"><div class="score-body">'692                            "<h3>Nothing tested yet</h3><p>Pick a market and a rule, then hit "693                            "<b>Run reality check</b>. A full run is a few seconds.</p>"694                            "</div></div>"695                        )696                        tiles_html = gr.HTML("")697 698                # The headline test gets the full width — it is the whole point.699                perm_plot = gr.Plot(value=empty_figure("The headline test appears here.", height=320))700                equity_plot = gr.Plot(value=empty_figure())701                with gr.Row():702                    dd_plot = gr.Plot(value=empty_figure(height=240))703                    exposure_plot = gr.Plot(value=empty_figure(height=200))704                with gr.Row():705                    wf_plot = gr.Plot(value=empty_figure(height=300))706                    components_plot = gr.Plot(value=empty_figure(height=260))707                detail_md = gr.Markdown("")708 709                strategy.change(710                    fn=param_controls, inputs=strategy, outputs=param_sliders711                ).then(712                    fn=lambda k: get_strategy(k).description, inputs=strategy, outputs=strategy_note713                )714 715                run_button.click(716                    fn=analyse,717                    inputs=[718                        symbol, start, end, strategy, *param_sliders,719                        commission, slippage, allow_short, n_perms, perm_method, wf_folds,720                    ],721                    outputs=[722                        verdict_html, tiles_html, perm_plot, equity_plot,723                        dd_plot, exposure_plot, wf_plot, components_plot, detail_md,724                    ],725                )726 727            with gr.Tab("Portfolio"):728                gr.Markdown(729                    "Cross-sectional strategies rank names against each other, so they get a "730                    "harder null: we keep every date's gross exposure, net exposure and position "731                    "count exactly as they were and randomise only **which name got which "732                    "weight**. A book that beats that is picking names. One that doesn't was "733                    "being paid for style exposure you can buy in an ETF — which the factor "734                    "regression below measures directly."735                )736                with gr.Row():737                    with gr.Column(scale=1):738                        xs_symbols = gr.Textbox(739                            value=", ".join(PORTFOLIO_UNIVERSE),740                            label="Universe",741                            lines=3,742                            info="Comma-separated. At least 4 names — fewer cannot be ranked.",743                        )744                        with gr.Row():745                            xs_start = gr.Textbox(value="2015-01-01", label="Start", scale=1)746                            xs_end = gr.Textbox(value="", label="End", scale=1)747 748                        xs_strategy = gr.Dropdown(749                            choices=XS_STRATEGY_CHOICES, value="xs_momentum", label="Strategy"750                        )751                        xs_note = gr.Markdown(get_xs_strategy("xs_momentum").description)752                        xs_sliders = [753                            gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)754                            for i in range(MAX_XS_PARAMS)755                        ]756                        xs_rebalance = gr.Radio(757                            ["D", "W", "M", "Q"], value="M", label="Rebalance",758                            info="Daily rebalancing of a real book is rarely affordable.",759                        )760 761                        with gr.Accordion("Costs, limits and testing", open=False):762                            xs_commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps)")763                            xs_slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps)")764                            xs_short = gr.Checkbox(value=True, label="Allow short positions")765                            xs_leverage = gr.Slider(0.1, 3.0, value=1.0, step=0.1, label="Gross leverage cap")766                            xs_maxw = gr.Slider(0.0, 1.0, value=0.25, step=0.05, label="Max weight per name")767                            xs_perms = gr.Slider(768                                0, 500, value=150, step=25, label="Name shuffles",769                            )770 771                        xs_button = gr.Button("Run portfolio check", variant="primary", size="lg")772 773                    with gr.Column(scale=2):774                        xs_verdict = gr.HTML(775                            '<div class="score-card"><div class="score-body">'776                            "<h3>Nothing tested yet</h3><p>Pick a universe and a ranking rule, then "777                            "hit <b>Run portfolio check</b>.</p></div></div>"778                        )779                        xs_tiles = gr.HTML("")780 781                xs_perm_plot = gr.Plot(value=empty_figure("The name-shuffle test appears here.", height=320))782                xs_equity_plot = gr.Plot(value=empty_figure())783                with gr.Row():784                    xs_attr_plot = gr.Plot(value=empty_figure(height=280))785                    xs_weights_plot = gr.Plot(value=empty_figure(height=240))786                xs_components_plot = gr.Plot(value=empty_figure(height=260))787                xs_detail = gr.Markdown("")788 789                xs_strategy.change(790                    fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders791                ).then(792                    fn=lambda k: get_xs_strategy(k).description, inputs=xs_strategy, outputs=xs_note793                )794                xs_button.click(795                    fn=analyse_portfolio,796                    inputs=[797                        xs_symbols, xs_start, xs_end, xs_strategy, *xs_sliders,798                        xs_rebalance, xs_commission, xs_slippage, xs_short,799                        xs_leverage, xs_maxw, xs_perms,800                    ],801                    outputs=[802                        xs_verdict, xs_tiles, xs_perm_plot, xs_equity_plot,803                        xs_attr_plot, xs_weights_plot, xs_components_plot, xs_detail,804                    ],805                )806 807            with gr.Tab("Arena"):808                gr.Markdown(809                    "Race every strategy on the same market, ranked by **evidence** rather than "810                    "return. Buy & hold and a coin flip stay in the field as controls."811                )812                with gr.Row():813                    arena_symbol = gr.Dropdown(814                        choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker", allow_custom_value=True815                    )816                    arena_start = gr.Textbox(value="2015-01-01", label="Start")817                    arena_short = gr.Checkbox(value=True, label="Allow shorts")818                    arena_perms = gr.Slider(0, 400, value=120, step=20, label="Shuffled markets per strategy")819                arena_button = gr.Button("Run the arena", variant="primary")820                arena_note = gr.Markdown("")821                arena_table = gr.Dataframe(interactive=False, wrap=True)822                arena_plot = gr.Plot(value=empty_figure(height=380))823 824                arena_button.click(825                    fn=race,826                    inputs=[arena_symbol, arena_start, arena_short, arena_perms],827                    outputs=[arena_table, arena_plot, arena_note],828                )829 830            with gr.Tab("How it works"):831                gr.Markdown(HOW_IT_WORKS)832 833        gr.Markdown(834            f"<sub>algotrader {__version__} · Apache-2.0 · "835            "Research tooling, not investment advice.</sub>"836        )837 838        demo.load(fn=param_controls, inputs=strategy, outputs=param_sliders)839        demo.load(fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders)840 841    return demo842 843 844if __name__ == "__main__":845    build_app().queue(max_size=24).launch(846        server_name="0.0.0.0",847        server_port=int(os.environ.get("PORT", 7860)),848        **_LAUNCH_KWARGS,849    )850