ParallelLLC/algorithmic_trading
2732
1"""Backtest Reality Check — the Hugging Face Space entrypoint.2 3Point it at a ticker and a trading rule. It runs the backtest, then spends the4rest of its time trying to prove the result was luck: shuffled markets,5selection-bias deflation, walk-forward, and a cost stress test.6 7Run locally with ``python app.py``.8"""9 10from __future__ import annotations11 12import logging13import os14from typing import Dict, List15 16import gradio as gr17import pandas as pd18 19from algotrader import __version__20from algotrader.charts import (21 arena_chart,22 attribution_chart,23 cross_permutation_chart,24 drawdown_chart,25 empty_figure,26 equity_chart,27 exposure_chart,28 permutation_chart,29 score_chart,30 walkforward_chart,31 weights_chart,32)33from algotrader.cross_sectional import XS_REGISTRY, get_xs_strategy34from algotrader.data import DEFAULT_UNIVERSE35from algotrader.lab import LabConfig, run_arena, run_lab36from algotrader.portfolio_lab import DEFAULT_UNIVERSE as PORTFOLIO_UNIVERSE37from algotrader.portfolio_lab import PortfolioLabConfig, run_portfolio_lab38from algotrader.strategies import REGISTRY, get_strategy39 40logging.basicConfig(level=logging.INFO, format="%(levelname)s %(name)s: %(message)s")41logger = logging.getLogger("app")42 43MAX_PARAMS = 344MAX_XS_PARAMS = 445STRATEGY_CHOICES = [(s.name, key) for key, s in REGISTRY.items()]46XS_STRATEGY_CHOICES = [(s.name, key) for key, s in XS_REGISTRY.items()]47 48GRADE_COLORS = {49 "A": "#0ca30c",50 "B": "#3987e5",51 "C": "#fab219",52 "D": "#ec835a",53 "F": "#d03b3b",54}55 56CSS = """57.gradio-container { max-width: 1280px !important; }58#hero h1 { font-size: 2.4rem; line-height: 1.1; margin: 0 0 .4rem 0; letter-spacing: -.02em; }59#hero p { color: #c3c2b7; margin: 0; font-size: 1.05rem; max-width: 60ch; }60.score-card {61 display: flex; gap: 24px; align-items: center; padding: 22px 24px;62 border: 1px solid rgba(255,255,255,.10); border-radius: 14px; background: #1a1a19;63}64.score-badge {65 min-width: 132px; text-align: center; padding: 14px 10px; border-radius: 12px;66 background: #0d0d0d; border: 1px solid rgba(255,255,255,.10);67}68.score-badge .grade { font-size: 3.2rem; font-weight: 700; line-height: 1; }69.score-badge .num { font-size: .95rem; color: #898781; margin-top: 6px; }70.score-body h3 { margin: 0 0 6px 0; font-size: 1.15rem; color: #fff; }71.score-body p { margin: 0; color: #c3c2b7; line-height: 1.55; }72.tiles { display: grid; grid-template-columns: repeat(auto-fit, minmax(132px,1fr)); gap: 10px; margin-top: 14px; }73.tile { padding: 12px 14px; border: 1px solid rgba(255,255,255,.10); border-radius: 10px; background: #1a1a19; }74.tile .label { font-size: .72rem; text-transform: uppercase; letter-spacing: .06em; color: #898781; }75.tile .value { font-size: 1.5rem; font-weight: 600; color: #fff; margin-top: 4px; }76.tile .sub { font-size: .75rem; color: #898781; margin-top: 2px; }77.flags { margin-top: 14px; padding: 0; list-style: none; }78.flags li {79 padding: 9px 12px; margin-bottom: 7px; border-radius: 8px; background: #1a1a19;80 border-left: 3px solid #ec835a; color: #c3c2b7; font-size: .9rem; line-height: 1.5;81}82.provenance { font-size: .82rem; color: #898781; margin-top: 10px; }83.provenance.sim { color: #fab219; }84"""85 86 87def _fmt_pct(value: float) -> str:88 return f"{value * 100:,.1f}%"89 90 91def param_controls(strategy_key: str):92 """Re-label the shared sliders to match the selected strategy."""93 strategy = get_strategy(strategy_key)94 updates = []95 for i in range(MAX_PARAMS):96 if i < len(strategy.params):97 spec = strategy.params[i]98 lo = spec.minimum if spec.minimum is not None else min(spec.grid)99 hi = spec.maximum if spec.maximum is not None else max(spec.grid)100 updates.append(101 gr.update(102 visible=True,103 label=spec.label,104 value=spec.cast(spec.default),105 minimum=lo,106 maximum=hi,107 step=spec.step or (1 if spec.kind == "int" else 0.1),108 )109 )110 else:111 updates.append(gr.update(visible=False))112 return tuple(updates)113 114 115def collect_params(strategy_key: str, *values) -> Dict[str, float]:116 strategy = get_strategy(strategy_key)117 return {spec.name: spec.cast(values[i]) for i, spec in enumerate(strategy.params[:MAX_PARAMS])}118 119 120def _score_card(report) -> str:121 v = report.verdict122 color = GRADE_COLORS.get(v["grade"], "#898781")123 market = report.market124 provenance = (125 f'<div class="provenance">Data: {market.source} · {market.symbol} · '126 f"{market.start.date()} to {market.end.date()} · {len(market.df):,} bars</div>"127 if market.is_real128 else f'<div class="provenance sim">⚠ {market.note}</div>'129 )130 flags = "".join(f"<li>{f}</li>" for f in v["flags"])131 flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""132 133 return f"""134<div class="score-card">135 <div class="score-badge">136 <div class="grade" style="color:{color}">{v['grade']}</div>137 <div class="num">{v['score']} / 100</div>138 </div>139 <div class="score-body">140 <h3>{report.strategy.name} on {market.symbol}</h3>141 <p>{v['verdict']}</p>142 </div>143</div>144{flags_html}145{provenance}146"""147 148 149def _tiles(report) -> str:150 m = report.backtest.metrics151 b = report.backtest.benchmark_metrics152 perm = report.permutation153 p_text = f"{perm.p_value:.3f}" if perm else "—"154 p_sub = "vs shuffled markets" if perm else "test skipped"155 pbo = report.pbo.get("pbo")156 pbo_text = f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a"157 158 cells = [159 ("Total return", _fmt_pct(m.get("total_return", 0)), f"buy & hold {_fmt_pct(b.get('total_return', 0))}"),160 ("CAGR", _fmt_pct(m.get("cagr", 0)), f"over {m.get('years', 0):.1f} years"),161 ("Sharpe", f"{m.get('sharpe', 0):.2f}", f"buy & hold {b.get('sharpe', 0):.2f}"),162 ("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),163 ("p-value", p_text, p_sub),164 ("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),165 ("Overfit prob.", pbo_text, "in-sample winner fails OOS"),166 ("Trades", f"{int(m.get('n_trades', 0)):,}", f"{m.get('turnover_ann', 0):.0f}x turnover/yr"),167 ]168 tiles = "".join(169 f'<div class="tile"><div class="label">{label}</div>'170 f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'171 for label, value, sub in cells172 )173 return f'<div class="tiles">{tiles}</div>'174 175 176def xs_param_controls(strategy_key: str):177 """Re-label the shared portfolio sliders for the selected strategy."""178 strategy = get_xs_strategy(strategy_key)179 updates = []180 for i in range(MAX_XS_PARAMS):181 if i < len(strategy.params):182 spec = strategy.params[i]183 updates.append(184 gr.update(185 visible=True,186 label=spec.label,187 value=spec.cast(spec.default),188 minimum=spec.minimum if spec.minimum is not None else min(spec.grid),189 maximum=spec.maximum if spec.maximum is not None else max(spec.grid),190 step=spec.step or (1 if spec.kind == "int" else 0.05),191 )192 )193 else:194 updates.append(gr.update(visible=False))195 return tuple(updates)196 197 198def _portfolio_card(report) -> str:199 v = report.verdict200 color = GRADE_COLORS.get(v["grade"], "#898781")201 panel = report.panel202 survivorship = report.survivorship203 204 provenance = (205 f'<div class="provenance">{len(panel.symbols)} symbols · '206 f"{panel.index[0].date()} to {panel.index[-1].date()} · {len(panel):,} bars · "207 f"rebalance {report.config.rebalance}</div>"208 )209 if panel.note:210 provenance += f'<div class="provenance sim">⚠ {panel.note}</div>'211 # The survivorship note is already in the flag list when it is a problem;212 # repeating it under the tiles just makes the card noisy.213 214 flags = "".join(f"<li>{f}</li>" for f in v["flags"])215 flags_html = f'<ul class="flags">{flags}</ul>' if flags else ""216 217 return f"""218<div class="score-card">219 <div class="score-badge">220 <div class="grade" style="color:{color}">{v['grade']}</div>221 <div class="num">{v['score']} / 100</div>222 </div>223 <div class="score-body">224 <h3>{report.strategy.name} across {len(panel.symbols)} names</h3>225 <p>{v['verdict']}</p>226 </div>227</div>228{flags_html}229{provenance}230"""231 232 233def _portfolio_tiles(report) -> str:234 m = report.backtest.metrics235 b = report.backtest.benchmark_metrics236 perm = report.permutation237 attribution = report.attribution238 pbo = report.pbo.get("pbo")239 240 cells = [241 ("Total return", _fmt_pct(m.get("total_return", 0)), f"equal weight {_fmt_pct(b.get('total_return', 0))}"),242 ("Sharpe", f"{m.get('sharpe', 0):.2f}", f"equal weight {b.get('sharpe', 0):.2f}"),243 ("Max drawdown", _fmt_pct(m.get("max_drawdown", 0)), f"{m.get('time_under_water_yrs', 0):.1f}y under water"),244 ("Name-shuffle p", f"{perm.p_value:.3f}" if perm else "—", "vs same book, random names"),245 ("Deflated Sharpe", f"{report.dsr.get('dsr', 0):.2f}", f"after {report.trials.get('n', 1)} variants"),246 (247 "Style alpha",248 f"{attribution.get('alpha_annual', 0) * 100:,.1f}%" if attribution.get("available") else "n/a",249 f"t = {attribution.get('alpha_t_stat', 0):.1f}" if attribution.get("available") else "not available",250 ),251 ("Overfit prob.", f"{pbo:.0%}" if pbo is not None and pbo == pbo else "n/a", "in-sample winner fails OOS"),252 ("Gross / net", f"{m.get('gross_exposure', 0):.2f}", f"net {m.get('net_exposure', 0):+.2f}"),253 ("Avg positions", f"{m.get('avg_positions', 0):.1f}", f"{m.get('turnover_ann', 0):.1f}x turnover/yr"),254 (255 "Survivorship",256 f"{report.survivorship.survival_rate:.0%}",257 f"{report.survivorship.n_delisted} of {report.survivorship.n_symbols} delisted",258 ),259 ]260 tiles = "".join(261 f'<div class="tile"><div class="label">{label}</div>'262 f'<div class="value">{value}</div><div class="sub">{sub}</div></div>'263 for label, value, sub in cells264 )265 return f'<div class="tiles">{tiles}</div>'266 267 268def _portfolio_detail(report) -> str:269 attribution, survivorship, wf = report.attribution, report.survivorship, report.walkforward270 lines = ["### Reading the evidence", ""]271 272 if report.permutation:273 lines += [274 f"**Did it pick the right names?** We rebuilt the book "275 f"{report.permutation.n_permutations} times, keeping every date's gross exposure, net "276 f"exposure and position count exactly as they were, and only randomising *which* asset "277 f"got which weight. The real book's Sharpe of {report.permutation.observed:.2f} sits at "278 f"the {report.permutation.percentile:.0f}th percentile of that null "279 f"(p = {report.permutation.p_value:.3f}). This test deliberately ignores trading costs: "280 "a shuffled book churns far more than a real one, and charging it for that would "281 "flatter the strategy for reasons unrelated to skill.",282 "",283 ]284 if attribution.get("available"):285 lines += [f"**Is it alpha or is it beta?** {attribution['note']}", ""]286 lines += [f"**Survivorship.** {survivorship.note}", ""]287 if survivorship.delisted_symbols:288 lines += [f"Stopped trading during the sample: `{'`, `'.join(survivorship.delisted_symbols)}`.", ""]289 if wf.get("folds"):290 lines += [291 f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "292 f"{wf.get('mean_is_sharpe', 0):.2f} against {wf.get('mean_oos_sharpe', 0):.2f} blind out "293 f"of sample — {wf.get('efficiency', 0):.0%} efficiency, with "294 f"{wf.get('oos_win_rate', 0):.0%} of folds profitable.",295 "",296 ]297 lines += [298 f"**Costs.** Sharpe is {report.cost_stress.get('sharpe_1x', 0):.2f} at the modelled friction "299 f"and {report.cost_stress.get('sharpe_3x', 0):.2f} at triple it "300 f"({report.cost_stress.get('ratio', 0):.0%} retained), on turnover of "301 f"{report.backtest.metrics.get('turnover_ann', 0):.1f}x a year.",302 "",303 "> Past performance, simulated or otherwise, does not predict future returns. "304 "This is research tooling, not investment advice.",305 ]306 return "\n".join(lines)307 308 309def analyse_portfolio(310 symbols: str,311 start: str,312 end: str,313 strategy_key: str,314 p1: float,315 p2: float,316 p3: float,317 p4: float,318 rebalance: str,319 commission: float,320 slippage: float,321 allow_short: bool,322 gross_leverage: float,323 max_weight: float,324 n_permutations: int,325 progress=gr.Progress(),326):327 """Portfolio Lab handler. Like the single-asset one, it never raises into the UI."""328 try:329 universe = [s.strip().upper() for s in (symbols or "").replace("\n", ",").split(",") if s.strip()]330 if len(universe) < 4:331 raise ValueError(332 "A cross-sectional strategy needs at least 4 symbols — it ranks names against "333 "each other, and there is nothing to rank in a list this short."334 )335 strategy = get_xs_strategy(strategy_key)336 values = (p1, p2, p3, p4)337 params = {338 spec.name: spec.cast(values[i])339 for i, spec in enumerate(strategy.params[:MAX_XS_PARAMS])340 }341 cfg = PortfolioLabConfig(342 symbols=universe,343 start=start or "2015-01-01",344 end=end or None,345 source="yahoo",346 strategy=strategy_key,347 params=params,348 commission_bps=float(commission),349 slippage_bps=float(slippage),350 allow_short=bool(allow_short),351 gross_leverage=float(gross_leverage),352 max_weight=float(max_weight) if max_weight else None,353 rebalance=rebalance,354 n_permutations=int(n_permutations),355 )356 report = run_portfolio_lab(cfg, progress=lambda f, m: progress(f, desc=m))357 except Exception as exc: # noqa: BLE001358 logger.exception("Portfolio lab run failed")359 message = (360 f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3>'361 f"<p>{exc}</p></div></div>"362 )363 blank = empty_figure("No results.")364 return message, "", blank, blank, blank, blank, blank, ""365 366 return (367 _portfolio_card(report),368 _portfolio_tiles(report),369 cross_permutation_chart(report),370 equity_chart(report, benchmark_label="Equal weight"),371 attribution_chart(report.attribution),372 weights_chart(report),373 score_chart(report.verdict, significance_label="Picks the right names"),374 _portfolio_detail(report),375 )376 377 378def _detail_markdown(report) -> str:379 dsr, wf, pbo, stress = report.dsr, report.walkforward, report.pbo, report.cost_stress380 mtr = dsr.get("min_track_record_years", float("inf"))381 mtr_text = f"{mtr:.1f} years" if mtr == mtr and mtr != float("inf") else "never, at this effect size"382 383 lines = [384 "### Reading the evidence",385 "",386 f"**Selection bias.** {report.trials.get('n', 1)} parameter variants of "387 f"*{report.strategy.name}* were backtested. The luckiest skill-free variant of that many "388 f"would be expected to show an annualised Sharpe of about "389 f"**{dsr.get('threshold_sr_annual', 0):.2f}** on its own. Yours was "390 f"**{report.backtest.sharpe:.2f}**, which puts the deflated probability of a real edge at "391 f"**{dsr.get('dsr', 0):.0%}**.",392 "",393 f"**Track record needed.** To call this Sharpe significant at 95% confidence given its "394 f"skew ({dsr.get('skew', 0):.2f}) and kurtosis ({dsr.get('kurtosis', 0):.1f}), you would need "395 f"about **{mtr_text}** of live returns.",396 "",397 f"**Costs.** At the modelled friction the Sharpe is {stress.get('sharpe_1x', 0):.2f}. "398 f"Triple the costs and it becomes {stress.get('sharpe_3x', 0):.2f} "399 f"({stress.get('ratio', 0):.0%} retained).",400 "",401 ]402 403 if wf.get("folds"):404 lines += [405 f"**Walk-forward.** Across {len(wf['folds'])} folds the tuned in-sample Sharpe averaged "406 f"{wf.get('mean_is_sharpe', 0):.2f} and the blind out-of-sample Sharpe averaged "407 f"{wf.get('mean_oos_sharpe', 0):.2f} — an efficiency of {wf.get('efficiency', 0):.0%}. "408 f"{wf.get('oos_win_rate', 0):.0%} of folds were profitable out of sample, and the tuner "409 f"picked a different parameter set in {wf.get('param_instability', 0):.0%} of them.",410 "",411 ]412 if pbo.get("note"):413 lines += [f"**Overfitting.** {pbo['note']}", ""]414 elif pbo.get("pbo") == pbo.get("pbo"):415 lines += [416 f"**Overfitting (CSCV).** Over {pbo.get('n_combinations', 0)} in/out splits of "417 f"{pbo.get('n_strategies', 0)} variants, the in-sample winner landed in the bottom half "418 f"out of sample **{pbo.get('pbo', 0):.0%}** of the time, and lost money outright "419 f"{pbo.get('prob_oos_loss', 0):.0%} of the time. The most frequently selected variant was "420 f"`{pbo.get('most_selected_label', 'n/a')}` "421 f"({pbo.get('selection_stability', 0):.0%} of splits).",422 "",423 ]424 425 lines += [426 "> Past performance, simulated or otherwise, does not predict future returns. "427 "This is research tooling, not investment advice.",428 ]429 return "\n".join(lines)430 431 432def analyse(433 symbol: str,434 start: str,435 end: str,436 strategy_key: str,437 p1: float,438 p2: float,439 p3: float,440 commission: float,441 slippage: float,442 allow_short: bool,443 n_permutations: int,444 perm_method: str,445 wf_folds: int,446 progress=gr.Progress(),447):448 """Main Lab handler. Never raises into the UI — it returns a readable message."""449 try:450 cfg = LabConfig(451 symbol=symbol or "SPY",452 start=start or "2015-01-01",453 end=end or None,454 source="yahoo",455 strategy=strategy_key,456 params=collect_params(strategy_key, p1, p2, p3),457 commission_bps=float(commission),458 slippage_bps=float(slippage),459 allow_short=bool(allow_short),460 n_permutations=int(n_permutations),461 permutation_method="block" if perm_method.startswith("Block") else "permute",462 wf_folds=int(wf_folds),463 )464 report = run_lab(cfg, progress=lambda f, m: progress(f, desc=m))465 except Exception as exc: # noqa: BLE001 - the UI must always say something useful466 logger.exception("Lab run failed")467 message = f'<div class="score-card"><div class="score-body"><h3>Could not run that</h3><p>{exc}</p></div></div>'468 blank = empty_figure("No results.")469 return message, "", blank, blank, blank, blank, blank, blank, ""470 471 return (472 _score_card(report),473 _tiles(report),474 permutation_chart(report),475 equity_chart(report),476 drawdown_chart(report),477 exposure_chart(report),478 walkforward_chart(report),479 score_chart(report.verdict),480 _detail_markdown(report),481 )482 483 484def race(symbol: str, start: str, allow_short: bool, n_permutations: int, progress=gr.Progress()):485 try:486 cfg = LabConfig(487 symbol=symbol or "SPY",488 start=start or "2015-01-01",489 source="yahoo",490 allow_short=bool(allow_short),491 )492 table, market, _ = run_arena(493 cfg,494 n_permutations=int(n_permutations),495 progress=lambda f, m: progress(f, desc=m),496 )497 except Exception as exc: # noqa: BLE001498 logger.exception("Arena run failed")499 return pd.DataFrame({"Error": [str(exc)]}), empty_figure("No results."), ""500 501 display = table.copy()502 for col in ("Return", "CAGR", "MaxDD"):503 display[col] = display[col].map(lambda v: f"{v * 100:,.1f}%")504 for col in ("Sharpe", "DSR", "Evidence"):505 display[col] = display[col].map(lambda v: f"{v:.2f}")506 display["p-value"] = display["p-value"].map(lambda v: "—" if v != v else f"{v:.3f}")507 display = display.drop(columns=["key"])508 509 provenance = (510 f"Data: {market.source} · {market.symbol} · {market.start.date()} to {market.end.date()}"511 if market.is_real512 else f"⚠ {market.note}"513 )514 note = (515 f"{provenance}\n\nRanked by **evidence** — `(1 − p) × deflated Sharpe` — not by return. "516 "*Buy & Hold* and *Coin Flip* are in the field on purpose: a leaderboard without a "517 "control group is marketing, not measurement."518 )519 return display, arena_chart(table), note520 521 522HOW_IT_WORKS = """523## Why most backtests are wrong524 525A backtest is a measurement taken with a ruler you built after seeing the thing you526are measuring. Four failure modes do almost all the damage, and this Space tests for527each one.528 529### 1. The market had no structure to find — permutation test530 531We take the real price series and shuffle it: each bar's gap, high, low, body and532volume are kept intact, but their **order** is destroyed. The result is a market with533the same volatility and the same fat tails, and no exploitable structure whatsoever.534Then we re-run *your exact rule* on hundreds of these shuffled markets.535 536If your Sharpe sits inside that cloud of results, your rule found nothing that a537coin-flip market would not also have handed it. The **p-value** is the share of538shuffled markets that did as well or better.539 540*Block mode* resamples contiguous chunks instead of single bars, preserving541short-horizon momentum and volatility clustering. It is a harder null, and trend542strategies should be held to it.543 544### 2. You tried 200 things and reported the best — Deflated Sharpe Ratio545 546If you test 200 worthless strategies, the best of them will show a Sharpe near 1.0547purely by chance. The **Deflated Sharpe Ratio** (Bailey & López de Prado, 2014) works548out what the luckiest of *N* skill-free variants would have scored, and asks whether549yours beats that bar — with an extra penalty for negative skew and fat tails, the550return shapes that flatter naive Sharpe ratios.551 552This Space counts the whole parameter grid as trials, because that is what a553researcher would really have run.554 555### 3. The parameters were fitted to the past — PBO and walk-forward556 557**Probability of Backtest Overfitting** (CSCV) cuts the timeline into chunks, and for558every way of splitting them half in-sample and half out-of-sample, checks whether the559in-sample winner stayed a winner. If the winner lands in the bottom half about half560the time, PBO ≈ 50% and your selection process has no skill at all.561 562**Walk-forward** re-tunes on a training window and trades the next window blind,563rolling forward. Efficiency is out-of-sample Sharpe over in-sample Sharpe: 100% means564the edge survived intact, 0% means it was entirely curve-fit.565 566### 4. The edge is smaller than the costs — stress test567 568Every result here is net of commission and slippage charged on exposure changes, plus569a borrow fee on short positions. We then re-run at **triple** the friction. A real edge570degrades; a fake one disappears.571 572---573 574## The Reality Score575 576| Weight | Component | What it measures |577|---:|---|---|578| 30% | Significance | How far outside the shuffled-market null the result sits |579| 25% | Selection | Deflated Sharpe — does it clear the best-of-N bar |580| 20% | Walk-forward | How much of the tuned Sharpe survived trading forward |581| 15% | Overfitting | 1 − PBO, from combinatorially symmetric cross-validation |582| 10% | Robustness | Sharpe retained when costs triple |583 584Grades: **A** ≥ 85 · **B** ≥ 70 · **C** ≥ 55 · **D** ≥ 40 · **F** below 40.585 586The scale is deliberately harsh. Most strategies people post online score below 40,587and the honest response to that is not to soften the scale.588 589## No look-ahead, by construction590 591A strategy emits a target exposure at each bar's close using only data up to that592bar. The engine holds `position[t] = target[t - lag]` with `lag ≥ 1`, so a signal593computed on Tuesday's close cannot earn Tuesday's move. That is the single line where594look-ahead could enter, and the test suite asserts it directly.595 596## Use it from Python597 598```python599from algotrader import LabConfig, run_lab600 601report = run_lab(LabConfig(symbol="SPY", strategy="sma_cross", params={"fast": 20, "slow": 100}))602print(report.verdict["grade"], report.verdict["score"])603print(report.permutation.p_value, report.dsr["dsr"], report.pbo["pbo"])604```605 606Or from the command line:607 608```bash609python -m algotrader.cli lab --symbol SPY --strategy donchian_breakout --permutations 500610python -m algotrader.cli arena --symbol BTC-USD611```612 613---614 615*Research tooling, not investment advice. Nothing here is a recommendation to trade.*616"""617 618 619# Gradio 6 moved `css` and `theme` from the Blocks constructor to launch().620# Spaces pin their own version, so pass them wherever the installed one wants.621_GRADIO_MAJOR = int(gr.__version__.split(".")[0])622_STYLE_KWARGS = {"css": CSS, "theme": gr.themes.Base()}623_BLOCKS_KWARGS = {} if _GRADIO_MAJOR >= 6 else _STYLE_KWARGS624# Gradio 6 also dropped launch(show_api=...).625_LAUNCH_KWARGS = dict(_STYLE_KWARGS) if _GRADIO_MAJOR >= 6 else {"show_api": False}626 627 628def build_app() -> gr.Blocks:629 with gr.Blocks(title="Backtest Reality Check", **_BLOCKS_KWARGS) as demo:630 with gr.Column(elem_id="hero"):631 gr.HTML(632 "<h1>Backtest Reality Check</h1>"633 "<p>Your backtest is probably lying to you. Pick a market and a trading rule — "634 "this runs it, then spends the rest of its effort trying to prove the result "635 "was luck.</p>"636 )637 638 with gr.Tabs():639 with gr.Tab("The Lab"):640 with gr.Row():641 with gr.Column(scale=1):642 symbol = gr.Dropdown(643 choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker",644 allow_custom_value=True,645 info="Any Yahoo Finance symbol. Falls back to a simulated market if offline.",646 )647 with gr.Row():648 start = gr.Textbox(value="2015-01-01", label="Start", scale=1)649 end = gr.Textbox(value="", label="End (blank = today)", scale=1)650 651 strategy = gr.Dropdown(652 choices=STRATEGY_CHOICES, value="sma_cross", label="Strategy"653 )654 strategy_note = gr.Markdown(get_strategy("sma_cross").description)655 656 param_sliders = [657 gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)658 for i in range(MAX_PARAMS)659 ]660 661 with gr.Accordion("Costs and testing", open=False):662 commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps per trade)")663 slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps per trade)")664 allow_short = gr.Checkbox(value=True, label="Allow short positions")665 n_perms = gr.Slider(666 0, 1000, value=250, step=50, label="Shuffled markets to test against",667 info="More is stricter and slower. 250 is plenty for a first look.",668 )669 perm_method = gr.Radio(670 ["Shuffle bars (standard)", "Block bootstrap (harder)"],671 value="Shuffle bars (standard)", label="Null market",672 )673 wf_folds = gr.Slider(2, 8, value=5, step=1, label="Walk-forward folds")674 675 run_button = gr.Button("Run reality check", variant="primary", size="lg")676 677 gr.Examples(678 label="Or try one of these",679 examples=[680 ["SPY", "sma_cross"],681 ["BTC-USD", "donchian_breakout"],682 ["NVDA", "rsi_reversion"],683 ["QQQ", "momentum"],684 ["SPY", "coin_flip"],685 ],686 inputs=[symbol, strategy],687 )688 689 with gr.Column(scale=2):690 verdict_html = gr.HTML(691 '<div class="score-card"><div class="score-body">'692 "<h3>Nothing tested yet</h3><p>Pick a market and a rule, then hit "693 "<b>Run reality check</b>. A full run is a few seconds.</p>"694 "</div></div>"695 )696 tiles_html = gr.HTML("")697 698 # The headline test gets the full width — it is the whole point.699 perm_plot = gr.Plot(value=empty_figure("The headline test appears here.", height=320))700 equity_plot = gr.Plot(value=empty_figure())701 with gr.Row():702 dd_plot = gr.Plot(value=empty_figure(height=240))703 exposure_plot = gr.Plot(value=empty_figure(height=200))704 with gr.Row():705 wf_plot = gr.Plot(value=empty_figure(height=300))706 components_plot = gr.Plot(value=empty_figure(height=260))707 detail_md = gr.Markdown("")708 709 strategy.change(710 fn=param_controls, inputs=strategy, outputs=param_sliders711 ).then(712 fn=lambda k: get_strategy(k).description, inputs=strategy, outputs=strategy_note713 )714 715 run_button.click(716 fn=analyse,717 inputs=[718 symbol, start, end, strategy, *param_sliders,719 commission, slippage, allow_short, n_perms, perm_method, wf_folds,720 ],721 outputs=[722 verdict_html, tiles_html, perm_plot, equity_plot,723 dd_plot, exposure_plot, wf_plot, components_plot, detail_md,724 ],725 )726 727 with gr.Tab("Portfolio"):728 gr.Markdown(729 "Cross-sectional strategies rank names against each other, so they get a "730 "harder null: we keep every date's gross exposure, net exposure and position "731 "count exactly as they were and randomise only **which name got which "732 "weight**. A book that beats that is picking names. One that doesn't was "733 "being paid for style exposure you can buy in an ETF — which the factor "734 "regression below measures directly."735 )736 with gr.Row():737 with gr.Column(scale=1):738 xs_symbols = gr.Textbox(739 value=", ".join(PORTFOLIO_UNIVERSE),740 label="Universe",741 lines=3,742 info="Comma-separated. At least 4 names — fewer cannot be ranked.",743 )744 with gr.Row():745 xs_start = gr.Textbox(value="2015-01-01", label="Start", scale=1)746 xs_end = gr.Textbox(value="", label="End", scale=1)747 748 xs_strategy = gr.Dropdown(749 choices=XS_STRATEGY_CHOICES, value="xs_momentum", label="Strategy"750 )751 xs_note = gr.Markdown(get_xs_strategy("xs_momentum").description)752 xs_sliders = [753 gr.Slider(label=f"Parameter {i + 1}", visible=False, minimum=0, maximum=100)754 for i in range(MAX_XS_PARAMS)755 ]756 xs_rebalance = gr.Radio(757 ["D", "W", "M", "Q"], value="M", label="Rebalance",758 info="Daily rebalancing of a real book is rarely affordable.",759 )760 761 with gr.Accordion("Costs, limits and testing", open=False):762 xs_commission = gr.Slider(0, 20, value=1, step=0.5, label="Commission (bps)")763 xs_slippage = gr.Slider(0, 50, value=2, step=0.5, label="Slippage (bps)")764 xs_short = gr.Checkbox(value=True, label="Allow short positions")765 xs_leverage = gr.Slider(0.1, 3.0, value=1.0, step=0.1, label="Gross leverage cap")766 xs_maxw = gr.Slider(0.0, 1.0, value=0.25, step=0.05, label="Max weight per name")767 xs_perms = gr.Slider(768 0, 500, value=150, step=25, label="Name shuffles",769 )770 771 xs_button = gr.Button("Run portfolio check", variant="primary", size="lg")772 773 with gr.Column(scale=2):774 xs_verdict = gr.HTML(775 '<div class="score-card"><div class="score-body">'776 "<h3>Nothing tested yet</h3><p>Pick a universe and a ranking rule, then "777 "hit <b>Run portfolio check</b>.</p></div></div>"778 )779 xs_tiles = gr.HTML("")780 781 xs_perm_plot = gr.Plot(value=empty_figure("The name-shuffle test appears here.", height=320))782 xs_equity_plot = gr.Plot(value=empty_figure())783 with gr.Row():784 xs_attr_plot = gr.Plot(value=empty_figure(height=280))785 xs_weights_plot = gr.Plot(value=empty_figure(height=240))786 xs_components_plot = gr.Plot(value=empty_figure(height=260))787 xs_detail = gr.Markdown("")788 789 xs_strategy.change(790 fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders791 ).then(792 fn=lambda k: get_xs_strategy(k).description, inputs=xs_strategy, outputs=xs_note793 )794 xs_button.click(795 fn=analyse_portfolio,796 inputs=[797 xs_symbols, xs_start, xs_end, xs_strategy, *xs_sliders,798 xs_rebalance, xs_commission, xs_slippage, xs_short,799 xs_leverage, xs_maxw, xs_perms,800 ],801 outputs=[802 xs_verdict, xs_tiles, xs_perm_plot, xs_equity_plot,803 xs_attr_plot, xs_weights_plot, xs_components_plot, xs_detail,804 ],805 )806 807 with gr.Tab("Arena"):808 gr.Markdown(809 "Race every strategy on the same market, ranked by **evidence** rather than "810 "return. Buy & hold and a coin flip stay in the field as controls."811 )812 with gr.Row():813 arena_symbol = gr.Dropdown(814 choices=DEFAULT_UNIVERSE, value="SPY", label="Ticker", allow_custom_value=True815 )816 arena_start = gr.Textbox(value="2015-01-01", label="Start")817 arena_short = gr.Checkbox(value=True, label="Allow shorts")818 arena_perms = gr.Slider(0, 400, value=120, step=20, label="Shuffled markets per strategy")819 arena_button = gr.Button("Run the arena", variant="primary")820 arena_note = gr.Markdown("")821 arena_table = gr.Dataframe(interactive=False, wrap=True)822 arena_plot = gr.Plot(value=empty_figure(height=380))823 824 arena_button.click(825 fn=race,826 inputs=[arena_symbol, arena_start, arena_short, arena_perms],827 outputs=[arena_table, arena_plot, arena_note],828 )829 830 with gr.Tab("How it works"):831 gr.Markdown(HOW_IT_WORKS)832 833 gr.Markdown(834 f"<sub>algotrader {__version__} · Apache-2.0 · "835 "Research tooling, not investment advice.</sub>"836 )837 838 demo.load(fn=param_controls, inputs=strategy, outputs=param_sliders)839 demo.load(fn=xs_param_controls, inputs=xs_strategy, outputs=xs_sliders)840 841 return demo842 843 844if __name__ == "__main__":845 build_app().queue(max_size=24).launch(846 server_name="0.0.0.0",847 server_port=int(os.environ.get("PORT", 7860)),848 **_LAUNCH_KWARGS,849 )850 