"""backtest-overfitting-deflated-sharpe.py — simulate strategy selection from noise and apply the deflated Sharpe ratio. What it does: draws N strategies of T daily returns with TRUE mean zero (pure noise, unit daily volatility scaled to a 16% annual figure), keeps the best in-sample Sharpe ratio, and shows how that 'best' Sharpe grows with N. Then computes the expected maximum Sharpe under the null (Bailey and Lopez de Prado, 'The Deflated Sharpe Ratio', 2014, eq. for E[max SR]) and the deflated Sharpe ratio DSR = Phi( (SR_hat - SR_0) * sqrt(T-1) / sqrt(1 - g3*SR_hat + (g4-1)/4*SR_hat^2) ), where SR_hat is the non-annualised (daily) Sharpe, g3 skewness, g4 kurtosis of the returns, and SR_0 = sqrt(V[SR]) * ((1-gamma) Z^-1(1-1/N) + gamma Z^-1(1-1/(N e))), gamma = 0.5772... Inputs : none (synthetic data, seed fixed). Outputs: two Markdown tables. Run : python code/backtest-overfitting-deflated-sharpe.py Needs : Python 3.13, numpy 2.3, scipy not required (normal CDF/quantile via math.erf and a bisection). """ import math import numpy as np SEED = 20260905 T = 1260 # five years of daily returns DAILY_VOL = 0.16 / math.sqrt(252) TRIALS = (1, 10, 100, 1000, 10000) REPS = 200 # repetitions per N to average the observed maximum GAMMA = 0.5772156649 def norm_cdf(x): return 0.5 * (1.0 + math.erf(x / math.sqrt(2.0))) def norm_ppf(p): lo, hi = -10.0, 10.0 for _ in range(200): mid = (lo + hi) / 2 if norm_cdf(mid) < p: lo = mid else: hi = mid return (lo + hi) / 2 def sharpe_daily(r): return r.mean() / r.std(ddof=1) def expected_max_sr(n, var_sr): if n == 1: return 0.0 return math.sqrt(var_sr) * ((1 - GAMMA) * norm_ppf(1 - 1 / n) + GAMMA * norm_ppf(1 - 1 / (n * math.e))) def dsr(sr_hat, sr0, t, skew, kurt): z = (sr_hat - sr0) * math.sqrt(t - 1) / math.sqrt(1 - skew * sr_hat + (kurt - 1) / 4 * sr_hat ** 2) return norm_cdf(z) def moments(r): m = r.mean() s = r.std(ddof=0) return ((r - m) ** 3).mean() / s ** 3, ((r - m) ** 4).mean() / s ** 4 def main(): rng = np.random.default_rng(SEED) print(f"T = {T} daily returns, true mean 0, annual vol 16%, {REPS} repetitions per N, seed {SEED}") print("| strategies tried N | mean of best in-sample Sharpe (annualised) | expected max under the null (formula) | " "share of runs where best Sharpe > 1.0 | DSR of the best strategy (mean) |") print("|---|---|---|---|---|") for n in TRIALS: best, dsrs = [], [] var_sr = None for _ in range(REPS): rets = rng.normal(0.0, DAILY_VOL, size=(n, T)) srs = rets.mean(axis=1) / rets.std(axis=1, ddof=1) k = int(np.argmax(srs)) sr_hat = float(srs[k]) # variance of the Sharpe estimates across the trials (the paper's V[SR_n]); 1/T when N == 1 var_sr = float(srs.var(ddof=1)) if n > 1 else 1.0 / T sk, ku = moments(rets[k]) sr0 = expected_max_sr(n, var_sr) best.append(sr_hat * math.sqrt(252)) dsrs.append(dsr(sr_hat, sr0, T, sk, ku)) print(f"| {n} | {np.mean(best):.2f} | {expected_max_sr(n, var_sr) * math.sqrt(252):.2f} | " f"{100 * np.mean(np.array(best) > 1.0):.0f}% | {np.mean(dsrs):.2f} |") # a single worked example the article walks through n = 1000 rets = rng.normal(0.0, DAILY_VOL, size=(n, T)) srs = rets.mean(axis=1) / rets.std(axis=1, ddof=1) k = int(np.argmax(srs)) sr_hat = float(srs[k]) sk, ku = moments(rets[k]) var_sr = float(srs.var(ddof=1)) sr0 = expected_max_sr(n, var_sr) print("\nworked example, N = 1000:") print(f" best daily Sharpe {sr_hat:.4f} (annualised {sr_hat * math.sqrt(252):.2f}); skew {sk:.3f}; kurtosis {ku:.3f}") print(f" variance of the 1000 Sharpe estimates {var_sr:.3e} (1/T would be {1 / T:.3e})") print(f" expected max Sharpe under the null SR0 = {sr0:.4f} daily ({sr0 * math.sqrt(252):.2f} annualised)") print(f" DSR = {dsr(sr_hat, sr0, T, sk, ku):.3f}; the naive p-value against zero (ignoring selection) would be " f"{1 - norm_cdf(sr_hat * math.sqrt(T - 1)):.4f}") print(f" the same statistic with N = 1 (no selection) would give DSR {dsr(sr_hat, 0.0, T, sk, ku):.3f}") if __name__ == "__main__": main()