"""log-returns-vs-simple-returns.py — simple versus log returns on FRED's S&P 500 series, and the errors that come from mixing them. Input : datasets/log-returns-vs-simple-returns.csv - the derived daily simple return series computed from FRED's SP500 daily close (FRED's ten-year rolling window) as it stood at the 2026-09-05 retrieval. The index levels themselves are not shipped. S&P Dow Jones Indices LLC states in the FRED series notes that "Reproduction of S&P 500 in any form is prohibited except with the prior written permission of S&P Dow Jones Indices LLC ("S&P")", so this file carries only the returns computed from the series, which are this site's own output. Source: S&P Dow Jones Indices LLC, https://fred.stlouisfed.org/series/SP500. Run with --download to pull SP500 from FRED and rebuild the return file end to end. FRED's ten-year window rolls forward, so a pull today covers a later sample than the article reports. Method: simple r_t = P_t/P_{t-1} - 1; log l_t = ln(P_t/P_{t-1}). Shows: (1) sum of log returns equals ln(P_T/P_0) exactly while the sum of simple returns does not; (2) mean simple return times 252 versus the geometric annual return; (3) the gap between simple and log grows with the size of the move; (4) the -50% then +100% asymmetry in numbers. Run : python code/log-returns-vs-simple-returns.py [--download] Needs : Python 3.13, pandas 3.0.2, numpy 2.3. """ import io import sys import urllib.request import numpy as np import pandas as pd CSV = "datasets/log-returns-vs-simple-returns.csv" def download(): """Pull SP500 from FRED and write only the derived returns; the index levels are never stored.""" req = urllib.request.Request("https://fred.stlouisfed.org/graph/fredgraph.csv?id=SP500", headers={"User-Agent": "prism-data-lab/1.0", "Accept": "*/*"}) txt = urllib.request.urlopen(req, timeout=60).read().decode() s = pd.read_csv(io.StringIO(txt), na_values=".", index_col=0).iloc[:, 0] s.index.name = "date" r = s.dropna().pct_change().rename("return_simple") r.reindex(s.index).to_frame().to_csv(CSV) def main(): if "--download" in sys.argv: download() col = pd.read_csv(CSV, parse_dates=["date"], index_col="date")["return_simple"] r = col.dropna() l = np.log1p(r) # ln(P_t/P_{t-1}) written in terms of the simple return n = len(r) years = n / 252 total = float((1.0 + r).prod() - 1.0) print(f"sample {col.index[0]:%Y-%m-%d} to {col.index[-1]:%Y-%m-%d}, {n} daily returns, {years:.2f} trading years") print("| quantity | simple returns | log returns |") print("|---|---|---|") print(f"| sum of daily returns | {100 * r.sum():.2f}% | {100 * l.sum():.2f}% |") print(f"| implied total return from the sum | {100 * r.sum():.2f}% (wrong) | {100 * (np.exp(l.sum()) - 1):.2f}% (exact) |") print(f"| actual total return P_T/P_0 - 1 | {100 * total:.2f}% | {100 * total:.2f}% |") print(f"| mean daily return x 252 | {100 * r.mean() * 252:.2f}% | {100 * l.mean() * 252:.2f}% |") print(f"| geometric annual return | {100 * ((1 + total) ** (1 / years) - 1):.2f}% | {100 * (np.exp(l.mean() * 252) - 1):.2f}% |") print(f"| daily std (ddof=1) | {100 * r.std(ddof=1):.3f}% | {100 * l.std(ddof=1):.3f}% |") print(f"| skewness | {r.skew():.2f} | {l.skew():.2f} |") print(f"| largest gap simple minus log on one day | {100 * (r - l).max():.3f} pp on {(r - l).idxmax():%Y-%m-%d} " f"(simple {100 * r[(r - l).idxmax()]:.2f}%) | |") print(f"| variance drag: mean simple - mean log (annualised) | {100 * (r.mean() - l.mean()) * 252:.2f} pp | " f"half the annual variance = {100 * 0.5 * r.var(ddof=1) * 252:.2f} pp |") print("\n| move | simple return | log return | gap (pp) |") print("|---|---|---|---|") for m in (0.001, 0.01, 0.05, 0.10, 0.25, 0.50, -0.10, -0.25, -0.50): print(f"| {100 * m:+.1f}% | {100 * m:+.2f}% | {100 * np.log1p(m):+.2f}% | {100 * (m - np.log1p(m)):+.2f} |") print("\nasymmetry: a -50% day followed by +50% leaves", f"{(1 - 0.5) * (1 + 0.5):.2f} of the start;", "the log returns are", f"{np.log(0.5):+.4f} and {np.log(1.5):+.4f}, summing to {np.log(0.5) + np.log(1.5):+.4f}") worst = r.idxmin() print(f"worst day in sample {worst:%Y-%m-%d}: simple {100 * r[worst]:.2f}%, log {100 * l[worst]:.2f}%; " f"recovery needed in simple terms {100 * (1 / (1 + r[worst]) - 1):.2f}%") if __name__ == "__main__": main()