"""sahm-rule-unemployment-recessions.py — compute the Sahm rule from UNRATE and compare it with NBER recession starts. Input : datasets/sahm-rule-unemployment-recessions.csv (FRED UNRATE %, USREC 0/1, SAHMREALTIME pp; monthly), retrieved 2026-09-05 via the public fredgraph CSV. Re-download with --download. Method: sahm_t = mean(UNRATE over t, t-1, t-2) - min over the previous 12 months of that 3-month mean. The rule signals when sahm_t >= 0.50 (the definition on the SAHMREALTIME series page). This uses today's revised UNRATE vintage; SAHMREALTIME is the real-time version and is printed alongside as a cross-check. Run : python code/sahm-rule-unemployment-recessions.py [--download] Needs : Python 3.13, pandas 3.0.2. """ import io import sys import urllib.request import pandas as pd CSV = "datasets/sahm-rule-unemployment-recessions.csv" def download(): frames = [] for sid in ("UNRATE", "USREC", "SAHMREALTIME"): req = urllib.request.Request(f"https://fred.stlouisfed.org/graph/fredgraph.csv?id={sid}", headers={"User-Agent": "prism-data-lab/1.0"}) txt = urllib.request.urlopen(req, timeout=60).read().decode() frames.append(pd.read_csv(io.StringIO(txt), na_values=".", index_col=0).iloc[:, 0].rename(sid)) df = pd.concat(frames, axis=1) df.index.name = "date" df[df.index >= "1948-01-01"].to_csv(CSV) def main(): if "--download" in sys.argv: download() df = pd.read_csv(CSV, parse_dates=["date"], index_col="date") u = df["UNRATE"] missing = u[u.isna()] # min_periods=2: a single missing month (FRED has none for 2025-10) averages the two months present m3 = u.rolling(3, min_periods=2).mean() low12 = m3.shift(1).rolling(12, min_periods=11).min() df["sahm"] = (m3 - low12).round(2) rec = df["USREC"] starts = rec.index[(rec == 1) & (rec.shift(1, fill_value=0) == 0)] ends = rec.index[(rec == 0) & (rec.shift(1, fill_value=0) == 1)] print(f"UNRATE months missing in the file: {[f'{d:%Y-%m}' for d in missing.index]}") print("| NBER recession start | first month Sahm >= 0.50 (from 3 months before the start) | lag (months) | Sahm that month | real-time series that month |") print("|---|---|---|---|---|") for s in starts: if s < pd.Timestamp("1960-01-01"): continue window = df.loc[s - pd.DateOffset(months=3): s + pd.DateOffset(months=24)] hit = window.index[window["sahm"] >= 0.5] if len(hit): h = hit[0] lag = (h.year - s.year) * 12 + h.month - s.month rt = df.loc[h, "SAHMREALTIME"] rt = f"{rt:.2f}" if pd.notna(rt) else "n/a" print(f"| {s:%Y-%m} | {h:%Y-%m} | {lag:+d} | {df.loc[h, 'sahm']:.2f} | {rt} |") else: print(f"| {s:%Y-%m} | none within window | - | - | - |") sig = df["sahm"] >= 0.5 print("\nmonths with Sahm >= 0.50 that are outside a recession, not within 12 months before a recession start," " and not within 18 months after a recession end (the rule stays elevated while unemployment is still high):") for d in df.index[sig]: before_start = any(0 <= (s - d).days <= 366 for s in starts) after_end = any(0 <= (d - e).days <= 548 for e in ends) if rec.loc[d] == 0 and not before_start and not after_end: print(f" {d:%Y-%m}: sahm {df.loc[d, 'sahm']:.2f}, UNRATE {df.loc[d, 'UNRATE']:.1f}, real-time {df.loc[d, 'SAHMREALTIME']}") tail = df.loc["2024-01-01":, ["UNRATE", "sahm", "SAHMREALTIME"]] print("\n| month | UNRATE | Sahm (revised data) | SAHMREALTIME |") print("|---|---|---|---|") for d, r in tail.iterrows(): print(f"| {d:%Y-%m} | {r['UNRATE']:.1f} | {r['sahm']:.2f} | {r['SAHMREALTIME']:.2f} |") print(f"\nlatest: {df.index[-1]:%Y-%m}; max Sahm since 2023: {df.loc['2023':, 'sahm'].max():.2f} in {df.loc['2023':, 'sahm'].idxmax():%Y-%m}") diff = (df["sahm"] - df["SAHMREALTIME"]).dropna() print(f"revised-vs-real-time difference: mean abs {diff.abs().mean():.3f} pp, max abs {diff.abs().max():.2f} pp " f"on {diff.abs().idxmax():%Y-%m}, months compared {len(diff)}, from {diff.index[0]:%Y-%m}") if __name__ == "__main__": main()