"""alfred-vintages-data-revisions.py — how much FRED's "final" numbers moved after first publication, measured from ALFRED's archived vintages of two series: total nonfarm payrolls (PAYEMS) and real GDP (GDPC1). Input : the ALFRED archive (https://alfred.stlouisfed.org), no API key. Two public endpoints are used: https://alfred.stlouisfed.org/series/downloaddata?seid= (HTML page; its vintage-date ', html, re.S) if not block: sys.exit(f"could not find the vintage list on the ALFRED page for {series}") return re.findall(r'value="(\d{4}-\d{2}-\d{2})"', block.group(0)) def download(series: str) -> None: allv = vintage_dates(series) want = [v for v in allv if v >= VINTAGES_FROM[series]] print(f"{series}: ALFRED lists {len(allv)} vintages ({allv[0]} to {allv[-1]}); fetching {len(want)}") table: dict[str, dict[str, str]] = {} cols: list[str] = [] for i in range(0, len(want), BATCH): chunk = want[i:i + BATCH] url = ("https://alfred.stlouisfed.org/graph/alfredgraph.csv?id=" + ",".join([series] * len(chunk)) + "&vintage_date=" + ",".join(chunk)) rows = list(csv.reader(get(url).splitlines())) head = rows[0][1:] expect = [f"{series}_{d.replace('-', '')}" for d in chunk] if head != expect: sys.exit(f"unexpected columns {head} for {expect}") cols += head for row in rows[1:]: if not row or row[0] < OBS_FROM: continue rec = table.setdefault(row[0], {}) for name, val in zip(head, row[1:]): rec[name] = "" if val in (".", "#N/A") else val time.sleep(1.5) with open(RAW[series], "w", encoding="utf-8", newline="") as f: w = csv.writer(f) w.writerow(["observation_date"] + cols) for d in sorted(table): w.writerow([d] + [table[d].get(c, "") for c in cols]) print(f" wrote {RAW[series]} ({len(table)} observations x {len(cols)} vintages, " f"retrieved {datetime.date.today().isoformat()})") # ------------------------------------------------------------------ load --- def load(series: str) -> tuple[list[str], list[str], dict[str, dict[str, float]]]: """(observation dates, vintage dates, {vintage: {observation: value}})""" with open(RAW[series], encoding="utf-8", newline="") as f: rows = list(csv.reader(f)) vint = [c.split("_")[1] for c in rows[0][1:]] vint = [f"{v[:4]}-{v[4:6]}-{v[6:]}" for v in vint] obs = [r[0] for r in rows[1:]] data: dict[str, dict[str, float]] = {v: {} for v in vint} for r in rows[1:]: for v, x in zip(vint, r[1:]): if x != "": data[v][r[0]] = float(x) return obs, vint, data def prev_period(d: str, months: int) -> str: y, m = int(d[:4]), int(d[5:7]) - months while m < 1: m += 12 y -= 1 return f"{y:04d}-{m:02d}-01" def estimates(obs: str, vint: list[str], data, value) -> list[tuple[str, float]]: """every (vintage, value) in publication order for vintages that contain obs""" out = [] for v in vint: x = value(data[v], obs) if x is not None: out.append((v, x)) return out def summarise(revs: list[float]) -> tuple[float, float, float]: return (statistics.fmean(revs), statistics.fmean(abs(r) for r in revs), statistics.pstdev(revs)) # ------------------------------------------------------------------ payrolls --- def payrolls(): obs, vint, data = load("PAYEMS") current = vint[-1] def change(series_v, m): p = prev_period(m, 1) if m in series_v and p in series_v: return series_v[m] - series_v[p] return None months = [m for m in obs if m >= "2010-01-01" and m in data[current]] rows = [] for m in months: est = estimates(m, vint, data, change) first_v, first = est[0] third_v, third = est[2] if len(est) >= 3 else ("", None) cur = est[-1][1] rows.append({"month": m[:7], "first_vintage": first_v, "first_change": first, "third_vintage": third_v, "third_change": third, "current_vintage": current, "current_change": cur, "n_vintages": len(est)}) print(f"PAYEMS: {len(vint)} vintages in the file ({vint[0]} to {current}); {len(months)} reference months " f"{months[0][:7]} to {months[-1][:7]}") with_third = [r for r in rows if r["third_change"] is not None] no2020 = [r for r in with_third if not r["month"].startswith("2020")] for label, sample in (("all months", with_third), ("excluding 2020", no2020)): r13 = [r["third_change"] - r["first_change"] for r in sample] r1c = [r["current_change"] - r["first_change"] for r in sample] m13, a13, s13 = summarise(r13) m1c, a1c, s1c = summarise(r1c) flips = sum(1 for r in sample if r["first_change"] * r["current_change"] < 0) print(f" {label:>15} (n={len(sample)}): first->third mean {m13:+.1f}k, mean abs {a13:.1f}k | " f"first->current mean {m1c:+.1f}k, mean abs {a1c:.1f}k | sign flips first vs current {flips}") flips_all = [r for r in rows if r["first_change"] * r["current_change"] < 0] print(f" sign flips, first print vs current, all {len(rows)} months: {len(flips_all)}") for r in flips_all: print(f" {r['month']}: first {r['first_change']:+.0f}k ({r['first_vintage']}), current " f"{r['current_change']:+.0f}k") big = max(no2020, key=lambda r: abs(r["current_change"] - r["first_change"])) print(f" largest first->current revision outside 2020: {big['month']} first {big['first_change']:+.0f}k " f"current {big['current_change']:+.0f}k ({big['current_change'] - big['first_change']:+.0f}k)") print("\n recent months (thousands): month | first print (vintage) | third (vintage) | current") for r in rows: if r["month"] >= "2025-01": third = f"{r['third_change']:+.0f} ({r['third_vintage']})" if r["third_change"] is not None else "n/a" print(f" {r['month']} | {r['first_change']:+.0f} ({r['first_vintage']}) | {third} | " f"{r['current_change']:+.0f}") print("\n calendar-year payroll change, sum of first prints vs current vintage (thousands)") years = [] for y in range(2010, int(current[:4]) + 1): ys = [r for r in rows if r["month"].startswith(str(y))] if not ys: continue first_sum = sum(r["first_change"] for r in ys) cur_sum = sum(r["current_change"] for r in ys) t3 = [r for r in ys if r["third_change"] is not None] sample_rev = sum(r["third_change"] - r["first_change"] for r in t3) later_rev = sum(r["current_change"] - r["third_change"] for r in t3) years.append((y, len(ys), first_sum, cur_sum)) print(f" {y} ({len(ys):>2} months): first prints {first_sum:+8.0f} | current {cur_sum:+8.0f} | " f"difference {cur_sum - first_sum:+6.0f} | first->third {sample_rev:+6.0f} | " f"third->current {later_rev:+6.0f} ({len(t3)} months with a third vintage)") print("\n annual benchmark revisions to the March level (thousands, SA): year | before (vintage) | " "after (vintage) | revision") bench = [] for y in range(2010, int(current[:4]) + 1): jan = f"{y}-01-01" after = next((v for v in vint if jan in data[v]), None) if after is None or vint.index(after) == 0: continue before = vint[vint.index(after) - 1] march = f"{y - 1}-03-01" lb, la = data[before][march], data[after][march] bench.append({"benchmark_year": y, "march_reference": march[:7], "vintage_before": before, "vintage_after": after, "level_before": lb, "level_after": la, "revision": la - lb, "revision_pct": 100 * (la - lb) / lb}) print(f" {y} | Mar {y - 1}: {lb:,.0f} ({before}) | {la:,.0f} ({after}) | {la - lb:+,.0f} " f"({100 * (la - lb) / lb:+.2f}%)") neg = sum(1 for b in bench if b["revision"] < 0) print(f" {len(bench)} benchmarks, {neg} negative; mean absolute {statistics.fmean(abs(b['revision']) for b in bench):.0f}k") with open(OUT_MONTHS, "w", encoding="utf-8", newline="") as f: w = csv.writer(f) w.writerow(["month", "first_vintage", "first_change_k", "third_vintage", "third_change_k", "current_vintage", "current_change_k", "rev_first_to_third_k", "rev_first_to_current_k", "n_vintages"]) for r in rows: t = r["third_change"] w.writerow([r["month"], r["first_vintage"], f"{r['first_change']:.0f}", r["third_vintage"], "" if t is None else f"{t:.0f}", r["current_vintage"], f"{r['current_change']:.0f}", "" if t is None else f"{t - r['first_change']:.0f}", f"{r['current_change'] - r['first_change']:.0f}", r["n_vintages"]]) with open(OUT_BENCH, "w", encoding="utf-8", newline="") as f: w = csv.writer(f) w.writerow(["benchmark_year", "march_reference", "vintage_before", "vintage_after", "level_before_k", "level_after_k", "revision_k", "revision_pct"]) for b in bench: w.writerow([b["benchmark_year"], b["march_reference"], b["vintage_before"], b["vintage_after"], f"{b['level_before']:.0f}", f"{b['level_after']:.0f}", f"{b['revision']:.0f}", f"{b['revision_pct']:.3f}"]) return rows, bench # ------------------------------------------------------------------ GDP --- def gdp(): obs, vint, data = load("GDPC1") current = vint[-1] def growth(series_v, q): p = prev_period(q, 3) if q in series_v and p in series_v: return 100 * ((series_v[q] / series_v[p]) ** 4 - 1) return None quarters = [q for q in obs if q >= "2010-04-01" and q in data[current]] rows = [] for q in quarters: est = estimates(q, vint, data, growth) if est[0][0] < VINTAGES_FROM["GDPC1"]: continue adv_v, adv = est[0] third_v, third = est[2] if len(est) >= 3 else ("", None) rows.append({"quarter": f"{q[:4]}Q{(int(q[5:7]) - 1) // 3 + 1}", "advance_vintage": adv_v, "advance": adv, "third_vintage": third_v, "third": third, "current_vintage": current, "current": est[-1][1], "n_vintages": len(est)}) print(f"\nGDPC1: {len(vint)} vintages in the file ({vint[0]} to {current}); quarters " f"{rows[0]['quarter']} to {rows[-1]['quarter']}") with_third = [r for r in rows if r["third"] is not None] no2020 = [r for r in with_third if not r["quarter"].startswith("2020")] for label, sample in (("all quarters", with_third), ("excluding 2020", no2020)): ra3 = [r["third"] - r["advance"] for r in sample] rac = [r["current"] - r["advance"] for r in sample] m3, a3, _ = summarise(ra3) mc, ac, sc = summarise(rac) flips = sum(1 for r in sample if r["advance"] * r["current"] < 0) within = sum(1 for x in rac if abs(x) <= 1.0) print(f" {label:>15} (n={len(sample)}): advance->third mean {m3:+.2f}pp, mean abs {a3:.2f}pp | " f"advance->current mean {mc:+.2f}pp, mean abs {ac:.2f}pp, sd {sc:.2f}pp, " f"within 1pp {within} | sign flips {flips}") for r in sample: if r["advance"] * r["current"] < 0: print(f" sign flip {r['quarter']}: advance {r['advance']:+.1f} ({r['advance_vintage']}) " f"current {r['current']:+.1f}") big = max(no2020, key=lambda r: abs(r["current"] - r["advance"])) print(f" largest advance->current revision outside 2020: {big['quarter']} advance {big['advance']:+.2f} " f"current {big['current']:+.2f}") print("\n recent quarters (annualised %): quarter | advance (vintage) | third vintage (date) | current") for r in rows: if r["quarter"] >= "2024Q1": t = f"{r['third']:+.1f} ({r['third_vintage']})" if r["third"] is not None else "n/a" print(f" {r['quarter']} | {r['advance']:+.1f} ({r['advance_vintage']}) | {t} | {r['current']:+.1f}" f" | {r['n_vintages']} vintages") # the level problem: one fixed quarter's level across every vintage fixed = "2015-01-01" lv = [(v, data[v][fixed]) for v in vint if fixed in data[v]] jumps = sorted(((100 * (b[1] / a[1] - 1), b[0], a[1], b[1]) for a, b in zip(lv, lv[1:])), key=lambda t: -abs(t[0]))[:3] print(f"\n level of 2015Q1 across {len(lv)} vintages: first {lv[0][1]:,.1f} ({lv[0][0]}), " f"current {lv[-1][1]:,.1f} ({lv[-1][0]}); largest single-vintage moves:") for pct, v, a, b in jumps: print(f" {v}: {a:,.1f} -> {b:,.1f} ({pct:+.2f}%)") g_first = growth(data[lv[0][0]], fixed) g_cur = growth(data[current], fixed) print(f" 2015Q1 growth: {g_first:+.2f}% in its first vintage here, {g_cur:+.2f}% now") with open(OUT_GDP, "w", encoding="utf-8", newline="") as f: w = csv.writer(f) w.writerow(["quarter", "advance_vintage", "advance_growth_saar_pct", "third_vintage", "third_vintage_growth_saar_pct", "current_vintage", "current_growth_saar_pct", "rev_advance_to_third_pp", "rev_advance_to_current_pp", "n_vintages"]) for r in rows: t = r["third"] w.writerow([r["quarter"], r["advance_vintage"], f"{r['advance']:.3f}", r["third_vintage"], "" if t is None else f"{t:.3f}", r["current_vintage"], f"{r['current']:.3f}", "" if t is None else f"{t - r['advance']:.3f}", f"{r['current'] - r['advance']:.3f}", r["n_vintages"]]) return rows # ------------------------------------------------------------------ figures --- def figures(rows, bench) -> None: import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt os.makedirs(IMG_DIR, exist_ok=True) paper, ink, soft, grid = "#ffffff", "#1b2430", "#5e646b", "#e3e2de" c1, c2 = "#2a78d6", "#eb6834" # fixed categorical order: first print, current vintage plt.rcParams.update({"font.family": "DejaVu Sans", "font.size": 10, "axes.edgecolor": grid, "axes.labelcolor": ink, "xtick.color": soft, "ytick.color": soft, "text.color": ink, "axes.spines.top": False, "axes.spines.right": False}) recent = [r for r in rows if r["month"] >= "2023-01"] x = [datetime.date(int(r["month"][:4]), int(r["month"][5:7]), 15) for r in recent] fig, ax = plt.subplots(figsize=(11, 4.6), dpi=120, facecolor=paper) ax.set_facecolor(paper) ax.grid(axis="y", color=grid, linewidth=0.8) ax.axhline(0, color=soft, linewidth=0.9) ax.plot(x, [r["first_change"] for r in recent], color=c1, linewidth=2, marker="o", markersize=4, label="first print (the vintage that first contained the month)") ax.plot(x, [r["current_change"] for r in recent], color=c2, linewidth=2, marker="o", markersize=4, label=f"current vintage ({recent[-1]['current_vintage']})") ax.set_ylabel("monthly change, thousands") ax.set_title("Total nonfarm payrolls, monthly change: as first published and as it stands now", loc="left", fontsize=11) ax.legend(frameon=False, loc="upper right") fig.tight_layout() fig.savefig(os.path.join(IMG_DIR, "payrolls-first-print-vs-current.png"), facecolor=paper) plt.close(fig) fig, ax = plt.subplots(figsize=(11, 4.2), dpi=120, facecolor=paper) ax.set_facecolor(paper) ax.grid(axis="y", color=grid, linewidth=0.8) ax.axhline(0, color=soft, linewidth=0.9) yrs = [b["benchmark_year"] for b in bench] revs = [b["revision"] for b in bench] ax.bar(yrs, revs, color=c1, width=0.7) for yv, rv in zip(yrs, revs): if abs(rv) >= 400: ax.annotate(f"{rv:+,.0f}", (yv, rv), textcoords="offset points", xytext=(0, -12 if rv < 0 else 4), ha="center", fontsize=9, color=ink) ax.set_xticks(yrs) ax.set_xticklabels([str(y) for y in yrs], rotation=0, fontsize=9) ax.set_ylabel("revision to prior March level, thousands") ax.set_title("Annual payroll benchmark revisions, by year published (PAYEMS, seasonally adjusted)", loc="left", fontsize=11) fig.tight_layout() fig.savefig(os.path.join(IMG_DIR, "payroll-benchmark-revisions.png"), facecolor=paper) plt.close(fig) print(f"\nwrote 2 figures to {IMG_DIR}") if __name__ == "__main__": if "--download" in sys.argv: for s in ("PAYEMS", "GDPC1"): download(s) payroll_rows, benchmarks = payrolls() gdp() if "--figures" in sys.argv: figures(payroll_rows, benchmarks)