""" cpi-basket-relative-importance.py - the CPI basket from the BLS relative importance tables, and what each component contributed to the index's change since the weights last reset. What it does 1. Reads BLS Table 1, "Relative importance of components in the Consumer Price Indexes: U.S. city average", for December 2020 through December 2025 (www.bls.gov/cpi/tables/relative-importance/.htm) and keeps 28 rows for CPI-U and CPI-W. Writes cpi-basket-relative-importance.csv. 2. Pulls December 2025 and August 2026 not-seasonally-adjusted index levels for 17 CPI-U series from the BLS public API v1 (no key; 25 series a query) and applies the method on the BLS relative importance page: a component's contribution to the all-items change is its relative importance at the start of the period times its own relative change. December 2025 to August 2026 sits inside one weight period (the December 2025 table uses 2024 weights), so no split is needed. Run python cpi-basket-relative-importance.py (fetches everything, writes the CSV, prints the tables) python cpi-basket-relative-importance.py --offline DIR (reads ri_.htm and bls_api_2025_2026.json from DIR) BLS answers www.bls.gov requests only when the User-Agent names the requester and a contact, so set CONTACT. Requires Python 3.10+ and pandas (read_html, with lxml). Not investment advice; a description of a public index. """ import argparse import io import json import os import sys import urllib.request import pandas as pd CONTACT = "Prism Data Lab research script (contact@prismdatalab.com)" YEARS = range(2020, 2026) ITEMS = ["All items", "Food and beverages", "Food", "Food at home", "Food away from home", "Housing", "Shelter", "Owners' equivalent rent of residences", "Rent of primary residence", "Apparel", "Transportation", "New vehicles", "Used cars and trucks", "Motor vehicle insurance", "Gasoline (all types)", "Medical care", "Medical care services", "Recreation", "Education and communication", "Other goods and services", "Energy", "Energy commodities", "Energy services", "Electricity", "All items less food and energy", "Commodities less food and energy commodities", "Services less energy services", "Transportation services"] SERIES = {"CUUR0000SA0": "All items", "CUUR0000SAF1": "Food", "CUUR0000SAF11": "Food at home", "CUUR0000SEFV": "Food away from home", "CUUR0000SA0E": "Energy", "CUUR0000SACL1E": "Commodities less food and energy commodities", "CUUR0000SASLE": "Services less energy services", "CUUR0000SAH1": "Shelter", "CUUR0000SEHC": "Owners' equivalent rent of residences", "CUUR0000SEHA": "Rent of primary residence", "CUUR0000SAM2": "Medical care services", "CUUR0000SAS4": "Transportation services", "CUUR0000SETE": "Motor vehicle insurance", "CUUR0000SETA02": "Used cars and trucks", "CUUR0000SETA01": "New vehicles", "CUUR0000SETB01": "Gasoline (all types)", "CUUR0000SEHF01": "Electricity"} AGGREGATES = ["Food", "Energy", "Commodities less food and energy commodities", "Services less energy services"] def get(url, data=None, headers=None): req = urllib.request.Request(url, data=data, headers=headers or {}) with urllib.request.urlopen(req, timeout=60) as r: return r.read() def table(year, offline): if offline: raw = open(os.path.join(offline, "ri_%d.htm" % year), encoding="utf-8", errors="replace").read() else: raw = get("https://www.bls.gov/cpi/tables/relative-importance/%d.htm" % year, headers={"User-Agent": CONTACT}).decode("utf-8", "replace") t = pd.read_html(io.StringIO(raw))[0] t.columns = ["item", "cpi_u", "cpi_w"] t["item"] = t["item"].astype(str).str.replace("’", "'").str.strip() return t.drop_duplicates("item", keep="first").set_index("item") # "All items" repeats at the foot of each table def api(offline): if offline: return json.load(open(os.path.join(offline, "bls_api_2025_2026.json"), encoding="utf-8"))["response"] body = json.dumps({"seriesid": list(SERIES), "startyear": "2025", "endyear": "2026"}).encode() return json.loads(get("https://api.bls.gov/publicAPI/v1/timeseries/data/", body, {"Content-Type": "application/json"})) def main(): p = argparse.ArgumentParser() p.add_argument("--offline", metavar="DIR") p.add_argument("--out", default="cpi-basket-relative-importance.csv") a = p.parse_args() rows = [] tables = {y: table(y, a.offline) for y in YEARS} for item in ITEMS: for y in YEARS: r = tables[y].loc[item] rows.append({"item": item, "december": y, "cpi_u": float(r["cpi_u"]), "cpi_w": float(r["cpi_w"])}) pd.DataFrame(rows).to_csv(a.out, index=False, float_format="%.3f") print("wrote %s (%d rows)" % (a.out, len(rows))) d = api(a.offline) assert d["status"] == "REQUEST_SUCCEEDED", d.get("message") idx = {s["seriesID"]: {(x["year"], x["period"]): float(x["value"]) for x in s["data"] if x["value"] not in ("-", "")} for s in d["Results"]["series"]} ri = tables[2025] total = idx["CUUR0000SA0"][("2026", "M08")] / idx["CUUR0000SA0"][("2025", "M12")] - 1 print("\nall items, NSA, December 2025 to August 2026: %+.3f%%" % (100 * total)) print("%-46s %8s %9s %9s %8s" % ("component (CPI-U)", "RI Dec25", "change", "contrib", "share")) contrib = {} for sid, item in SERIES.items(): if item == "All items": continue ch = idx[sid][("2026", "M08")] / idx[sid][("2025", "M12")] - 1 c = float(ri.loc[item, "cpi_u"]) * ch # percentage points of the all-items change contrib[item] = c print("%-46s %8.3f %+8.2f%% %+8.3f %+7.1f%%" % (item, ri.loc[item, "cpi_u"], 100 * ch, c, c / (100 * total) * 100)) s = sum(contrib[k] for k in AGGREGATES) print("\nfood + energy + core goods + core services = %+.3f points against %+.3f for all items (residual %+.4f)" % (s, 100 * total, s - 100 * total)) if __name__ == "__main__": sys.exit(main())