"""ai-in-10-k-risk-factors-edgar-full-text.py — count EDGAR full-text-search documents matching AI phrases in 10-K filings, by year, and list a sample of matching filings. What it does: queries https://efts.sec.gov/LATEST/search-index for exact phrases restricted to form 10-K and a calendar-year date range, records the 'total' hit count the API returns (documents, which can include exhibits, so a filing may count more than once), and writes the table to CSV. With --fetch it also writes the 40 newest 10-K hits for "generative artificial intelligence" since 2025-01-01, with accession numbers, to a second CSV. Inputs : none; --fetch queries EDGAR live (33 requests, well under the SEC's 10/sec limit), otherwise the script reads datasets/ai-in-10-k-risk-factors-edgar-full-text.csv (counts) and datasets/ai-in-10-k-risk-factors-edgar-full-text-filings.csv (sample), both written 2026-09-06. Outputs : the two CSVs and three Markdown tables. Needs : Python 3.13 standard library; pandas 3.0.2 for the tables. """ import datetime as dt import json import os import sys import time import urllib.parse import urllib.request import pandas as pd CSV = "datasets/ai-in-10-k-risk-factors-edgar-full-text.csv" FILINGS = "datasets/ai-in-10-k-risk-factors-edgar-full-text-filings.csv" PHRASES = ["artificial intelligence", "generative artificial intelligence", "machine learning", "large language model"] YEARS = (2019, 2020, 2021, 2022, 2023, 2024, 2025, 2026) UA = os.environ.get("EDGAR_UA", "prism-data-lab/1.0 (research; contact via site)") BASE = "https://efts.sec.gov/LATEST/search-index?q=" def get(url: str): req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Encoding": "identity"}) with urllib.request.urlopen(req, timeout=60) as r: return json.loads(r.read().decode("utf-8")) def query(phrase: str, start: str, end: str): return get(BASE + urllib.parse.quote(f'"{phrase}"') + f"&forms=10-K&dateRange=custom&startdt={start}&enddt={end}") def main(): today = dt.date(2026, 9, 6) if "--fetch" in sys.argv: today = dt.date.today() rows = [] for p in PHRASES: for y in YEARS: end = today.isoformat() if y == today.year else f"{y}-12-31" rows.append({"phrase": p, "year": y, "documents": int(query(p, f"{y}-01-01", end)["hits"]["total"]["value"]), "retrieved": today.isoformat()}) time.sleep(0.25) pd.DataFrame(rows).to_csv(CSV, index=False) hits = query("generative artificial intelligence", "2025-01-01", today.isoformat())["hits"]["hits"][:40] sample = [] for h in hits: s = h["_source"] adsh = h["_id"].split(":")[0] cik = str(s["ciks"][0]).lstrip("0") sample.append({"filed": s["file_date"], "form": s["form"], "entity": s["display_names"][0].split(" (")[0].strip(), "cik": cik, "accession": adsh, "url": f"https://www.sec.gov/Archives/edgar/data/{cik}/{adsh.replace('-', '')}/", "query": "generative artificial intelligence", "retrieved": today.isoformat()}) pd.DataFrame(sample).sort_values("filed", ascending=False).to_csv(FILINGS, index=False) df = pd.read_csv(CSV) wide = df.pivot(index="year", columns="phrase", values="documents")[PHRASES] print(f"retrieved {df['retrieved'].iloc[0]}; {today.year} is year-to-date through that day") print("| filing year | " + " | ".join(f'"{p}"' for p in PHRASES) + " |") print("|---|" + "---|" * len(PHRASES)) for y, r in wide.iterrows(): print(f"| {y} | " + " | ".join(f"{int(v):,}" for v in r.values) + " |") g = wide["generative artificial intelligence"] a = wide["artificial intelligence"] print(f"\n'artificial intelligence' documents: {int(a.loc[2019]):,} in 2019 -> {int(a.loc[2025]):,} in 2025 " f"({a.loc[2025] / a.loc[2019]:.1f}x); 'generative artificial intelligence': {int(g.loc[2022])} in 2022 -> {int(g.loc[2025]):,} in 2025") print(f"share of 'artificial intelligence' documents that also match 'generative artificial intelligence', 2025: {100 * g.loc[2025] / a.loc[2025]:.1f}%") sm = pd.read_csv(FILINGS) print(f"\nsample: {len(sm)} filings, {sm['cik'].nunique()} distinct filers, filed {sm['filed'].min()} to {sm['filed'].max()}; " f"forms: {sm['form'].value_counts().to_dict()}; filed in 2026: {int((sm['filed'] >= '2026-01-01').sum())}") print("| filed | entity | form | accession |") print("|---|---|---|---|") for _, r in sm.head(12).iterrows(): print(f"| {r['filed']} | {r['entity']} | {r['form']} | {r['accession']} |") if __name__ == "__main__": main()