"""sec-edgar-python-submissions-fts-xbrl.py — the three free EDGAR JSON endpoints in one small tool: submissions, full-text search, XBRL company facts. What it does: --cik 320193 --filings 10-K recent filings of one company with accession numbers and primary-document URLs (https://data.sec.gov/submissions/CIK##########.json) --fts "generative artificial intelligence" --forms 10-K --since 2025-01-01 full-text search hits with accession numbers (https://efts.sec.gov/LATEST/search-index) --cik 320193 --fact Revenues one us-gaap concept from company facts, fiscal-year rows only, deduplicated so each period is taken from the filing that first reported it (https://data.sec.gov/api/xbrl/companyfacts/CIK##########.json) Requirements: the SEC asks for a descriptive User-Agent with contact information and no more than 10 requests per second (stated on the SEC's 'Accessing EDGAR Data' page). Set EDGAR_UA to your own string. Run : python code/sec-edgar-python-submissions-fts-xbrl.py --cik 320193 --filings 10-K --max 5 Needs: Python 3.13 standard library only. """ import argparse import datetime as dt import json import os import urllib.parse import urllib.request UA = os.environ.get("EDGAR_UA", "prism-data-lab/1.0 (research; contact via site)") def get_json(url: str): req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Encoding": "identity"}) with urllib.request.urlopen(req, timeout=60) as r: return json.loads(r.read().decode("utf-8")) def filings(cik: str, form: str | None, max_results: int): cik10 = str(int(cik)).zfill(10) j = get_json(f"https://data.sec.gov/submissions/CIK{cik10}.json") rec = j["filings"]["recent"] out = [] for i in range(len(rec["accessionNumber"])): if form and rec["form"][i] != form: continue acc = rec["accessionNumber"][i] out.append({"entity": j["name"], "form": rec["form"][i], "filed": rec["filingDate"][i], "period": rec["reportDate"][i], "accession": acc, "url": f"https://www.sec.gov/Archives/edgar/data/{int(cik)}/{acc.replace('-', '')}/{rec['primaryDocument'][i]}"}) if len(out) >= max_results: break return out def fts(phrase: str, forms: str | None, since: str | None, max_results: int): url = "https://efts.sec.gov/LATEST/search-index?q=" + urllib.parse.quote(f'"{phrase}"') if forms: url += "&forms=" + urllib.parse.quote(forms) if since: url += f"&dateRange=custom&startdt={since}&enddt={dt.date.today().isoformat()}" j = get_json(url) total = j["hits"]["total"]["value"] hits = [] for h in j["hits"]["hits"][:max_results]: s = h["_source"] adsh = h["_id"].split(":")[0] cik = str(s["ciks"][0]).lstrip("0") hits.append({"entity": s["display_names"][0], "form": s["form"], "filed": s["file_date"], "accession": adsh, "cik": cik, "url": f"https://www.sec.gov/Archives/edgar/data/{cik}/{adsh.replace('-', '')}/"}) return total, hits def fact(cik: str, concept: str): cik10 = str(int(cik)).zfill(10) j = get_json(f"https://data.sec.gov/api/xbrl/companyfacts/CIK{cik10}.json") f = j["facts"]["us-gaap"][concept] unit = next(iter(f["units"])) seen, rows = set(), [] for v in sorted(f["units"][unit], key=lambda v: (v["end"], v["filed"])): if v.get("fp") != "FY" or v.get("form") != "10-K": continue key = (v.get("start"), v["end"]) if key in seen: continue # later 10-Ks restate the same period; keep the first report seen.add(key) rows.append({"entity": j["entityName"], "concept": concept, "unit": unit, "start": v.get("start", ""), "end": v["end"], "value": v["val"], "fy": v["fy"], "accession": v["accn"], "filed": v["filed"]}) return rows def main(): ap = argparse.ArgumentParser() ap.add_argument("--cik") ap.add_argument("--filings") ap.add_argument("--fact") ap.add_argument("--fts") ap.add_argument("--forms") ap.add_argument("--since") ap.add_argument("--max", type=int, default=10) a = ap.parse_args() if a.cik and a.filings: print(json.dumps(filings(a.cik, a.filings, a.max), indent=1)) elif a.cik and a.fact: print(json.dumps(fact(a.cik, a.fact), indent=1)) elif a.fts: total, hits = fts(a.fts, a.forms, a.since, a.max) print(json.dumps({"total_documents": total, "hits": hits}, indent=1)) else: ap.print_help() if __name__ == "__main__": main()