"""target-10-k-2026-ai-disclosure.py — pull one 10-K from EDGAR by accession number and print every sentence that mentions artificial intelligence, with counts by section. What it does: looks up the filing's primary document through the submissions API (https://data.sec.gov/submissions/CIK##########.json), downloads it from the EDGAR archive, strips the HTML to text, splits it into sentences, and prints (1) the number of sentences containing "artificial intelligence", "generative artificial intelligence", and "agentic", (2) the count inside Item 1A (Risk Factors) versus the rest of the document, and (3) each matching sentence with a running number so it can be quoted by number. Inputs : --cik (default 27419, Target Corporation) and --accession (default 0000027419-26-000016, the 10-K filed 2026-03-11 for the fiscal year ended 2026-01-31). Queries EDGAR live: two requests. Needs : Python 3.13 standard library only. Run: python code/target-10-k-2026-ai-disclosure.py """ import argparse import html import json import os import re import urllib.request UA = os.environ.get("EDGAR_UA", "prism-data-lab/1.0 (research; contact via site)") def get(url: str) -> str: req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Encoding": "identity"}) with urllib.request.urlopen(req, timeout=60) as r: return r.read().decode("utf-8", "replace") def main(): ap = argparse.ArgumentParser() ap.add_argument("--cik", default="27419") ap.add_argument("--accession", default="0000027419-26-000016") a = ap.parse_args() sub = json.loads(get(f"https://data.sec.gov/submissions/CIK{str(int(a.cik)).zfill(10)}.json")) rec = sub["filings"]["recent"] i = rec["accessionNumber"].index(a.accession) url = f"https://www.sec.gov/Archives/edgar/data/{int(a.cik)}/{a.accession.replace('-', '')}/{rec['primaryDocument'][i]}" print(f"{sub['name']} | {rec['form'][i]} filed {rec['filingDate'][i]} for period {rec['reportDate'][i]} | {a.accession}\n{url}") raw = get(url) text = re.sub(r"<[^>]+>", " ", raw) text = html.unescape(text).replace("\xa0", " ") text = re.sub(r"\s+", " ", text) # Item 1A runs from its heading to the Item 1B heading; the table of contents also carries both strings, so # take the LAST occurrence of the Item 1A heading before the first Item 1B heading that follows it. heads_1a = [m.start() for m in re.finditer(r"Item 1A\.?\s+Risk Factors", text, flags=re.I)] start = heads_1a[-1] if heads_1a else 0 m1b = re.search(r"Item 1B\.?\s+Unresolved Staff Comments", text[start:], flags=re.I) end = start + m1b.start() if m1b else len(text) sentences = re.split(r"(?<=[.;])\s+(?=[A-Z(])", text) pos, hits = 0, [] for s in sentences: at = text.find(s, pos) pos = at + len(s) if re.search(r"artificial intelligence", s, flags=re.I): hits.append((at, s.strip())) in_1a = [h for h in hits if start <= h[0] < end] print(f"document length: {len(text):,} characters; Item 1A spans characters {start:,} to {end:,} ({end - start:,})") print(f"sentences mentioning 'artificial intelligence': {len(hits)}; inside Item 1A: {len(in_1a)}; elsewhere: {len(hits) - len(in_1a)}") print(f"sentences with 'generative artificial intelligence': {sum('generative artificial intelligence' in s.lower() for _, s in hits)}; " f"with 'agentic': {sum('agentic' in s.lower() for _, s in hits)}; with 'machine learning': {len(re.findall(r'machine learning', text, flags=re.I))} (whole document)") for n, (at, s) in enumerate(hits, 1): where = "1A" if start <= at < end else "other" print(f"\n[{n}] ({where}) {s}") if __name__ == "__main__": main()