#!/usr/bin/env python3 """Generate config/sources.d/49-edgar-issuers.yaml — SEC EDGAR `submissions` sensors for every S&P 500, Nasdaq-100 and TSX 60 issuer not yet covered by the registry, plus the largest US-listed pharma/biotech (SIC 2834/2836), semiconductor (3674), software (7372) and bank (6021/6022) issuers by market cap. Builds on scripts/gen-edgar.py (apex/norm/slug/pretty_name/DOMAINS/MANUAL) but: * matches existing organizations across ALL registry files (founding file + every fragment, incl. 40–46) by manual map, ticker alias, apex domain, normalized name, then unambiguous name prefix; * skips every CIK already present anywhere in the registry (no duplicate sensor URLs); * resolves the company website from EDGAR `website`, Wikidata (P856 by ticker), then the DOMAINS maps; * sets `country:` from the business address (foreign private issuers keep their own country). Inputs (all fetched by `prepare`): /tmp/tickers.json https://www.sec.gov/files/company_tickers.json (ordered by market cap) /tmp/tickers_exch.json https://www.sec.gov/files/company_tickers_exchange.json /tmp/sp500.json ["MMM", …] (Wikipedia list of S&P 500 companies) /tmp/ndx.json ["AAPL", …] (api.nasdaq.com nasdaq100 list) /tmp/tsx60.json [["AEM", "Agnico Eagle Mines Limited"], …] (Wikipedia S&P/TSX 60) /tmp/wd.csv Wikidata SPARQL export: item,itemLabel,ticker,exchLabel,website Usage (repo root): python3 scripts/gen-edgar2.py fetch [--sector-limit 150] [--scan 900] # → /tmp/edgar2_meta.json (≤ 6 req/s) python3 scripts/gen-edgar2.py emit > config/sources.d/49-edgar-issuers.yaml """ import csv import glob import json import re import sys import time from urllib.parse import urlparse import requests import yaml import importlib.util as _ilu # noqa: E402 _spec = _ilu.spec_from_file_location("gen_edgar", __file__.rsplit("/", 1)[0] + "/gen-edgar.py") _g = _ilu.module_from_spec(_spec) _spec.loader.exec_module(_g) DOMAINS1, MANUAL1, apex, norm, pretty_name, slug = _g.DOMAINS, _g.MANUAL, _g.apex, _g.norm, _g.pretty_name, _g.slug UA = "WebSensor registry generator (contact@websensor.io)" META = "/tmp/edgar2_meta.json" FORMS = "[8-K, 10-K, 10-Q, 6-K, 20-F, 40-F, S-1, SC 13D, DEF 14A]" SECTOR_SICS = {"2834", "2836", "3674", "7372", "6021", "6022"} US_STATES = set("AL AK AZ AR CA CO CT DE FL GA HI ID IL IN IA KS KY LA ME MD MA MI MN MS MO MT NE NV NH NJ NM NY NC ND OH OK OR PA RI SC SD TN TX UT VT VA WA WV WI WY DC PR GU VI X1".split()) COUNTRY_BY_DESC = { "canada": "CA", "netherlands": "NL", "united kingdom": "GB", "ireland": "IE", "switzerland": "CH", "germany": "DE", "france": "FR", "japan": "JP", "china": "CN", "taiwan": "TW", "korea": "KR", "india": "IN", "brazil": "BR", "israel": "IL", "singapore": "SG", "hong kong": "HK", "bermuda": "BM", "cayman islands": "KY", "luxembourg": "LU", "denmark": "DK", "sweden": "SE", "finland": "FI", "norway": "NO", "spain": "ES", "italy": "IT", "belgium": "BE", "australia": "AU", "mexico": "MX", "argentina": "AR", "chile": "CL", "colombia": "CO", "peru": "PE", "panama": "PA", "south africa": "ZA", "new zealand": "NZ", "austria": "AT", "portugal": "PT", "greece": "GR", "poland": "PL", "jersey": "JE", "guernsey": "GG", "isle of man": "IM", "monaco": "MC", "united arab emirates": "AE", "indonesia": "ID", "philippines": "PH", "thailand": "TH", "malaysia": "MY", "vietnam": "VN", "turkey": "TR", "russia": "RU", "puerto rico": "US", "virgin islands": "US", "marshall islands": "MH", "liberia": "LR", "cyprus": "CY", "malta": "MT", "czech": "CZ", "hungary": "HU", "uruguay": "UY", "macau": "MO", "kazakhstan": "KZ", "saudi arabia": "SA", } # TSX 60 ticker → SEC ticker (US listing) when the symbols differ; None = no SEC registration (skipped). TSX_TO_SEC = {"ABX": "B", "BIP.UN": "BIP", "CCO": "CCJ", "CNR": "CNI", "GIB.A": "GIB", "K": "KGC", "MG": "MGA", "PPL": "PBA", "RCI.B": "RCI", "TECK.B": "TECK", "T": "TU", "ATD": None, "CTC.A": None, "CCL.B": None, "CSU": None, "DOL": None, "EMA": None, "FFH": None, "FM": None, "WN": None, "H": None, "IFC": None, "L": None, "MRU": None, "NA": None, "POW": None, "SAP": None, "TOU": None, "WSP": None} # Ticker → existing registry id when domain/name matching cannot decide (checked against the real id set). MANUAL2 = dict(MANUAL1) MANUAL2.update({"GOOG": "google", "BRK-A": "berkshire-hathaway", "FOX": "fox-corporation", "FOXA": "fox-corporation", "NWS": "news-corp", "NWSA": "news-corp", "PARA": "paramount", "PSKY": "paramount-skydance", "WBD": "warner-bros-discovery", "EA": "electronic-arts", "TTWO": "take-two", "RBLX": "roblox", "U": "unity", "SPOT": "spotify", "ABNB": "airbnb", "BKNG": "booking-holdings", "EXPE": "expedia", "MAR": "marriott", "HLT": "hilton", "RY": "rbc", "TD": "td-bank", "BMO": "bmo", "BNS": "scotiabank", "CM": "cibc", "CNQ": "canadian-natural", "CNI": "cn", "CP": "cpkc", "ENB": "enbridge", "TRP": "tc-energy", "SU": "suncor", "BCE": "bce", "RCI": "rogers", "TU": "telus", "TRI": "thomson-reuters", "MFC": "manulife", "SLF": "sun-life", "OTEX": "opentext", "GIB": "cgi", "QSR": "restaurant-brands", "NTR": "nutrien", "WCN": "waste-connections", "CVE": "cenovus", "IMO": "imperial-oil", "FTS": "fortis", "CCJ": "cameco", "MGA": "magna", "CLS": "celestica", "CAE": "cae", "ORCL": "oracle", "IBM": "ibm", "CSCO": "cisco", "HPE": "hpe", "HPQ": "hp", "DELL": "dell", "NTAP": "netapp", "PSTG": "pure-storage", "CRWV": "coreweave", "NBIS": "nebius", "PLTR": "palantir", "SNPS": "synopsys", "CDNS": "cadence", "ANSS": "ansys", "ADP": "adp", "PAYX": "paychex", "OKTA": "okta", "TWLO": "twilio", "HUBS": "hubspot", "ZM": "zoom", "DOCU": "docusign", "BOX": "box", "DBX": "dropbox", "GTLB": "gitlab", "S": "sentinelone", "TENB": "tenable", "RPD": "rapid7", "CYBR": "cyberark", "QLYS": "qualys", "VRNS": "varonis", "CHKP": "check-point", "AKAM": "akamai", "FSLY": "fastly", "DOCN": "digitalocean", "GDDY": "godaddy", "WIX": "wix", "SQSP": "squarespace", "SHOP": "shopify", "SQ": "block", "XYZ": "block", "MELI": "mercadolibre", "SE": "sea-limited", "BABA": "alibaba", "JD": "jd-com", "PDD": "pdd-holdings", "BIDU": "baidu", "NTES": "netease", "TCEHY": "tencent", "SONY": "sony", "NTDOY": "nintendo", "TSM": "tsmc", "ASML": "asml", "AMAT": "applied-materials", "LRCX": "lam-research", "KLAC": "kla", "TXN": "texas-instruments", "ADI": "analog-devices", "MRVL": "marvell", "NXPI": "nxp", "ON": "onsemi", "MCHP": "microchip", "STM": "stmicroelectronics", "IFNNY": "infineon", "SWKS": "skyworks", "QRVO": "qorvo", "MPWR": "monolithic-power", "GFS": "globalfoundries", "SMCI": "supermicro", "WDC": "western-digital", "STX": "seagate", "SNDK": "sandisk", "AVGO": "broadcom", "MU": "micron", "ARM": "arm", "QCOM": "qualcomm", "INTC": "intel", "AMD": "amd", "NVDA": "nvidia", "AAPL": "apple", "MSFT": "microsoft", "GOOGL": "google", "META": "meta-ai", "AMZN": "amazon", "NFLX": "netflix", "TSLA": "tesla", "UBER": "uber", "LYFT": "lyft", "DASH": "doordash", "RDDT": "reddit", "PINS": "pinterest", "SNAP": "snap", "DUOL": "duolingo", "COIN": "coinbase", "HOOD": "robinhood", "CRCL": "circle", "MSTR": "strategy", "V": "visa", "MA": "mastercard", "AXP": "american-express", "PYPL": "paypal", "FI": "fiserv", "FIS": "fis", "GPN": "global-payments", "ADYEY": "adyen", "JPM": "jpmorgan", "BAC": "bank-of-america", "WFC": "wells-fargo", "C": "citi", "GS": "goldman-sachs", "MS": "morgan-stanley", "SCHW": "charles-schwab", "BLK": "blackrock", "BX": "blackstone", "KKR": "kkr", "APO": "apollo", "USB": "us-bancorp", "PNC": "pnc", "TFC": "truist", "COF": "capital-one", "BK": "bny", "STT": "state-street", "NTRS": "northern-trust", "ICE": "ice", "CME": "cme-group", "NDAQ": "nasdaq", "CBOE": "cboe", "SPGI": "sp-global", "MCO": "moodys", "MSCI": "msci", "HSBC": "hsbc", "BCS": "barclays", "UBS": "ubs", "DB": "deutsche-bank", "SAN": "santander", "ING": "ing", "BBVA": "bbva", "MUFG": "mufg", "SMFG": "smbc", "MFG": "mizuho", "NMR": "nomura", "LYG": "lloyds", "NWG": "natwest", "TM": "toyota", "HMC": "honda", "F": "ford", "GM": "gm", "STLA": "stellantis", "RIVN": "rivian", "LCID": "lucid", "PFE": "pfizer", "MRNA": "moderna", "MRK": "merck", "LLY": "eli-lilly", "AZN": "astrazeneca", "NVS": "novartis", "GSK": "gsk", "SNY": "sanofi", "NVO": "novo-nordisk", "RHHBY": "roche", "JNJ": "johnson-johnson", "ABBV": "abbvie", "AMGN": "amgen", "GILD": "gilead", "BMY": "bristol-myers-squibb", "REGN": "regeneron", "VRTX": "vertex", "BIIB": "biogen", "BNTX": "biontech", "TAK": "takeda", "BAYRY": "bayer", "ISRG": "intuitive-surgical", "MDT": "medtronic", "ABT": "abbott", "SYK": "stryker", "BSX": "boston-scientific", "TMO": "thermo-fisher", "DHR": "danaher", "UNH": "unitedhealth", "CVS": "cvs-health", "CI": "cigna", "HUM": "humana", "ELV": "elevance", "CNC": "centene", "MOH": "molina", "T": "att", "VZ": "verizon", "TMUS": "t-mobile", "CMCSA": "comcast", "CHTR": "charter", "DIS": "disney", "LUMN": "lumen", "DAL": "delta", "UAL": "united-airlines", "AAL": "american-airlines", "LUV": "southwest", "BA": "boeing", "LMT": "lockheed-martin", "RTX": "rtx", "NOC": "northrop-grumman", "GD": "general-dynamics", "GE": "ge-aerospace", "RKLB": "rocket-lab", "ASTS": "ast-spacemobile", "SPCX": "spacex", "XOM": "exxonmobil", "CVX": "chevron", "COP": "conocophillips", "SHEL": "shell", "BP": "bp", "TTE": "totalenergies", "EQNR": "equinor", "WMT": "walmart", "COST": "costco", "TGT": "target", "HD": "home-depot", "LOW": "lowes", "KR": "kroger", "NKE": "nike", "SBUX": "starbucks", "MCD": "mcdonalds", "KO": "coca-cola", "PEP": "pepsico", "PG": "procter-gamble", "UL": "unilever", "NSRGY": "nestle"}) # Tickers that must NOT be folded into a look-alike source (subsidiary / sister company with a similar name). NEVER_MATCH = {"CCEP", "TXT"} # Tickers skipped altogether (shell / successor registrants with an empty filing history). SKIP_TICKERS = {"CONE"} # Ticker → registry id when the slug of the display name is unusable (initials, accents, generic words). IDS2 = {"FNB": "fnb-corporation", "ITUB": "itau-unibanco", "AOS": "a-o-smith", "NWSA": "news-corp", "SSNC": "ssc-technologies", "YOU": "clear-secure", "MSTR": "strategy-microstrategy", "WRB": "wr-berkley", "DHI": "dr-horton", "MTB": "mt-bank", "PCG": "pge-corporation", "GWW": "ww-grainger", "VMRK": "vivmark-residential"} # Ticker → display name when EDGAR's registrant name is ugly (raw upper-case, legal suffixes, state fragments). NAMES2 = {"SSNC": "SS&C Technologies", "BBD": "Banco Bradesco", "ITUB": "Itaú Unibanco", "AOS": "A. O. Smith", "DHI": "D.R. Horton", "WRB": "W. R. Berkley", "CRH": "CRH", "NWSA": "News Corp", "MTB": "M&T Bank", "PCG": "PG&E", "GWW": "W.W. Grainger", "ZION": "Zions Bancorporation", "KEY": "KeyCorp", "LEN": "Lennar", "OKE": "ONEOK", "PHM": "PulteGroup", "HBAN": "Huntington Bancshares", "GLW": "Corning", "CFG": "Citizens Financial Group", "UBSI": "United Bankshares", "ONB": "Old National Bancorp", "FNB": "F.N.B. Corporation", "CBSH": "Commerce Bancshares", "ABVX": "Abivax", "ASND": "Ascendis Pharma", "GMAB": "Genmab", "MFG": "Mizuho Financial Group", "SHG": "Shinhan Financial Group", "LVS": "Las Vegas Sands", "ODFL": "Old Dominion Freight Line", "MKC": "McCormick & Company", "MCK": "McKesson", "DVA": "DaVita", "COO": "The Cooper Companies", "EXPD": "Expeditors International", "IEX": "IDEX", "IDXX": "IDEXX Laboratories", "IQV": "IQVIA", "AME": "AMETEK", "MTD": "Mettler-Toledo", "STE": "STERIS", "CSGP": "CoStar Group", "SSB": "SouthState", "NTES": "NetEase", "GFS": "GlobalFoundries", "FORM": "FormFactor", "KDP": "Keurig Dr Pepper", "PM": "Philip Morris International", "MO": "Altria Group", "BF-B": "Brown-Forman", "BRO": "Brown & Brown", "MRSH": "Marsh McLennan", "FCNCA": "First Citizens BancShares", "EWBC": "East West Bancorp", "CFR": "Cullen/Frost Bankers", "HWC": "Hancock Whitney", "HOMB": "Home BancShares", "PNFP": "Pinnacle Financial Partners", "BPOP": "Popular, Inc.", "PB": "Prosperity Bancshares", "UMBF": "UMB Financial", "WAL": "Western Alliance Bancorporation", "WTFC": "Wintrust Financial", "BOKF": "BOK Financial", "ASB": "Associated Banc-Corp", "AUB": "Atlantic Union Bankshares", "ABCB": "Ameris Bancorp", "COLB": "Columbia Banking System", "GBCI": "Glacier Bancorp", "ALLY": "Ally Financial", "VLY": "Valley National Bancorp", "SYF": "Synchrony Financial", "ASX": "ASE Technology Holding", "UMC": "United Microelectronics", "STM": "STMicroelectronics", "TSEM": "Tower Semiconductor", "SIMO": "Silicon Motion", "MTSI": "MACOM Technology Solutions", "SLAB": "Silicon Labs", "AAOI": "Applied Optoelectronics", "YMM": "Full Truck Alliance", "GLBE": "Global-e", "APPF": "AppFolio", "BSY": "Bentley Systems", "YOU": "CLEAR (Clear Secure)", "GWRE": "Guidewire", "MANH": "Manhattan Associates", "MBLY": "Mobileye", "PAYC": "Paycom", "PCTY": "Paylocity", "PCOR": "Procore", "TTAN": "ServiceTitan", "SAIL": "SailPoint", "IFNNY": "Infineon Technologies", "ATEYY": "Advantest", "NBIS": "Nebius Group", "MSTR": "Strategy (MicroStrategy)", "IT": "Gartner", "J": "Jacobs", "L": "Loews Corporation", "TT": "Trane Technologies", "TEL": "TE Connectivity", "ARE": "Alexandria Real Estate Equities", "DOC": "Healthpeak Properties", "AES": "The AES Corporation", "APA": "APA Corporation", "EQT": "EQT Corporation", "PPL": "PPL Corporation", "NVR": "NVR, Inc.", "UDR": "UDR, Inc.", "BXP": "BXP, Inc.", "CDW": "CDW Corporation", "FICO": "Fair Isaac (FICO)", "GEN": "Gen Digital", "ON": "onsemi", "SMCI": "Supermicro", "SNDK": "Sandisk", "ECHO": "EchoStar", "PTC": "PTC Inc.", "B": "Barrick Mining", "MAA": "Mid-America Apartment Communities", "VMRK": "Vivmark Residential (Equity Residential)", "EXE": "Expand Energy", "FIX": "Comfort Systems USA", "EME": "EMCOR Group", "PWR": "Quanta Services", "TDY": "Teledyne Technologies", "GL": "Globe Life", "SW": "Smurfit Westrock", "ROP": "Roper Technologies", "HPE": "Hewlett Packard Enterprise", "FDXF": "FedEx Freight", "NXT": "Nextpower (Nextracker)", "Q": "Qnity Electronics", "TPL": "Texas Pacific Land", "AEM": "Agnico Eagle Mines", "FNV": "Franco-Nevada", "KGC": "Kinross Gold", "WPM": "Wheaton Precious Metals", "TECK": "Teck Resources", "PBA": "Pembina Pipeline", "MGA": "Magna International", "GIL": "Gildan Activewear", "FSV": "FirstService", "WCN": "Waste Connections", "FER": "Ferrovial", "CCEP": "Coca-Cola Europacific Partners", "WTW": "Willis Towers Watson", "PNR": "Pentair", "AMCR": "Amcor", "APTV": "Aptiv", "ALLE": "Allegion", "JCI": "Johnson Controls", "SPCX": "SpaceX"} # Ticker → website domain when EDGAR, Wikidata and gen-edgar DOMAINS are silent. DOMAINS2 = {"BRK-B": "berkshirehathaway.com", "GOOG": "abc.xyz", "PSKY": "paramount.com", "FOXA": "foxcorporation.com", "NWSA": "newscorp.com", "LEN-B": "lennar.com", "UHAL-B": "uhaul.com", "TKO": "tkogrp.com", "CRH": "crh.com", "TEL": "te.com", "APH": "amphenol.com", "TT": "tranetechnologies.com", "JCI": "johnsoncontrols.com", "ETN": "eaton.com", "ACN": "accenture.com", "MDT": "medtronic.com", "GRMN": "garmin.com", "STE": "steris.com", "ALLE": "allegion.com", "PNR": "pentair.com", "NVT": "nvent.com", "AON": "aon.com", "WTW": "wtwco.com", "CB": "chubb.com", "IR": "irco.com", "AER": "aercap.com", "ICLR": "iclr.com", "MPWR": "monolithicpower.com", "WAB": "wabteccorp.com", "FTV": "fortive.com", "HUBB": "hubbell.com", "L": "loews.com", "ERIE": "erieinsurance.com", "ABVX": "abivax.com", "ATEYY": "advantest.com", "ASND": "ascendispharma.com", "BLTE": "belitebio.com", "CGON": "cgoncology.com", "COGT": "cogentbio.com", "COLB": "columbiabankingsystem.com", "CRDO": "credosemi.com", "DNTH": "dianthustx.com", "FORM": "formfactor.com", "YMM": "fulltruckalliance.com", "GMAB": "genmab.com", "GLBE": "global-e.com", "IBRX": "immunitybio.com", "IMVT": "immunovant.com", "IFNNY": "infineon.com", "INSM": "insmed.com", "IONS": "ionis.com", "KNSA": "kiniksa.com", "LGND": "ligand.com", "MDGL": "madrigalpharma.com", "NXT": "nextracker.com", "ORKA": "orukatx.com", "PRAX": "praxismedicines.com", "PTGX": "protagonist-inc.com", "RVMD": "revmed.com", "RPRX": "royaltypharma.com", "SAIL": "sailpoint.com", "SRRK": "scholarrock.com", "TTAN": "servicetitan.com", "SYRE": "spyre.com", "TVTX": "travere.com", "UTHR": "unither.com", "VLY": "valley.com", "PCVX": "vaxcyte.com", "FER": "ferrovial.com", "FSV": "firstservice.com", "MRSH": "marshmclennan.com", "WCN": "wasteconnections.com", "WPM": "wheatonpm.com", "CCEP": "cocacolaep.com", "TXT": "textron.com", "ASX": "aseglobal.com", "AME": "ametek.com", "KDP": "keurigdrpepper.com", "GL": "globelifeinsurance.com", "DOC": "healthpeak.com", "SW": "smurfitwestrock.com", "MSTR": "strategy.com", "EXE": "expandenergy.com", "SYF": "synchrony.com", "ROP": "ropertech.com", "FNB": "fnb-online.com", "ZETA": "zetaglobal.com", "AES": "aes.com", "APA": "apacorp.com", "B": "barrick.com", "BXP": "bxp.com", "CDW": "cdw.com", "ECHO": "echostar.com", "EQT": "eqt.com", "FICO": "fico.com", "GEN": "gendigital.com", "KEY": "key.com", "MAA": "maac.com", "NVR": "nvrinc.com", "ON": "onsemi.com", "PCG": "pgecorp.com", "PPL": "pplweb.com", "PTC": "ptc.com", "SNDK": "sandisk.com", "SMCI": "supermicro.com", "UDR": "udr.com"} def load_registry() -> dict: """Ids, apex domains, normalized names, aliases and CIKs across the founding file and every fragment.""" by_domain: dict[str, str] = {} by_name: dict[str, str] = {} by_alias: dict[str, str] = {} ids: set[str] = set() ciks: set[str] = set() urls: set[str] = set() name_of: dict[str, str] = {} later_domains: dict[str, str] = {} later_names: dict[str, str] = {} files = ["config/sources.yaml"] + sorted(glob.glob("config/sources.d/*.yaml")) later_ids: set[str] = set() # declared in fragments merged AFTER this one — cannot be extended from here for f in files: if f.endswith("49-edgar-issuers.yaml"): continue text = open(f).read() ciks.update(f"{int(c):010d}" for c in re.findall(r"CIK(\d+)\.json", text)) d = yaml.safe_load(text) or {} base = f.split("/")[-1] later = base[:2].isdigit() and (int(base[:2]) > 49 or base.startswith("49b")) for s in d.get("sources", []): for sen in s.get("sensors") or []: urls.add(sen.get("url", "")) if later: if not s.get("extend"): later_ids.add(s["id"]) later_domains.setdefault(apex(s["domain"]), s["id"]) later_names.setdefault(norm(s["name"]), s["id"]) continue for a in s.get("aliases") or []: by_alias.setdefault(str(a).lower(), s["id"]) if s.get("extend"): continue ids.add(s["id"]) name_of[s["id"]] = norm(s["name"]) by_domain.setdefault(apex(s["domain"]), s["id"]) by_name.setdefault(norm(s["name"]), s["id"]) return {"by_domain": by_domain, "by_name": by_name, "by_alias": by_alias, "ids": ids, "ciks": ciks, "urls": urls, "name_of": name_of, "later_ids": later_ids, "later_domains": later_domains, "later_names": later_names} def sec_universe() -> tuple[list[dict], dict[str, dict]]: """company_tickers.json in market-cap order (list) + ticker → record (with exchange).""" ordered = [v for _, v in sorted(json.load(open("/tmp/tickers.json")).items(), key=lambda kv: int(kv[0]))] exch = {} for cik, name, ticker, ex in json.load(open("/tmp/tickers_exch.json"))["data"]: exch[ticker] = ex by_ticker = {} for v in ordered: v = dict(v) v["exchange"] = exch.get(v["ticker"]) by_ticker.setdefault(v["ticker"], v) return ordered, by_ticker def wikidata_sites() -> dict[str, list[tuple[str, str]]]: """ticker → [(normalized label, website)…] — several issuers can share a symbol across exchanges.""" out: dict[str, list[tuple[str, str]]] = {} try: for r in csv.DictReader(open("/tmp/wd.csv")): t = r["ticker"].upper().replace(".", "-") if r.get("website"): out.setdefault(t, []).append((norm(r.get("itemLabel") or ""), r["website"])) except FileNotFoundError: pass return out def wd_pick(t: str, name: str, wd: dict[str, list[tuple[str, str]]]) -> str | None: """Wikidata website for ticker t only when the Wikidata label shares a real word with the EDGAR name.""" toks = {w for w in norm(name).split() if len(w) >= 4} for label, site in wd.get(t, []): if toks & {w for w in label.split() if len(w) >= 4}: return site return None def fetch_one(sess: requests.Session, cik: str) -> dict | None: r = sess.get(f"https://data.sec.gov/submissions/CIK{cik}.json", headers={"User-Agent": UA, "Accept": "application/json", "Host": "data.sec.gov"}, timeout=30) if r.status_code != 200: print(f"CIK{cik}: HTTP {r.status_code}", file=sys.stderr) return None j = r.json() recent = (j.get("filings") or {}).get("recent") or {} forms = recent.get("form") or [] biz = (j.get("addresses") or {}).get("business") or {} return { "cik": cik, "name": j.get("name"), "website": j.get("website") or j.get("investorWebsite") or "", "sic": str(j.get("sic") or ""), "sicDescription": j.get("sicDescription"), "exchanges": j.get("exchanges"), "tickers": j.get("tickers"), "state": j.get("stateOfIncorporation"), "stateDesc": j.get("stateOfIncorporationDescription"), "bizState": biz.get("stateOrCountry"), "bizStateDesc": biz.get("stateOrCountryDescription"), "recent": len(forms), "hasForeign": any(f in ("6-K", "20-F", "40-F") for f in forms[:200]), } def fetch(argv: list[str]) -> None: sector_limit = int(argv[argv.index("--sector-limit") + 1]) if "--sector-limit" in argv else 150 scan = int(argv[argv.index("--scan") + 1]) if "--scan" in argv else 900 reg = load_registry() ordered, by_ticker = sec_universe() try: meta = json.load(open(META)) except FileNotFoundError: meta = {"index": {}, "sector": {}} sess = requests.Session() def polite_get(cik: str) -> dict | None: t0 = time.time() m = fetch_one(sess, cik) dt = time.time() - t0 if dt < 0.17: time.sleep(0.17 - dt) return m # 1) index universe: S&P 500 ∪ Nasdaq-100 ∪ TSX 60 (mapped to their SEC ticker) index_tickers: dict[str, str] = {} for t in json.load(open("/tmp/sp500.json")): index_tickers.setdefault(t.replace(".", "-"), "sp500") for t in json.load(open("/tmp/ndx.json")): index_tickers.setdefault(t.replace(".", "-"), "ndx") tsx_unmatched = [] for t, name in json.load(open("/tmp/tsx60.json")): sec_t = TSX_TO_SEC.get(t, t) if sec_t is None: continue if sec_t not in by_ticker: # name match against SEC names (Canadian issuers are usually cross-listed under the same symbol) nm = norm(name) cands = [v["ticker"] for v in ordered if norm(v["title"]) == nm] if cands: sec_t = cands[0] else: tsx_unmatched.append((t, name)) continue index_tickers.setdefault(sec_t, "tsx60") if tsx_unmatched: print(f"TSX 60 without SEC registration: {tsx_unmatched}", file=sys.stderr) todo = [t for t in index_tickers if t in by_ticker and f"{int(by_ticker[t]['cik_str']):010d}" not in reg["ciks"]] missing = [t for t in index_tickers if t not in by_ticker] if missing: print(f"index tickers not in company_tickers.json: {missing}", file=sys.stderr) print(f"index universe: {len(index_tickers)} tickers, {len(todo)} not yet covered by CIK", file=sys.stderr) seen_ciks: set[str] = set() for t in todo: cik = f"{int(by_ticker[t]['cik_str']):010d}" if cik in seen_ciks: continue seen_ciks.add(cik) if t in meta["index"]: continue m = polite_get(cik) if m: m["group"] = index_tickers[t] m["ticker"] = t m["exchange"] = by_ticker[t]["exchange"] meta["index"][t] = m json.dump(meta, open(META, "w"), indent=1) print(f"{len(meta['index'])} index issuers in {META}", file=sys.stderr) # 2) sector universe: walk company_tickers.json (market-cap order) and keep SIC 2834/2836/3674/7372/6021/6022 covered = set(reg["ciks"]) | {m["cik"] for m in meta["index"].values()} scanned = meta.setdefault("scanned", {}) n_sector = len(meta["sector"]) seen_here: set[str] = set() for v in ordered: if n_sector >= sector_limit or len(scanned) >= scan: break cik = f"{int(v['cik_str']):010d}" t = v["ticker"] if cik in covered or cik in seen_here or t in meta["sector"]: continue seen_here.add(cik) if cik in scanned: m = scanned[cik] if m is None or m.get("sic") not in SECTOR_SICS: continue else: m = polite_get(cik) scanned[cik] = {"sic": m["sic"], "name": m["name"]} if m else None if len(scanned) % 25 == 0: json.dump(meta, open(META, "w"), indent=1) print(f" scanned {len(scanned)} · sector picks {n_sector}", file=sys.stderr) if not m or m["sic"] not in SECTOR_SICS: continue m["group"] = "sector" m["ticker"] = t m["exchange"] = by_ticker.get(t, {}).get("exchange") meta["sector"][t] = m n_sector += 1 json.dump(meta, open(META, "w"), indent=1) print(f"{len(meta['sector'])} sector issuers ({len(scanned)} scanned) in {META}", file=sys.stderr) COUNTRY_OVERRIDE = {"BBD": "BR", "SHG": "KR", "NXPI": "NL", "KGC": "CA", "YMM": "CN", "NTES": "CN", "KNSA": "US", "TEL": "IE", "GIL": "CA", "ASX": "TW", "FER": "NL", "CRDO": "US", "SIMO": "TW", "ROIV": "US", "RPRX": "US", "AMCR": "CH", "ITUB": "BR", "MFG": "JP", "ATEYY": "JP", "IFNNY": "DE", "STM": "CH", "ABVX": "FR", "ASND": "DK", "GMAB": "DK", "GLBE": "IL", "TSEM": "IL", "UMC": "TW", "APTV": "IE", "CCEP": "GB", "WTW": "GB", "PNR": "GB", "ALLE": "IE", "CRH": "IE", "JCI": "IE", "STE": "IE", "TT": "IE", "ALKS": "IE", "FSV": "CA", "WCN": "CA", "MGA": "CA", "CLS": "CA", "AEM": "CA", "FNV": "CA", "WPM": "CA", "TECK": "CA", "PBA": "CA"} def country_of(m: dict) -> str: if m.get("ticker") in COUNTRY_OVERRIDE: return COUNTRY_OVERRIDE[m["ticker"]] code = (m.get("bizState") or m.get("state") or "").upper() desc = (m.get("bizStateDesc") or m.get("stateDesc") or "").lower() if code in US_STATES: return "US" for k, v in COUNTRY_BY_DESC.items(): if k in desc: return v if re.fullmatch(r"[AB][0-9]", code) or code == "Z4": return "CA" # EDGAR province codes A0–B0 / Z4 = Canada if desc or code: print(f"# unknown country {code!r} {desc!r} ({m.get('ticker')} {m.get('name')}) — defaulting to US", file=sys.stderr) return "US" def categories(m: dict) -> str: """Registry categories from the SIC code / description (whole-word matching — "Toilet" is not "oil").""" sic = m.get("sic") or "" d = (m.get("sicDescription") or "").lower() def has(*words: str) -> bool: return any(re.search(rf"\b{w}\b", d) for w in words) def sic_in(*prefixes: str) -> bool: return any(sic.startswith(p) for p in prefixes) if sic in ("2834", "2836", "2835", "8731") or has("pharmaceutical", "biological", "in vitro"): return "[pharma, filings]" if sic == "3674" or has("semiconductors"): return "[semiconductors, filings]" if sic in ("7372", "7371", "7370", "7373", "7374") or has("software", "computer programming", "data processing"): return "[enterprise, filings]" if sic_in("6798") or has("real estate investment trusts"): return "[real-estate, filings]" if sic_in("65") or has("real estate"): return "[real-estate, filings]" if sic_in("63", "64") or has("insurance"): return "[finance, insurance, filings]" if sic_in("60", "61", "62", "67") or has("bank", "banks", "finance", "investment", "loan", "credit", "brokers", "dealers"): return "[finance, filings]" if sic_in("5122") or has("wholesale-drugs", "drugs proprietaries"): return "[health, filings]" if sic_in("384", "385", "80") or has("medical", "surgical", "dental", "health", "hospitals", "orthopedic"): return "[health, filings]" if sic_in("2711", "2721", "2731", "2741", "4832", "4833", "4841", "7311", "7812", "7819", "7822") or has("newspapers", "periodicals", "books", "television", "radio broadcasting", "advertising", "motion picture", "cable"): return "[media, filings]" if sic_in("48") or has("telephone", "communications", "telegraph"): return "[telecom, filings]" if sic_in("4953", "4955", "4959") or has("refuse", "hazardous waste"): return "[infrastructure, filings]" if sic_in("13", "29", "46", "49", "5171", "5172") or has("petroleum", "natural gas", "gas transmission", "gas distribution", "electric", "energy", "crude", "power"): return "[energy, filings]" if sic_in("10", "12", "14") or has("mining", "gold", "silver", "ores", "quarrying", "coal"): return "[mining, filings]" if sic_in("2300", "23", "3021", "3140", "3021") or has("apparel", "footwear", "shoes"): return "[retail, filings]" if sic_in("52", "53", "54", "56", "57", "59", "5500", "5531") or has("retail", "stores", "convenience"): return "[retail, filings]" if sic_in("58") or has("eating places", "restaurants"): return "[retail, food, filings]" if sic_in("20", "21", "01", "02", "07") or has("food", "beverages", "tobacco", "cigarettes", "agricultural", "grain mill"): return "[food, filings]" if sic_in("70", "79") or has("hotels", "motels", "amusement", "gaming", "casino"): return "[entertainment, travel, filings]" if sic_in("37", "40", "42", "44", "45", "47", "3711", "3713", "3714", "3715", "3720", "3721", "3724", "3728", "3730") or has("motor vehicle", "motor vehicles", "aircraft", "air transportation", "railroad", "trucking", "ships", "shipping", "freight", "courier", "airports", "pipelines"): return "[transport, filings]" if sic_in("15", "16", "17") or has("construction", "contractors", "operative builders"): return "[real-estate, construction, filings]" if sic_in("28") or has("chemicals", "plastics", "fertilizers", "industrial gases", "cosmetics", "soap"): return "[chemicals, filings]" if sic_in("3571", "3572", "3576", "3577", "3578", "3579", "3661", "3663", "3669", "3670", "3672", "3677", "3678", "3679", "3812", "3821", "3822", "3823", "3824", "3825", "3826", "3827", "3829", "3861", "3357", "3679", "5045", "5065") or has("computer", "electronic", "instruments", "measuring", "optical", "photographic", "navigation", "communications equipment"): return "[technology, filings]" if sic_in("33", "34", "35", "36", "38", "39", "24", "25", "26", "30", "31", "32") or has("machinery", "equipment", "manufacturing", "industries", "products", "metal"): return "[manufacturing, filings]" if sic_in("50", "51") or has("wholesale"): return "[commerce, filings]" return "[enterprise, filings]" def resolve_domain(t: str, m: dict, wd: dict[str, list[tuple[str, str]]]) -> tuple[str | None, str | None]: """→ (apex domain, host for homepage) from EDGAR website, Wikidata, DOMAINS maps.""" cands = [] if t in DOMAINS2: cands.append("https://" + DOMAINS2[t]) if m.get("website"): cands.append(m["website"]) for tk in [t] + list(m.get("tickers") or []): site = wd_pick(tk, m.get("name") or "", wd) if site and site not in cands: cands.append(site) for src in (DOMAINS2, DOMAINS1): if t in src: cands.append("https://" + src[t]) for c in cands: host = urlparse(c if c.startswith("http") else "https://" + c).hostname if host and "." in host and not host.endswith(("sec.gov", "wikipedia.org")): host = host.lower() return apex(host), host return None, None def emit() -> None: meta = json.load(open(META)) reg = load_registry() wd = wikidata_sites() all_ids = set(reg["ids"]) entries = list(meta["index"].values()) + list(meta["sector"].values()) print("# config/sources.d/49-edgar-issuers.yaml — SEC EDGAR issuers (2026-09-11, generated by scripts/gen-edgar2.py from") print("# data.sec.gov submissions + company_tickers.json; validated live). One `edgar` sensor per issuer for every S&P 500,") print("# Nasdaq-100 and TSX 60 company not yet covered by an earlier fragment, plus the largest US-listed pharma/biotech") print("# (SIC 2834/2836), semiconductor (3674), software (7372) and bank (6021/6022) issuers by market cap. Forms watched:") print("# 8-K, 10-K, 10-Q, 6-K, 20-F, 40-F, S-1, SC 13D, DEF 14A (insider forms excluded). Organizations that already exist") print("# anywhere in the registry are extended (matched by ticker alias, apex domain or name); the others are declared with") print("# their own website as domain, `country:` from the business address (foreign private issuers keep theirs).") print("sources:") used: set[str] = set() used_ciks: set[str] = set(reg["ciks"]) stats: dict[str, list[str]] = {} skipped = [] for m in sorted(entries, key=lambda x: (x.get("group") != "sector", pretty_name(x["ticker"], x["name"] or "").lower())): t = m["ticker"] if m["cik"] in used_ciks or t in SKIP_TICKERS: continue ap, host = resolve_domain(t, m, wd) target = None why = "" nm = norm(m["name"] or "") toks = {w for w in nm.split() if len(w) >= 4} cand = MANUAL2.get(t) if cand and cand in all_ids: target, why = cand, "manual" if not target and ap and ap in reg["by_domain"]: target, why = reg["by_domain"][ap], f"domain {ap}" if not target and nm in reg["by_name"]: target, why = reg["by_name"][nm], "name" if not target and t.lower() in reg["by_alias"] and reg["by_alias"][t.lower()] in all_ids: # a ticker alias only counts when the source's own name shares a real word with the issuer's name sid = reg["by_alias"][t.lower()] if toks & {w for w in reg["name_of"].get(sid, "").split() if len(w) >= 4}: target, why = sid, f"alias {t.lower()}" if not target: cands = {sid for n, sid in reg["by_name"].items() if len(n) >= 6 and (nm.startswith(n + " ") or n.startswith(nm + " "))} if len(cands) == 1: target, why = cands.pop(), "name prefix" if not target and slug(m["name"] or "") in all_ids: target, why = slug(m["name"] or ""), "slug" if t in NEVER_MATCH: target = None if not target: later = (ap and reg["later_domains"].get(ap)) or reg["later_names"].get(nm) or (MANUAL2.get(t) if MANUAL2.get(t) in reg["later_ids"] else None) if later: skipped.append(f"{t} ({m['name']}): already declared as `{later}` in a later fragment — add its EDGAR sensor there") continue if target: print(f"# match {t:6} {m['name'][:40]:40} → {target} ({why})", file=sys.stderr) sensor = f' - {{ name: edgar filings, url: "https://data.sec.gov/submissions/CIK{m["cik"]}.json", type: REST_API, connector: edgar, tier: B, config: {{ forms: {FORMS} }} }}' sector = categories(m).strip("[]").split(",")[0] if target: if target in used: skipped.append(f"{t} → {target} already extended in this file") continue used.add(target) used_ciks.add(m["cik"]) print(f" - id: {target}") print(" extend: true") print(" categories: [filings]") print(f" aliases: [{t.lower()}]") print(" sensors:") print(sensor) stats.setdefault(f"extended/{sector}", []).append(t) continue if not ap: skipped.append(f"{t} ({m['name']}): no website (EDGAR/Wikidata/DOMAINS)") continue name = NAMES2.get(t) or pretty_name(t, m["name"] or t) sid = IDS2.get(t) or slug(re.sub(r"\s*\(.*?\)|\b(s\.?a\.?|a/s|n\.?v\.?|plc|public limited company|l\.?p\.?|ag|se|inc\.?|corp\.?|co\.?|ltd\.?|the)\b|,.*$", " ", name, flags=re.I)) if sid in all_ids or sid in used: sid = f"{sid}-{t.lower()}" if sid in all_ids or sid in used: skipped.append(f"{t}: id collision {sid}") continue used.add(sid) used_ciks.add(m["cik"]) country = country_of(m) print(f" - id: {sid}") print(f" name: {json.dumps(name, ensure_ascii=False)}") print(f" domain: {ap}") print(f" homepage: https://{'www.' + host if host.count('.') == 1 else host}") print(f" categories: {categories(m)}") print(" tier: B") print(" entity_type: company") print(f" aliases: [{t.lower()}]") print(f" country: {country}") print(f" notes: \"SIC {m.get('sic') or 'n/a'} {m.get('sicDescription') or ''} · {', '.join(m.get('exchanges') or [])} · CIK {int(m['cik'])}{' · foreign private issuer' if m.get('hasForeign') else ''}\"") print(" discover: { rss: true }") print(" sensors:") print(sensor) stats.setdefault(f"new/{sector}", []).append(t) for k in sorted(stats): print(f"{k}: {len(stats[k])} — {' '.join(stats[k])}", file=sys.stderr) for s in skipped: print(f"# skipped: {s}", file=sys.stderr) print(f"total sensors: {sum(len(v) for v in stats.values())}", file=sys.stderr) if __name__ == "__main__": if sys.argv[1:2] == ["fetch"]: fetch(sys.argv[2:]) else: emit()