/** * Shared helpers for the operators2 connector batch (code-backed facility parsers). * Every helper is deterministic and only reads what the page publishes: no geocoding, no guessed figures. */ import { load, type CheerioAPI } from "cheerio"; import type { ExtractedRecord, Parser, RawDocument } from "@dci/connectors"; import { jsonLd, metaDescription, pageTitle, registerParser } from "@dci/connectors"; import { cleanText, countryFromText, htmlToText, parseAreaHa, parseAreaSqm, parseMw, slugify } from "@dci/core"; /** Accumulates a facility record with a per-field extraction method (provenance). */ export class Rec { data: Record = {}; methods: Record = {}; set(field: string, value: unknown, method: string): this { if (value == null || value === "" || (Array.isArray(value) && !value.length)) return this; if (typeof value === "number" && !Number.isFinite(value)) return this; this.data[field] = typeof value === "string" ? cleanText(value) : value; this.methods[field] = method; return this; } has(field: string): boolean { return this.data[field] != null; } done(key: string, url: string, certainty = 0.85, kind: ExtractedRecord["kind"] = "facility"): ExtractedRecord { if (!this.has("website")) this.set("website", url, "url:page"); return { kind, key, data: this.data, methods: this.methods, certainty, url }; } } /** Page basics: cheerio root, visible body text (nav/header/footer removed), title, h1, meta description. */ export interface Page { $: CheerioAPI; html: string; text: string; title: string | null; h1: string | null; desc: string | null; url: string } export function page(doc: RawDocument): Page { const html = doc.text; const $ = load(html); const h1 = cleanText($("h1").first().text()); const $body = load(html); $body("script,style,noscript,svg,iframe,nav,header,footer,aside,form,[role=navigation],[class*=cookie],[id*=cookie],[class*=menu],[class*=breadcrumb]").remove(); const text = htmlToText($body("body").html() ?? "").replace(/[ \t ]+/g, " ").replace(/\n{2,}/g, "\n").trim(); return { $, html, text, title: pageTitle(html), h1, desc: metaDescription(html), url: doc.finalUrl || doc.url }; } export function isHtml(doc: RawDocument): boolean { return !doc.error && doc.status === 200 && Boolean(doc.text) && /html/i.test(doc.contentType ?? "text/html"); } export function first(text: string, re: RegExp, group = 1): string | null { const m = text.match(re); return m ? cleanText(m[group] ?? m[0]) : null; } export function all(text: string, re: RegExp): RegExpMatchArray[] { return [...text.matchAll(re)]; } /** Normalise area strings so parseAreaSqm understands "23 000 m2", "1,650m 2", "5000㎡", "141,000 SF", "450 SQM", "3,000 NTM". */ export function normArea(s: string): string { return s .replace(/(\d)[  ](?=\d{3}\b)/g, "$1") .replace(/㎡/g, " sqm").replace(/m\s?²|m\s2\b/gi, " sqm").replace(/\bSQM\b/g, "sqm").replace(/\bNTM\b/g, "sqm") .replace(/\bft²|ft2\b/gi, " sq ft").replace(/\b(SF|sf)\b/g, "sq ft").replace(/sq\.\s*ft\./gi, "sq ft").replace(/square\s+met(?:er|re)s?/gi, "sqm"); } export function sqm(s: string | null | undefined): number | null { return s ? parseAreaSqm(normArea(s)) : null; } export function ha(s: string | null | undefined): number | null { return s ? parseAreaHa(s) : null; } export function mw(s: string | null | undefined): number | null { if (!s) return null; const v = parseMw(s.replace(/\+/g, "").replace(/MWs\b/gi, "MW")); // "1.125 MW" on an English page is a decimal, not a European thousands separator (a 1,125 MW single facility is not plausible). if (v != null && v >= 1000 && /(^|[^\d,])\d\.\d{3}\s*MW/i.test(s)) return v / 1000; return v; } /** * MW explicitly labelled as IT / critical load for the facility described by `text`. * "At least" figures keep their raw shape — "10+MW", "20MW+", "150+ MW" — and `mw()` / `parseMw` read the stated figure. */ export function itMw(text: string): number | null { const pats = [ /(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:critical\s+)?(?:IT|critical)\s+(?:capacity|load|power)\b/i, /(?:IT|critical)\s+(?:load|capacity|power)(?:\s+capacity)?(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)/i, /(\d[\d.,]*\+?\s*MWs?\+?)\s+of\s+(?:critical\s+)?(?:IT|critical)\s+(?:load|power|capacity)/i, /(\d[\d.,]*\+?\s*MWs?\+?)\s+(?:critical|IT)\b/i, ]; for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } } return null; } /** MW labelled as (total) capacity / power without an IT qualifier. */ export function capacityMw(text: string): number | null { const pats = [/(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:capacity|power)\b/i, /(?:total\s+)?(?:capacity|power)(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)(?![\w])/i]; for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } } return null; } export function pue(text: string): number | null { const m = text.match(/(?:PUE|Power Usage Effectiveness)[^\d<≤\n]{0,30}[<≤]?\s*(\d[.,]\d{1,2})\b/i) ?? text.match(/[<≤]?\s*(\d[.,]\d{1,2})\s*(?:design\s+|average\s+(?:annuali[sz]ed\s+)?|target\s+)?PUE\b/i) ?? text.match(/sub\s*-?\s*(\d[.,]\d{1,2})\s*PUE/i); if (!m?.[1]) return null; const v = Number(m[1].replace(",", ".")); return v >= 1 && v <= 3 ? v : null; } const ROMAN: Record = { "1": "I", "2": "II", "3": "III", "4": "IV", i: "I", ii: "II", iii: "III", iv: "IV" }; export function tier(text: string): string | null { const m = text.match(/\b(?:Uptime(?:\s+Institute)?\s+)?(Tier|Rated)[\s-]*(IV|III|II|I|[1-4])\b(?![\d.])/i); if (!m) return null; const r = ROMAN[m[2]!.toLowerCase()] ?? m[2]!; return m[1]!.toLowerCase() === "rated" ? `Rated ${m[2]}` : `Tier ${r}`; } const CERTS: Array<[RegExp, string]> = [ [/ISO\s*\/?\s*IEC\s*27001|ISO\s*27001/i, "ISO 27001"], [/ISO\s*9001/i, "ISO 9001"], [/ISO\s*14001/i, "ISO 14001"], [/ISO\s*22301/i, "ISO 22301"], [/ISO\s*50001/i, "ISO 50001"], [/ISO\s*45001/i, "ISO 45001"], [/SOC\s*1\b/i, "SOC 1"], [/SOC\s*2\b/i, "SOC 2"], [/SOC\s*3\b/i, "SOC 3"], [/PCI[\s-]*DSS|PCI\s+compliant/i, "PCI DSS"], [/HIPAA/i, "HIPAA"], [/HITRUST/i, "HITRUST"], [/FedRAMP/i, "FedRAMP"], [/\bLEED\b/, "LEED"], [/ENERGY\s*STAR/i, "ENERGY STAR"], [/BREEAM/i, "BREEAM"], [/ISAE\s*3402/i, "ISAE 3402"], [/SSAE\s*1[68]/i, "SSAE 18"], [/T[ÜU]V/, "TÜV"], [/OCP[\s-]*Ready/i, "OCP Ready"], [/IGBC/i, "IGBC"], [/DGX[\s-]*Ready/i, "NVIDIA DGX-Ready"], [/Cyber\s*Essentials/i, "Cyber Essentials"], [/ASD\s+CCSL|SCEC/i, "ASD/SCEC"], [/NIST\s*800/i, "NIST 800-53"], ]; export function certifications(text: string): string[] { const out: string[] = []; for (const [re, name] of CERTS) if (re.test(text) && !out.includes(name)) out.push(name); return out; } /** "123 Main Street Suite 100, Springfield, IL 62701" → parts. US/CA street addresses only. */ export function usAddress(text: string): { street: string; city: string; region: string; postal: string } | null { const m = text.match(/(\d{1,6}(?:-\d+)?\s+[A-Z0-9][^\n,]{3,70}?)[,\s]+-?\s*([A-Z][a-zA-Z.'\- ]{2,40}?),\s*([A-Z]{2})\s+(\d{5}(?:-\d{4})?|[A-Z]\d[A-Z] ?\d[A-Z]\d)\b/); if (!m) return null; return { street: cleanText(m[1])!, city: cleanText(m[2])!, region: m[3]!, postal: m[4]! }; } /** Deterministic metro → ISO-2 lookup for operators whose pages name the city but not the country. */ export const CITY_COUNTRY: Record = { tokyo: "JP", osaka: "JP", inzai: "JP", saitama: "JP", "chiba new town": "JP", keihanna: "JP", shiohama: "JP", yoshikawa: "JP", mumbai: "IN", "navi mumbai": "IN", chennai: "IN", hyderabad: "IN", bengaluru: "IN", bangalore: "IN", delhi: "IN", noida: "IN", pune: "IN", kolkata: "IN", ahmedabad: "IN", jaipur: "IN", rabale: "IN", siruseri: "IN", singapore: "SG", johor: "MY", "johor bahru": "MY", "kuala lumpur": "MY", cyberjaya: "MY", batam: "ID", jakarta: "ID", manila: "PH", bangkok: "TH", seoul: "KR", busan: "KR", incheon: "KR", beijing: "CN", shanghai: "CN", shenzhen: "CN", "hong kong": "HK", taipei: "TW", sydney: "AU", melbourne: "AU", brisbane: "AU", perth: "AU", canberra: "AU", adelaide: "AU", darwin: "AU", geelong: "AU", maddington: "AU", auckland: "NZ", wellington: "NZ", london: "GB", slough: "GB", manchester: "GB", birmingham: "GB", edinburgh: "GB", newcastle: "GB", reading: "GB", harlow: "GB", farnborough: "GB", didcot: "GB", cardiff: "GB", croydon: "GB", "milton keynes": "GB", maidenhead: "GB", rotherham: "GB", fareham: "GB", northolt: "GB", elstree: "GB", enfield: "GB", hayes: "GB", frankfurt: "DE", offenbach: "DE", berlin: "DE", munich: "DE", mainz: "DE", hamburg: "DE", amsterdam: "NL", "schiphol-rijk": "NL", rotterdam: "NL", eindhoven: "NL", groningen: "NL", brussels: "BE", paris: "FR", marseille: "FR", magny: "FR", madrid: "ES", barcelona: "ES", milan: "IT", noviglio: "IT", zurich: "CH", geneva: "CH", vienna: "AT", warsaw: "PL", dublin: "IE", luxembourg: "LU", stockholm: "SE", oslo: "NO", stavanger: "NO", copenhagen: "DK", helsinki: "FI", reykjavik: "IS", athens: "GR", lisbon: "PT", prague: "CZ", johannesburg: "ZA", "cape town": "ZA", durban: "ZA", midrand: "ZA", samrand: "ZA", isando: "ZA", lagos: "NG", nairobi: "KE", kigali: "RW", accra: "GH", casablanca: "MA", rabat: "MA", cairo: "EG", "são paulo": "BR", "sao paulo": "BR", "rio de janeiro": "BR", brasília: "BR", brasilia: "BR", curitiba: "BR", "porto alegre": "BR", fortaleza: "BR", campinas: "BR", barueri: "BR", "santana de parnaíba": "BR", tamboré: "BR", "santana de parnaiba": "BR", querétaro: "MX", queretaro: "MX", "mexico city": "MX", bogotá: "CO", bogota: "CO", santiago: "CL", "buenos aires": "AR", lima: "PE", toronto: "CA", montreal: "CA", montréal: "CA", vancouver: "CA", calgary: "CA", beauharnois: "CA", chicago: "US", "los angeles": "US", houston: "US", atlanta: "US", dallas: "US", "fort worth": "US", phoenix: "US", seattle: "US", denver: "US", "new york": "US", "san jose": "US", "santa clara": "US", "silicon valley": "US", "northern virginia": "US", ashburn: "US", austin: "US", "san antonio": "US", miami: "US", boston: "US", philadelphia: "US", minneapolis: "US", "kansas city": "US", "st. louis": "US", "st louis": "US", "salt lake city": "US", "las vegas": "US", portland: "US", hillsboro: "US", columbus: "US", cleveland: "US", pittsburgh: "US", indianapolis: "US", memphis: "US", nashville: "US", charlotte: "US", tampa: "US", orlando: "US", sacramento: "US", "san diego": "US", reno: "US", quincy: "US", "san francisco": "US", omaha: "US", detroit: "US", milwaukee: "US", baltimore: "US", richmond: "US", raleigh: "US", cincinnati: "US", secaucus: "US", piscataway: "US", chaska: "US", "wichita falls": "US", "el paso": "US", jacksonville: "US", albuquerque: "US", "colorado springs": "US", "bluffdale": "US", "lithia springs": "US", }; export function countryFromCity(city: string | null | undefined): string | null { if (!city) return null; const k = city.toLowerCase().replace(/\s+/g, " ").trim(); if (CITY_COUNTRY[k]) return CITY_COUNTRY[k]!; for (const [name, iso] of Object.entries(CITY_COUNTRY)) if (new RegExp(`(^|[^a-z])${name.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}([^a-z]|$)`, "i").test(k)) return iso; return null; } /** First known metro named in a text (longest names first). */ export function cityFromText(text: string): string | null { const s = text.toLowerCase(); for (const name of Object.keys(CITY_COUNTRY).sort((a, b) => b.length - a.length)) if (new RegExp(`(^|[^a-z])${name.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}([^a-z]|$)`, "i").test(s)) return name.replace(/\b\w/g, (c) => c.toUpperCase()); return null; } /** Country from a text via country names first, then metro lookup. */ export function countryOf(...texts: Array): { iso: string; method: string } | null { const t = safeCountryText(texts.filter(Boolean).join(" \n ")); const c = countryFromText(t); if (c) return { iso: c, method: "regex:country-name" }; const city = cityFromText(t); const iso = countryFromCity(city); return iso ? { iso, method: "lookup:city-country" } : null; } /** Explicit lifecycle wording on the page (hero region). Operational wording wins over pipeline wording. */ export function statusOf(text: string, limit = 3000): string | null { const head = text.slice(0, limit); if (/\b(now open|in operation|(is|now|fully|became|currently|already) operational|operational since|fully let|fully leased|opened in (19|20)\d\d|ready for service now|RFS now|live since)\b/i.test(head) && !/\b(will be operational|becomes? operational|operational (in|by) (Q[1-4]|\d{4}|20\d\d))/i.test(head)) return "operational"; if (/\b(under construction|under development|currently being built|being constructed|construction (is )?under ?way|in the process of building|currently building|construction (has )?(started|begun|commenced)|breaking ground|broke ground|early stages of construction|scheduled to (deliver|open)|RFS (within|in) \d)/i.test(head)) return "under_construction"; if (/\b(coming soon|planned (for|to open)|in development|future (site|campus|phase)|announced|proposed|secured land|land (and power )?(acquired|secured))\b/i.test(head)) return "announced"; return null; } /** Remove phrases that would mis-resolve to a country ("Latin America" → US, "Georgia" the US state → GE). */ export function safeCountryText(t: string): string { let s = t.replace(/\b(Latin|South|North|Central) America\b/gi, " ").replace(/\bAmericas\b/gi, " ").replace(/\bAsia[- ]Pacific\b/gi, " "); if (/\bAtlanta\b|\bGA\b|\bUnited States\b|\bU\.S\./.test(s)) s = s.replace(/\bGeorgia\b/g, " "); return s; } /** Split "1300 Park Center Drive Austin" at the last street-type word. */ export function splitStreetCity(s: string): { street: string; city: string } | null { const m = s.match(/^(.*\b(?:Drive|Dr\.?|Road|Rd\.?|Street|St\.?|Avenue|Ave\.?|Boulevard|Blvd\.?|Parkway|Pkwy\.?|Way|Lane|Ln\.?|Court|Ct\.?|Highway|Hwy\.?|Trail|Place|Pl\.?|Circle|Cir\.?|Loop|Center|Centre)(?:\s+(?:Suite|Ste\.?|#)\s*[A-Z0-9-]+)?)\s+([A-Z][a-zA-Z.'\- ]{2,40})$/); return m ? { street: m[1]!.trim(), city: m[2]!.trim() } : null; } /** pt-BR / es numbers: "9.693,39" → "9693.39", "15.535" → "15535". */ export function latinNumbers(s: string): string { return s.replace(/(\d)\.(?=\d{3}\b)/g, "$1").replace(/(\d),(\d{1,2})\b/g, "$1.$2"); } export function titleCase(slug: string): string { return slug.replace(/[-_]+/g, " ").replace(/\b\w/g, (c) => c.toUpperCase()).trim(); } export function seg(url: string): string[] { try { return new URL(url).pathname.split("/").filter(Boolean); } catch { return []; } } export function key(op: string, ...parts: Array): string { return `${op}:${parts.filter(Boolean).map((p) => slugify(String(p))).join(":")}`; } /** JSON-LD objects of a type that carry a PostalAddress (LocalBusiness / Place / Organization…). */ export function ldWithAddress(html: string, type?: string): Array & { address: Record; geo?: Record }> { return jsonLd(html, type).filter((o) => o.address && typeof o.address === "object") as Array & { address: Record; geo?: Record }>; } /** Leaf of the JSON-LD BreadcrumbList ("Richmond, VA") — the page's own location line, more reliable than a facility-name h1. */ export function breadcrumbLeaf(html: string): string | null { for (const o of jsonLd(html, "BreadcrumbList")) { const items = o.itemListElement; if (!Array.isArray(items) || !items.length) continue; const last = items[items.length - 1] as Record | undefined; const item = last?.item as Record | string | undefined; const name = cleanText(String(last?.name ?? (typeof item === "object" ? item?.name : "") ?? "")); if (name && !/^home$/i.test(name)) return name; } return null; } export function ldCountry(a: Record): string | null { const c = a.addressCountry; if (!c) return null; const s = typeof c === "string" ? c : String((c as Record).name ?? ""); return /^[A-Z]{2}$/.test(s.trim()) ? s.trim() : countryFromText(s); } export function applyLdAddress(r: Rec, a: Record, method = "json-ld:PostalAddress"): void { r.set("address", a.streetAddress as string, method).set("city", a.addressLocality as string, method).set("regionName", a.addressRegion as string, method).set("postalCode", a.postalCode as string, method).set("countryIso2", ldCountry(a), method); } export function applyLdGeo(r: Rec, o: Record, precision = "exact", method = "json-ld:GeoCoordinates"): void { const g = o.geo as Record | undefined; if (!g) return; const lat = Number(g.latitude), lng = Number(g.longitude); if (Number.isFinite(lat) && Number.isFinite(lng) && (lat !== 0 || lng !== 0)) r.set("lat", lat, method).set("lng", lng, method).set("geoPrecision", precision, method); } /** * Version shared by every operators2 parser — part of the effective extractor version (docs/CONNECTORS.md § 6). * v2 (2026-09-12): "MW+" / "+MW" stat figures are read (itMw, capacityMw, stat regexes); EdgeConneX city from the breadcrumb. */ export const OPERATORS2_PARSER_VERSION = "v2"; /** Register a facility parser with the standard shape. */ export function facilityParser(name: string, parse: (p: Page, doc: RawDocument) => ExtractedRecord[], version = OPERATORS2_PARSER_VERSION): Parser { const parser: Parser = { name, version, pageTypes: ["facility_page"], parse: (doc) => (isHtml(doc) ? parse(page(doc), doc) : []) }; registerParser(parser); return parser; } /** Description from meta description, falling back to the first long paragraph of the page. */ export function describe(p: Page, min = 80): string | null { if (p.desc && p.desc.length >= 40 && !/meta description not set/i.test(p.desc)) return p.desc.slice(0, 600); const para = p.text.split("\n").map((l) => l.trim()).find((l) => l.length >= min && /[a-z]{3}/.test(l) && !/cookie|javascript/i.test(l)); return para ? para.slice(0, 600) : null; }