import { load, type CheerioAPI } from "cheerio"; import type { ExtractedRecord, RawDocument } from "@dci/connectors"; import { metaDescription } from "@dci/connectors"; import { cleanText, countryFromText, htmlToText, parseMw, parseAreaHa, parseAreaSqm } from "@dci/core"; /** * Shared helpers for operator connectors (batch 1). Pure functions over page HTML/text; no network. * Rule: we only surface what the page publishes. No geocoding, no guessed capacities. */ export const US_STATES = new Set(["AL", "AK", "AZ", "AR", "CA", "CO", "CT", "DE", "FL", "GA", "HI", "ID", "IL", "IN", "IA", "KS", "KY", "LA", "ME", "MD", "MA", "MI", "MN", "MS", "MO", "MT", "NE", "NV", "NH", "NJ", "NM", "NY", "NC", "ND", "OH", "OK", "OR", "PA", "RI", "SC", "SD", "TN", "TX", "UT", "VT", "VA", "WA", "WV", "WI", "WY", "DC"]); export const CA_PROVINCES = new Set(["AB", "BC", "MB", "NB", "NL", "NS", "NT", "NU", "ON", "PE", "QC", "SK", "YT"]); const US_STATE_NAMES: Record = { alabama: "AL", alaska: "AK", arizona: "AZ", arkansas: "AR", california: "CA", colorado: "CO", connecticut: "CT", delaware: "DE", florida: "FL", georgia: "GA", hawaii: "HI", idaho: "ID", illinois: "IL", indiana: "IN", iowa: "IA", kansas: "KS", kentucky: "KY", louisiana: "LA", maine: "ME", maryland: "MD", massachusetts: "MA", michigan: "MI", minnesota: "MN", mississippi: "MS", missouri: "MO", montana: "MT", nebraska: "NE", nevada: "NV", "new hampshire": "NH", "new jersey": "NJ", "new mexico": "NM", "new york": "NY", "north carolina": "NC", "north dakota": "ND", ohio: "OH", oklahoma: "OK", oregon: "OR", pennsylvania: "PA", "rhode island": "RI", "south carolina": "SC", "south dakota": "SD", tennessee: "TN", texas: "TX", utah: "UT", vermont: "VT", virginia: "VA", washington: "WA", "west virginia": "WV", wisconsin: "WI", wyoming: "WY" }; /** "Nevada" → "NV"; "VA" → "VA"; unknown → null. */ export function usStateCode(s: string | null | undefined): string | null { if (!s) return null; const t = s.trim(); if (US_STATES.has(t.toUpperCase()) && t.length === 2) return t.toUpperCase(); return US_STATE_NAMES[t.toLowerCase()] ?? null; } export interface ParsedAddress { address: string | null; city: string | null; region: string | null; postal: string | null; countryIso2: string | null } const STREET_SUFFIX = /^(.*\b(?:Drive|Dr|Road|Rd|Parkway|Pkwy|Court|Ct|Boulevard|Blvd|Street|St|Avenue|Ave|Way|Lane|Ln|Place|Pl|Circle|Cir|Trail|Trl|Highway|Hwy|Loop|Terrace|Ter|Plaza|Pike|Turnpike|Tpke|Expressway|Expy|Freeway|Fwy|Route|Rte|Square|Sq|Row|Run|Path|Alley|Center|Centre)\.?)(?:\s*\([^)]*\))?\s+(.+)$/i; /** North-American postal address: "21715 Filigree Court, Ashburn, VA 20147" / "22435 Glenn Drive Sterling, VA 20164 USA" / "123 Rue X, Montréal, QC H1A 1A1". */ export function parseNaAddress(raw: string | null | undefined): ParsedAddress | null { const s = cleanText(raw?.replace(/<[^>]+>/g, " ")); if (!s) return null; // "Itasca,IL 60143" (a
collapsed to nothing after the comma) is accepted: the state is anchored by the postal code that follows const m = s.match(/^(.+?)(?:,\s*|\s+)([A-Z]{2})[,\s]+(\d{5}(?:-\d{4})?|[A-Z]\d[A-Z]\s?\d[A-Z]\d)\b(?:[\s,]*(?:USA|U\.S\.A\.|United States|US|Canada))?\.?$/); if (!m) return null; const head = m[1]!.trim(); const region = m[2]!; const postal = m[3]!; const isCa = CA_PROVINCES.has(region) && /^[A-Z]\d[A-Z]/.test(postal); if (!isCa && !US_STATES.has(region)) return null; let street: string | null = head; let city: string | null = null; const lastComma = head.lastIndexOf(","); if (lastComma > 0) { street = head.slice(0, lastComma).trim(); city = head.slice(lastComma + 1).trim(); } else { const sm = head.match(STREET_SUFFIX); if (sm) { street = sm[1]!.trim(); city = sm[2]!.trim(); } } return { address: street || null, city: city || null, region, postal, countryIso2: isCa ? "CA" : "US" }; } /** European-style: "Hanauer Landstrasse 302, 60314 Frankfurt am Main, Germany" / "Huibertgatweg 2 9979 XZ Eemshaven , Netherlands". */ export function parseEuAddress(raw: string | null | undefined): ParsedAddress | null { const s = cleanText(raw?.replace(/<[^>]+>/g, " ")); if (!s) return null; const m = s.match(/^(.+?),?\s+((?:[A-Z]{1,2}-)?\d{4,6}(?:\s?[A-Z]{2})?)\s+([^,]+?)\s*,\s*([A-Za-z .'-]+)$/); if (m) { const countryIso2 = countryFromText(m[4]!); if (countryIso2) return { address: m[1]!.trim(), city: m[3]!.trim(), region: null, postal: m[2]!.trim(), countryIso2 }; } // Comma-separated with the country last: "…, Vienna, 1210, Austria" / "665 Ajax Avenue, Slough Trading Estate, Slough, SL1 4BG, United Kingdom" / "Street 1, City, Country" const parts = s.split(",").map((p) => p.trim()).filter(Boolean); if (parts.length >= 3) { const countryIso2 = countryFromText(parts[parts.length - 1]!); if (countryIso2 && !/\d/.test(parts[parts.length - 1]!)) { const rest = parts.slice(0, -1); const isPostal = (p: string) => /\d/.test(p) && p.length <= 10 && !/^\d+\s+[A-Za-z]/.test(p); let postal: string | null = null, city: string | null = null; if (rest.length >= 2 && isPostal(rest[rest.length - 1]!)) { postal = rest.pop()!; city = rest.pop() ?? null; } else if (rest.length >= 2) { city = rest.pop() ?? null; } // "Frankfurt am Main 60486" / "60486 Frankfurt am Main" inside the city token const cm = city?.match(/^(?:((?:[A-Z]{1,2}-)?\d{4,6})\s+(.+)|(.+?)\s+((?:[A-Z]{1,2}-)?\d{4,6}))$/); if (cm) { postal ??= (cm[1] ?? cm[4]) ?? null; city = (cm[2] ?? cm[3]) ?? city; } if (city && /\d/.test(city) && !/^\d/.test(city)) { /* keep */ } const street = rest.join(", "); if (city && /\d/.test(street + (postal ?? ""))) return { address: street || null, city, region: null, postal, countryIso2 }; } } return null; } /** Best-effort structured address; falls back to the raw string + country detected in it. */ export function parseAddress(raw: string | null | undefined): ParsedAddress | null { const s = cleanText(raw?.replace(/<[^>]+>/g, " ")); if (!s) return null; return parseNaAddress(s) ?? parseEuAddress(s) ?? { address: s, city: null, region: null, postal: null, countryIso2: countryFromText(s) }; } /** "Sterling, VA" / "Frankfurt, Germany" / "London, UK" → city, region, country. */ export function parseCityLine(raw: string | null | undefined): { city: string | null; region: string | null; countryIso2: string | null } { const s = cleanText(raw); if (!s) return { city: null, region: null, countryIso2: null }; const parts = s.split(",").map((p) => p.trim()).filter(Boolean); const city = parts[0] ?? null; const tail = parts[parts.length - 1] ?? ""; if (parts.length >= 2) { const st = usStateCode(tail); if (st) return { city, region: st, countryIso2: "US" }; if (CA_PROVINCES.has(tail.toUpperCase()) && tail.length === 2) return { city, region: tail.toUpperCase(), countryIso2: "CA" }; const c = countryFromText(tail); if (c) return { city, region: parts.length >= 3 ? parts[1] ?? null : null, countryIso2: c }; } return { city, region: null, countryIso2: countryFromText(s) }; } /** First MW figure appearing within `window` chars after a label. */ export function mwAfter(text: string, label: RegExp, window = 160): number | null { const m = text.match(label); if (!m || m.index == null) return null; const seg = text.slice(m.index + m[0].length, m.index + m[0].length + window); return parseMw(seg); } /** First MW figure immediately followed by `ctx` (anchored right after the figure, e.g. /\s*of critical IT load/). */ export function mwWithContext(text: string, ctx: RegExp, after = 60): number | null { const anchored = new RegExp(`^(?:${ctx.source})`, ctx.flags.replace("g", "")); for (const m of text.matchAll(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\s*(?:MW|megawatts?)\b\+?/gi)) { const i = m.index ?? 0; const tail = text.slice(i + m[0].length, i + m[0].length + after); if (anchored.test(tail)) return parseMw(m[0]); } return null; } /** "800K square feet" / "2.4 million sqft" / "105,000 SQ FT" / "18,000 sq. m." → m². */ export function areaSqm(text: string | null | undefined): number | null { if (!text) return null; const k = text.match(/(\d+(?:\.\d+)?)\s*[Kk]\b\s*(?:square\s*feet|sq\.?\s*ft|sqft|SF)\b/i); if (k) return Math.round(Number(k[1]) * 1000 * 0.09290304); const mm = text.match(/(\d+(?:\.\d+)?)\s*[Mm]\b\s*(?:square\s*feet|sq\.?\s*ft|sqft|SF)\b/); if (mm) return Math.round(Number(mm[1]) * 1_000_000 * 0.09290304); const sf = text.match(/(\d{1,3}(?:,\d{3})+|\d+)\s*SF\b/); if (sf) return Math.round(Number(sf[1]!.replace(/,/g, "")) * 0.09290304); return parseAreaSqm(text.replace(/sq\.\s*m\./i, "sqm")); } export function areaHa(text: string | null | undefined): number | null { return parseAreaHa(text ?? ""); } /** Facility code like DC2, FRA5, IAD12, NVA8, ATL01A. */ export const CODE_RE = /^[A-Z]{2,4}\d{1,3}[A-Z]?$/; export function extractCode(s: string | null | undefined): string | null { if (!s) return null; const m = s.toUpperCase().match(/\b([A-Z]{2,4}\d{1,3}[A-Z]?)\b/); return m ? m[1]! : null; } export function lastSlug(url: string): string { const u = url.split(/[?#]/)[0]!.replace(/\/+$/, ""); return decodeURIComponent(u.slice(u.lastIndexOf("/") + 1)).toLowerCase(); } export function pageHtml(doc: RawDocument): string { return doc.text; } export function $of(doc: RawDocument): CheerioAPI { return load(doc.text); } export function bodyText(html: string): string { return htmlToText(html); } export function description(html: string): string | null { return metaDescription(html); } /** Standard "coming soon / under construction" detection on a facility page's headline area only. */ export function statusFromText(text: string | null | undefined): "announced" | "under_construction" | null { if (!text) return null; if (/\b(under construction|currently being built|breaking ground|broke ground)\b/i.test(text)) return "under_construction"; if (/\b(coming soon|opening (in|soon)|future (campus|data center|facility)|planned (campus|data center|facility)|now pre-?leasing|is developing|under development|in development)\b/i.test(text)) return "announced"; return null; } /** Drop nulls / empty strings so `methods` only lists fields that exist. */ export function compact(data: Record, methods: Record): { data: Record; methods: Record } { const d: Record = {}; const m: Record = {}; for (const [k, v] of Object.entries(data)) { if (v == null || v === "" || (Array.isArray(v) && !v.length)) continue; d[k] = v; if (methods[k]) m[k] = methods[k]!; } return { data: d, methods: m }; } export function record(kind: ExtractedRecord["kind"], key: string, url: string, data: Record, methods: Record, certainty = 0.9): ExtractedRecord { const c = compact(data, methods); return { kind, key, url, data: { ...c.data, website: url }, methods: c.methods, certainty, pageType: "facility_page" }; } /** Certification-looking tokens from a list of strings (drops "target"/"pending" items). */ export function certificationsFrom(items: string[]): string[] { const out = new Set(); for (const raw of items) { const t = cleanText(raw); if (!t || /\b(target|pending|planned|in progress)\b/i.test(t)) continue; const m = t.match(/\b(ISO\/IEC\s?\d{4,5}(?:-\d)?|ISO\s?\d{4,5}(?:-\d)?|SOC\s?[123](?:\s?Type\s?(?:I{1,2}|[12]))?|PCI[- ]?DSS|HIPAA|HITRUST|FedRAMP(?:\s\w+)?|NIST\s?800-53(?:\/FISMA\s\w+)?|FISMA(?:\s\w+)?|LEED(?:\s\w+)?|BREEAM(?:\s"?\w+"?)?|Uptime(?:\sInstitute)?\s(?:Tier\s(?:I{1,3}V?|[1-4])[\w\s]*)|HECVAT|ITAR|GDPR|C5|TISAX|EN\s?50600|SS\s?564|MTCS(?:\s\w+)?|OSPAR|CSA\sSTAR|ENERGY\sSTAR|ISAE\s?3402|IRAP|TIA-942(?:\s\w+)?)\b/i); if (m) out.add(m[1]!.replace(/\s+/g, " ").trim()); } return [...out]; } export const KNOWN_CLOUDS = [/\bAWS\b|Amazon Web Services/i, /\bAzure\b|Microsoft/i, /Google Cloud|\bGCP\b|\bGoogle\b/i, /IBM Cloud/i, /Oracle(?:\sCloud|\sFastConnect)?/i, /Alibaba Cloud/i, /Salesforce/i, /SAP/i, /Tencent/i, /Huawei Cloud/i]; export const CLOUD_NAMES = ["AWS", "Microsoft Azure", "Google Cloud", "IBM Cloud", "Oracle Cloud", "Alibaba Cloud", "Salesforce", "SAP", "Tencent Cloud", "Huawei Cloud"]; export function cloudsIn(text: string | null | undefined): string[] { if (!text) return []; const out: string[] = []; KNOWN_CLOUDS.forEach((re, i) => { if (re.test(text)) out.push(CLOUD_NAMES[i]!); }); return out; } export function ixpsIn(text: string | null | undefined): string[] { if (!text) return []; const out = new Set(); for (const m of text.matchAll(/\b(Any2(?:\s?Exchange)?(?:\s(?:West|East|Denver|Chicago|Boston|Miami|Atlanta|New York|Los Angeles|Silicon Valley|Northern Virginia))?|LINX(?:\s\w+)?|DE-CIX(?:\s\w+)?|AMS-IX(?:\s\w+)?|NYIIX|SIX(?:\sSeattle)?|Equinix IX|Equinix Internet Exchange|MICE|France-IX|LONAP|NL-ix|JPNAP|JPIX|BBIX|HKIX|SGIX|Megaport IX|NAPAfrica|JINX|Community IX|MIX(?:\sMilan)?|ECIX|Netnod|SwissIX|CIXP|VIX|ESPANIX|GigaPIX|PTT|IX\.br)\b/g)) out.add(m[1]!.replace(/\s+/g, " ")); return [...out]; } /** Text of the first element matching any selector. */ export function firstText($: CheerioAPI, ...selectors: string[]): string | null { for (const s of selectors) { const el = $(s).first(); if (el.length) { const t = cleanText(el.text()); if (t) return t; } } return null; } /** Cooling keywords → short claim string (only when explicit). */ export function coolingFrom(text: string | null | undefined): string | null { if (!text) return null; const m = text.match(/\b(direct[- ]to[- ]chip liquid cooling|liquid[- ]cooling|immersion cooling|water[- ]free cooling|zero[- ]water cooling|closed[- ]loop (?:chilled[- ]water|water[- ]cooling|cooling)(?: systems?)?|air[- ]cooled chillers?|free cooling|evaporative cooling|adiabatic cooling|rear[- ]door heat exchangers?|air[- ]side economiz(?:ation|ers?))\b/i); return m ? m[1]!.replace(/\s+/g, " ") : null; } /** The clause (bounded by ". ", ", " or " and ") that carries a renewable / carbon-free energy claim; null when none. */ export function renewableFrom(text: string | null | undefined): string | null { if (!text) return null; const re = /\b(100%\s+(?:renewable|carbon[- ]free|green)|powered by (?:100% )?renewable|renewable (?:energy|power|electricity)(?:\s+(?:credits|options|certificates|sourc\w+|match\w*|procurement|supply))?|carbon[- ]free (?:energy|electricity)|wind (?:RECs?|power|energy)|solar (?:power|PPA|energy)|hydro(?:electric)?(?: power)?|green[- ]e\s+RECs?|RECs\s+from|vPPAs?|power purchase agreement)/i; const m = text.match(re); if (!m || m.index == null) return null; const bounds = [". ", ", ", " and ", "; "].map((sep) => { const p = text.lastIndexOf(sep, m.index!); return p < 0 ? 0 : p + sep.length; }); const start = Math.max(0, m.index - 80, ...bounds); let end = text.length; for (const sep of [". ", "; ", ", "]) { const p = text.indexOf(sep, m.index + m[0].length); if (p > 0 && p < end) end = p; } end = Math.min(end, m.index + m[0].length + 120); const clause = cleanText(text.slice(start, end)); return clause ? clause.slice(0, 200) : null; } export { cleanText, countryFromText, parseMw };