/** * Query interpretation for /search: pulls structured filters (country, operator, metro, status, min / max MW, * facility type, AI, entity = project, opening year) out of free text and returns the remaining text. * Pure function — unit-tested. */ import type { FacilityStatus, FacilityType, SearchResponse } from "@dci/core"; import { COUNTRY_ALIASES, normalizeStatus, parseMw } from "@dci/core"; export type Interpreted = NonNullable; export interface InterpretOptions { /** operator candidates already matched by trigram against the whole query (best first) */ operators?: Array<{ slug: string; name: string; similarity?: number }>; /** extra country names from the countries table: [{ iso2, name }] */ countries?: Array<{ iso2: string; name: string }>; /** metro / market names (and aliases) from the metros table */ metros?: Array<{ slug: string; name: string; aliases?: string[] }>; } const STATUS_PHRASES: Array<[RegExp, FacilityStatus]> = [ [/\b(under[- ]construction|being built|in construction|construction)\b/i, "under_construction"], [/\b(partially[- ]operational)\b/i, "partially_operational"], [/\b(operational|live|open(ed)?|existing|operating|in service)\b/i, "operational"], [/\b(planned|upcoming|future|announced|coming soon|pipeline|in development)\b/i, "announced"], [/\b(proposed)\b/i, "proposed"], [/\b(permitting|planning application|zoning)\b/i, "permitting"], [/\b(approved)\b/i, "approved"], [/\b(rumou?red)\b/i, "rumored"], [/\b(delayed|on hold|paused)\b/i, "delayed"], [/\b(cancell?ed|scrapped)\b/i, "cancelled"], [/\b(closed|decommissioned)\b/i, "closed"], [/\b(expansion)\b/i, "expansion"], ]; const TYPE_PHRASES: Array<[RegExp, FacilityType]> = [ [/\b(ai|artificial intelligence|gpu|ai[- ]ready)\b/i, "ai"], [/\b(hyperscale|hyperscaler)\b/i, "hyperscale"], [/\b(colocation|colo|retail colo)\b/i, "colocation"], [/\b(cloud region|cloud)\b/i, "cloud_region"], [/\b(edge)\b/i, "edge"], [/\b(carrier hotel)\b/i, "carrier_hotel"], [/\b(internet exchange|ixp?)\b/i, "internet_exchange"], [/\b(hpc|supercomput(er|ing))\b/i, "hpc"], [/\b(sovereign)\b/i, "sovereign_cloud"], [/\b(wholesale)\b/i, "wholesale"], [/\b(enterprise)\b/i, "enterprise"], [/\b(government)\b/i, "government"], ]; const STOP = new Set(["data", "center", "centers", "centre", "centres", "datacenter", "datacenters", "datacentre", "datacentres", "dc", "facility", "facilities", "in", "at", "near", "the", "of", "and", "with", "over", "above", "more", "than", "least", "min", "minimum", "campus", "campuses", "site", "sites", "under", "below", "less", "max", "maximum", "up", "to", "by", "opening", "opens", "open", "project", "projects"]); const MW_UNIT = String.raw`(\d+(?:[.,]\d+)?\s*\+?\s*(?:gw|gigawatts?|mw|megawatts?))\b`; const MAX_RE = new RegExp(String.raw`(?:<=?|under|below|less than|at most|max(?:imum)?|up to|smaller than|no more than)\s*${MW_UNIT}`, "i"); const MIN_RE = new RegExp(String.raw`(?:>=?|over|above|more than|at least|min(?:imum)?|\+|larger than|bigger than)?\s*${MW_UNIT}`, "i"); const AI_RE = /\b(ai|artificial intelligence|gpu|gpus|ai[- ]ready|ai[- ]factory|ai[- ]factories)\b/i; const PROJECT_RE = /\b(projects?|pipeline projects?)\b/i; const YEAR_RE = /\b(before|by|until|prior to|pre|after|from|since|post|opening|opens|open(?:ed|ing)? in|in|for|due)?\s*((?:19|20)\d{2})\b/i; function esc(s: string): string { return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } function strip(text: string, re: RegExp): string { return text.replace(re, " ").replace(/\s+/g, " ").trim(); } export function interpretQuery(raw: string, opts: InterpretOptions = {}): Interpreted { const out: Interpreted = {}; let text = raw.replace(/\s+/g, " ").trim(); if (!text) return out; // 1. Power figures: "<500 MW" / "under 500 MW" → maxMw; "100 MW", "over 50mw", "> 1 GW", "100+ MW" → minMw const maxMatch = text.match(MAX_RE); if (maxMatch) { const mw = parseMw(maxMatch[1]!.replace("+", "")); if (mw != null && mw > 0) out.maxMw = mw; text = strip(text, new RegExp(esc(maxMatch[0]), "i")); } const mwMatch = text.match(MIN_RE); if (mwMatch) { const mw = parseMw(mwMatch[1]!.replace("+", "")); if (mw != null && mw > 0) out.minMw = mw; text = strip(text, new RegExp(esc(mwMatch[0]), "i")); } // 2. Opening year phrases: "opening 2028", "before 2028", "after 2027" (a year right after a MW figure was consumed above) const yearMatch = text.match(YEAR_RE); if (yearMatch) { const y = Number(yearMatch[2]); const cue = (yearMatch[1] ?? "").toLowerCase(); if (y >= 1990 && y <= 2060) { out.year = y; out.yearOp = /^(before|by|until|prior to|pre)$/.test(cue) ? "before" : /^(after|from|since|post)$/.test(cue) ? "after" : "in"; text = strip(text, new RegExp(esc(yearMatch[0]), "i")); } } // 3. Operator (candidates supplied by the caller from a trigram lookup on the full query) for (const op of opts.operators ?? []) { const re = new RegExp(`(^|[^a-z0-9])${esc(op.name)}([^a-z0-9]|$)`, "i"); if (re.test(text) || (op.similarity ?? 0) >= 0.8) { out.operator = op.slug; if (re.test(text)) text = strip(text, re); else if ((op.similarity ?? 0) >= 0.85) text = ""; break; } } // 4. Metro / market names (longest first, word boundaries, aliases included) if (opts.metros?.length) { const cands = opts.metros.flatMap((m) => [m.name, ...(m.aliases ?? [])].filter((n) => n && n.length >= 3).map((n) => ({ slug: m.slug, n }))).sort((a, b) => b.n.length - a.n.length); for (const c of cands) { const re = new RegExp(`(^|[^\\p{L}\\p{N}])${esc(c.n)}([^\\p{L}\\p{N}]|$)`, "iu"); if (re.test(text)) { out.metro = c.slug; text = strip(text, re); break; } } } // 5. Country: bare ISO2 token in caps, known aliases, or the countries table const iso2Token = raw.match(/(^|\s)([A-Z]{2})(\s|$)/); if (iso2Token && /^(US|UK|CA|DE|FR|NL|GB|IE|SG|JP|AU|IN|BR|ES|IT|SE|NO|FI|DK|PL|CH|AT|BE|PT|MX|ZA|AE|SA|KR|CN|HK|TW|MY|ID|TH|VN|PH|NZ|CL|AR|CO|IL|TR|QA|KE|NG|EG)$/.test(iso2Token[2]!)) { const code = iso2Token[2]! === "UK" ? "GB" : iso2Token[2]!; out.countryIso2 = code; text = strip(text, new RegExp(`(^|\\s)${iso2Token[2]}(\\s|$)`)); } if (!out.countryIso2) { const lower = text.toLowerCase(); const keys = Object.keys(COUNTRY_ALIASES).filter((k) => k !== "usa_" && k.length > 2).sort((a, b) => b.length - a.length); for (const k of keys) { const re = new RegExp(`(^|[^a-z])${esc(k)}([^a-z]|$)`, "i"); if (re.test(lower)) { out.countryIso2 = COUNTRY_ALIASES[k]!; text = strip(text, re); break; } } } if (!out.countryIso2 && opts.countries?.length) { for (const c of [...opts.countries].sort((a, b) => b.name.length - a.name.length)) { const re = new RegExp(`(^|[^a-z])${esc(c.name)}([^a-z]|$)`, "i"); if (re.test(text)) { out.countryIso2 = c.iso2; text = strip(text, re); break; } } } // 6. Entity: "projects" (plural is the strongest signal) — "announced" / "planned" alone only set the status if (PROJECT_RE.test(text)) { out.entity = "project"; text = strip(text, PROJECT_RE); } // 7. Status for (const [re, st] of STATUS_PHRASES) { if (re.test(text)) { out.status = normalizeStatus(st) ?? st; text = strip(text, re); break; } } // 8. AI flag (kept alongside the facility type so callers can match is_ai / ai_evidence rather than the type column) if (AI_RE.test(text)) out.ai = true; // 9. Facility type for (const [re, ty] of TYPE_PHRASES) { if (re.test(text)) { out.facilityType = ty; text = strip(text, re); break; } } // 10. Remaining text (drop generic words) const rest = text.split(" ").filter((t) => t && !STOP.has(t.toLowerCase())).join(" ").trim(); if (rest) out.text = rest; return out; } export function hasFilters(i: Interpreted): boolean { return Boolean(i.countryIso2 || i.operator || i.metro || i.status || i.minMw != null || i.maxMw != null || i.facilityType || i.ai || i.entity || i.year != null); }