/** * Pure reconciliation logic (no database): facility candidate scoring, thresholds, field-merge policy, * completeness and confidence. Everything here is unit-tested; facilities.ts wires it to Postgres. * Algorithm + thresholds are documented in docs/RECONCILIATION.md. */ import { compareFacilityCodes, computeConfidence, COUNTRY_ALIASES, extractFacilityCodes, haversineKm, jaccard, matchRadiusKm, normalizeName, normalizeNameKeepCodes, PIPELINE_STATUSES, PRECISION_RANK, validLatLng, type ConfidenceLevel, type GeoPrecision, type SourceKind, } from "@dci/core"; import { authority, isPrimaryKind } from "./common.js"; import { findCanonicalOperator } from "./canonical-operators.js"; export const AUTO_MERGE_THRESHOLD = 0.92; export const PENDING_THRESHOLD = 0.6; /** Minimal projection of a stored facility used for scoring. */ export interface FacilityCandidate { id: string; name: string; normalizedName: string; aliases?: string[]; operatorId?: string | null; /** canonical operator name when known — lets the name signal ignore operator tokens ("Colt DCS", "Sabey") */ operatorName?: string | null; countryIso2?: string | null; city?: string | null; address?: string | null; lat?: number | null; lng?: number | null; geoPrecision?: string | null; externalIds?: Record | null; } /** Incoming record projection. */ export interface FacilityProbe { name: string; aliases?: string[]; operatorId?: string | null; operatorName?: string | null; countryIso2?: string | null; city?: string | null; address?: string | null; lat?: number | null; lng?: number | null; geoPrecision?: GeoPrecision | null; } export interface MatchScore { score: number; reasons: string[]; blocked: boolean; // different known operators without exact name+address distanceKm: number | null; } export type MatchDecision = "merge" | "pending" | "create"; export function decide(score: number): MatchDecision { if (score >= AUTO_MERGE_THRESHOLD) return "merge"; if (score >= PENDING_THRESHOLD) return "pending"; return "create"; } export function normalizeAddress(a: string | null | undefined): string { if (!a) return ""; return a .normalize("NFKD") .replace(/[̀-ͯ]/g, "") .toLowerCase() .replace(/\b(street|st)\b/g, "st") .replace(/\b(road|rd)\b/g, "rd") .replace(/\b(avenue|ave)\b/g, "ave") .replace(/\b(boulevard|blvd)\b/g, "blvd") .replace(/\b(drive|dr)\b/g, "dr") .replace(/\b(court|ct)\b/g, "ct") .replace(/\b(lane|ln)\b/g, "ln") .replace(/\b(place|pl)\b/g, "pl") .replace(/\b(parkway|pkwy)\b/g, "pkwy") .replace(/\b(highway|hwy)\b/g, "hwy") .replace(/\b(north|n|south|s|east|e|west|w)\b/g, (m) => m[0]!) .replace(/\b(suite|ste|unit|floor|fl)\b\s*\S+/g, " ") .replace(/[^a-z0-9]+/g, " ") .replace(/\s+/g, " ") .trim(); } function addressSimilarity(a: string | null | undefined, b: string | null | undefined): number | null { const na = normalizeAddress(a); const nb = normalizeAddress(b); if (!na || !nb) return null; if (na === nb) return 1; const ta = new Set(na.split(" ")); const tb = new Set(nb.split(" ")); const j = jaccard(ta, tb); // house number agreement is a strong hint — "21721 Filigree Ct" (US) or "Sägereistrasse 35" (Europe); a leading // token is always a candidate, other numeric tokens only when short enough not to be a postal code const houseNumbers = (s: string): Set => { const toks = s.split(" "); const out = new Set(); for (const [i, t] of toks.entries()) if (/^\d+[a-z]?$/.test(t) && (i === 0 || t.replace(/[a-z]$/, "").length <= 4)) out.add(t); return out; }; const ha = houseNumbers(na), hb = houseNumbers(nb); if (ha.size && hb.size) { const shared = [...ha].some((n) => hb.has(n)); return shared ? Math.max(j, 0.75) : Math.min(j, 0.4); } return j; } // --------------------------------------------------------------------------------------------------------------- // Name comparison /** Ordinals / roman numerals are compared as digits: "Campus II" = "Campus 2" = "Campus Two". */ const ORDINAL_TOKENS: Record = { one: "1", i: "1", two: "2", ii: "2", three: "3", iii: "3", four: "4", iv: "4", five: "5", v: "5", six: "6", vi: "6", seven: "7", vii: "7", eight: "8", viii: "8", nine: "9", ix: "9", ten: "10", x: "10" }; /** Tokens that distinguish sibling buildings of one site: directions, ordinals, single letters, small numbers. */ const SIBLING_WORDS = new Set(["north", "south", "east", "west", "northern", "southern", "eastern", "western", "a", "b", "c", "d", "e", "f"]); const isSiblingToken = (t: string): boolean => SIBLING_WORDS.has(t) || /^\d{1,3}$/.test(t); /** Canonical token set of a facility name (noise words removed, ordinals as digits). */ export function nameTokens(name: string | null | undefined): Set { if (!name) return new Set(); return new Set(normalizeName(name).split(" ").filter(Boolean).map((t) => ORDINAL_TOKENS[t] ?? t)); } const setsEqual = (a: Set, b: Set): boolean => a.size === b.size && [...a].every((t) => b.has(t)); const without = (a: Set, drop: Set): Set => new Set([...a].filter((t) => !drop.has(t))); const countryTokenCache = new Map(); /** Token sequences naming a country ("united kingdom", "uk", "denmark") — stripped from names before comparison. */ function countryAliasTokens(iso2: string | null | undefined): string[][] { if (!iso2) return []; const hit = countryTokenCache.get(iso2); if (hit) return hit; const out: string[][] = []; for (const [alias, code] of Object.entries(COUNTRY_ALIASES)) if (code === iso2) { const t = normalizeName(alias).split(" ").filter(Boolean); if (t.length) out.push(t); } countryTokenCache.set(iso2, out); return out; } /** Token sequences naming an operator (canonical name + curated aliases + the stored name). */ function operatorAliasTokens(names: Array): string[][] { const out: string[][] = []; for (const n of names) { if (!n) continue; const canon = findCanonicalOperator(n); for (const a of [n, ...(canon ? [canon.name, ...canon.aliases] : [])]) { const t = normalizeName(a).split(" ").filter(Boolean); if (t.length) out.push(t); } } return out; } /** Remove every alias whose tokens are all present ("united kingdom" is removed only when both tokens are there). */ function stripSequences(tokens: Set, sequences: string[][]): Set { let out = tokens; for (const seq of sequences) if (seq.every((t) => out.has(t))) out = without(out, new Set(seq)); return out; } export interface NameSimilarity { score: number; /** identical names (canonical token sets, or code-preserving normalization) */ exact: boolean; alias: boolean; /** the names only differ by sibling tokens: "Telehouse North" / "Telehouse East", "Inzai 3" / "Inzai 4", "Campus I" / "Campus II" */ siblings: boolean; /** identical once operator and country tokens are removed: "Google Fredericia" / "Fredericia, Denmark" (operator Google) */ residualEqual: boolean; /** at least one name is nothing but the operator's name ("Pulsant", "BCX", "Digital Realty Data Center") */ operatorOnly: boolean; /** number of tokens in the shorter full name */ minTokens: number; } export function nameSimilarity(probe: FacilityProbe, cand: FacilityCandidate): NameSimilarity { const pt = nameTokens(probe.name); const ct = nameTokens(cand.name); const minTokens = Math.min(pt.size, ct.size); const none: NameSimilarity = { score: 0, exact: false, alias: false, siblings: false, residualEqual: false, operatorOnly: false, minTokens }; // names that normalize to nothing (non-Latin scripts) are compared verbatim — never "equal because both empty" if (!pt.size || !ct.size) { const rp = probe.name.trim().toLowerCase(), rc = cand.name.trim().toLowerCase(); return rp && rp === rc ? { ...none, exact: true, score: 1 } : none; } const opSeqs = operatorAliasTokens([probe.operatorName, cand.operatorName]); const ctySeqs = [...countryAliasTokens(probe.countryIso2), ...countryAliasTokens(cand.countryIso2)]; const pr = stripSequences(stripSequences(pt, opSeqs), ctySeqs); const cr = stripSequences(stripSequences(ct, opSeqs), ctySeqs); // a name that is just a brand ("Digital Realty Data Center") says nothing about which building it is const brandOnly = (tokens: Set, raw: string) => !tokens.size || !!findCanonicalOperator(raw) || (opSeqs.length === 0 && !!findCanonicalOperator([...tokens].join(" "))); const operatorOnly = brandOnly(pr, probe.name) || brandOnly(cr, cand.name); // siblings: identical apart from direction / ordinal / letter tokens, with a shared non-sibling stem const diff = new Set([...[...pt].filter((t) => !ct.has(t)), ...[...ct].filter((t) => !pt.has(t))]); const stem = [...pt].filter((t) => ct.has(t) && !isSiblingToken(t)); const siblings = diff.size > 0 && stem.length > 0 && [...diff].every(isSiblingToken); const exact = setsEqual(pt, ct) || (normalizeNameKeepCodes(probe.name) !== "" && normalizeNameKeepCodes(probe.name) === normalizeNameKeepCodes(cand.name)); if (exact) return { ...none, score: 1, exact: true, operatorOnly }; if (siblings) return { ...none, score: 0.2, siblings: true, operatorOnly }; const residualEqual = pr.size > 0 && cr.size > 0 && setsEqual(pr, cr); if (residualEqual) return { ...none, score: 0.9, residualEqual: true, operatorOnly }; // alias hit: only aliases that carry more than a brand ("meta" alone must not link every Meta site) const meaningful = (a: string) => { const t = nameTokens(a); return t.size >= 2 || (t.size === 1 && [...t][0]!.length >= 5 && !findCanonicalOperator(a)); }; const probeAliases = [probe.name, ...(probe.aliases ?? [])].filter(meaningful).map((a) => normalizeName(a)); const candAliases = [cand.name, ...(cand.aliases ?? [])].filter(meaningful).map((a) => normalizeName(a)); if (probeAliases.some((a) => candAliases.includes(a))) return { ...none, score: 0.9, alias: true, operatorOnly }; // token overlap on the residual names (operator / country tokens ignored when known) const a = pr.size ? pr : pt, b = cr.size ? cr : ct; const j = jaccard(a, b); const inter = [...a].filter((t) => b.has(t)).length; const containment = a.size && b.size ? inter / Math.min(a.size, b.size) : 0; // containment ("Equinix DC12" inside "Equinix DC12 Building B") beats jaccard, but a single shared token is weak const cScore = containment * (Math.min(a.size, b.size) >= 2 ? 0.85 : 0.6); return { ...none, score: Math.max(j, cScore), operatorOnly }; } /** Code prefixes that name a building inside a site rather than the site itself. */ const BUILDING_CODE_PREFIXES = new Set(["DC", "B", "BLDG", "BUILDING", "HALL", "DH", "PH", "PHASE", "BLD", "BLOCK", "H"]); /** true when one name carries a building code the other lacks, or when exactly one of the two names says "campus". */ function subBuildingCodes(a: string, b: string): boolean { const ca = extractFacilityCodes(a), cb = extractFacilityCodes(b); const onlyA = ca.filter((c) => !cb.includes(c)), onlyB = cb.filter((c) => !ca.includes(c)); const building = (codes: string[]) => codes.some((c) => BUILDING_CODE_PREFIXES.has(c.replace(/\d.*$/, ""))); if ((building(onlyA) && !onlyB.length) || (building(onlyB) && !onlyA.length)) return true; const campusA = /\bcampus\b/i.test(a), campusB = /\bcampus\b/i.test(b); return campusA !== campusB && (onlyA.length > 0 || onlyB.length > 0); } /** true when the un-attributed side's name starts with the other side's operator ("Google Data Center Fredericia" vs operator Google). */ function operatorImpliedByName(probe: FacilityProbe, cand: FacilityCandidate): boolean { const pair = !probe.operatorId && cand.operatorId && cand.operatorName ? [probe.name, cand.operatorName] : probe.operatorId && !cand.operatorId && probe.operatorName ? [cand.name, probe.operatorName] : null; if (!pair) return false; const tokens = nameTokens(pair[0]); return operatorAliasTokens([pair[1]]).some((seq) => seq.every((t) => tokens.has(t))); } /** * Weighted score of an incoming record against a stored candidate. * Weights: name 0.40, facility codes 0.15, operator 0.20, distance 0.15, address 0.10 — weights for signals * that are unavailable on either side are redistributed over the available ones. Rules on top (docs/RECONCILIATION.md): * operator+code and operator+site raise the name signal; sibling tokens and conflicting codes are hard negatives * (score < pending); an exact multi-token name at the same spot with no operator conflict auto-merges. */ export function scoreFacilityMatch(probe: FacilityProbe, cand: FacilityCandidate): MatchScore { const reasons: string[] = []; const parts: Array<{ w: number; v: number }> = []; const name = nameSimilarity(probe, cand); if (name.exact) reasons.push("name:exact"); else if (name.alias) reasons.push("name:alias"); else if (name.residualEqual) reasons.push("name:site-equal"); else if (name.siblings) reasons.push("name:siblings"); else reasons.push(`name:${name.score.toFixed(2)}`); if (name.operatorOnly) reasons.push("name:operator-only"); // facility codes (DC12, FR5, LONDON6, Inzai 3…): shared → 1, same prefix with different numbers → 0, else no signal const codes = compareFacilityCodes(extractFacilityCodes(probe.name), extractFacilityCodes(cand.name)); if (codes != null) reasons.push(codes ? "codes:match" : "codes:conflict"); // operator let op = 0.5; let blocked = false; if (probe.operatorId && cand.operatorId) { if (probe.operatorId === cand.operatorId) { op = 1; reasons.push("operator:same"); } else { op = 0; reasons.push("operator:different"); const addr = addressSimilarity(probe.address, cand.address); blocked = !(name.exact && addr === 1); } } else if (operatorImpliedByName(probe, cand)) { op = 0.9; reasons.push("operator:name-implied"); } else reasons.push("operator:unknown"); // distance let geo: number | null = null; let distanceKm: number | null = null; const radius = matchRadiusKm(probe.geoPrecision ?? "unknown", (cand.geoPrecision as GeoPrecision | undefined) ?? "unknown"); let sameCity = false; if (validLatLng(probe.lat, probe.lng) && validLatLng(cand.lat, cand.lng)) { distanceKm = haversineKm(probe.lat as number, probe.lng as number, cand.lat as number, cand.lng as number); geo = distanceKm <= radius ? 1 - (distanceKm / radius) * 0.4 : Math.max(0, 1 - distanceKm / (radius * 4)); reasons.push(`distance:${distanceKm.toFixed(2)}km/${radius}km`); } else if (probe.city && cand.city && normalizeName(probe.city) === normalizeName(cand.city)) { geo = 0.6; sameCity = true; reasons.push("city:same"); } // "same place": inside twice the match radius, or the same city without coordinates const samePlace = distanceKm != null ? distanceKm <= 2 * radius : sameCity; const geoCompatible = geo == null || samePlace; const addr = addressSimilarity(probe.address, cand.address); if (addr != null) reasons.push(`address:${addr.toFixed(2)}`); let nameScore = name.score; // Strong identifier rule: same (or name-implied) operator + same facility code + geographically compatible → exact. if (codes === 1 && op >= 0.9 && geoCompatible) { nameScore = Math.max(nameScore, 0.95); reasons.push("rule:operator+code"); } // Same site rule: names identical once the operator / country words are removed ("Fredericia" = "Fredericia, Denmark"). if (name.residualEqual && op >= 0.9 && geoCompatible) { nameScore = Math.max(nameScore, op === 1 ? 1 : 0.95); reasons.push("rule:operator+site"); } // Conflicting codes (DC12 vs DC13, LONDON6 vs LONDON5) and sibling tokens (North vs East, I vs II) are hard negatives. if (codes === 0) nameScore = Math.min(nameScore, 0.45); if (name.siblings) nameScore = Math.min(nameScore, 0.2); // A brand-only name ("Pulsant", "Switch") is not evidence for one particular building of that brand. if (name.operatorOnly && !(name.exact && codes === 1)) nameScore = Math.min(nameScore, 0.3); parts.push({ w: 0.4, v: nameScore }); if (codes != null) parts.push({ w: 0.15, v: codes }); parts.push({ w: 0.2, v: op }); if (geo != null) parts.push({ w: 0.15, v: geo }); if (addr != null) parts.push({ w: 0.1, v: addr }); const wsum = parts.reduce((s, p) => s + p.w, 0); let score = parts.reduce((s, p) => s + p.w * p.v, 0) / wsum; // Exact multi-token name at the same spot, no operator conflict, no address disagreement → the same building. // ("Meta Odense Data Center" ×2 in Odense, "Adeo Datacenter ApS" ×2 at 0 m, "Equinix ZRH5" from two sources.) const nameIsSpecific = name.minTokens >= 2 || addr === 1 || (distanceKm != null && distanceKm <= 0.1); if (name.exact && !name.operatorOnly && op !== 0 && samePlace && (addr == null || addr >= 0.75) && nameIsSpecific && codes !== 0) { score = Math.max(score, op === 1 ? 0.93 : 0.92); reasons.push("rule:exact-name+site"); } // An alias may come from a rename or an earlier fold: it can send a pair to review, never auto-merge it. if (name.alias && score >= AUTO_MERGE_THRESHOLD) { score = AUTO_MERGE_THRESHOLD - 0.01; reasons.push("rule:alias-review"); } // Campus vs one of its buildings ("DATA4 MAD1" vs "MAD1-DC01", "Teraco JB3 Isando Campus" vs "JB3 Data Centre"): // a shared site code with an extra building code on one side is a containment, not an identity → review. if (codes === 1 && subBuildingCodes(probe.name, cand.name) && score >= AUTO_MERGE_THRESHOLD) { score = AUTO_MERGE_THRESHOLD - 0.01; reasons.push("rule:campus-vs-building"); } // country mismatch is decisive if (probe.countryIso2 && cand.countryIso2 && probe.countryIso2 !== cand.countryIso2) { score = Math.min(score, 0.3); reasons.push("country:different"); } // far apart with precise coordinates on both sides → not the same building if (distanceKm != null && distanceKm > 4 * radius) { score = Math.min(score, 0.5); reasons.push("distance:too-far"); } if (codes === 0 || name.siblings) { score = Math.min(score, PENDING_THRESHOLD - 0.01); reasons.push(codes === 0 ? "rule:code-conflict" : "rule:siblings"); } if (blocked) { score = Math.min(score, PENDING_THRESHOLD - 0.01); reasons.push("blocked:different-operators"); } return { score: Math.round(score * 1000) / 1000, reasons, blocked, distanceKm }; } export function bestMatch(probe: FacilityProbe, candidates: FacilityCandidate[]): { candidate: FacilityCandidate; match: MatchScore } | null { let best: { candidate: FacilityCandidate; match: MatchScore } | null = null; for (const c of candidates) { const m = scoreFacilityMatch(probe, c); if (!best || m.score > best.match.score) best = { candidate: c, match: m }; } return best; } // --------------------------------------------------------------------------------------------------------------- // Field merge policy export interface FieldObservation { value: unknown; sourceKind?: string | null; confidence?: string | null; isEstimate?: boolean | null; observedAt?: string | null; // ISO sourceId?: string | null; url?: string | null; } export type MergeReason = "empty" | "same-source" | "higher-authority" | "equal-authority-stale" | "estimate-vs-measured" | "lower-authority" | "keep" | "equal"; /** A stored value observed by a peer source within this window is considered fresh and is not churned by equal-authority peers. */ export const STALE_AFTER_DAYS = 30; /** * Decide whether an incoming observation replaces the stored one. * - empty stored → take * - the same source re-observing the same page → take (a source may correct itself) * - measured beats estimate; higher authority beats lower * - equal authority → keep the stored value while it is fresh (< 30 days); replace it once stale. * (Two pages of one source describing the same facility must not make values ping-pong on every crawl.) */ export function shouldReplace(incoming: FieldObservation, stored: FieldObservation | null): { replace: boolean; reason: MergeReason } { if (stored == null || stored.value == null || stored.value === "") return { replace: incoming.value != null, reason: "empty" }; if (incoming.value == null) return { replace: false, reason: "keep" }; if (JSON.stringify(incoming.value) === JSON.stringify(stored.value)) return { replace: false, reason: "equal" }; const sameSource = !!incoming.sourceId && !!stored.sourceId && incoming.sourceId === stored.sourceId; const sameUrl = !incoming.url || !stored.url || incoming.url === stored.url; if (sameSource && sameUrl) return { replace: true, reason: "same-source" }; if (!!incoming.isEstimate !== !!stored.isEstimate) return { replace: !incoming.isEstimate, reason: "estimate-vs-measured" }; const ai = authority(incoming.sourceKind, incoming.confidence, incoming.isEstimate); const as = authority(stored.sourceKind, stored.confidence, stored.isEstimate); if (ai > as) return { replace: true, reason: "higher-authority" }; if (ai < as) return { replace: false, reason: "lower-authority" }; const ti = incoming.observedAt ? Date.parse(incoming.observedAt) : Date.now(); const ts = stored.observedAt ? Date.parse(stored.observedAt) : 0; return ti - ts > STALE_AFTER_DAYS * 86_400_000 ? { replace: true, reason: "equal-authority-stale" } : { replace: false, reason: "keep" }; } /** MW policy: same as shouldReplace but only operator/government/filing/utility sources may overwrite a primary-sourced figure. */ export function shouldReplaceMw(incoming: FieldObservation, stored: FieldObservation | null): { replace: boolean; reason: MergeReason } { const base = shouldReplace(incoming, stored); if (!base.replace || !stored || stored.value == null) return base; if (base.reason === "same-source" || base.reason === "estimate-vs-measured") return base; if (isPrimaryKind(stored.sourceKind) && !isPrimaryKind(incoming.sourceKind)) return { replace: false, reason: "lower-authority" }; return base; } /** Geo: the incoming point may only replace a stored one when its precision is at least as good. */ export function shouldReplaceGeo(incomingPrecision: GeoPrecision, storedPrecision: string | null | undefined, storedHasCoords: boolean, incomingIsSameSource = false): boolean { if (!storedHasCoords) return true; const sp = storedPrecision && storedPrecision in PRECISION_RANK ? PRECISION_RANK[storedPrecision as GeoPrecision] : 0; if (incomingIsSameSource && PRECISION_RANK[incomingPrecision] === sp) return true; return PRECISION_RANK[incomingPrecision] > sp || (PRECISION_RANK[incomingPrecision] === sp && !incomingIsSameSource); } // --------------------------------------------------------------------------------------------------------------- // Completeness & confidence export interface CompletenessInput { geoPrecision?: string | null; hasCoords: boolean; operatorId?: string | null; address?: string | null; status?: string | null; itCapacityMw?: number | null; totalPowerMw?: number | null; plannedPowerMw?: number | null; mwIsEstimate?: boolean; facilityType?: string | null; openedOn?: string | null; website?: string | null; description?: string | null; carriersCount?: number | null; ixpCount?: number | null; } /** 0–100. Weights: geo 15, operator 10, address 10, status 10, MW 20, type 5, opened 10, website 5, description 5, tenants/IXPs 10. */ export function completenessScore(f: CompletenessInput): number { let s = 0; if (f.hasCoords) { const p = f.geoPrecision ?? "unknown"; s += p === "exact" || p === "parcel" || p === "street" ? 15 : p === "unknown" ? 4 : 8; } if (f.operatorId) s += 10; if (f.address && f.address.trim().length > 4) s += 10; if (f.status && f.status !== "unknown") s += 10; const mw = f.itCapacityMw ?? f.totalPowerMw ?? null; if (mw != null && mw > 0) s += f.mwIsEstimate ? 14 : 20; else if (f.plannedPowerMw != null && f.plannedPowerMw > 0) s += f.mwIsEstimate ? 8 : 12; if (f.facilityType && f.facilityType !== "unknown") s += 5; if (f.openedOn) s += 10; if (f.website) s += 5; if (f.description && f.description.trim().length > 40) s += 5; if ((f.carriersCount ?? 0) > 0 || (f.ixpCount ?? 0) > 0) s += 10; return Math.max(0, Math.min(100, s)); } export interface ConfidenceSummaryInput { /** distinct source kinds with current provenance for this facility */ sourceKinds: string[]; /** number of distinct sources with current provenance (defaults to the number of kinds) */ sourceCount?: number; lastVerifiedIso?: string | null; /** true when every MW observation is an estimate and no non-MW primary source exists */ onlyEstimates?: boolean; } const KIND_ORDER: SourceKind[] = ["operator", "government", "filing", "utility", "cloud_provider", "registry", "dataset", "community", "secondary", "news"]; /** Kinds whose single word is enough for `high`: the operator itself, a government / regulator, a filing, a utility, a cloud provider. */ const AUTHORITATIVE_KINDS: ReadonlySet = new Set(["operator", "government", "filing", "utility", "cloud_provider"]); /** Crowd-sourced or second-hand kinds: alone, they are `unverified`; corroborated by a second source, `moderate`. */ const COMMUNITY_KINDS: ReadonlySet = new Set(["community", "secondary", "news"]); /** * Facility-level confidence: best source kind as base, corroborated by the number of distinct source kinds. * - `high` only when an authoritative kind (operator / government / filing / utility / cloud provider) backs the record; * a registry or dataset alone is `moderate`; * - a single community / secondary / news source is `unverified` (OpenStreetMap features seen nowhere else); * - `verified` needs ≥ 2 independent authoritative kinds. */ export function facilityConfidence(i: ConfidenceSummaryInput): ConfidenceLevel { const kinds = i.sourceKinds.filter((k): k is SourceKind => (KIND_ORDER as string[]).includes(k)); if (!kinds.length) return "unverified"; const best = KIND_ORDER.find((k) => kinds.includes(k))!; const sourceCount = i.sourceCount ?? kinds.length; if (COMMUNITY_KINDS.has(best) && sourceCount <= 1) return i.onlyEstimates ? "estimated" : "unverified"; const ageDays = i.lastVerifiedIso ? Math.max(0, (Date.now() - Date.parse(i.lastVerifiedIso)) / 86_400_000) : 0; const authoritative = kinds.filter((k) => AUTHORITATIVE_KINDS.has(k)); // "verified" requires ≥ 2 independent authoritative kinds agreeing on the record const corroborations = authoritative.length >= 2 ? authoritative.length : Math.min(1, kinds.length - 1); const level = computeConfidence({ sourceKind: best, corroborations, ageDays, isEstimate: !!i.onlyEstimates }); if (!authoritative.length && (level === "high" || level === "verified")) return "moderate"; return level; } export function isPipelineStatus(status: string | null | undefined): boolean { return !!status && (PIPELINE_STATUSES as readonly string[]).includes(status); } export function bestMw(f: { itCapacityMw?: number | null; totalPowerMw?: number | null; plannedPowerMw?: number | null }): number | null { return f.itCapacityMw ?? f.totalPowerMw ?? f.plannedPowerMw ?? null; }