SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
2 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%

Parsers: EdgeConneX breadcrumb city, QTS figure-before-label + own-caption map pins, '+'-suffixed MW across operators2 (v2), news main-text teaser/footer stripping + operator scope + portfolio money filter (news_v3), quoted project names; all fixture gaps promoted to assertions

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Simon-Pierre Boucher committed 20 days ago (Sep 12, 2026) parent 86241c4

15 changed files +263 −98

modified apps/worker/fixtures/datacenterdynamics/sb-gruppe-bucharest.expected.json +5 −11
@@ -3,7 +3,7 @@
3 3 "documentId": "doc_555b8c8f69f1bdbb",
4 4 "fetchedAt": "2026-09-12T02:09:08.275Z",
5 5 "contentType": "text/html; charset=utf-8",
6 − "note": "NEW_BUILD: 'The 12MW facility has secured planning permission' → news_event + project (12 MW, Bucharest, RO). Two attribution errors are flagged in _todo: the $9.4bn is S+B Gruppe's whole portfolio ('total investment volume of €8.1 billion ($9.4bn)'), and 'Microsoft' only appears in the related-articles footer ('Microsoft expands plans for La Porte…'). Author byline redacted.",
6 + "note": "NEW_BUILD: 'The 12MW facility has secured planning permission' → news_event + project (12 MW, Bucharest, RO). Two attribution traps, both handled since 2026-09-12: the $9.4bn is S+B Gruppe's whole portfolio ('total investment volume of €8.1 billion ($9.4bn)') → no investment; 'Microsoft' only appears in the 'More in Construction & Site Selection' teasers, which mainText now drops → no operator (S+B Gruppe is not in the lexicon). The project keeps its name from the lead ('Named ‘NIMB,’'). Author byline redacted.",
7 7 "counts": {
8 8 "news_event": 1,
9 9 "project": 1
@@ -25,11 +25,8 @@
25 25 "countryIso2": "RO",
26 26 "status": "announced",
27 27 "projectClass": "NEW_BUILD",
28 − "_todo": {
29 − "operatorName": null,
30 − "investmentUsd": null,
31 − "_reason": "extract-project: operator 'Microsoft' comes from the 'More in Construction & Site Selection' footer teasers that mainText() keeps; the $9.4bn is the developer's portfolio volume, not this project's investment (the article gives no figure for it). S+B Gruppe is not in the operators lexicon, so the correct operator is null (or 'S+B Gruppe' once added)"
32 − }
28 + "operatorName": null,
29 + "investmentUsd": null
33 30 }
34 31 ],
35 32 "thirdParty": true,
@@ -40,10 +37,7 @@
40 37 "country": "RO",
41 38 "city": "Bucharest",
42 39 "status": "announced",
43 − "_todo": {
44 − "operator": null,
45 − "investmentUsd": null,
46 − "_reason": "see projects[0]._todo"
47 − }
40 + "operator": null,
41 + "investmentUsd": null
48 42 }
49 43 }
modified apps/worker/fixtures/edgeconnex/richmond-va.expected.json +4 −7
@@ -3,7 +3,7 @@
3 3 "documentId": "doc_498ed95c603598af",
4 4 "fetchedAt": "2026-09-11T08:11:23.445Z",
5 5 "contentType": "text/html; charset=UTF-8",
6 − "note": "No JSON-LD LocalBusiness with a street on this page → h1 fallback. h1 is 'Richmond Data Center' and the breadcrumb says 'Richmond, VA': the parser currently copies the whole h1 into city (known gap, see _todo).",
6 + "note": "No JSON-LD LocalBusiness with a street on this page → metro-level fallback. h1 is 'Richmond Data Center' (facility name); the JSON-LD BreadcrumbList leaf 'Richmond, VA' carries city + state (fixed 2026-09-12: the parser used to copy the whole h1 into city).",
7 7 "counts": {
8 8 "facility": 1
9 9 },
@@ -12,6 +12,8 @@
12 12 "key": "edgeconnex:richmond-va",
13 13 "name": "EdgeConneX Richmond Data Center",
14 14 "operatorName": "EdgeConneX",
15 + "city": "Richmond",
16 + "regionName": "VA",
15 17 "countryIso2": "US",
16 18 "itCapacityMw": null,
17 19 "certifications": [
@@ -20,12 +22,7 @@
20 22 "PCI DSS",
21 23 "HIPAA",
22 24 "ENERGY STAR"
23 − ],
24 − "_todo": {
25 − "city": "Richmond",
26 − "regionName": "VA",
27 − "_reason": "edgeconnex_facility_v1 h1 fallback: /^([A-Z][^,]+?)(?:,\\s*([A-Z][^,]+))?$/ captures 'Richmond Data Center' as the city; the breadcrumb 'Richmond, VA' carries city + state"
28 − }
25 + ]
29 26 }
30 27 ]
31 28 }
modified apps/worker/fixtures/nextdc/p2-perth.expected.json +3 −6
@@ -3,7 +3,7 @@
3 3 "documentId": "doc_10f36a288aedf0b3",
4 4 "fetchedAt": "2026-09-11T08:31:56.476Z",
5 5 "contentType": "text/html; charset=UTF-8",
6 − "note": "Same stat block as B2 but '20MW+ IT Capacity': the trailing '+' breaks /(\\d[\\d.,]*\\s*MW)\\s*\\n?\\s*IT Capacity/ and itMw() finds no 'IT' qualifier → itCapacityMw missing (see _todo). 12,000 m², 6,300 racks, Tier IV are fine.",
6 + "note": "Same stat block as B2 but '20MW+ IT Capacity': the trailing '+' (an 'at least' figure) is accepted by the stat regex since 2026-09-12 and parsed as the stated 20 MW. 12,000 m², 6,300 racks, Tier IV.",
7 7 "counts": {
8 8 "facility": 1
9 9 },
@@ -14,13 +14,10 @@
14 14 "operatorName": "NEXTDC",
15 15 "city": "Perth",
16 16 "countryIso2": "AU",
17 + "itCapacityMw": 20,
17 18 "buildingSqm": 12000,
18 19 "rackCount": 6300,
19 − "tier": "Tier IV",
20 − "_todo": {
21 − "itCapacityMw": 20,
22 − "_reason": "nextdc_facility_v1 stat regex does not allow the '+' suffix ('20MW+' then 'IT Capacity'); P1 Malaga '10+MW' and S5 '80+MW' have the same shape"
23 − }
20 + "tier": "Tier IV"
24 21 }
25 22 ]
26 23 }
modified apps/worker/fixtures/qts/phoenix-2.expected.json +3 −6
@@ -3,7 +3,7 @@
3 3 "documentId": "doc_de2f9d8b562c5c33",
4 4 "fetchedAt": "2026-09-11T08:46:05.593Z",
5 5 "contentType": "text/html; charset=UTF-8",
6 − "note": "'80 acre campus' → 32.37 ha; address 'DC2 - 1200 N 40th St, Phoenix, AZ 85008'. The page states '210 MW+ of critical campus capacity' (figure BEFORE the label) which qts_facility_v1 does not pick up (see _todo). No Google-Maps marker on this page → geo null.",
6 + "note": "'80 acre campus' → 32.37 ha; address 'DC2 - 1200 N 40th St, Phoenix, AZ 85008'. The page states '210 MW+ of critical campus capacity' (figure BEFORE the label) → totalPowerMw 210 (read since 2026-09-12). No Google-Maps marker on this page → geo null.",
7 7 "counts": {
8 8 "facility": 1
9 9 },
@@ -18,11 +18,8 @@
18 18 "countryIso2": "US",
19 19 "siteAreaHa": 32.37,
20 20 "itCapacityMw": null,
21 − "geo": null,
22 − "_todo": {
23 − "totalPowerMw": 210,
24 − "_reason": "qts_facility_v1 only matches 'Critical campus capacity <n> MW' (label first) and '<n> MW of critical/total power|capacity|IT load'; '210 MW+ of critical campus capacity' matches neither"
25 − }
21 + "totalPowerMw": 210,
22 + "geo": null
26 23 }
27 24 ]
28 25 }
modified apps/worker/fixtures/qts/phoenix-4.expected.json +2 −5
@@ -3,7 +3,7 @@
3 3 "documentId": "doc_3b14b8129dd879b3",
4 4 "fetchedAt": "2026-09-11T08:45:56.574Z",
5 5 "contentType": "text/html; charset=UTF-8",
6 − "note": "Address '11753 W Lower Buckeye Rd, Tolleson, AZ 85353' parses correctly, but the Elementor Google-Maps widget on this page carries address:'41.84273925236679, -87.66755675771267' — Chicago, ~2 300 km from Tolleson — which the parser stores as 'exact' coordinates (see _todo).",
6 + "note": "Address '11753 W Lower Buckeye Rd, Tolleson, AZ 85353' parses correctly. The Elementor Google-Maps widget on this page carries a pin at 41.84273925236679, -87.66755675771267 captioned '2800 S Ashland Ave, Chicago, IL 60608' — Chicago, ~2 300 km from Tolleson, copied from another template. Since 2026-09-12 a pin is used only when its caption names this facility's postal code or city + state → geo null (never another site's point).",
7 7 "counts": {
8 8 "facility": 1
9 9 },
@@ -17,10 +17,7 @@
17 17 "regionName": "AZ",
18 18 "postalCode": "85353",
19 19 "countryIso2": "US",
20 − "_todo": {
21 − "geo": null,
22 − "_reason": "qts_facility_v1 trusts the Elementor google_maps marker without checking it against the parsed state (AZ): 41.84, -87.67 is Chicago. Correct output is no coordinates (never fake / wrong 'exact' geo)"
23 − }
20 + "geo": null
24 21 }
25 22 ]
26 23 }
modified apps/worker/src/connectors/news/extract-project.test.ts +43 −1
@@ -1,6 +1,6 @@
1 1 import { describe, expect, it } from "vitest";
2 2 import type { ConnectorContext, RawDocument } from "@dci/connectors";
3 −import { extractAnnouncement, parseAllMoney, parseExpectedOpening, parsePhaseCount, qualifiesAsProject } from "./extract-project.js";
3 +import { extractAnnouncement, parseAllMoney, parseExpectedOpening, parsePhaseCount, projectMoneyFigures, qualifiesAsProject } from "./extract-project.js";
4 4 import { detectOperators, OPERATORS } from "./operators-lexicon.js";
5 5 import { detectLocation } from "./locations-lexicon.js";
6 6 import { newsArticleParser } from "./article-parser.js";
@@ -360,8 +360,50 @@ describe("news_article_v1", () => {
360 360 const off = ARTICLE.replace(/data center|Data Centers|192MW|hyperscale|campus/g, "bakery");
361 361 expect(await newsArticleParser.parse(htmlDoc("https://example.com/news/bakery", off), ctx())).toEqual([]);
362 362 });
363 + it("an operator named only in related-articles / footer teasers never becomes the project's operator", async () => {
364 + // the story is about an unknown developer; Microsoft, Google, Equinix and AWS appear only in teaser blocks (DCD "More in …" shape)
365 + const html = `<!doctype html><html><head><title>Riverside Estates to build 30MW data center in Leesburg | Trade Press</title>
366 +<meta property="article:published_time" content="2026-09-10T09:30:00Z"></head><body><nav>Home News</nav><main>
367 +<article class="card"><a href="/t1">Microsoft expands plans for La Porte, Indiana, data center campus</a></article>
368 +<article itemprop="mainEntity"><h1>Riverside Estates to build 30MW data center in Leesburg</h1>
369 +<p>Property developer Riverside Estates has secured planning permission for a 30MW data center in Leesburg, Virginia. Named ‘Aire Park One,’ the facility is expected to open in 2028 and will offer 4,000 sqm of white space to cloud providers and financial institutions.</p>
370 +<p>Founded in 1998, Riverside Estates manages an estate of 600,000 sqm across the region and a total investment volume of £2.1 billion ($2.7bn). The Leesburg site is its first data center.</p>
371 +<div class="block-auto_featured_content"><h2>More in Construction &amp; Site Selection</h2><a href="/t2">Google to build 300MW data center campus in Aragon, Spain</a></div></article>
372 +<div class="related-articles"><a href="/t3">Equinix opens LD14 in Slough</a></div></main>
373 +<aside><a href="/t4">Trending: Amazon Web Services buys 1GW site in Ohio</a></aside>
374 +<footer><p>Microsoft, Google and Equinix are trademarks of their owners.</p></footer></body></html>`;
375 + const recs = await newsArticleParser.parse(htmlDoc("https://example.com/news/riverside-leesburg", html), ctx(), { keepText: false });
376 + expect(recs.map((r) => r.kind)).toEqual(["news_event", "project"]);
377 + const [ev, pr] = recs;
378 + expect(ev!.data.operators).toEqual([]);
379 + expect(pr!.data).toMatchObject({ operatorName: null, name: "Aire Park One", plannedMw: 30, city: "Leesburg", countryIso2: "US", status: "announced", investmentUsd: null });
380 + expect(pr!.methods?.operatorName).toBe("none");
381 + // the same article with the developer in the lexicon keeps its operator (the teasers still count for nothing)
382 + const known = html.replace(/Riverside Estates/g, "Vantage Data Centers");
383 + const a = extractAnnouncement("Vantage Data Centers to build 30MW data center in Leesburg", (await articleContentOf(known)).text, { publishedAt: "2026-09-10", now: NOW });
384 + expect(a.operator?.name).toBe("Vantage Data Centers");
385 + expect(a.operators.map((o) => o.name)).toEqual(["Vantage Data Centers"]);
386 + });
387 + it("company-background money (portfolio volume, revenue) is never the project's investment", () => {
388 + const text = "S+B Gruppe manages an estate of 414,000 sqm across central and eastern Europe and a total investment volume of €8.1 billion ($9.4bn). The 12MW facility will cost $60 million.";
389 + expect(parseAllMoney(text)).toHaveLength(3);
390 + expect(projectMoneyFigures(text)).toEqual([{ amount: 60_000_000, currency: "USD" }]);
391 + expect(projectMoneyFigures("The company reported revenue of $4.2 billion last year.")).toEqual([]);
392 + expect(projectMoneyFigures("Vantage will invest $2 billion in the Frederick campus.")).toEqual([{ amount: 2_000_000_000, currency: "USD" }]);
393 + const a = x("S+B Gruppe to build data center in Bucharest, Romania", "Real estate developer S+B Gruppe has secured planning permission to build a 12MW data center in Bucharest. Named ‘NIMB,’ the facility is expected to open in 2028. Founded in 1986 in Vienna, S+B Gruppe manages an estate of more than 414,000 sqm and a total investment volume of €8.1 billion ($9.4bn).");
394 + expect(a.money).toBeNull();
395 + expect(a.investmentUsd).toBeNull();
396 + expect(a.explicitName).toBe("NIMB");
397 + expect(a.operator).toBeNull();
398 + expect(a.classification.mayCreateProject).toBe(true);
399 + });
363 400 });
364 401
402 +async function articleContentOf(html: string) {
403 + const { articleContent } = await import("./article-parser.js");
404 + return articleContent(htmlDoc("https://example.com/news/x", html));
405 +}
406 +
365 407 describe("news_planning_pdf_v1", () => {
366 408 const REPORT = `<html><head><title>Staff Report — Rezoning Application REZ 2026-0017</title></head><body><main>
367 409 <h1>Planning Commission Staff Report</h1>
modified apps/worker/src/connectors/news/extract-project.ts +37 −5
@@ -52,7 +52,9 @@ export interface Announcement {
52 52 hqGuarded: boolean;
53 53 }
54 54
55 −export const EXTRACTOR_VERSION = "news_v2";
55 +export const EXTRACTOR_VERSION = "news_v3";
56 +/** Operators are looked up in the title and the first 4 000 characters of the story only — never in trailing teasers or footers. */
57 +export const OPERATOR_SCOPE_CHARS = 4000;
56 58
57 59 /**
58 60 * Company-domicile mentions must not locate a project: "Denver-based Vantage plans a campus in Abilene, Texas" is in
@@ -176,7 +178,8 @@ const CURRENCY_START_RE = /(US\$|USD|\$|CA\$|C\$|€|EUR|£|GBP|A\$|AUD|¥|JPY|S
176 178 /** Approximate rates — gating only (is this ≥ $50M?), never persisted. */
177 179 const APPROX_USD: Record<string, number> = { USD: 1, EUR: 1.1, GBP: 1.3, CAD: 0.73, AUD: 0.66, SGD: 0.75, INR: 0.012, JPY: 0.0067, CHF: 1.12, SEK: 0.095, NOK: 0.093, DKK: 0.147 };
178 180
179 −export function parseAllMoney(text: string): Money[] {
181 +/** Every money figure in a text (largest first); `keep(start)` may veto a figure from its position in `text`. */
182 +export function parseAllMoney(text: string, keep: (start: number) => boolean = () => true): Money[] {
180 183 const out: Money[] = [];
181 184 const seen = new Set<string>();
182 185 CURRENCY_START_RE.lastIndex = 0;
@@ -187,6 +190,7 @@ export function parseAllMoney(text: string): Money[] {
187 190 const money = parseMoney(window);
188 191 // < $1M is not a project investment; > $500B is an industry-wide statistic ("$31.6 trillion AI buildout")
189 192 if (!money || money.amount < 1_000_000 || money.amount > 500_000_000_000) continue;
193 + if (!keep(m.index ?? 0)) continue;
190 194 const k = `${money.currency}:${money.amount}`;
191 195 if (seen.has(k)) continue;
192 196 seen.add(k);
@@ -195,6 +199,25 @@ export function parseAllMoney(text: string): Money[] {
195 199 return out.sort((a, b) => b.amount - a.amount);
196 200 }
197 201
202 +/**
203 + * Company-background wording before a money figure, in the same sentence: the developer's portfolio, balance sheet or
204 + * history ("a total investment volume of €8.1 billion ($9.4bn)", "assets under management of…", "revenue of…") — never
205 + * this project's budget.
206 + */
207 +export const MONEY_PORTFOLIO_RE = /\b(investment volume|portfolio|assets under management|under management|AUM|market (?:cap|capitali[sz]ation)|net worth|revenues?|turnover|balance sheet|(?:manages|owns|operates|holds|controls) an? (?:estate|portfolio|footprint)|estate of|founded in \d{4}|since (?:its )?(?:founding|inception)|to date|over the (?:past|last) (?:\d+|two|three|five|ten) years)\b/i;
208 +
209 +/** The sentence fragment before position `at` (from the previous sentence break, at most `max` chars). */
210 +function sentenceBefore(text: string, at: number, max = 300): string {
211 + const seg = text.slice(Math.max(0, at - max), at);
212 + const brk = Math.max(seg.lastIndexOf(". "), seg.lastIndexOf("\n"), seg.lastIndexOf("? "), seg.lastIndexOf("! "));
213 + return brk >= 0 ? seg.slice(brk + 1) : seg;
214 +}
215 +
216 +/** Money figures that may describe the announced project: company-background figures (portfolio volume, revenue…) are skipped. */
217 +export function projectMoneyFigures(scope: string): Money[] {
218 + return parseAllMoney(scope, (start) => !MONEY_PORTFOLIO_RE.test(sentenceBefore(scope, start)));
219 +}
220 +
198 221 export function approxUsd(m: Money | null): number | null {
199 222 if (!m) return null;
200 223 const r = APPROX_USD[m.currency];
@@ -302,8 +325,15 @@ export function findExplicitName(title: string | null, text: string, operator: O
302 325 return name.replace(/’/g, "'");
303 326 }
304 327 }
328 + // "Named ‘NIMB,’ the firm…", "dubbed 'Project Sail'" — a quoted proper name given to the facility in the lead
329 + const q = text.slice(0, 1500).match(QUOTED_NAME_RE);
330 + if (q) {
331 + const name = cleanText(q[1]!.replace(/’/g, "'"));
332 + if (name && name.length >= 3 && name.length <= 60 && /^[A-Z0-9]/.test(name) && !GENERIC_WORD.test(name) && !UNIT_WORD.test(name) && !detectOperators(name).length) return name;
333 + }
305 334 return null;
306 335 }
336 +const QUOTED_NAME_RE = /\b(?:named|dubbed|called|known as|branded|code-?named|titled)\s+[‘'"“]([^‘’'"“”\n]{2,60}?)[,.]?[’'"”]/i;
307 337
308 338 export function cleanTitle(title: string | null): string {
309 339 if (!title) return "";
@@ -405,7 +435,7 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op
405 435 // money: same title → lead → body preference, largest figure in the winning scope
406 436 let money: Money | null = null;
407 437 for (const [scopeName, scope] of [["title", title], ["lead", `${title} ${head}`], ["body", `${title} ${body}`]] as Array<[string, string]>) {
408 − const all = parseAllMoney(scope);
438 + const all = projectMoneyFigures(scope);
409 439 if (all.length) { money = all[0]!; methods.investment = `regex:money_v1:${scopeName}`; break; }
410 440 }
411 441 const investmentUsd = money && money.currency === "USD" ? money.amount : null;
@@ -441,8 +471,10 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op
441 471 if (location) { methods.location = location.method; }
442 472 const hqGuarded = hqTitle.stripped || hqLead.stripped || hqBody.stripped;
443 473
444 − const operators = detectOperators(`${title}\n${body}`);
445 − let operator = primaryOperator(title, body);
474 + // operators: title + the first OPERATOR_SCOPE_CHARS of the story only (a "More in …" teaser or a footer never names the developer)
475 + const opScope = text.slice(0, OPERATOR_SCOPE_CHARS);
476 + const operators = detectOperators(`${title}\n${opScope}`);
477 + let operator = primaryOperator(title, opScope);
446 478 // the headline's subject is a company we do not know → lexicon operators in the body are not the developer
447 479 if (operator && !detectOperators(title).length && unknownTitleSubject(title)) { operator = null; methods.operatorName = "none:unknown-title-subject"; }
448 480 else if (operator) methods.operatorName = "lexicon:operator";
modified apps/worker/src/connectors/operators1/qts.ts +40 −10
@@ -1,10 +1,11 @@
1 −import { load } from "cheerio";
1 +import { load, type CheerioAPI } from "cheerio";
2 2 import type { ExtractedRecord, Parser, RawDocument } from "@dci/connectors";
3 −import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, firstText, lastSlug, mwAfter, mwWithContext, parseCityLine, parseEuAddress, parseNaAddress, record, statusFromText, usStateCode } from "./shared.js";
3 +import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, firstText, lastSlug, mwAfter, mwWithContext, parseCityLine, parseEuAddress, parseNaAddress, record, statusFromText, usStateCode, type ParsedAddress } from "./shared.js";
4 4
5 5 /**
6 6 * QTS (q.com/data-centers/<slug>/). WordPress + Elementor; two templates:
7 − * US campus: <h1>Ashburn 3</h1> … "Campus footprint 25 acre campus" … "Critical campus capacity 80 MW+" … <h2>Ashburn 3 (ASH3)</h2>
7 + * US campus: <h1>Ashburn 3</h1> … "Campus footprint 25 acre campus" … "Critical campus capacity 80 MW+" (or, figure first,
8 + * "210 MW+ of critical campus capacity") … <h2>Ashburn 3 (ASH3)</h2>
8 9 * <p>ASH3 DC1: 22291 Shellhorn Rd<br>Ashburn, VA 20147</p>; addresses also in Elementor hotspot tooltips.
9 10 * EU site: <h1>Eemshaven, Netherlands</h1> … "36 MW total power" … "Huibertgatweg 2 9979 XZ Eemshaven, Netherlands".
10 11 * Language duplicates (-nl, calatorao/vimercate ES/IT originals) are excluded in YAML; "-en" is stripped from the key.
@@ -12,9 +13,37 @@ import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, fir
12 13 const NA_ADDR = /(\d{2,6}\s[A-Za-z0-9 .'-]+?\b(?:Rd|Road|Dr|Drive|Blvd|Boulevard|St|Street|Ave|Avenue|Pkwy|Parkway|Way|Ln|Lane|Ct|Court|Hwy|Highway|Trail|Trl|Pl|Place|Cir|Circle|Loop|Pike|Route|Rte)\.?)\s*,?\s*\n?\s*([A-Za-z .'-]+?),\s*([A-Z]{2})\s+(\d{5})\b/;
13 14 const EU_ADDR = /([A-Z][A-Za-z .'-]+?\s\d+[a-z]?)\s*,?\s*\n?\s*((?:[A-Z]{1,2}-)?\d{4,6}(?:\s?[A-Z]{2})?)\s+([A-Za-z .'-]+?)\s*,\s*([A-Za-z ]+)\b/;
14 15
16 +const escapeRe = (s: string) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
17 +
18 +/**
19 + * Elementor Google-Maps widget: data-pins='[{"address":"<lat>, <lng>","position":{"lat":…,"lng":…},"desc":"<p>2800 S Ashland Ave, Chicago, IL 60608</p>"}]'.
20 + * The widget is copied between page templates (the Phoenix 4 page carried the Chicago pin), so a pin is used only when
21 + * its own caption names the facility's postal code, or its city and state. No matching caption → no coordinates: a
22 + * facility never inherits another site's point.
23 + */
24 +export function mapPin($: CheerioAPI, pa: ParsedAddress | null): { lat: number; lng: number } | null {
25 + if (!pa) return null;
26 + const pins: Array<Record<string, unknown>> = [];
27 + $("[data-pins]").each((_, el) => {
28 + try { const v = JSON.parse($(el).attr("data-pins") ?? "[]") as unknown; if (Array.isArray(v)) pins.push(...(v as Array<Record<string, unknown>>)); } catch { /* not JSON */ }
29 + });
30 + for (const pin of pins) {
31 + const pos = pin.position as { lat?: unknown; lng?: unknown } | undefined;
32 + const fromAddr = typeof pin.address === "string" ? pin.address.match(/^\s*(-?\d{1,2}\.\d{3,}),\s*(-?\d{1,3}\.\d{3,})\s*$/) : null;
33 + const lat = Number(pos?.lat ?? fromAddr?.[1]), lng = Number(pos?.lng ?? fromAddr?.[2]);
34 + if (!Number.isFinite(lat) || !Number.isFinite(lng) || (lat === 0 && lng === 0)) continue;
35 + const caption = cleanText(String(pin.desc ?? "").replace(/<[^>]+>/g, " ")) ?? "";
36 + if (!caption) continue;
37 + const samePostal = !!pa.postal && new RegExp(`\\b${escapeRe(pa.postal)}\\b`).test(caption);
38 + const sameCity = !!pa.city && !!pa.region && new RegExp(`\\b${escapeRe(pa.city)}\\b[,\\s]+${escapeRe(pa.region)}\\b`, "i").test(caption);
39 + if (samePostal || sameCity) return { lat, lng };
40 + }
41 + return null;
42 +}
43 +
15 44 export const qtsFacility: Parser = {
16 45 name: "qts_facility_v1",
17 − version: "1.0.0",
46 + version: "1.1.0",
18 47 pageTypes: ["facility_page"],
19 48 parse(doc: RawDocument): ExtractedRecord[] {
20 49 const url = doc.finalUrl;
@@ -28,7 +57,8 @@ export const qtsFacility: Parser = {
28 57 const code = codeMatch?.[2] ?? null;
29 58 const slug = lastSlug(url).replace(/-(en|nl|es|it|fi)$/, "");
30 59
31 − const totalPowerMw = mwAfter(text, /Critical campus capacity/i, 80) ?? mwWithContext(text, /\s*(?:of\s+)?(?:total|totaal)\b/i) ?? mwWithContext(text, /\s*(?:aan|of)\s+(?:totaal|total)/i) ?? mwWithContext(text, /\s*(?:of\s+)?(?:critical|total)\s+(?:power|capacity|IT load)/i);
60 + // label first ("Critical campus capacity 80 MW+") or figure first ("210 MW+ of critical campus capacity", "36 MW total power")
61 + const totalPowerMw = mwAfter(text, /Critical campus capacity/i, 80) ?? mwWithContext(text, /\s*(?:of\s+)?(?:total|totaal)\b/i) ?? mwWithContext(text, /\s*(?:aan|of)\s+(?:totaal|total)/i) ?? mwWithContext(text, /\s*(?:of\s+)?(?:critical|total)\s+(?:campus\s+)?(?:power|capacity|IT load)/i);
32 62 const acre = text.match(/(\d+(?:\.\d+)?)\s*[- ]?acre\b/i);
33 63 const siteAreaHa = acre ? areaHa(`${acre[1]} acres`) : areaHa(text.match(/(\d+(?:\.\d+)?)\s*hectares?\b/i)?.[0]);
34 64
@@ -52,9 +82,9 @@ export const qtsFacility: Parser = {
52 82 const proudState = proud ? usStateCode(proud[1]) : null;
53 83 const countryIso2 = pa?.countryIso2 ?? cl.countryIso2 ?? (proudState ? "US" : null) ?? countryFromText(title.replace(/-\s*QTS.*$/, "").replace(/-/g, " ")) ?? null;
54 84 const buildings = [...text.matchAll(/\b([A-Z]{2,5}\d{0,2}\s+DC\d+):\s*([^\n]+)\n?([^\n]*\d{5})?/g)].map((m) => cleanText(`${m[1]}: ${m[2]} ${m[3] ?? ""}`) ?? "").filter(Boolean);
55 − // Elementor Google-Maps widget: "address":"<lat>, <lng>" (operator-placed marker) → coordinates
56 − const gm = doc.text.match(/address(?:&quot;|"):(?:&quot;|")\s*(-?\d{1,2}\.\d{3,}),\s*(-?\d{1,3}\.\d{3,})\s*(?:&quot;|")/);
57 − const lat = gm ? Number(gm[1]) : null, lng = gm ? Number(gm[2]) : null;
85 + // Elementor Google-Maps pin, only when its caption is this facility's address (see mapPin)
86 + const pin = mapPin($, pa);
87 + const lat = pin?.lat ?? null, lng = pin?.lng ?? null;
58 88 // "200,000 ft2 facility" / "22,000 m2 total"
59 89 const bld = text.match(/(\d{1,3}(?:,\d{3})+|\d+)\s*(?:ft2|ft²|sq\.?\s?ft|square feet)\s+(?:facility|building|data center)/i);
60 90 const bldM = text.match(/(\d{1,3}(?:[,.]\d{3})+|\d+)\s*(?:m2|m²|sqm|sq\.?\s?m\.?)\s+(?:total|facility|building|of)/i);
@@ -69,14 +99,14 @@ export const qtsFacility: Parser = {
69 99 const data: Record<string, unknown> = {
70 100 name: `QTS ${h1.split(",")[0]!.trim()}`, code, aliases: [h1, ...(code ? [`QTS ${code}`] : [])], // keep the site number ("QTS Ashburn 3") — several campuses share a city
71 101 address: pa?.address ?? null, city: pa?.city ?? cl.city, regionName: pa?.region ?? cl.region ?? proudState, postalCode: pa?.postal ?? null, countryIso2,
72 − lat, lng, geoPrecision: "exact",
102 + lat, lng, geoPrecision: pin ? "exact" : null,
73 103 totalPowerMw, siteAreaHa, buildingSqm,
74 104 status, description: [desc, buildings.length ? `Buildings: ${buildings.join("; ")}.` : null].filter(Boolean).join(" ") || null,
75 105 externalIds: { qts_slug: slug },
76 106 };
77 107 const methods: Record<string, string> = {
78 108 name: "selector:h1", code: "selector:h2+regex:(CODE)", address: addrRaw && tooltips.length ? "elementor:hotspot_tooltip+parse" : "regex:address_v1", city: pa?.city ? "regex:address_v1" : "selector:h1+parse", regionName: "regex:address_v1", postalCode: "regex:address_v1", countryIso2: "regex:address_v1>country",
79 − lat: "elementor:google_maps.address", lng: "elementor:google_maps.address",
109 + lat: "elementor:google_maps.pin(caption=address)", lng: "elementor:google_maps.pin(caption=address)", geoPrecision: "elementor:google_maps.pin(caption=address)",
80 110 totalPowerMw: "regex:critical_campus_capacity_v1>mw", siteAreaHa: "regex:acre_campus>ha", buildingSqm: "regex:facility_sqft>sqm", status: "regex:status_v1", description: desc === metaDesc ? "meta:description" : "selector:p(intro)", externalIds: "url:slug",
81 111 };
82 112 return [record("facility", `qts:${slug}`, url, data, methods, 0.85)];
modified apps/worker/src/connectors/operators2/americas.ts +23 −14
@@ -2,7 +2,8 @@
2 2 import type { ExtractedRecord } from "@dci/connectors";
3 3 import { extractGeo } from "@dci/connectors";
4 4 import { countryFromText } from "@dci/core";
5 −import { Rec, all, applyLdAddress, applyLdGeo, capacityMw, certifications, countryFromCity, countryOf, describe, facilityParser, first, ha, itMw, key, latinNumbers, ldWithAddress, mw, pue, safeCountryText, seg, splitStreetCity, sqm, statusOf, tier, titleCase, usAddress } from "./shared.js";
5 +import { usStateCode } from "../operators1/shared.js";
6 +import { Rec, all, applyLdAddress, applyLdGeo, breadcrumbLeaf, capacityMw, certifications, countryFromCity, countryOf, describe, facilityParser, first, ha, itMw, key, latinNumbers, ldWithAddress, mw, pue, safeCountryText, seg, splitStreetCity, sqm, statusOf, tier, titleCase, usAddress } from "./shared.js";
6 7
7 8 /** DataBank — /data-centers/<metro>/[<campus>/]<facility>/ ; "<Name> (<CODE>)" heading followed by the street address; stats in .c-figure-stats / .c-info-table__stats. Campus pages (code before name) yield nothing. */
8 9 facilityParser("databank_facility_v1", (p) => {
@@ -28,7 +29,10 @@ facilityParser("databank_facility_v1", (p) => {
28 29 return [r.done(key("databank", code), p.url)];
29 30 });
30 31
31 −/** EdgeConneX — metro pages; one JSON-LD LocalBusiness per facility (address only). Without JSON-LD, one metro-level record (city/country from H1, facility codes as aliases). */
32 +/**
33 + * EdgeConneX — metro pages; one JSON-LD LocalBusiness per facility (address only). Without JSON-LD, one metro-level record
34 + * (city / state from the BreadcrumbList leaf "Richmond, VA", else from the h1 minus its "Data Center" suffix; facility codes as aliases).
35 + */
32 36 facilityParser("edgeconnex_facility_v1", (p) => {
33 37 const out: ExtractedRecord[] = [];
34 38 const metro = seg(p.url).pop() ?? "";
@@ -43,17 +47,22 @@ facilityParser("edgeconnex_facility_v1", (p) => {
43 47 }
44 48 if (out.length) return out;
45 49 const h1 = p.h1 ?? "";
46 − const loc = h1.match(/^([A-Z][^,]+?)(?:,\s*([A-Z][^,]+))?$/);
47 − if (!loc || !/edgeconnex/i.test(p.title ?? "")) return [];
50 + if (!h1 || !/edgeconnex/i.test(p.title ?? "")) return [];
51 + // h1 "Richmond Data Center" is the facility name; the breadcrumb leaf "Richmond, VA" is the location line
52 + const crumb = breadcrumbLeaf(p.html);
53 + const locLine = crumb && crumb.length <= 60 ? crumb : h1.replace(/\s+Data\s+Cent(?:er|re)s?$/i, "");
54 + const locMethod = crumb && crumb.length <= 60 ? "json-ld:BreadcrumbList" : "selector:h1";
55 + const loc = locLine.match(/^([A-Z][^,]+?)(?:,\s*([A-Z][^,]+))?$/);
56 + if (!loc) return [];
48 57 const r = new Rec();
49 − r.set("name", `EdgeConneX ${h1}`, "selector:h1").set("city", loc[1], "selector:h1");
58 + r.set("name", `EdgeConneX ${h1}`, "selector:h1").set("city", loc[1], locMethod);
50 59 const region = loc[2]?.trim() ?? null;
51 − const c = countryOf(region, h1, p.desc, p.text.slice(0, 1500));
60 + const c = region && usStateCode(region) ? { iso: "US", method: "lookup:us-state" } : countryOf(region, h1, p.desc, p.text.slice(0, 1500));
52 61 if (c) r.set("countryIso2", c.iso, c.method);
53 − if (region && c?.iso === "US") r.set("regionName", region, "selector:h1");
62 + if (region && c?.iso === "US") r.set("regionName", region, locMethod);
54 63 const codes = [...new Set([...p.text.matchAll(/\b([A-Z]{3}\d{2})\b/g)].map((m) => m[1]!))];
55 64 r.set("aliases", codes, "regex:facility_codes");
56 − const single = codes.length === 1 ? p.text.match(new RegExp(`${codes[0]}:\\s*(\\d[\\d.]*\\s*MW)`)) : null;
65 + const single = codes.length === 1 ? p.text.match(new RegExp(`${codes[0]}:\\s*(\\d[\\d.]*\\+?\\s*MW\\+?)`)) : null;
57 66 const st = statusOf(p.text, 2500); if (st) r.set("status", st, "regex:status_v1");
58 67 if (single) r.set(st && st !== "operational" ? "plannedPowerMw" : "itCapacityMw", mw(single[1]), "regex:code_mw_row");
59 68 r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
@@ -104,7 +113,7 @@ facilityParser("serverfarm_facility_v1", (p) => {
104 113 r.set("name", `Serverfarm ${h1.replace(/\s*data center\s*$/i, "")}`, "selector:h1").set("code", code, "selector:h1").set("city", cityPart, "selector:h1");
105 114 const c = countryOf(cityPart) ?? countryOf(p.text.slice(0, 3000)); if (c) r.set("countryIso2", c.iso, c.method);
106 115 const g = extractGeo(p.html); if (g && g.method === "attr:data-lat") r.set("lat", g.lat, g.method).set("lng", g.lng, g.method).set("geoPrecision", "exact", g.method);
107 − r.set("itCapacityMw", mw(first(p.text, /\bIT\s+(\d[\d.,]*\s*MW)/)) ?? itMw(p.text), "regex:it_mw_serverfarm").set("pue", pue(p.text), "regex:pue_v1");
116 + r.set("itCapacityMw", mw(first(p.text, /\bIT\s+(\d[\d.,]*\+?\s*MW\+?)/)) ?? itMw(p.text), "regex:it_mw_serverfarm").set("pue", pue(p.text), "regex:pue_v1");
108 117 const carriers = first(p.text, /Carriers\s*\n?\s*([A-Z][^\n]{5,300}?)(?:\n|$)/); if (carriers && /,/.test(carriers)) r.set("carriers", carriers.split(/,\s*/).map((s) => s.trim()).filter((s) => s.length > 1 && s.length < 40), "regex:carriers_list");
109 118 r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
110 119 return [r.done(key("serverfarm", code ?? cityPart), p.url)];
@@ -147,7 +156,7 @@ facilityParser("stream_facility_v1", (p) => {
147 156 if (status) r.set("status", /leased|operational|in service|complete|sold|stabili[sz]ed/i.test(status) ? "operational" : /construction/i.test(status) ? "under_construction" : /planned|future|available|development|land/i.test(status) ? "announced" : null, "regex:status_label");
148 157 if (/hyperscale/i.test(ptype)) r.set("facilityType", "hyperscale", "regex:product_type"); else if (/wholesale|build-to-suit/i.test(ptype)) r.set("facilityType", "wholesale", "regex:product_type"); else if (/private|enterprise/i.test(ptype)) r.set("facilityType", "enterprise", "regex:product_type");
149 158 r.set("itCapacityMw", itMw(p.text), "regex:it_mw_v1");
150 − if (!r.has("itCapacityMw")) r.set("totalPowerMw", mw(first(p.text, /^\s*(\d[\d.]*\s*MW)\s*\n\s*(?:Campus|Power|Capacity)/m)), "regex:hero_mw_stat");
159 + if (!r.has("itCapacityMw")) r.set("totalPowerMw", mw(first(p.text, /^\s*(\d[\d.]*\+?\s*MW\+?)\s*\n\s*(?:Campus|Power|Capacity)/m)), "regex:hero_mw_stat");
151 160 r.set("buildingSqm", sqm(first(p.text, /([\d,]+\s*(?:Square Feet|SF))\s*\n?\s*(?:Data Center|Building|$)/im) ?? first(p.text, /([\d,]+\s*Square Feet)/i)), "regex:sqft_v1").set("siteAreaHa", ha(first(p.text, /([\d.]+\s*Acres)/i)), "regex:acres_v1");
152 161 r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
153 162 return [r.done(key("stream", seg(p.url).pop()), p.url)];
@@ -162,7 +171,7 @@ facilityParser("skybox_facility_v1", (p) => {
162 171 r.set("name", name.replace(/™/g, ""), "html:title").set("countryIso2", "US", "default:operator");
163 172 const h1 = (p.h1 ?? "").replace(/™/g, "");
164 173 const dev = /PowerCampus|Build to Suit|Development/i.test(h1);
165 − const hero = p.text.match(/^\s*([A-Za-z™ ]{2,40})\n\s*(\d[\d.,]*\s*MW)\n(?:\s*([\d,]+\s*SF)\n)?(?:\s*([\d,.]+\s*Acres)\n)?/m);
174 + const hero = p.text.match(/^\s*([A-Za-z™ ]{2,40})\n\s*(\d[\d.,]*\+?\s*MW\+?)\n(?:\s*([\d,]+\s*SF)\n)?(?:\s*([\d,.]+\s*Acres)\n)?/m);
166 175 const cityRaw = h1.replace(/^PowerCampus\s*/i, "").trim() || hero?.[1]?.trim() || "";
167 176 r.set("city", cityRaw ? titleCase(cityRaw.toLowerCase()) : null, "selector:h1");
168 177 if (hero) { r.set(dev ? "plannedPowerMw" : "totalPowerMw", mw(hero[2]), "regex:hero_stats").set("buildingSqm", sqm(hero[3] ?? null), "regex:hero_stats").set("siteAreaHa", ha(hero[4] ?? null), "regex:hero_stats"); }
@@ -197,7 +206,7 @@ facilityParser("elementcritical_facility_v1", (p) => {
197 206 const r = new Rec();
198 207 const name = (p.h1 ?? p.title ?? "").replace(/\s*data center services\s*$/i, "").replace(/\s*[-|].*$/, "").trim();
199 208 r.set("name", `Element Critical ${name}`, "selector:h1").set("address", loc[1], "regex:location_row").set("city", loc[2], "regex:location_row").set("regionName", loc[3], "regex:location_row").set("postalCode", loc[4], "regex:location_row").set("countryIso2", "US", "regex:location_row");
200 − r.set("buildingSqm", sqm(first(p.text, /Property Size\s*\n?\s*([\d,]+\s*Sq\.?\s*Ft\.?)/i)), "regex:property_size").set("totalPowerMw", mw(first(p.text, /Power\s*\n?\s*(\d[\d.]*\s*MW)/)) ?? capacityMw(p.text), "regex:power_row");
209 + r.set("buildingSqm", sqm(first(p.text, /Property Size\s*\n?\s*([\d,]+\s*Sq\.?\s*Ft\.?)/i)), "regex:property_size").set("totalPowerMw", mw(first(p.text, /Power\s*\n?\s*(\d[\d.]*\+?\s*MW\+?)/)) ?? capacityMw(p.text), "regex:power_row");
201 210 r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
202 211 return [r.done(key("elementcritical", loc[1]), p.url)];
203 212 });
@@ -209,7 +218,7 @@ facilityParser("threesixtyfive_facility_v1", (p) => {
209 218 const r = new Rec();
210 219 const name = (p.title ?? "").split(/\s*[-|]\s*/)[0]?.trim() || titleCase(seg(p.url).pop() ?? "");
211 220 r.set("name", `365 Data Centers ${name.replace(/\s*data center$/i, "")}`, "html:title").set("address", addr.street, "regex:us_address_v1").set("city", addr.city, "regex:us_address_v1").set("regionName", addr.region, "regex:us_address_v1").set("postalCode", addr.postal, "regex:us_address_v1").set("countryIso2", "US", "regex:us_address_v1");
212 − r.set("totalPowerMw", mw(first(p.text, /(\d[\d.]*\s*MW) of (?:critical\s+)?power/i)) ?? capacityMw(p.text), "regex:mw_of_power").set("buildingSqm", sqm(first(p.text, /([\d,]+\s*sq\.?\s*ft)\s+(?:data center|facility)/i)), "regex:sqft_v1");
221 + r.set("totalPowerMw", mw(first(p.text, /(\d[\d.]*\+?\s*MW\+?) of (?:critical\s+)?power/i)) ?? capacityMw(p.text), "regex:mw_of_power").set("buildingSqm", sqm(first(p.text, /([\d,]+\s*sq\.?\s*ft)\s+(?:data center|facility)/i)), "regex:sqft_v1");
213 222 r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
214 223 return [r.done(key("365dc", seg(p.url).pop()), p.url)];
215 224 });
@@ -252,7 +261,7 @@ facilityParser("odata_facility_v1", (p) => {
252 261 const r = new Rec();
253 262 r.set("name", `ODATA ${code}`, "html:title").set("code", code, "html:title");
254 263 const t = latinNumbers(p.text);
255 − r.set("itCapacityMw", mw(first(t, /IT power\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(t), "regex:it_power_row").set("buildingSqm", sqm(first(t, /Floor area\s*\n?\s*([\d.,]+\s*m²)/i)), "regex:floor_area_row");
264 + r.set("itCapacityMw", mw(first(t, /IT power\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(t), "regex:it_power_row").set("buildingSqm", sqm(first(t, /Floor area\s*\n?\s*([\d.,]+\s*m²)/i)), "regex:floor_area_row");
256 265 const g = extractGeo(p.html); if (g && g.method === "link:google-maps") r.set("lat", g.lat, g.method).set("lng", g.lng, g.method).set("geoPrecision", "exact", g.method);
257 266 const loc = p.text.match(/Located in (?:the )?(?:[A-Z][\w ]+? area in )?([A-Z][^,.\n(]{2,40}?)\s*[,.(]/);
258 267 if (loc) r.set("city", loc[1], "regex:located_in");
modified apps/worker/src/connectors/operators2/apac.ts +8 −8
@@ -19,7 +19,7 @@ facilityParser("airtrunk_facility_v1", (p) => {
19 19 return [r.done(key("airtrunk", m[1]), p.url)];
20 20 });
21 21
22 −/** NEXTDC — /data-centres/<metro>-data-centres/<code>-<city> ; IT capacity, technical space, rack capacity, Tier III. */
22 +/** NEXTDC — /data-centres/<metro>-data-centres/<code>-<city> ; IT capacity ("12MW", "20MW+", "10+MW"), technical space, rack capacity, Tier III. */
23 23 facilityParser("nextdc_facility_v1", (p) => {
24 24 const slug = seg(p.url).pop() ?? "";
25 25 const m = slug.match(/^([a-z]{1,2}\d{1,2})-([a-z-]+)$/);
@@ -27,7 +27,7 @@ facilityParser("nextdc_facility_v1", (p) => {
27 27 const code = m[1]!.toUpperCase();
28 28 const r = new Rec();
29 29 r.set("name", `NEXTDC ${code} ${titleCase(m[2]!)}`, "url:slug").set("code", code, "url:slug").set("city", titleCase(m[2]!), "url:slug").set("countryIso2", "AU", "default:operator");
30 − r.set("itCapacityMw", mw(first(p.text, /(\d[\d.,]*\s*MW)\s*\n?\s*IT Capacity/i)) ?? itMw(p.text), "regex:it_capacity_stat").set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*m\s?2)\s*\n?\s*Technical Space/i)), "regex:technical_space_stat");
30 + r.set("itCapacityMw", mw(first(p.text, /(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Capacity/i)) ?? itMw(p.text), "regex:it_capacity_stat").set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*m\s?2)\s*\n?\s*Technical Space/i)), "regex:technical_space_stat");
31 31 const racks = first(p.text, /([\d,]+)\s*\n?\s*Rack Capacity/i); if (racks) r.set("rackCount", Number(racks.replace(/,/g, "")), "regex:rack_capacity_stat");
32 32 const addr = p.text.match(/\n\s*(\d{1,4}[-–]?\d{0,4}\s+[A-Z][^\n,]{3,60}),?\s*\n?\s*([A-Z][a-zA-Z ]+?)\s+(NSW|VIC|QLD|SA|WA|ACT|NT|TAS)\s+(\d{4})\b/);
33 33 if (addr) r.set("address", addr[1], "regex:au_address").set("regionName", addr[3], "regex:au_address").set("postalCode", addr[4], "regex:au_address");
@@ -62,7 +62,7 @@ facilityParser("digitaledge_facility_v1", (p) => {
62 62 const city = first(p.text, new RegExp(`${code}\\s*\\n?\\s*([A-Z][a-zA-Z ]+?)\\s+Data Center`)) ?? first(p.title ?? "", /\b[A-Z]{3,5}\d\s+([A-Z][a-zA-Z ]+?)\s+Data Center/);
63 63 r.set("name", `Digital Edge ${code}${city ? ` ${city}` : ""}`, "selector:h1").set("code", code, "selector:h1").set("city", city, "regex:code_city");
64 64 const c = countryOf(p.desc, city, p.text.slice(0, 2000)); if (c) r.set("countryIso2", c.iso, c.method);
65 − const full = first(p.text, /(\d[\d.,]*\s*MW) campus at full capacity/i); if (full) r.set("plannedPowerMw", mw(full), "regex:full_capacity_campus");
65 + const full = first(p.text, /(\d[\d.,]*\+?\s*MW\+?) campus at full capacity/i); if (full) r.set("plannedPowerMw", mw(full), "regex:full_capacity_campus");
66 66 r.set("itCapacityMw", itMw(p.text.slice(0, 3000)), "regex:it_mw_v1").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
67 67 const st = statusOf(p.text.slice(0, 3000)) ?? (/expected to be complete|will be constructed/i.test(p.text.slice(0, 3000)) ? "under_construction" : null); if (st) r.set("status", st, "regex:status_v1");
68 68 return [r.done(key("digitaledge", code), p.url)];
@@ -81,12 +81,12 @@ facilityParser("cdc_campus_v1", (p) => {
81 81 if (out.some((o) => o.key === key("cdc", city, campus))) continue;
82 82 const r = new Rec();
83 83 r.set("name", `CDC ${campus}`, "regex:campus_block").set("campusName", `CDC ${campus}`, "regex:campus_block").set("city", city, "selector:h1").set("countryIso2", iso, "lookup:city-country");
84 − const operating = first(block, /Current capacity \(operating\)\s*\n?\s*(\d[\d.,]*\s*MW)/i) ?? first(block, /(?:over|with) (\d[\d.,]*\s*MW) of capacity/i);
85 − const planned = first(block, /(\d[\d.,]*\s*MW) of planned capacity/i) ?? (/under construction|will be home|early stages/i.test(block) ? first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i) : null);
84 + const operating = first(block, /Current capacity \(operating\)\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i) ?? first(block, /(?:over|with) (\d[\d.,]*\+?\s*MW\+?) of capacity/i);
85 + const planned = first(block, /(\d[\d.,]*\+?\s*MW\+?) of planned capacity/i) ?? (/under construction|will be home|early stages/i.test(block) ? first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i) : null);
86 86 r.set("totalPowerMw", mw(operating), "regex:campus_block").set("plannedPowerMw", mw(planned), "regex:campus_block");
87 87 if (!operating && planned) r.set("status", /under construction|early stages of construction/i.test(block) ? "under_construction" : "announced", "regex:campus_block");
88 88 else if (/under construction/i.test(block) && !/operational/i.test(block)) r.set("status", "under_construction", "regex:campus_block");
89 − if (!operating && !planned) { const total = first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i); r.set("totalPowerMw", mw(total), "regex:campus_block"); }
89 + if (!operating && !planned) { const total = first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i); r.set("totalPowerMw", mw(total), "regex:campus_block"); }
90 90 const codes = [...new Set([...block.matchAll(/\b([A-Z]{2}\d)\b/g)].map((x) => x[1]!))]; if (codes.length) r.set("aliases", codes, "regex:campus_block");
91 91 const blurb = block.split("\n").map((l) => l.trim()).find((l) => l.length > 30);
92 92 r.set("facilityType", "hyperscale", "default:operator").set("description", blurb ?? describe(p), "regex:campus_block").set("certifications", certifications(p.text), "regex:certifications_v1");
@@ -111,7 +111,7 @@ facilityParser("pdg_facility_v1", (p) => {
111 111 const r = new Rec();
112 112 r.set("name", `PDG ${code} ${city}`, "regex:city_code_heading").set("code", code, "regex:city_code_heading").set("city", city, "regex:city_code_heading");
113 113 const c = countryFromCity(city) ?? countryFromUrl; r.set("countryIso2", c, countryFromCity(city) ? "lookup:city-country" : "url:segment>country");
114 − r.set("totalPowerMw", mw(first(block, /Capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? capacityMw(block), "regex:capacity_row").set("buildingSqm", sqm(first(block, /Colocation Area\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:colocation_area_row").set("siteAreaHa", sqm(first(block, /land area of ([\d,.]+\s*sqm)/i)) != null ? sqm(first(block, /land area of ([\d,.]+\s*sqm)/i))! / 10_000 : null, "regex:land_area>ha");
114 + r.set("totalPowerMw", mw(first(block, /Capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? capacityMw(block), "regex:capacity_row").set("buildingSqm", sqm(first(block, /Colocation Area\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:colocation_area_row").set("siteAreaHa", sqm(first(block, /land area of ([\d,.]+\s*sqm)/i)) != null ? sqm(first(block, /land area of ([\d,.]+\s*sqm)/i))! / 10_000 : null, "regex:land_area>ha");
115 115 const st = statusOf(block); if (st) r.set("status", st, "regex:status_v1");
116 116 r.set("facilityType", "hyperscale", "default:operator").set("certifications", certifications(block), "regex:certifications_v1").set("description", first(block, /ABOUT THE FACILITY[^\n]*\n\s*([^\n]{40,600})/i) ?? describe(p), "regex:about_facility");
117 117 out.push(r.done(key("pdg", code), p.url, 0.8));
@@ -128,7 +128,7 @@ facilityParser("macquarie_facility_v1", (p) => {
128 128 const city = titleCase(parts[1]!);
129 129 r.set("name", `Macquarie ${code} ${city}`, "url:slug").set("code", code, "url:slug").set("city", city, "url:slug").set("countryIso2", "AU", "default:operator");
130 130 const campus = first(p.title ?? "", /-\s*([A-Z][A-Za-z ]+Campus)/); if (campus) r.set("campusName", `Macquarie ${campus}`, "html:title");
131 − const it = first(p.text, /(\d[\d.,]*\s*MW)\s*\n?\s*IT Load/i); if (it) r.set("itCapacityMw", mw(it), "regex:it_load_stat");
131 + const it = first(p.text, /(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Load/i); if (it) r.set("itCapacityMw", mw(it), "regex:it_load_stat");
132 132 r.set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*(?:m²|sqm|m2))\s*\n?\s*Technical space/i)), "regex:technical_space_stat").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
133 133 if (/government|sovereign/i.test(`${p.title ?? ""} ${p.desc ?? ""}`)) r.set("facilityType", "colocation", "default:operator");
134 134 const st = statusOf(p.text.slice(0, 3000)); if (st) r.set("status", st, "regex:status_v1");
modified apps/worker/src/connectors/operators2/emea.ts +5 −5
@@ -95,7 +95,7 @@ facilityParser("greenmountain_facility_v1", (p) => {
95 95 r.set("name", `Green Mountain ${code}`, "selector:h1").set("code", code, "selector:h1");
96 96 const place = code.split("-")[1]!; if (!/^(east|west|north|south|central)$/i.test(place)) r.set("city", place, "selector:h1");
97 97 const c = countryOf(p.desc, p.text.slice(0, 2500)); r.set("countryIso2", c?.iso ?? "NO", c ? c.method : "default:operator");
98 − r.set("itCapacityMw", mw(first(p.text, /Maximum IT capacity:\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(p.text), "regex:it_mw_v1").set("buildingSqm", sqm(first(p.text, /([\d ]{3,9}\s*m2)\s*\n?\s*(?:Total Campus Footprint|\(\d)/i) ?? first(p.text, /approx ([\d ]{3,9}\s*m2)/i)), "regex:campus_footprint");
98 + r.set("itCapacityMw", mw(first(p.text, /Maximum IT capacity:\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(p.text), "regex:it_mw_v1").set("buildingSqm", sqm(first(p.text, /([\d ]{3,9}\s*m2)\s*\n?\s*(?:Total Campus Footprint|\(\d)/i) ?? first(p.text, /approx ([\d ]{3,9}\s*m2)/i)), "regex:campus_footprint");
99 99 r.set("pue", pue(p.text), "regex:pue_v1").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
100 100 if (/100 ?% renewable|hydro ?power/i.test(p.text)) r.set("renewableClaim", first(p.text, /([^.\n]*100 ?% renewable[^.\n]*)/i) ?? "Renewable (hydro) powered", "regex:renewable_claim");
101 101 const st = statusOf(`${p.desc ?? ""}\n${p.text}`, 2000); if (st) r.set("status", st, "regex:status_v1");
@@ -171,7 +171,7 @@ facilityParser("kaodata_facility_v1", (p) => {
171 171 const base = (r: Rec) => { if (addr) r.set("address", addr[1], "regex:address_row").set("city", addr[2], "regex:address_row").set("postalCode", addr[3], "regex:address_row"); else r.set("city", campus, "url:segment"); r.set("countryIso2", "GB", "default:operator"); if (/100 ?% renewable/i.test(p.text)) r.set("renewableClaim", "100% renewable energy", "regex:renewable_claim"); };
172 172 const c = new Rec();
173 173 c.set("name", `Kao Data ${campus}`, "url:segment").set("campusName", `Kao Data ${campus}`, "url:segment"); base(c);
174 − c.set("plannedPowerMw", mw(first(p.text, /(?:ITE?|IT-?)\s*load of (\d[\d.]*\s*MW)/i)), "regex:ite_load_full_build").set("siteAreaHa", ha(first(p.text, /(\d+\s*acres?)/i)), "regex:acres_v1").set("buildingSqm", sqm(first(p.text, /Technical Space:\s*([\d,]+\s*m2)/i)), "regex:technical_space");
174 + c.set("plannedPowerMw", mw(first(p.text, /(?:ITE?|IT-?)\s*load of (\d[\d.]*\+?\s*MW\+?)/i)), "regex:ite_load_full_build").set("siteAreaHa", ha(first(p.text, /(\d+\s*acres?)/i)), "regex:acres_v1").set("buildingSqm", sqm(first(p.text, /Technical Space:\s*([\d,]+\s*m2)/i)), "regex:technical_space");
175 175 c.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
176 176 out.push(c.done(key("kaodata", campus), p.url, 0.8));
177 177 for (const m of p.text.matchAll(/\b(K[A-Z]{2,4}-\d{2})\s*\((\d[\d.]*)\s*MW\s+([A-Za-z ]+?)\)/g)) {
@@ -191,7 +191,7 @@ facilityParser("pulsant_facility_v1", (p) => {
191 191 if (!m) return [];
192 192 const r = new Rec();
193 193 r.set("name", `Pulsant ${h1}`, "selector:h1").set("code", m[2], "selector:h1").set("city", m[1], "selector:h1").set("countryIso2", "GB", "default:operator");
194 − r.set("itCapacityMw", mw(first(p.text, /Total IT power\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(p.text), "regex:total_it_power_row").set("buildingSqm", sqm(first(p.text, /Total building size\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:building_size_row");
194 + r.set("itCapacityMw", mw(first(p.text, /Total IT power\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(p.text), "regex:total_it_power_row").set("buildingSqm", sqm(first(p.text, /Total building size\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:building_size_row");
195 195 const ren = first(p.text, /Renewable energy procurement\s*\n?\s*(\d+\s*%)/i); if (ren) r.set("renewableClaim", `${ren} renewable energy procurement`, "regex:renewable_row");
196 196 const cool = first(p.text, /Cooling:\s*([^\n]{2,30})/i); if (cool) r.set("coolingType", `Cooling ${cool}`, "regex:cooling_row");
197 197 r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
@@ -218,7 +218,7 @@ facilityParser("virtus_facility_v1", (p) => {
218 218 r.set("countryIso2", c, "url:segment>country");
219 219 const idx = code ? p.text.search(new RegExp(`\\b${code}\\b`)) : -1;
220 220 const seq = idx >= 0 ? p.text.slice(idx, idx + 3000) : p.text.slice(0, 3000);
221 − if (code) r.set("itCapacityMw", mw(first(seq, /(\d[\d.,]*\s*MW) of IT load/i)) ?? itMw(seq), "regex:mw_of_it_load").set("buildingSqm", sqm(first(seq, /([\d,.]+\s*m2) (?:of )?net technical/i)), "regex:net_technical_sqm");
221 + if (code) r.set("itCapacityMw", mw(first(seq, /(\d[\d.,]*\+?\s*MW\+?) of IT load/i)) ?? itMw(seq), "regex:mw_of_it_load").set("buildingSqm", sqm(first(seq, /([\d,.]+\s*m2) (?:of )?net technical/i)), "regex:net_technical_sqm");
222 222 const st = statusOf(seq, 2500); if (st) r.set("status", st, "regex:status_v1");
223 223 r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
224 224 return [r.done(key("virtus", code ?? slug), p.url, code ? 0.85 : 0.7)];
@@ -263,7 +263,7 @@ facilityParser("adc_facility_v1", (p) => {
263 263 const city = m[2]!.trim();
264 264 r.set("name", `Africa Data Centres ${m[1]} ${city}`, "regex:code_city").set("code", m[1], "regex:code_city").set("city", city, "regex:code_city");
265 265 const c = countryOf(p.text.slice(0, 4000), city); if (c) r.set("countryIso2", c.iso, c.method);
266 − r.set("itCapacityMw", mw(first(p.text, /(\d[\d.]*\s*MW) of (?:IT\s+)?power/i)) ?? itMw(p.text), "regex:mw_of_power").set("plannedPowerMw", mw(first(p.text, /IT load of (\d[\d.]*\s*MW) upon completion/i) ?? first(p.text, /(?:developed|expanded) to [\d,]+ square met(?:re|er)s and (\d[\d.]*\s*MW)/i)), "regex:upon_completion");
266 + r.set("itCapacityMw", mw(first(p.text, /(\d[\d.]*\+?\s*MW\+?) of (?:IT\s+)?power/i)) ?? itMw(p.text), "regex:mw_of_power").set("plannedPowerMw", mw(first(p.text, /IT load of (\d[\d.]*\+?\s*MW\+?) upon completion/i) ?? first(p.text, /(?:developed|expanded) to [\d,]+ square met(?:re|er)s and (\d[\d.]*\+?\s*MW\+?)/i)), "regex:upon_completion");
267 267 r.set("buildingSqm", sqm(first(p.text, /([\d,]+\s*(?:m2|m²|square met(?:re|er)s)) of (?:IT space|secured rack space)/i)), "regex:it_space_sqm").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description");
268 268 return [r.done(key("adc", m[1]), p.url)];
269 269 });
modified apps/worker/src/connectors/operators2/shared.ts +29 −8
@@ -59,20 +59,23 @@ export function mw(s: string | null | undefined): number | null {
59 59 return v;
60 60 }
61 61
62 −/** MW explicitly labelled as IT / critical load for the facility described by `text`. */
62 +/**
63 + * MW explicitly labelled as IT / critical load for the facility described by `text`.
64 + * "At least" figures keep their raw shape — "10+MW", "20MW+", "150+ MW" — and `mw()` / `parseMw` read the stated figure.
65 + */
63 66 export function itMw(text: string): number | null {
64 67 const pats = [
65 − /(\d[\d.,]*\+?\s*MWs?)\s*(?:of\s+)?(?:total\s+)?(?:critical\s+)?(?:IT|critical)\s+(?:capacity|load|power)\b/i,
66 − /(?:IT|critical)\s+(?:load|capacity|power)(?:\s+capacity)?(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW)/i,
67 − /(\d[\d.,]*\+?\s*MWs?)\s+of\s+(?:critical\s+)?(?:IT|critical)\s+(?:load|power|capacity)/i,
68 − /(\d[\d.,]*\+?\s*MWs?)\s+(?:critical|IT)\b/i,
68 + /(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:critical\s+)?(?:IT|critical)\s+(?:capacity|load|power)\b/i,
69 + /(?:IT|critical)\s+(?:load|capacity|power)(?:\s+capacity)?(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)/i,
70 + /(\d[\d.,]*\+?\s*MWs?\+?)\s+of\s+(?:critical\s+)?(?:IT|critical)\s+(?:load|power|capacity)/i,
71 + /(\d[\d.,]*\+?\s*MWs?\+?)\s+(?:critical|IT)\b/i,
69 72 ];
70 73 for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } }
71 74 return null;
72 75 }
73 76 /** MW labelled as (total) capacity / power without an IT qualifier. */
74 77 export function capacityMw(text: string): number | null {
75 − const pats = [/(\d[\d.,]*\+?\s*MWs?)\s*(?:of\s+)?(?:total\s+)?(?:capacity|power)\b/i, /(?:total\s+)?(?:capacity|power)(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW)\b/i];
78 + const pats = [/(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:capacity|power)\b/i, /(?:total\s+)?(?:capacity|power)(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)(?![\w])/i];
76 79 for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } }
77 80 return null;
78 81 }
@@ -176,6 +179,18 @@ export function key(op: string, ...parts: Array<string | null | undefined>): str
176 179 export function ldWithAddress(html: string, type?: string): Array<Record<string, unknown> & { address: Record<string, unknown>; geo?: Record<string, unknown> }> {
177 180 return jsonLd(html, type).filter((o) => o.address && typeof o.address === "object") as Array<Record<string, unknown> & { address: Record<string, unknown>; geo?: Record<string, unknown> }>;
178 181 }
182 +/** Leaf of the JSON-LD BreadcrumbList ("Richmond, VA") — the page's own location line, more reliable than a facility-name h1. */
183 +export function breadcrumbLeaf(html: string): string | null {
184 + for (const o of jsonLd(html, "BreadcrumbList")) {
185 + const items = o.itemListElement;
186 + if (!Array.isArray(items) || !items.length) continue;
187 + const last = items[items.length - 1] as Record<string, unknown> | undefined;
188 + const item = last?.item as Record<string, unknown> | string | undefined;
189 + const name = cleanText(String(last?.name ?? (typeof item === "object" ? item?.name : "") ?? ""));
190 + if (name && !/^home$/i.test(name)) return name;
191 + }
192 + return null;
193 +}
179 194 export function ldCountry(a: Record<string, unknown>): string | null {
180 195 const c = a.addressCountry;
181 196 if (!c) return null;
@@ -192,9 +207,15 @@ export function applyLdGeo(r: Rec, o: Record<string, unknown>, precision = "exac
192 207 if (Number.isFinite(lat) && Number.isFinite(lng) && (lat !== 0 || lng !== 0)) r.set("lat", lat, method).set("lng", lng, method).set("geoPrecision", precision, method);
193 208 }
194 209
210 +/**
211 + * Version shared by every operators2 parser — part of the effective extractor version (docs/CONNECTORS.md § 6).
212 + * v2 (2026-09-12): "MW+" / "+MW" stat figures are read (itMw, capacityMw, stat regexes); EdgeConneX city from the breadcrumb.
213 + */
214 +export const OPERATORS2_PARSER_VERSION = "v2";
215 +
195 216 /** Register a facility parser with the standard shape. */
196 −export function facilityParser(name: string, parse: (p: Page, doc: RawDocument) => ExtractedRecord[]): Parser {
197 − const parser: Parser = { name, version: "v1", pageTypes: ["facility_page"], parse: (doc) => (isHtml(doc) ? parse(page(doc), doc) : []) };
217 +export function facilityParser(name: string, parse: (p: Page, doc: RawDocument) => ExtractedRecord[], version = OPERATORS2_PARSER_VERSION): Parser {
218 + const parser: Parser = { name, version, pageTypes: ["facility_page"], parse: (doc) => (isHtml(doc) ? parse(page(doc), doc) : []) };
198 219 registerParser(parser);
199 220 return parser;
200 221 }
modified docs/CONNECTORS.md +6 −6
@@ -82,8 +82,8 @@ export interface Parser {
82 82 A parser is a pure function of the archived document: **no network, no geocoding, no guessed figures** — it surfaces what the page publishes, with a per-field `methods` map for provenance (`{ itCapacityMw: "regex:it_mw_v1", lat: "json-ld:GeoCoordinates" }`). Parsers live in `apps/worker/src/connectors/<group>/<file>.ts` and are registered by the group's `index.ts`:
83 83
84 84 - `operators1/` (Equinix, Digital Realty, NTT, CyrusOne, QTS, Vantage, STACK, CoreSite, Switch): each file exports a `Parser` object; `operators1/index.ts` `register()` calls `registerParser` for each. Helpers in `operators1/shared.ts` (`parseNaAddress`, `parseEuAddress`, `parseCityLine`, `mwAfter`, `mwWithContext`, `areaSqm`, `certificationsFrom`, `record(kind, key, url, data, methods, certainty)` which drops empty fields and sets `pageType: facility_page`).
85 −- `operators2/` (DataBank, EdgeConneX, TierPoint, NEXTDC, AirTrunk, atNorth, CloudHQ, …, grouped in `americas.ts` / `emea.ts` / `apac.ts`): parsers are declared with **`facilityParser(name, (page, doc) => ExtractedRecord[])`** from `operators2/shared.ts`, which registers on import and hands you a `Page` (`$`, `html`, visible `text` with nav/header/footer removed, `title`, `h1`, `desc`, `url`). Build records with the **`Rec`** builder: `new Rec().set(field, value, method)` ignores null/empty/NaN and records the method; `.done(key, url, certainty = 0.85, kind = "facility")`. Other helpers: `itMw`, `capacityMw`, `mw` (handles `1.125 MW` decimals and `MWs`), `sqm`, `ha`, `pue`, `tier`, `certifications`, `usAddress`, `ldWithAddress` / `applyLdAddress` / `applyLdGeo` (JSON-LD), `countryOf` / `countryFromCity` (deterministic metro → ISO table, not geocoding), `statusOf`, `key(op, ...parts)`. `operators2/index.ts` `register()` only checks that every name in `AMERICAS_PARSERS` / `EMEA_PARSERS` / `APAC_PARSERS` is registered.
86 −- `news/` (`news_article_v1`, `news_planning_pdf_v1`, `news_edgar_fts_v1` + the `news_edgar_fts` implementation): article → `news_event` (+ `project` when `qualifiesAsProject`) via `articleContent(doc)` → `extractAnnouncement(title, text)` (`extract-project.ts`, unit-tested in `extract-project.test.ts`). Params: `keepText` (public-sector sources only), `minProjectMw` (5), `minInvestmentUsd` (50 000 000), `minAcres` (100), `lenient`.
85 +- `operators2/` (DataBank, EdgeConneX, TierPoint, NEXTDC, AirTrunk, atNorth, CloudHQ, …, grouped in `americas.ts` / `emea.ts` / `apac.ts`): parsers are declared with **`facilityParser(name, (page, doc) => ExtractedRecord[], version = OPERATORS2_PARSER_VERSION)`** from `operators2/shared.ts`, which registers on import and hands you a `Page` (`$`, `html`, visible `text` with nav/header/footer removed, `title`, `h1`, `desc`, `url`). Build records with the **`Rec`** builder: `new Rec().set(field, value, method)` ignores null/empty/NaN and records the method; `.done(key, url, certainty = 0.85, kind = "facility")`. Other helpers: `itMw`, `capacityMw`, `mw` (handles `1.125 MW` decimals, `MWs`, and "at least" figures `10+MW` / `20MW+` — pass the raw string, `parseMw` reads the stated figure), `sqm`, `ha`, `pue`, `tier`, `certifications`, `usAddress`, `ldWithAddress` / `applyLdAddress` / `applyLdGeo` / `breadcrumbLeaf` (JSON-LD), `countryOf` / `countryFromCity` (deterministic metro → ISO table, not geocoding), `statusOf`, `key(op, ...parts)`. `operators2/index.ts` `register()` only checks that every name in `AMERICAS_PARSERS` / `EMEA_PARSERS` / `APAC_PARSERS` is registered. A stat regex that reads a figure before its label must accept the `+` suffix: `/(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Capacity/`.
86 +- `news/` (`news_article_v1`, `news_planning_pdf_v1`, `news_edgar_fts_v1` + the `news_edgar_fts` implementation): article → `news_event` (+ `project` when `qualifiesAsProject`) via `articleContent(doc)` → `extractAnnouncement(title, text)` (`extract-project.ts`, unit-tested in `extract-project.test.ts`). `articleContent` uses `mainText` (`packages/connectors/src/extract.ts`): the largest `<article>` / `<main>` block after removing nav / header / footer / aside / related / recommended / trending / popular / sidebar / comments blocks, cut at the first trailing teaser heading ("More in …", "Related articles", "Tags"). Operators are looked up in the title plus the first `OPERATOR_SCOPE_CHARS` (4 000) characters only; money figures in a company-background sentence (`MONEY_PORTFOLIO_RE`: "investment volume", "revenue", "assets under management"…) are never the project's investment. Params: `keepText` (public-sector sources only), `minProjectMw` (5), `minInvestmentUsd` (50 000 000), `minAcres` (100), `lenient`.
87 87 - `cloud/` (AWS, Azure, GCP, Oracle, IBM, Alibaba, Tencent, Meta regions) and `datasets/` (PeeringDB, Wikidata, World Bank, OSM Overpass): mostly `implementation`s.
88 88
89 89 `apps/worker/src/connectors/index.ts` `registerAllConnectors()` imports every `<group>/index.ts` found on disk and calls its `register()`; a group that throws is reported and the others still load. Make `register()` idempotent (guard with `listParsers()` or a module flag) — the worker, the CLI, `try-connector` and the fixture tests may all call it. Add a new group by creating `apps/worker/src/connectors/<group>/index.ts` with `export function register(): void`.
@@ -151,8 +151,8 @@ Rules for the body: keep it byte-for-byte (no reformatting, no CRLF normalisatio
151 151 "itCapacityMw": 0.675, "totalPowerMw": null, // null = absent or null; numbers are compared exactly
152 152 "mentions.countriesIso2": ["US"], // dotted paths reach nested fields
153 153 "_todo": { // CORRECT values the parser does not produce yet → it.fails
154 − "city": "Richmond",
155 − "_reason": "h1 fallback copies 'Richmond Data Center' into city"
154 + "totalPowerMw": 36,
155 + "_reason": "the figure precedes its label ('36 MW total power'); the stat regex only reads label-first"
156 156 }
157 157 }
158 158 ],
@@ -176,7 +176,7 @@ pnpm --filter @dci/worker test # everythi
176 176 pnpm --filter @dci/worker exec vitest run src/connectors/fixtures.test.ts -t databank # one connector
177 177 ```
178 178
179 −Golden coverage on 2026-09-12: 21 fixtures over 18 connectors (equinix, digitalrealty, stack, databank ×2, ntt, edgeconnex ×2, cyrusone, qts ×2, vantage, coresite, airtrunk, nextdc ×2, atnorth, cloudhq, datacenterfrontier, datacenterdynamics, loudoun-county). No planning PDF fixture yet: the archive holds no document with a PDF content type.
179 +Golden coverage on 2026-09-12: 21 fixtures over 18 connectors (equinix, digitalrealty, stack, databank ×2, ntt, edgeconnex ×2, cyrusone, qts ×2, vantage, coresite, airtrunk, nextdc ×2, atnorth, cloudhq, datacenterfrontier, datacenterdynamics, loudoun-county), no open `_todo` (the nine gaps recorded on 2026-09-12 — EdgeConneX city from the h1, NEXTDC `20MW+`, QTS figure-first capacity and template-copied map pin, DCD teaser operator and portfolio money — were fixed the same day and promoted). No planning PDF fixture yet: the archive holds no document with a PDF content type.
180 180
181 181 ## 4. Live dry-run
182 182
@@ -200,7 +200,7 @@ Two different things share the word:
200 200 The **effective extractor version** stored on every document (`documents.extractor_version`) is computed by `effectiveExtractorVersion(cfg, connector)` in `apps/worker/src/configs.ts`:
201 201
202 202 ```
203 −<cfg.parserVersion>+<sha256("news_article_v1@news_v2|databank_facility_v1@v1|impl:peeringdb@v1")[0:10]>
203 +<cfg.parserVersion>+<sha256("news_article_v1@news_v3|databank_facility_v1@v2|impl:peeringdb@v1")[0:10]>
204 204 ```
205 205
206 206 i.e. the YAML `parserVersion` plus a hash of every referenced `Parser.name@Parser.version` (and `impl:<name>@<connector.parserVersion>`). Consequences:
modified packages/connectors/src/extract.test.ts +24 −1
@@ -1,6 +1,6 @@
1 1 import { describe, expect, it } from "vitest";
2 2 import { load } from "cheerio";
3 −import { applyTransform, embeddedJson, evalRule, extractAddress, extractGeo, jsonLd, jsonPath, pdfText, publishedDate, walkJson } from "./extract.js";
3 +import { applyTransform, cutTeasers, embeddedJson, evalRule, extractAddress, extractGeo, jsonLd, jsonPath, mainText, pdfText, publishedDate, walkJson } from "./extract.js";
4 4
5 5 const HTML = `<html><head><title>Page</title>
6 6 <meta property="og:title" content="OG Title"><meta name="description" content="Desc">
@@ -90,6 +90,29 @@ describe("geo / address / date", () => {
90 90 });
91 91 });
92 92
93 +describe("mainText", () => {
94 + const story = `<p>${"The developer secured planning permission for a 12MW data center in Bucharest. ".repeat(6)}</p>`;
95 + it("keeps the largest <article>, drops related / recommended / trending / sidebar / aside / footer blocks and cuts trailing teasers", () => {
96 + const html = `<html><body><nav>Home</nav><main>
97 +<article class="card"><a>Microsoft expands plans for La Porte campus</a></article>
98 +<article itemprop="mainEntity"><h1>S+B Gruppe to build data center in Bucharest</h1>${story}
99 +<div class="block-auto_featured_content"><h2>More in Construction &amp; Site Selection</h2><a>Ignis to build DayOne's 300MW data center</a></div></article>
100 +<div class="related-articles"><a>Google onboarding case study</a></div><div id="sidebar-popular"><a>Equinix opens LD14</a></div>
101 +<section class="recommended-for-you"><a>Vantage raises $2bn</a></section><div class="trending-now"><a>Digital Realty results</a></div></main>
102 +<aside><a>AWS buys site</a></aside><footer>Amazon Web Services</footer></body></html>`;
103 + const t = mainText(html);
104 + expect(t).toContain("S+B Gruppe to build data center in Bucharest");
105 + expect(t).toContain("planning permission");
106 + for (const leak of ["Microsoft", "Ignis", "More in", "Google", "Equinix", "Vantage", "Digital Realty", "AWS", "Amazon"]) expect(t, leak).not.toContain(leak);
107 + });
108 + it("cutTeasers only cuts after the story has started and only on a heading line", () => {
109 + expect(cutTeasers("Related\nshort lead")).toBe("Related\nshort lead");
110 + const body = "x".repeat(400);
111 + expect(cutTeasers(`${body}\n Tags \n Bucharest\n Comments`)).toBe(body);
112 + expect(cutTeasers(`${body}\nMore in the article follows here with a long sentence about the project itself.`)).toContain("More in the article");
113 + });
114 +});
115 +
93 116 describe("pdfText limits", () => {
94 117 it("rejects oversized PDFs before parsing", async () => {
95 118 await expect(pdfText(Buffer.alloc(16 * 1024 * 1024), 60, { maxBytes: 15 * 1024 * 1024 })).rejects.toThrow(/too large/);
modified packages/connectors/src/extract.ts +31 −5
@@ -126,16 +126,42 @@ export function metaDescription(html: string): string | null {
126 126 return cleanText($('meta[name="description"]').attr("content") ?? $('meta[property="og:description"]').attr("content") ?? null);
127 127 }
128 128
129 −/** Main text of an article-like page (heuristic: largest <article>/<main>/content block). */
129 +/** Blocks that are never the story: chrome, related / recommended / trending teasers, sidebars, comments. */
130 +export const NON_CONTENT_SELECTOR = "script,style,noscript,nav,header,footer,aside,form,iframe,svg,[role=navigation],[role=complementary],[class*=related],[class*=recommend],[class*=trending],[class*=popular],[class*=sidebar],[class*=read-more],[class*=readmore],[class*=more-stories],[class*=also-like],[class*=comments],[id*=related],[id*=comments]";
131 +/** Heading that opens a trailing teaser section inside the story element itself ("More in Construction & Site Selection", "Related articles", "Tags"). */
132 +export const TEASER_HEADING_RE = /^\s*(?:More (?:in|from|on|like this)\b.*|Related(?: (?:articles?|stories|news|content|posts?|coverage|reading|links?))?|Recommended(?: for you| reading| stories| articles)?|You (?:may|might) also like|Most (?:read|popular|viewed)|Trending(?: now| stories)?|Popular (?:now|posts|stories|articles)|Latest (?:news|stories|posts|articles)|Read (?:more|next|also)|See also|Further reading|Editor'?s picks|Sponsored(?: content)?|Tags|Comments|Leave a (?:comment|reply))\s*:?\s*$/i;
133 +/** Minimum story length before a teaser heading is allowed to cut the text (a short lead titled "Related" is not a teaser section). */
134 +const TEASER_CUT_MIN_CHARS = 300;
135 +
136 +/** Drop everything from the first teaser heading onwards (teasers are the last thing in a story element). */
137 +export function cutTeasers(text: string): string {
138 + const lines = text.split("\n");
139 + let consumed = 0;
140 + for (let i = 0; i < lines.length; i++) {
141 + const line = lines[i]!;
142 + if (consumed >= TEASER_CUT_MIN_CHARS && line.trim().length <= 60 && TEASER_HEADING_RE.test(line)) return lines.slice(0, i).join("\n").trimEnd();
143 + consumed += line.length;
144 + }
145 + return text;
146 +}
147 +
148 +/**
149 + * Main text of an article-like page: the largest <article> / <main> / content block, after removing navigation, footer,
150 + * aside, related / recommended / trending / popular / sidebar blocks, and cut at the first trailing teaser heading — so a
151 + * "More in …" list of other headlines never leaks into the story (and never names its operator).
152 + */
130 153 export function mainText(html: string): string {
131 154 const $ = load(html);
132 − $("script,style,noscript,nav,header,footer,aside,form,iframe,svg").remove();
155 + $(NON_CONTENT_SELECTOR).remove();
133 156 const candidates = ["article", "main", '[role="main"]', ".article-body", ".press-release", ".entry-content", ".post-content", ".content", "#content", "body"];
134 157 for (const c of candidates) {
135 − const el = $(c).first();
136 − if (el.length) { const t = htmlToText(el.html() ?? ""); if (t.length > 300) return t; }
158 + // several matches (the story plus teaser <article class="card"> cards): the largest one is the story
159 + const texts = $(c).toArray().map((el) => htmlToText($(el).html() ?? ""));
160 + if (!texts.length) continue;
161 + const t = cutTeasers(texts.sort((a, b) => b.length - a.length)[0]!);
162 + if (t.length > 300) return t;
137 163 }
138 − return htmlToText($("body").html() ?? html);
164 + return cutTeasers(htmlToText($("body").html() ?? html));
139 165 }
140 166
141 167 /** Publication date from meta tags / JSON-LD / <time>. */
142 168