Parsers: EdgeConneX breadcrumb city, QTS figure-before-label + own-caption map pins, '+'-suffixed MW across operators2 (v2), news main-text teaser/footer stripping + operator scope + portfolio money filter (news_v3), quoted project names; all fixture gaps promoted to assertions
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
15 changed files +263 −98
modified
apps/worker/fixtures/datacenterdynamics/sb-gruppe-bucharest.expected.json
+5 −11
@@ -3,7 +3,7 @@ | ||
| 3 | 3 | "documentId": "doc_555b8c8f69f1bdbb", |
| 4 | 4 | "fetchedAt": "2026-09-12T02:09:08.275Z", |
| 5 | 5 | "contentType": "text/html; charset=utf-8", |
| 6 | − "note": "NEW_BUILD: 'The 12MW facility has secured planning permission' → news_event + project (12 MW, Bucharest, RO). Two attribution errors are flagged in _todo: the $9.4bn is S+B Gruppe's whole portfolio ('total investment volume of €8.1 billion ($9.4bn)'), and 'Microsoft' only appears in the related-articles footer ('Microsoft expands plans for La Porte…'). Author byline redacted.", | |
| 6 | + "note": "NEW_BUILD: 'The 12MW facility has secured planning permission' → news_event + project (12 MW, Bucharest, RO). Two attribution traps, both handled since 2026-09-12: the $9.4bn is S+B Gruppe's whole portfolio ('total investment volume of €8.1 billion ($9.4bn)') → no investment; 'Microsoft' only appears in the 'More in Construction & Site Selection' teasers, which mainText now drops → no operator (S+B Gruppe is not in the lexicon). The project keeps its name from the lead ('Named ‘NIMB,’'). Author byline redacted.", | |
| 7 | 7 | "counts": { |
| 8 | 8 | "news_event": 1, |
| 9 | 9 | "project": 1 |
@@ -25,11 +25,8 @@ | ||
| 25 | 25 | "countryIso2": "RO", |
| 26 | 26 | "status": "announced", |
| 27 | 27 | "projectClass": "NEW_BUILD", |
| 28 | − "_todo": { | |
| 29 | − "operatorName": null, | |
| 30 | − "investmentUsd": null, | |
| 31 | − "_reason": "extract-project: operator 'Microsoft' comes from the 'More in Construction & Site Selection' footer teasers that mainText() keeps; the $9.4bn is the developer's portfolio volume, not this project's investment (the article gives no figure for it). S+B Gruppe is not in the operators lexicon, so the correct operator is null (or 'S+B Gruppe' once added)" | |
| 32 | − } | |
| 28 | + "operatorName": null, | |
| 29 | + "investmentUsd": null | |
| 33 | 30 | } |
| 34 | 31 | ], |
| 35 | 32 | "thirdParty": true, |
@@ -40,10 +37,7 @@ | ||
| 40 | 37 | "country": "RO", |
| 41 | 38 | "city": "Bucharest", |
| 42 | 39 | "status": "announced", |
| 43 | − "_todo": { | |
| 44 | − "operator": null, | |
| 45 | − "investmentUsd": null, | |
| 46 | − "_reason": "see projects[0]._todo" | |
| 47 | − } | |
| 40 | + "operator": null, | |
| 41 | + "investmentUsd": null | |
| 48 | 42 | } |
| 49 | 43 | } |
modified
apps/worker/fixtures/edgeconnex/richmond-va.expected.json
+4 −7
@@ -3,7 +3,7 @@ | ||
| 3 | 3 | "documentId": "doc_498ed95c603598af", |
| 4 | 4 | "fetchedAt": "2026-09-11T08:11:23.445Z", |
| 5 | 5 | "contentType": "text/html; charset=UTF-8", |
| 6 | − "note": "No JSON-LD LocalBusiness with a street on this page → h1 fallback. h1 is 'Richmond Data Center' and the breadcrumb says 'Richmond, VA': the parser currently copies the whole h1 into city (known gap, see _todo).", | |
| 6 | + "note": "No JSON-LD LocalBusiness with a street on this page → metro-level fallback. h1 is 'Richmond Data Center' (facility name); the JSON-LD BreadcrumbList leaf 'Richmond, VA' carries city + state (fixed 2026-09-12: the parser used to copy the whole h1 into city).", | |
| 7 | 7 | "counts": { |
| 8 | 8 | "facility": 1 |
| 9 | 9 | }, |
@@ -12,6 +12,8 @@ | ||
| 12 | 12 | "key": "edgeconnex:richmond-va", |
| 13 | 13 | "name": "EdgeConneX Richmond Data Center", |
| 14 | 14 | "operatorName": "EdgeConneX", |
| 15 | + "city": "Richmond", | |
| 16 | + "regionName": "VA", | |
| 15 | 17 | "countryIso2": "US", |
| 16 | 18 | "itCapacityMw": null, |
| 17 | 19 | "certifications": [ |
@@ -20,12 +22,7 @@ | ||
| 20 | 22 | "PCI DSS", |
| 21 | 23 | "HIPAA", |
| 22 | 24 | "ENERGY STAR" |
| 23 | − ], | |
| 24 | − "_todo": { | |
| 25 | − "city": "Richmond", | |
| 26 | − "regionName": "VA", | |
| 27 | − "_reason": "edgeconnex_facility_v1 h1 fallback: /^([A-Z][^,]+?)(?:,\\s*([A-Z][^,]+))?$/ captures 'Richmond Data Center' as the city; the breadcrumb 'Richmond, VA' carries city + state" | |
| 28 | − } | |
| 25 | + ] | |
| 29 | 26 | } |
| 30 | 27 | ] |
| 31 | 28 | } |
modified
apps/worker/fixtures/nextdc/p2-perth.expected.json
+3 −6
@@ -3,7 +3,7 @@ | ||
| 3 | 3 | "documentId": "doc_10f36a288aedf0b3", |
| 4 | 4 | "fetchedAt": "2026-09-11T08:31:56.476Z", |
| 5 | 5 | "contentType": "text/html; charset=UTF-8", |
| 6 | − "note": "Same stat block as B2 but '20MW+ IT Capacity': the trailing '+' breaks /(\\d[\\d.,]*\\s*MW)\\s*\\n?\\s*IT Capacity/ and itMw() finds no 'IT' qualifier → itCapacityMw missing (see _todo). 12,000 m², 6,300 racks, Tier IV are fine.", | |
| 6 | + "note": "Same stat block as B2 but '20MW+ IT Capacity': the trailing '+' (an 'at least' figure) is accepted by the stat regex since 2026-09-12 and parsed as the stated 20 MW. 12,000 m², 6,300 racks, Tier IV.", | |
| 7 | 7 | "counts": { |
| 8 | 8 | "facility": 1 |
| 9 | 9 | }, |
@@ -14,13 +14,10 @@ | ||
| 14 | 14 | "operatorName": "NEXTDC", |
| 15 | 15 | "city": "Perth", |
| 16 | 16 | "countryIso2": "AU", |
| 17 | + "itCapacityMw": 20, | |
| 17 | 18 | "buildingSqm": 12000, |
| 18 | 19 | "rackCount": 6300, |
| 19 | − "tier": "Tier IV", | |
| 20 | − "_todo": { | |
| 21 | − "itCapacityMw": 20, | |
| 22 | − "_reason": "nextdc_facility_v1 stat regex does not allow the '+' suffix ('20MW+' then 'IT Capacity'); P1 Malaga '10+MW' and S5 '80+MW' have the same shape" | |
| 23 | − } | |
| 20 | + "tier": "Tier IV" | |
| 24 | 21 | } |
| 25 | 22 | ] |
| 26 | 23 | } |
modified
apps/worker/fixtures/qts/phoenix-2.expected.json
+3 −6
@@ -3,7 +3,7 @@ | ||
| 3 | 3 | "documentId": "doc_de2f9d8b562c5c33", |
| 4 | 4 | "fetchedAt": "2026-09-11T08:46:05.593Z", |
| 5 | 5 | "contentType": "text/html; charset=UTF-8", |
| 6 | − "note": "'80 acre campus' → 32.37 ha; address 'DC2 - 1200 N 40th St, Phoenix, AZ 85008'. The page states '210 MW+ of critical campus capacity' (figure BEFORE the label) which qts_facility_v1 does not pick up (see _todo). No Google-Maps marker on this page → geo null.", | |
| 6 | + "note": "'80 acre campus' → 32.37 ha; address 'DC2 - 1200 N 40th St, Phoenix, AZ 85008'. The page states '210 MW+ of critical campus capacity' (figure BEFORE the label) → totalPowerMw 210 (read since 2026-09-12). No Google-Maps marker on this page → geo null.", | |
| 7 | 7 | "counts": { |
| 8 | 8 | "facility": 1 |
| 9 | 9 | }, |
@@ -18,11 +18,8 @@ | ||
| 18 | 18 | "countryIso2": "US", |
| 19 | 19 | "siteAreaHa": 32.37, |
| 20 | 20 | "itCapacityMw": null, |
| 21 | − "geo": null, | |
| 22 | − "_todo": { | |
| 23 | − "totalPowerMw": 210, | |
| 24 | − "_reason": "qts_facility_v1 only matches 'Critical campus capacity <n> MW' (label first) and '<n> MW of critical/total power|capacity|IT load'; '210 MW+ of critical campus capacity' matches neither" | |
| 25 | − } | |
| 21 | + "totalPowerMw": 210, | |
| 22 | + "geo": null | |
| 26 | 23 | } |
| 27 | 24 | ] |
| 28 | 25 | } |
modified
apps/worker/fixtures/qts/phoenix-4.expected.json
+2 −5
@@ -3,7 +3,7 @@ | ||
| 3 | 3 | "documentId": "doc_3b14b8129dd879b3", |
| 4 | 4 | "fetchedAt": "2026-09-11T08:45:56.574Z", |
| 5 | 5 | "contentType": "text/html; charset=UTF-8", |
| 6 | − "note": "Address '11753 W Lower Buckeye Rd, Tolleson, AZ 85353' parses correctly, but the Elementor Google-Maps widget on this page carries address:'41.84273925236679, -87.66755675771267' — Chicago, ~2 300 km from Tolleson — which the parser stores as 'exact' coordinates (see _todo).", | |
| 6 | + "note": "Address '11753 W Lower Buckeye Rd, Tolleson, AZ 85353' parses correctly. The Elementor Google-Maps widget on this page carries a pin at 41.84273925236679, -87.66755675771267 captioned '2800 S Ashland Ave, Chicago, IL 60608' — Chicago, ~2 300 km from Tolleson, copied from another template. Since 2026-09-12 a pin is used only when its caption names this facility's postal code or city + state → geo null (never another site's point).", | |
| 7 | 7 | "counts": { |
| 8 | 8 | "facility": 1 |
| 9 | 9 | }, |
@@ -17,10 +17,7 @@ | ||
| 17 | 17 | "regionName": "AZ", |
| 18 | 18 | "postalCode": "85353", |
| 19 | 19 | "countryIso2": "US", |
| 20 | − "_todo": { | |
| 21 | − "geo": null, | |
| 22 | − "_reason": "qts_facility_v1 trusts the Elementor google_maps marker without checking it against the parsed state (AZ): 41.84, -87.67 is Chicago. Correct output is no coordinates (never fake / wrong 'exact' geo)" | |
| 23 | − } | |
| 20 | + "geo": null | |
| 24 | 21 | } |
| 25 | 22 | ] |
| 26 | 23 | } |
modified
apps/worker/src/connectors/news/extract-project.test.ts
+43 −1
@@ -1,6 +1,6 @@ | ||
| 1 | 1 | import { describe, expect, it } from "vitest"; |
| 2 | 2 | import type { ConnectorContext, RawDocument } from "@dci/connectors"; |
| 3 | −import { extractAnnouncement, parseAllMoney, parseExpectedOpening, parsePhaseCount, qualifiesAsProject } from "./extract-project.js"; | |
| 3 | +import { extractAnnouncement, parseAllMoney, parseExpectedOpening, parsePhaseCount, projectMoneyFigures, qualifiesAsProject } from "./extract-project.js"; | |
| 4 | 4 | import { detectOperators, OPERATORS } from "./operators-lexicon.js"; |
| 5 | 5 | import { detectLocation } from "./locations-lexicon.js"; |
| 6 | 6 | import { newsArticleParser } from "./article-parser.js"; |
@@ -360,8 +360,50 @@ describe("news_article_v1", () => { | ||
| 360 | 360 | const off = ARTICLE.replace(/data center|Data Centers|192MW|hyperscale|campus/g, "bakery"); |
| 361 | 361 | expect(await newsArticleParser.parse(htmlDoc("https://example.com/news/bakery", off), ctx())).toEqual([]); |
| 362 | 362 | }); |
| 363 | + it("an operator named only in related-articles / footer teasers never becomes the project's operator", async () => { | |
| 364 | + // the story is about an unknown developer; Microsoft, Google, Equinix and AWS appear only in teaser blocks (DCD "More in …" shape) | |
| 365 | + const html = `<!doctype html><html><head><title>Riverside Estates to build 30MW data center in Leesburg | Trade Press</title> | |
| 366 | +<meta property="article:published_time" content="2026-09-10T09:30:00Z"></head><body><nav>Home News</nav><main> | |
| 367 | +<article class="card"><a href="/t1">Microsoft expands plans for La Porte, Indiana, data center campus</a></article> | |
| 368 | +<article itemprop="mainEntity"><h1>Riverside Estates to build 30MW data center in Leesburg</h1> | |
| 369 | +<p>Property developer Riverside Estates has secured planning permission for a 30MW data center in Leesburg, Virginia. Named ‘Aire Park One,’ the facility is expected to open in 2028 and will offer 4,000 sqm of white space to cloud providers and financial institutions.</p> | |
| 370 | +<p>Founded in 1998, Riverside Estates manages an estate of 600,000 sqm across the region and a total investment volume of £2.1 billion ($2.7bn). The Leesburg site is its first data center.</p> | |
| 371 | +<div class="block-auto_featured_content"><h2>More in Construction & Site Selection</h2><a href="/t2">Google to build 300MW data center campus in Aragon, Spain</a></div></article> | |
| 372 | +<div class="related-articles"><a href="/t3">Equinix opens LD14 in Slough</a></div></main> | |
| 373 | +<aside><a href="/t4">Trending: Amazon Web Services buys 1GW site in Ohio</a></aside> | |
| 374 | +<footer><p>Microsoft, Google and Equinix are trademarks of their owners.</p></footer></body></html>`; | |
| 375 | + const recs = await newsArticleParser.parse(htmlDoc("https://example.com/news/riverside-leesburg", html), ctx(), { keepText: false }); | |
| 376 | + expect(recs.map((r) => r.kind)).toEqual(["news_event", "project"]); | |
| 377 | + const [ev, pr] = recs; | |
| 378 | + expect(ev!.data.operators).toEqual([]); | |
| 379 | + expect(pr!.data).toMatchObject({ operatorName: null, name: "Aire Park One", plannedMw: 30, city: "Leesburg", countryIso2: "US", status: "announced", investmentUsd: null }); | |
| 380 | + expect(pr!.methods?.operatorName).toBe("none"); | |
| 381 | + // the same article with the developer in the lexicon keeps its operator (the teasers still count for nothing) | |
| 382 | + const known = html.replace(/Riverside Estates/g, "Vantage Data Centers"); | |
| 383 | + const a = extractAnnouncement("Vantage Data Centers to build 30MW data center in Leesburg", (await articleContentOf(known)).text, { publishedAt: "2026-09-10", now: NOW }); | |
| 384 | + expect(a.operator?.name).toBe("Vantage Data Centers"); | |
| 385 | + expect(a.operators.map((o) => o.name)).toEqual(["Vantage Data Centers"]); | |
| 386 | + }); | |
| 387 | + it("company-background money (portfolio volume, revenue) is never the project's investment", () => { | |
| 388 | + const text = "S+B Gruppe manages an estate of 414,000 sqm across central and eastern Europe and a total investment volume of €8.1 billion ($9.4bn). The 12MW facility will cost $60 million."; | |
| 389 | + expect(parseAllMoney(text)).toHaveLength(3); | |
| 390 | + expect(projectMoneyFigures(text)).toEqual([{ amount: 60_000_000, currency: "USD" }]); | |
| 391 | + expect(projectMoneyFigures("The company reported revenue of $4.2 billion last year.")).toEqual([]); | |
| 392 | + expect(projectMoneyFigures("Vantage will invest $2 billion in the Frederick campus.")).toEqual([{ amount: 2_000_000_000, currency: "USD" }]); | |
| 393 | + const a = x("S+B Gruppe to build data center in Bucharest, Romania", "Real estate developer S+B Gruppe has secured planning permission to build a 12MW data center in Bucharest. Named ‘NIMB,’ the facility is expected to open in 2028. Founded in 1986 in Vienna, S+B Gruppe manages an estate of more than 414,000 sqm and a total investment volume of €8.1 billion ($9.4bn)."); | |
| 394 | + expect(a.money).toBeNull(); | |
| 395 | + expect(a.investmentUsd).toBeNull(); | |
| 396 | + expect(a.explicitName).toBe("NIMB"); | |
| 397 | + expect(a.operator).toBeNull(); | |
| 398 | + expect(a.classification.mayCreateProject).toBe(true); | |
| 399 | + }); | |
| 363 | 400 | }); |
| 364 | 401 | |
| 402 | +async function articleContentOf(html: string) { | |
| 403 | + const { articleContent } = await import("./article-parser.js"); | |
| 404 | + return articleContent(htmlDoc("https://example.com/news/x", html)); | |
| 405 | +} | |
| 406 | + | |
| 365 | 407 | describe("news_planning_pdf_v1", () => { |
| 366 | 408 | const REPORT = `<html><head><title>Staff Report — Rezoning Application REZ 2026-0017</title></head><body><main> |
| 367 | 409 | <h1>Planning Commission Staff Report</h1> |
modified
apps/worker/src/connectors/news/extract-project.ts
+37 −5
@@ -52,7 +52,9 @@ export interface Announcement { | ||
| 52 | 52 | hqGuarded: boolean; |
| 53 | 53 | } |
| 54 | 54 | |
| 55 | −export const EXTRACTOR_VERSION = "news_v2"; | |
| 55 | +export const EXTRACTOR_VERSION = "news_v3"; | |
| 56 | +/** Operators are looked up in the title and the first 4 000 characters of the story only — never in trailing teasers or footers. */ | |
| 57 | +export const OPERATOR_SCOPE_CHARS = 4000; | |
| 56 | 58 | |
| 57 | 59 | /** |
| 58 | 60 | * Company-domicile mentions must not locate a project: "Denver-based Vantage plans a campus in Abilene, Texas" is in |
@@ -176,7 +178,8 @@ const CURRENCY_START_RE = /(US\$|USD|\$|CA\$|C\$|€|EUR|£|GBP|A\$|AUD|¥|JPY|S | ||
| 176 | 178 | /** Approximate rates — gating only (is this ≥ $50M?), never persisted. */ |
| 177 | 179 | const APPROX_USD: Record<string, number> = { USD: 1, EUR: 1.1, GBP: 1.3, CAD: 0.73, AUD: 0.66, SGD: 0.75, INR: 0.012, JPY: 0.0067, CHF: 1.12, SEK: 0.095, NOK: 0.093, DKK: 0.147 }; |
| 178 | 180 | |
| 179 | −export function parseAllMoney(text: string): Money[] { | |
| 181 | +/** Every money figure in a text (largest first); `keep(start)` may veto a figure from its position in `text`. */ | |
| 182 | +export function parseAllMoney(text: string, keep: (start: number) => boolean = () => true): Money[] { | |
| 180 | 183 | const out: Money[] = []; |
| 181 | 184 | const seen = new Set<string>(); |
| 182 | 185 | CURRENCY_START_RE.lastIndex = 0; |
@@ -187,6 +190,7 @@ export function parseAllMoney(text: string): Money[] { | ||
| 187 | 190 | const money = parseMoney(window); |
| 188 | 191 | // < $1M is not a project investment; > $500B is an industry-wide statistic ("$31.6 trillion AI buildout") |
| 189 | 192 | if (!money || money.amount < 1_000_000 || money.amount > 500_000_000_000) continue; |
| 193 | + if (!keep(m.index ?? 0)) continue; | |
| 190 | 194 | const k = `${money.currency}:${money.amount}`; |
| 191 | 195 | if (seen.has(k)) continue; |
| 192 | 196 | seen.add(k); |
@@ -195,6 +199,25 @@ export function parseAllMoney(text: string): Money[] { | ||
| 195 | 199 | return out.sort((a, b) => b.amount - a.amount); |
| 196 | 200 | } |
| 197 | 201 | |
| 202 | +/** | |
| 203 | + * Company-background wording before a money figure, in the same sentence: the developer's portfolio, balance sheet or | |
| 204 | + * history ("a total investment volume of €8.1 billion ($9.4bn)", "assets under management of…", "revenue of…") — never | |
| 205 | + * this project's budget. | |
| 206 | + */ | |
| 207 | +export const MONEY_PORTFOLIO_RE = /\b(investment volume|portfolio|assets under management|under management|AUM|market (?:cap|capitali[sz]ation)|net worth|revenues?|turnover|balance sheet|(?:manages|owns|operates|holds|controls) an? (?:estate|portfolio|footprint)|estate of|founded in \d{4}|since (?:its )?(?:founding|inception)|to date|over the (?:past|last) (?:\d+|two|three|five|ten) years)\b/i; | |
| 208 | + | |
| 209 | +/** The sentence fragment before position `at` (from the previous sentence break, at most `max` chars). */ | |
| 210 | +function sentenceBefore(text: string, at: number, max = 300): string { | |
| 211 | + const seg = text.slice(Math.max(0, at - max), at); | |
| 212 | + const brk = Math.max(seg.lastIndexOf(". "), seg.lastIndexOf("\n"), seg.lastIndexOf("? "), seg.lastIndexOf("! ")); | |
| 213 | + return brk >= 0 ? seg.slice(brk + 1) : seg; | |
| 214 | +} | |
| 215 | + | |
| 216 | +/** Money figures that may describe the announced project: company-background figures (portfolio volume, revenue…) are skipped. */ | |
| 217 | +export function projectMoneyFigures(scope: string): Money[] { | |
| 218 | + return parseAllMoney(scope, (start) => !MONEY_PORTFOLIO_RE.test(sentenceBefore(scope, start))); | |
| 219 | +} | |
| 220 | + | |
| 198 | 221 | export function approxUsd(m: Money | null): number | null { |
| 199 | 222 | if (!m) return null; |
| 200 | 223 | const r = APPROX_USD[m.currency]; |
@@ -302,8 +325,15 @@ export function findExplicitName(title: string | null, text: string, operator: O | ||
| 302 | 325 | return name.replace(/’/g, "'"); |
| 303 | 326 | } |
| 304 | 327 | } |
| 328 | + // "Named ‘NIMB,’ the firm…", "dubbed 'Project Sail'" — a quoted proper name given to the facility in the lead | |
| 329 | + const q = text.slice(0, 1500).match(QUOTED_NAME_RE); | |
| 330 | + if (q) { | |
| 331 | + const name = cleanText(q[1]!.replace(/’/g, "'")); | |
| 332 | + if (name && name.length >= 3 && name.length <= 60 && /^[A-Z0-9]/.test(name) && !GENERIC_WORD.test(name) && !UNIT_WORD.test(name) && !detectOperators(name).length) return name; | |
| 333 | + } | |
| 305 | 334 | return null; |
| 306 | 335 | } |
| 336 | +const QUOTED_NAME_RE = /\b(?:named|dubbed|called|known as|branded|code-?named|titled)\s+[‘'"“]([^‘’'"“”\n]{2,60}?)[,.]?[’'"”]/i; | |
| 307 | 337 | |
| 308 | 338 | export function cleanTitle(title: string | null): string { |
| 309 | 339 | if (!title) return ""; |
@@ -405,7 +435,7 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op | ||
| 405 | 435 | // money: same title → lead → body preference, largest figure in the winning scope |
| 406 | 436 | let money: Money | null = null; |
| 407 | 437 | for (const [scopeName, scope] of [["title", title], ["lead", `${title} ${head}`], ["body", `${title} ${body}`]] as Array<[string, string]>) { |
| 408 | − const all = parseAllMoney(scope); | |
| 438 | + const all = projectMoneyFigures(scope); | |
| 409 | 439 | if (all.length) { money = all[0]!; methods.investment = `regex:money_v1:${scopeName}`; break; } |
| 410 | 440 | } |
| 411 | 441 | const investmentUsd = money && money.currency === "USD" ? money.amount : null; |
@@ -441,8 +471,10 @@ export function extractAnnouncement(rawTitle: string | null, rawText: string, op | ||
| 441 | 471 | if (location) { methods.location = location.method; } |
| 442 | 472 | const hqGuarded = hqTitle.stripped || hqLead.stripped || hqBody.stripped; |
| 443 | 473 | |
| 444 | − const operators = detectOperators(`${title}\n${body}`); | |
| 445 | − let operator = primaryOperator(title, body); | |
| 474 | + // operators: title + the first OPERATOR_SCOPE_CHARS of the story only (a "More in …" teaser or a footer never names the developer) | |
| 475 | + const opScope = text.slice(0, OPERATOR_SCOPE_CHARS); | |
| 476 | + const operators = detectOperators(`${title}\n${opScope}`); | |
| 477 | + let operator = primaryOperator(title, opScope); | |
| 446 | 478 | // the headline's subject is a company we do not know → lexicon operators in the body are not the developer |
| 447 | 479 | if (operator && !detectOperators(title).length && unknownTitleSubject(title)) { operator = null; methods.operatorName = "none:unknown-title-subject"; } |
| 448 | 480 | else if (operator) methods.operatorName = "lexicon:operator"; |
modified
apps/worker/src/connectors/operators1/qts.ts
+40 −10
@@ -1,10 +1,11 @@ | ||
| 1 | −import { load } from "cheerio"; | |
| 1 | +import { load, type CheerioAPI } from "cheerio"; | |
| 2 | 2 | import type { ExtractedRecord, Parser, RawDocument } from "@dci/connectors"; |
| 3 | −import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, firstText, lastSlug, mwAfter, mwWithContext, parseCityLine, parseEuAddress, parseNaAddress, record, statusFromText, usStateCode } from "./shared.js"; | |
| 3 | +import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, firstText, lastSlug, mwAfter, mwWithContext, parseCityLine, parseEuAddress, parseNaAddress, record, statusFromText, usStateCode, type ParsedAddress } from "./shared.js"; | |
| 4 | 4 | |
| 5 | 5 | /** |
| 6 | 6 | * QTS (q.com/data-centers/<slug>/). WordPress + Elementor; two templates: |
| 7 | − * US campus: <h1>Ashburn 3</h1> … "Campus footprint 25 acre campus" … "Critical campus capacity 80 MW+" … <h2>Ashburn 3 (ASH3)</h2> | |
| 7 | + * US campus: <h1>Ashburn 3</h1> … "Campus footprint 25 acre campus" … "Critical campus capacity 80 MW+" (or, figure first, | |
| 8 | + * "210 MW+ of critical campus capacity") … <h2>Ashburn 3 (ASH3)</h2> | |
| 8 | 9 | * <p>ASH3 DC1: 22291 Shellhorn Rd<br>Ashburn, VA 20147</p>; addresses also in Elementor hotspot tooltips. |
| 9 | 10 | * EU site: <h1>Eemshaven, Netherlands</h1> … "36 MW total power" … "Huibertgatweg 2 9979 XZ Eemshaven, Netherlands". |
| 10 | 11 | * Language duplicates (-nl, calatorao/vimercate ES/IT originals) are excluded in YAML; "-en" is stripped from the key. |
@@ -12,9 +13,37 @@ import { areaHa, areaSqm, bodyText, cleanText, countryFromText, description, fir | ||
| 12 | 13 | const NA_ADDR = /(\d{2,6}\s[A-Za-z0-9 .'-]+?\b(?:Rd|Road|Dr|Drive|Blvd|Boulevard|St|Street|Ave|Avenue|Pkwy|Parkway|Way|Ln|Lane|Ct|Court|Hwy|Highway|Trail|Trl|Pl|Place|Cir|Circle|Loop|Pike|Route|Rte)\.?)\s*,?\s*\n?\s*([A-Za-z .'-]+?),\s*([A-Z]{2})\s+(\d{5})\b/; |
| 13 | 14 | const EU_ADDR = /([A-Z][A-Za-z .'-]+?\s\d+[a-z]?)\s*,?\s*\n?\s*((?:[A-Z]{1,2}-)?\d{4,6}(?:\s?[A-Z]{2})?)\s+([A-Za-z .'-]+?)\s*,\s*([A-Za-z ]+)\b/; |
| 14 | 15 | |
| 16 | +const escapeRe = (s: string) => s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); | |
| 17 | + | |
| 18 | +/** | |
| 19 | + * Elementor Google-Maps widget: data-pins='[{"address":"<lat>, <lng>","position":{"lat":…,"lng":…},"desc":"<p>2800 S Ashland Ave, Chicago, IL 60608</p>"}]'. | |
| 20 | + * The widget is copied between page templates (the Phoenix 4 page carried the Chicago pin), so a pin is used only when | |
| 21 | + * its own caption names the facility's postal code, or its city and state. No matching caption → no coordinates: a | |
| 22 | + * facility never inherits another site's point. | |
| 23 | + */ | |
| 24 | +export function mapPin($: CheerioAPI, pa: ParsedAddress | null): { lat: number; lng: number } | null { | |
| 25 | + if (!pa) return null; | |
| 26 | + const pins: Array<Record<string, unknown>> = []; | |
| 27 | + $("[data-pins]").each((_, el) => { | |
| 28 | + try { const v = JSON.parse($(el).attr("data-pins") ?? "[]") as unknown; if (Array.isArray(v)) pins.push(...(v as Array<Record<string, unknown>>)); } catch { /* not JSON */ } | |
| 29 | + }); | |
| 30 | + for (const pin of pins) { | |
| 31 | + const pos = pin.position as { lat?: unknown; lng?: unknown } | undefined; | |
| 32 | + const fromAddr = typeof pin.address === "string" ? pin.address.match(/^\s*(-?\d{1,2}\.\d{3,}),\s*(-?\d{1,3}\.\d{3,})\s*$/) : null; | |
| 33 | + const lat = Number(pos?.lat ?? fromAddr?.[1]), lng = Number(pos?.lng ?? fromAddr?.[2]); | |
| 34 | + if (!Number.isFinite(lat) || !Number.isFinite(lng) || (lat === 0 && lng === 0)) continue; | |
| 35 | + const caption = cleanText(String(pin.desc ?? "").replace(/<[^>]+>/g, " ")) ?? ""; | |
| 36 | + if (!caption) continue; | |
| 37 | + const samePostal = !!pa.postal && new RegExp(`\\b${escapeRe(pa.postal)}\\b`).test(caption); | |
| 38 | + const sameCity = !!pa.city && !!pa.region && new RegExp(`\\b${escapeRe(pa.city)}\\b[,\\s]+${escapeRe(pa.region)}\\b`, "i").test(caption); | |
| 39 | + if (samePostal || sameCity) return { lat, lng }; | |
| 40 | + } | |
| 41 | + return null; | |
| 42 | +} | |
| 43 | + | |
| 15 | 44 | export const qtsFacility: Parser = { |
| 16 | 45 | name: "qts_facility_v1", |
| 17 | − version: "1.0.0", | |
| 46 | + version: "1.1.0", | |
| 18 | 47 | pageTypes: ["facility_page"], |
| 19 | 48 | parse(doc: RawDocument): ExtractedRecord[] { |
| 20 | 49 | const url = doc.finalUrl; |
@@ -28,7 +57,8 @@ export const qtsFacility: Parser = { | ||
| 28 | 57 | const code = codeMatch?.[2] ?? null; |
| 29 | 58 | const slug = lastSlug(url).replace(/-(en|nl|es|it|fi)$/, ""); |
| 30 | 59 | |
| 31 | − const totalPowerMw = mwAfter(text, /Critical campus capacity/i, 80) ?? mwWithContext(text, /\s*(?:of\s+)?(?:total|totaal)\b/i) ?? mwWithContext(text, /\s*(?:aan|of)\s+(?:totaal|total)/i) ?? mwWithContext(text, /\s*(?:of\s+)?(?:critical|total)\s+(?:power|capacity|IT load)/i); | |
| 60 | + // label first ("Critical campus capacity 80 MW+") or figure first ("210 MW+ of critical campus capacity", "36 MW total power") | |
| 61 | + const totalPowerMw = mwAfter(text, /Critical campus capacity/i, 80) ?? mwWithContext(text, /\s*(?:of\s+)?(?:total|totaal)\b/i) ?? mwWithContext(text, /\s*(?:aan|of)\s+(?:totaal|total)/i) ?? mwWithContext(text, /\s*(?:of\s+)?(?:critical|total)\s+(?:campus\s+)?(?:power|capacity|IT load)/i); | |
| 32 | 62 | const acre = text.match(/(\d+(?:\.\d+)?)\s*[- ]?acre\b/i); |
| 33 | 63 | const siteAreaHa = acre ? areaHa(`${acre[1]} acres`) : areaHa(text.match(/(\d+(?:\.\d+)?)\s*hectares?\b/i)?.[0]); |
| 34 | 64 | |
@@ -52,9 +82,9 @@ export const qtsFacility: Parser = { | ||
| 52 | 82 | const proudState = proud ? usStateCode(proud[1]) : null; |
| 53 | 83 | const countryIso2 = pa?.countryIso2 ?? cl.countryIso2 ?? (proudState ? "US" : null) ?? countryFromText(title.replace(/-\s*QTS.*$/, "").replace(/-/g, " ")) ?? null; |
| 54 | 84 | const buildings = [...text.matchAll(/\b([A-Z]{2,5}\d{0,2}\s+DC\d+):\s*([^\n]+)\n?([^\n]*\d{5})?/g)].map((m) => cleanText(`${m[1]}: ${m[2]} ${m[3] ?? ""}`) ?? "").filter(Boolean); |
| 55 | − // Elementor Google-Maps widget: "address":"<lat>, <lng>" (operator-placed marker) → coordinates | |
| 56 | − const gm = doc.text.match(/address(?:"|"):(?:"|")\s*(-?\d{1,2}\.\d{3,}),\s*(-?\d{1,3}\.\d{3,})\s*(?:"|")/); | |
| 57 | − const lat = gm ? Number(gm[1]) : null, lng = gm ? Number(gm[2]) : null; | |
| 85 | + // Elementor Google-Maps pin, only when its caption is this facility's address (see mapPin) | |
| 86 | + const pin = mapPin($, pa); | |
| 87 | + const lat = pin?.lat ?? null, lng = pin?.lng ?? null; | |
| 58 | 88 | // "200,000 ft2 facility" / "22,000 m2 total" |
| 59 | 89 | const bld = text.match(/(\d{1,3}(?:,\d{3})+|\d+)\s*(?:ft2|ft²|sq\.?\s?ft|square feet)\s+(?:facility|building|data center)/i); |
| 60 | 90 | const bldM = text.match(/(\d{1,3}(?:[,.]\d{3})+|\d+)\s*(?:m2|m²|sqm|sq\.?\s?m\.?)\s+(?:total|facility|building|of)/i); |
@@ -69,14 +99,14 @@ export const qtsFacility: Parser = { | ||
| 69 | 99 | const data: Record<string, unknown> = { |
| 70 | 100 | name: `QTS ${h1.split(",")[0]!.trim()}`, code, aliases: [h1, ...(code ? [`QTS ${code}`] : [])], // keep the site number ("QTS Ashburn 3") — several campuses share a city |
| 71 | 101 | address: pa?.address ?? null, city: pa?.city ?? cl.city, regionName: pa?.region ?? cl.region ?? proudState, postalCode: pa?.postal ?? null, countryIso2, |
| 72 | − lat, lng, geoPrecision: "exact", | |
| 102 | + lat, lng, geoPrecision: pin ? "exact" : null, | |
| 73 | 103 | totalPowerMw, siteAreaHa, buildingSqm, |
| 74 | 104 | status, description: [desc, buildings.length ? `Buildings: ${buildings.join("; ")}.` : null].filter(Boolean).join(" ") || null, |
| 75 | 105 | externalIds: { qts_slug: slug }, |
| 76 | 106 | }; |
| 77 | 107 | const methods: Record<string, string> = { |
| 78 | 108 | name: "selector:h1", code: "selector:h2+regex:(CODE)", address: addrRaw && tooltips.length ? "elementor:hotspot_tooltip+parse" : "regex:address_v1", city: pa?.city ? "regex:address_v1" : "selector:h1+parse", regionName: "regex:address_v1", postalCode: "regex:address_v1", countryIso2: "regex:address_v1>country", |
| 79 | − lat: "elementor:google_maps.address", lng: "elementor:google_maps.address", | |
| 109 | + lat: "elementor:google_maps.pin(caption=address)", lng: "elementor:google_maps.pin(caption=address)", geoPrecision: "elementor:google_maps.pin(caption=address)", | |
| 80 | 110 | totalPowerMw: "regex:critical_campus_capacity_v1>mw", siteAreaHa: "regex:acre_campus>ha", buildingSqm: "regex:facility_sqft>sqm", status: "regex:status_v1", description: desc === metaDesc ? "meta:description" : "selector:p(intro)", externalIds: "url:slug", |
| 81 | 111 | }; |
| 82 | 112 | return [record("facility", `qts:${slug}`, url, data, methods, 0.85)]; |
modified
apps/worker/src/connectors/operators2/americas.ts
+23 −14
@@ -2,7 +2,8 @@ | ||
| 2 | 2 | import type { ExtractedRecord } from "@dci/connectors"; |
| 3 | 3 | import { extractGeo } from "@dci/connectors"; |
| 4 | 4 | import { countryFromText } from "@dci/core"; |
| 5 | −import { Rec, all, applyLdAddress, applyLdGeo, capacityMw, certifications, countryFromCity, countryOf, describe, facilityParser, first, ha, itMw, key, latinNumbers, ldWithAddress, mw, pue, safeCountryText, seg, splitStreetCity, sqm, statusOf, tier, titleCase, usAddress } from "./shared.js"; | |
| 5 | +import { usStateCode } from "../operators1/shared.js"; | |
| 6 | +import { Rec, all, applyLdAddress, applyLdGeo, breadcrumbLeaf, capacityMw, certifications, countryFromCity, countryOf, describe, facilityParser, first, ha, itMw, key, latinNumbers, ldWithAddress, mw, pue, safeCountryText, seg, splitStreetCity, sqm, statusOf, tier, titleCase, usAddress } from "./shared.js"; | |
| 6 | 7 | |
| 7 | 8 | /** DataBank — /data-centers/<metro>/[<campus>/]<facility>/ ; "<Name> (<CODE>)" heading followed by the street address; stats in .c-figure-stats / .c-info-table__stats. Campus pages (code before name) yield nothing. */ |
| 8 | 9 | facilityParser("databank_facility_v1", (p) => { |
@@ -28,7 +29,10 @@ facilityParser("databank_facility_v1", (p) => { | ||
| 28 | 29 | return [r.done(key("databank", code), p.url)]; |
| 29 | 30 | }); |
| 30 | 31 | |
| 31 | −/** EdgeConneX — metro pages; one JSON-LD LocalBusiness per facility (address only). Without JSON-LD, one metro-level record (city/country from H1, facility codes as aliases). */ | |
| 32 | +/** | |
| 33 | + * EdgeConneX — metro pages; one JSON-LD LocalBusiness per facility (address only). Without JSON-LD, one metro-level record | |
| 34 | + * (city / state from the BreadcrumbList leaf "Richmond, VA", else from the h1 minus its "Data Center" suffix; facility codes as aliases). | |
| 35 | + */ | |
| 32 | 36 | facilityParser("edgeconnex_facility_v1", (p) => { |
| 33 | 37 | const out: ExtractedRecord[] = []; |
| 34 | 38 | const metro = seg(p.url).pop() ?? ""; |
@@ -43,17 +47,22 @@ facilityParser("edgeconnex_facility_v1", (p) => { | ||
| 43 | 47 | } |
| 44 | 48 | if (out.length) return out; |
| 45 | 49 | const h1 = p.h1 ?? ""; |
| 46 | − const loc = h1.match(/^([A-Z][^,]+?)(?:,\s*([A-Z][^,]+))?$/); | |
| 47 | − if (!loc || !/edgeconnex/i.test(p.title ?? "")) return []; | |
| 50 | + if (!h1 || !/edgeconnex/i.test(p.title ?? "")) return []; | |
| 51 | + // h1 "Richmond Data Center" is the facility name; the breadcrumb leaf "Richmond, VA" is the location line | |
| 52 | + const crumb = breadcrumbLeaf(p.html); | |
| 53 | + const locLine = crumb && crumb.length <= 60 ? crumb : h1.replace(/\s+Data\s+Cent(?:er|re)s?$/i, ""); | |
| 54 | + const locMethod = crumb && crumb.length <= 60 ? "json-ld:BreadcrumbList" : "selector:h1"; | |
| 55 | + const loc = locLine.match(/^([A-Z][^,]+?)(?:,\s*([A-Z][^,]+))?$/); | |
| 56 | + if (!loc) return []; | |
| 48 | 57 | const r = new Rec(); |
| 49 | − r.set("name", `EdgeConneX ${h1}`, "selector:h1").set("city", loc[1], "selector:h1"); | |
| 58 | + r.set("name", `EdgeConneX ${h1}`, "selector:h1").set("city", loc[1], locMethod); | |
| 50 | 59 | const region = loc[2]?.trim() ?? null; |
| 51 | − const c = countryOf(region, h1, p.desc, p.text.slice(0, 1500)); | |
| 60 | + const c = region && usStateCode(region) ? { iso: "US", method: "lookup:us-state" } : countryOf(region, h1, p.desc, p.text.slice(0, 1500)); | |
| 52 | 61 | if (c) r.set("countryIso2", c.iso, c.method); |
| 53 | − if (region && c?.iso === "US") r.set("regionName", region, "selector:h1"); | |
| 62 | + if (region && c?.iso === "US") r.set("regionName", region, locMethod); | |
| 54 | 63 | const codes = [...new Set([...p.text.matchAll(/\b([A-Z]{3}\d{2})\b/g)].map((m) => m[1]!))]; |
| 55 | 64 | r.set("aliases", codes, "regex:facility_codes"); |
| 56 | − const single = codes.length === 1 ? p.text.match(new RegExp(`${codes[0]}:\\s*(\\d[\\d.]*\\s*MW)`)) : null; | |
| 65 | + const single = codes.length === 1 ? p.text.match(new RegExp(`${codes[0]}:\\s*(\\d[\\d.]*\\+?\\s*MW\\+?)`)) : null; | |
| 57 | 66 | const st = statusOf(p.text, 2500); if (st) r.set("status", st, "regex:status_v1"); |
| 58 | 67 | if (single) r.set(st && st !== "operational" ? "plannedPowerMw" : "itCapacityMw", mw(single[1]), "regex:code_mw_row"); |
| 59 | 68 | r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
@@ -104,7 +113,7 @@ facilityParser("serverfarm_facility_v1", (p) => { | ||
| 104 | 113 | r.set("name", `Serverfarm ${h1.replace(/\s*data center\s*$/i, "")}`, "selector:h1").set("code", code, "selector:h1").set("city", cityPart, "selector:h1"); |
| 105 | 114 | const c = countryOf(cityPart) ?? countryOf(p.text.slice(0, 3000)); if (c) r.set("countryIso2", c.iso, c.method); |
| 106 | 115 | const g = extractGeo(p.html); if (g && g.method === "attr:data-lat") r.set("lat", g.lat, g.method).set("lng", g.lng, g.method).set("geoPrecision", "exact", g.method); |
| 107 | − r.set("itCapacityMw", mw(first(p.text, /\bIT\s+(\d[\d.,]*\s*MW)/)) ?? itMw(p.text), "regex:it_mw_serverfarm").set("pue", pue(p.text), "regex:pue_v1"); | |
| 116 | + r.set("itCapacityMw", mw(first(p.text, /\bIT\s+(\d[\d.,]*\+?\s*MW\+?)/)) ?? itMw(p.text), "regex:it_mw_serverfarm").set("pue", pue(p.text), "regex:pue_v1"); | |
| 108 | 117 | const carriers = first(p.text, /Carriers\s*\n?\s*([A-Z][^\n]{5,300}?)(?:\n|$)/); if (carriers && /,/.test(carriers)) r.set("carriers", carriers.split(/,\s*/).map((s) => s.trim()).filter((s) => s.length > 1 && s.length < 40), "regex:carriers_list"); |
| 109 | 118 | r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 110 | 119 | return [r.done(key("serverfarm", code ?? cityPart), p.url)]; |
@@ -147,7 +156,7 @@ facilityParser("stream_facility_v1", (p) => { | ||
| 147 | 156 | if (status) r.set("status", /leased|operational|in service|complete|sold|stabili[sz]ed/i.test(status) ? "operational" : /construction/i.test(status) ? "under_construction" : /planned|future|available|development|land/i.test(status) ? "announced" : null, "regex:status_label"); |
| 148 | 157 | if (/hyperscale/i.test(ptype)) r.set("facilityType", "hyperscale", "regex:product_type"); else if (/wholesale|build-to-suit/i.test(ptype)) r.set("facilityType", "wholesale", "regex:product_type"); else if (/private|enterprise/i.test(ptype)) r.set("facilityType", "enterprise", "regex:product_type"); |
| 149 | 158 | r.set("itCapacityMw", itMw(p.text), "regex:it_mw_v1"); |
| 150 | − if (!r.has("itCapacityMw")) r.set("totalPowerMw", mw(first(p.text, /^\s*(\d[\d.]*\s*MW)\s*\n\s*(?:Campus|Power|Capacity)/m)), "regex:hero_mw_stat"); | |
| 159 | + if (!r.has("itCapacityMw")) r.set("totalPowerMw", mw(first(p.text, /^\s*(\d[\d.]*\+?\s*MW\+?)\s*\n\s*(?:Campus|Power|Capacity)/m)), "regex:hero_mw_stat"); | |
| 151 | 160 | r.set("buildingSqm", sqm(first(p.text, /([\d,]+\s*(?:Square Feet|SF))\s*\n?\s*(?:Data Center|Building|$)/im) ?? first(p.text, /([\d,]+\s*Square Feet)/i)), "regex:sqft_v1").set("siteAreaHa", ha(first(p.text, /([\d.]+\s*Acres)/i)), "regex:acres_v1"); |
| 152 | 161 | r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 153 | 162 | return [r.done(key("stream", seg(p.url).pop()), p.url)]; |
@@ -162,7 +171,7 @@ facilityParser("skybox_facility_v1", (p) => { | ||
| 162 | 171 | r.set("name", name.replace(/™/g, ""), "html:title").set("countryIso2", "US", "default:operator"); |
| 163 | 172 | const h1 = (p.h1 ?? "").replace(/™/g, ""); |
| 164 | 173 | const dev = /PowerCampus|Build to Suit|Development/i.test(h1); |
| 165 | − const hero = p.text.match(/^\s*([A-Za-z™ ]{2,40})\n\s*(\d[\d.,]*\s*MW)\n(?:\s*([\d,]+\s*SF)\n)?(?:\s*([\d,.]+\s*Acres)\n)?/m); | |
| 174 | + const hero = p.text.match(/^\s*([A-Za-z™ ]{2,40})\n\s*(\d[\d.,]*\+?\s*MW\+?)\n(?:\s*([\d,]+\s*SF)\n)?(?:\s*([\d,.]+\s*Acres)\n)?/m); | |
| 166 | 175 | const cityRaw = h1.replace(/^PowerCampus\s*/i, "").trim() || hero?.[1]?.trim() || ""; |
| 167 | 176 | r.set("city", cityRaw ? titleCase(cityRaw.toLowerCase()) : null, "selector:h1"); |
| 168 | 177 | if (hero) { r.set(dev ? "plannedPowerMw" : "totalPowerMw", mw(hero[2]), "regex:hero_stats").set("buildingSqm", sqm(hero[3] ?? null), "regex:hero_stats").set("siteAreaHa", ha(hero[4] ?? null), "regex:hero_stats"); } |
@@ -197,7 +206,7 @@ facilityParser("elementcritical_facility_v1", (p) => { | ||
| 197 | 206 | const r = new Rec(); |
| 198 | 207 | const name = (p.h1 ?? p.title ?? "").replace(/\s*data center services\s*$/i, "").replace(/\s*[-|].*$/, "").trim(); |
| 199 | 208 | r.set("name", `Element Critical ${name}`, "selector:h1").set("address", loc[1], "regex:location_row").set("city", loc[2], "regex:location_row").set("regionName", loc[3], "regex:location_row").set("postalCode", loc[4], "regex:location_row").set("countryIso2", "US", "regex:location_row"); |
| 200 | − r.set("buildingSqm", sqm(first(p.text, /Property Size\s*\n?\s*([\d,]+\s*Sq\.?\s*Ft\.?)/i)), "regex:property_size").set("totalPowerMw", mw(first(p.text, /Power\s*\n?\s*(\d[\d.]*\s*MW)/)) ?? capacityMw(p.text), "regex:power_row"); | |
| 209 | + r.set("buildingSqm", sqm(first(p.text, /Property Size\s*\n?\s*([\d,]+\s*Sq\.?\s*Ft\.?)/i)), "regex:property_size").set("totalPowerMw", mw(first(p.text, /Power\s*\n?\s*(\d[\d.]*\+?\s*MW\+?)/)) ?? capacityMw(p.text), "regex:power_row"); | |
| 201 | 210 | r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 202 | 211 | return [r.done(key("elementcritical", loc[1]), p.url)]; |
| 203 | 212 | }); |
@@ -209,7 +218,7 @@ facilityParser("threesixtyfive_facility_v1", (p) => { | ||
| 209 | 218 | const r = new Rec(); |
| 210 | 219 | const name = (p.title ?? "").split(/\s*[-|]\s*/)[0]?.trim() || titleCase(seg(p.url).pop() ?? ""); |
| 211 | 220 | r.set("name", `365 Data Centers ${name.replace(/\s*data center$/i, "")}`, "html:title").set("address", addr.street, "regex:us_address_v1").set("city", addr.city, "regex:us_address_v1").set("regionName", addr.region, "regex:us_address_v1").set("postalCode", addr.postal, "regex:us_address_v1").set("countryIso2", "US", "regex:us_address_v1"); |
| 212 | − r.set("totalPowerMw", mw(first(p.text, /(\d[\d.]*\s*MW) of (?:critical\s+)?power/i)) ?? capacityMw(p.text), "regex:mw_of_power").set("buildingSqm", sqm(first(p.text, /([\d,]+\s*sq\.?\s*ft)\s+(?:data center|facility)/i)), "regex:sqft_v1"); | |
| 221 | + r.set("totalPowerMw", mw(first(p.text, /(\d[\d.]*\+?\s*MW\+?) of (?:critical\s+)?power/i)) ?? capacityMw(p.text), "regex:mw_of_power").set("buildingSqm", sqm(first(p.text, /([\d,]+\s*sq\.?\s*ft)\s+(?:data center|facility)/i)), "regex:sqft_v1"); | |
| 213 | 222 | r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 214 | 223 | return [r.done(key("365dc", seg(p.url).pop()), p.url)]; |
| 215 | 224 | }); |
@@ -252,7 +261,7 @@ facilityParser("odata_facility_v1", (p) => { | ||
| 252 | 261 | const r = new Rec(); |
| 253 | 262 | r.set("name", `ODATA ${code}`, "html:title").set("code", code, "html:title"); |
| 254 | 263 | const t = latinNumbers(p.text); |
| 255 | − r.set("itCapacityMw", mw(first(t, /IT power\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(t), "regex:it_power_row").set("buildingSqm", sqm(first(t, /Floor area\s*\n?\s*([\d.,]+\s*m²)/i)), "regex:floor_area_row"); | |
| 264 | + r.set("itCapacityMw", mw(first(t, /IT power\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(t), "regex:it_power_row").set("buildingSqm", sqm(first(t, /Floor area\s*\n?\s*([\d.,]+\s*m²)/i)), "regex:floor_area_row"); | |
| 256 | 265 | const g = extractGeo(p.html); if (g && g.method === "link:google-maps") r.set("lat", g.lat, g.method).set("lng", g.lng, g.method).set("geoPrecision", "exact", g.method); |
| 257 | 266 | const loc = p.text.match(/Located in (?:the )?(?:[A-Z][\w ]+? area in )?([A-Z][^,.\n(]{2,40}?)\s*[,.(]/); |
| 258 | 267 | if (loc) r.set("city", loc[1], "regex:located_in"); |
modified
apps/worker/src/connectors/operators2/apac.ts
+8 −8
@@ -19,7 +19,7 @@ facilityParser("airtrunk_facility_v1", (p) => { | ||
| 19 | 19 | return [r.done(key("airtrunk", m[1]), p.url)]; |
| 20 | 20 | }); |
| 21 | 21 | |
| 22 | −/** NEXTDC — /data-centres/<metro>-data-centres/<code>-<city> ; IT capacity, technical space, rack capacity, Tier III. */ | |
| 22 | +/** NEXTDC — /data-centres/<metro>-data-centres/<code>-<city> ; IT capacity ("12MW", "20MW+", "10+MW"), technical space, rack capacity, Tier III. */ | |
| 23 | 23 | facilityParser("nextdc_facility_v1", (p) => { |
| 24 | 24 | const slug = seg(p.url).pop() ?? ""; |
| 25 | 25 | const m = slug.match(/^([a-z]{1,2}\d{1,2})-([a-z-]+)$/); |
@@ -27,7 +27,7 @@ facilityParser("nextdc_facility_v1", (p) => { | ||
| 27 | 27 | const code = m[1]!.toUpperCase(); |
| 28 | 28 | const r = new Rec(); |
| 29 | 29 | r.set("name", `NEXTDC ${code} ${titleCase(m[2]!)}`, "url:slug").set("code", code, "url:slug").set("city", titleCase(m[2]!), "url:slug").set("countryIso2", "AU", "default:operator"); |
| 30 | − r.set("itCapacityMw", mw(first(p.text, /(\d[\d.,]*\s*MW)\s*\n?\s*IT Capacity/i)) ?? itMw(p.text), "regex:it_capacity_stat").set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*m\s?2)\s*\n?\s*Technical Space/i)), "regex:technical_space_stat"); | |
| 30 | + r.set("itCapacityMw", mw(first(p.text, /(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Capacity/i)) ?? itMw(p.text), "regex:it_capacity_stat").set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*m\s?2)\s*\n?\s*Technical Space/i)), "regex:technical_space_stat"); | |
| 31 | 31 | const racks = first(p.text, /([\d,]+)\s*\n?\s*Rack Capacity/i); if (racks) r.set("rackCount", Number(racks.replace(/,/g, "")), "regex:rack_capacity_stat"); |
| 32 | 32 | const addr = p.text.match(/\n\s*(\d{1,4}[-–]?\d{0,4}\s+[A-Z][^\n,]{3,60}),?\s*\n?\s*([A-Z][a-zA-Z ]+?)\s+(NSW|VIC|QLD|SA|WA|ACT|NT|TAS)\s+(\d{4})\b/); |
| 33 | 33 | if (addr) r.set("address", addr[1], "regex:au_address").set("regionName", addr[3], "regex:au_address").set("postalCode", addr[4], "regex:au_address"); |
@@ -62,7 +62,7 @@ facilityParser("digitaledge_facility_v1", (p) => { | ||
| 62 | 62 | const city = first(p.text, new RegExp(`${code}\\s*\\n?\\s*([A-Z][a-zA-Z ]+?)\\s+Data Center`)) ?? first(p.title ?? "", /\b[A-Z]{3,5}\d\s+([A-Z][a-zA-Z ]+?)\s+Data Center/); |
| 63 | 63 | r.set("name", `Digital Edge ${code}${city ? ` ${city}` : ""}`, "selector:h1").set("code", code, "selector:h1").set("city", city, "regex:code_city"); |
| 64 | 64 | const c = countryOf(p.desc, city, p.text.slice(0, 2000)); if (c) r.set("countryIso2", c.iso, c.method); |
| 65 | − const full = first(p.text, /(\d[\d.,]*\s*MW) campus at full capacity/i); if (full) r.set("plannedPowerMw", mw(full), "regex:full_capacity_campus"); | |
| 65 | + const full = first(p.text, /(\d[\d.,]*\+?\s*MW\+?) campus at full capacity/i); if (full) r.set("plannedPowerMw", mw(full), "regex:full_capacity_campus"); | |
| 66 | 66 | r.set("itCapacityMw", itMw(p.text.slice(0, 3000)), "regex:it_mw_v1").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 67 | 67 | const st = statusOf(p.text.slice(0, 3000)) ?? (/expected to be complete|will be constructed/i.test(p.text.slice(0, 3000)) ? "under_construction" : null); if (st) r.set("status", st, "regex:status_v1"); |
| 68 | 68 | return [r.done(key("digitaledge", code), p.url)]; |
@@ -81,12 +81,12 @@ facilityParser("cdc_campus_v1", (p) => { | ||
| 81 | 81 | if (out.some((o) => o.key === key("cdc", city, campus))) continue; |
| 82 | 82 | const r = new Rec(); |
| 83 | 83 | r.set("name", `CDC ${campus}`, "regex:campus_block").set("campusName", `CDC ${campus}`, "regex:campus_block").set("city", city, "selector:h1").set("countryIso2", iso, "lookup:city-country"); |
| 84 | − const operating = first(block, /Current capacity \(operating\)\s*\n?\s*(\d[\d.,]*\s*MW)/i) ?? first(block, /(?:over|with) (\d[\d.,]*\s*MW) of capacity/i); | |
| 85 | − const planned = first(block, /(\d[\d.,]*\s*MW) of planned capacity/i) ?? (/under construction|will be home|early stages/i.test(block) ? first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i) : null); | |
| 84 | + const operating = first(block, /Current capacity \(operating\)\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i) ?? first(block, /(?:over|with) (\d[\d.,]*\+?\s*MW\+?) of capacity/i); | |
| 85 | + const planned = first(block, /(\d[\d.,]*\+?\s*MW\+?) of planned capacity/i) ?? (/under construction|will be home|early stages/i.test(block) ? first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i) : null); | |
| 86 | 86 | r.set("totalPowerMw", mw(operating), "regex:campus_block").set("plannedPowerMw", mw(planned), "regex:campus_block"); |
| 87 | 87 | if (!operating && planned) r.set("status", /under construction|early stages of construction/i.test(block) ? "under_construction" : "announced", "regex:campus_block"); |
| 88 | 88 | else if (/under construction/i.test(block) && !/operational/i.test(block)) r.set("status", "under_construction", "regex:campus_block"); |
| 89 | − if (!operating && !planned) { const total = first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i); r.set("totalPowerMw", mw(total), "regex:campus_block"); } | |
| 89 | + if (!operating && !planned) { const total = first(block, /Total capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i); r.set("totalPowerMw", mw(total), "regex:campus_block"); } | |
| 90 | 90 | const codes = [...new Set([...block.matchAll(/\b([A-Z]{2}\d)\b/g)].map((x) => x[1]!))]; if (codes.length) r.set("aliases", codes, "regex:campus_block"); |
| 91 | 91 | const blurb = block.split("\n").map((l) => l.trim()).find((l) => l.length > 30); |
| 92 | 92 | r.set("facilityType", "hyperscale", "default:operator").set("description", blurb ?? describe(p), "regex:campus_block").set("certifications", certifications(p.text), "regex:certifications_v1"); |
@@ -111,7 +111,7 @@ facilityParser("pdg_facility_v1", (p) => { | ||
| 111 | 111 | const r = new Rec(); |
| 112 | 112 | r.set("name", `PDG ${code} ${city}`, "regex:city_code_heading").set("code", code, "regex:city_code_heading").set("city", city, "regex:city_code_heading"); |
| 113 | 113 | const c = countryFromCity(city) ?? countryFromUrl; r.set("countryIso2", c, countryFromCity(city) ? "lookup:city-country" : "url:segment>country"); |
| 114 | − r.set("totalPowerMw", mw(first(block, /Capacity\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? capacityMw(block), "regex:capacity_row").set("buildingSqm", sqm(first(block, /Colocation Area\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:colocation_area_row").set("siteAreaHa", sqm(first(block, /land area of ([\d,.]+\s*sqm)/i)) != null ? sqm(first(block, /land area of ([\d,.]+\s*sqm)/i))! / 10_000 : null, "regex:land_area>ha"); | |
| 114 | + r.set("totalPowerMw", mw(first(block, /Capacity\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? capacityMw(block), "regex:capacity_row").set("buildingSqm", sqm(first(block, /Colocation Area\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:colocation_area_row").set("siteAreaHa", sqm(first(block, /land area of ([\d,.]+\s*sqm)/i)) != null ? sqm(first(block, /land area of ([\d,.]+\s*sqm)/i))! / 10_000 : null, "regex:land_area>ha"); | |
| 115 | 115 | const st = statusOf(block); if (st) r.set("status", st, "regex:status_v1"); |
| 116 | 116 | r.set("facilityType", "hyperscale", "default:operator").set("certifications", certifications(block), "regex:certifications_v1").set("description", first(block, /ABOUT THE FACILITY[^\n]*\n\s*([^\n]{40,600})/i) ?? describe(p), "regex:about_facility"); |
| 117 | 117 | out.push(r.done(key("pdg", code), p.url, 0.8)); |
@@ -128,7 +128,7 @@ facilityParser("macquarie_facility_v1", (p) => { | ||
| 128 | 128 | const city = titleCase(parts[1]!); |
| 129 | 129 | r.set("name", `Macquarie ${code} ${city}`, "url:slug").set("code", code, "url:slug").set("city", city, "url:slug").set("countryIso2", "AU", "default:operator"); |
| 130 | 130 | const campus = first(p.title ?? "", /-\s*([A-Z][A-Za-z ]+Campus)/); if (campus) r.set("campusName", `Macquarie ${campus}`, "html:title"); |
| 131 | − const it = first(p.text, /(\d[\d.,]*\s*MW)\s*\n?\s*IT Load/i); if (it) r.set("itCapacityMw", mw(it), "regex:it_load_stat"); | |
| 131 | + const it = first(p.text, /(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Load/i); if (it) r.set("itCapacityMw", mw(it), "regex:it_load_stat"); | |
| 132 | 132 | r.set("buildingSqm", sqm(first(p.text, /([\d,.]+\s*(?:m²|sqm|m2))\s*\n?\s*Technical space/i)), "regex:technical_space_stat").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 133 | 133 | if (/government|sovereign/i.test(`${p.title ?? ""} ${p.desc ?? ""}`)) r.set("facilityType", "colocation", "default:operator"); |
| 134 | 134 | const st = statusOf(p.text.slice(0, 3000)); if (st) r.set("status", st, "regex:status_v1"); |
modified
apps/worker/src/connectors/operators2/emea.ts
+5 −5
@@ -95,7 +95,7 @@ facilityParser("greenmountain_facility_v1", (p) => { | ||
| 95 | 95 | r.set("name", `Green Mountain ${code}`, "selector:h1").set("code", code, "selector:h1"); |
| 96 | 96 | const place = code.split("-")[1]!; if (!/^(east|west|north|south|central)$/i.test(place)) r.set("city", place, "selector:h1"); |
| 97 | 97 | const c = countryOf(p.desc, p.text.slice(0, 2500)); r.set("countryIso2", c?.iso ?? "NO", c ? c.method : "default:operator"); |
| 98 | − r.set("itCapacityMw", mw(first(p.text, /Maximum IT capacity:\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(p.text), "regex:it_mw_v1").set("buildingSqm", sqm(first(p.text, /([\d ]{3,9}\s*m2)\s*\n?\s*(?:Total Campus Footprint|\(\d)/i) ?? first(p.text, /approx ([\d ]{3,9}\s*m2)/i)), "regex:campus_footprint"); | |
| 98 | + r.set("itCapacityMw", mw(first(p.text, /Maximum IT capacity:\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(p.text), "regex:it_mw_v1").set("buildingSqm", sqm(first(p.text, /([\d ]{3,9}\s*m2)\s*\n?\s*(?:Total Campus Footprint|\(\d)/i) ?? first(p.text, /approx ([\d ]{3,9}\s*m2)/i)), "regex:campus_footprint"); | |
| 99 | 99 | r.set("pue", pue(p.text), "regex:pue_v1").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 100 | 100 | if (/100 ?% renewable|hydro ?power/i.test(p.text)) r.set("renewableClaim", first(p.text, /([^.\n]*100 ?% renewable[^.\n]*)/i) ?? "Renewable (hydro) powered", "regex:renewable_claim"); |
| 101 | 101 | const st = statusOf(`${p.desc ?? ""}\n${p.text}`, 2000); if (st) r.set("status", st, "regex:status_v1"); |
@@ -171,7 +171,7 @@ facilityParser("kaodata_facility_v1", (p) => { | ||
| 171 | 171 | const base = (r: Rec) => { if (addr) r.set("address", addr[1], "regex:address_row").set("city", addr[2], "regex:address_row").set("postalCode", addr[3], "regex:address_row"); else r.set("city", campus, "url:segment"); r.set("countryIso2", "GB", "default:operator"); if (/100 ?% renewable/i.test(p.text)) r.set("renewableClaim", "100% renewable energy", "regex:renewable_claim"); }; |
| 172 | 172 | const c = new Rec(); |
| 173 | 173 | c.set("name", `Kao Data ${campus}`, "url:segment").set("campusName", `Kao Data ${campus}`, "url:segment"); base(c); |
| 174 | − c.set("plannedPowerMw", mw(first(p.text, /(?:ITE?|IT-?)\s*load of (\d[\d.]*\s*MW)/i)), "regex:ite_load_full_build").set("siteAreaHa", ha(first(p.text, /(\d+\s*acres?)/i)), "regex:acres_v1").set("buildingSqm", sqm(first(p.text, /Technical Space:\s*([\d,]+\s*m2)/i)), "regex:technical_space"); | |
| 174 | + c.set("plannedPowerMw", mw(first(p.text, /(?:ITE?|IT-?)\s*load of (\d[\d.]*\+?\s*MW\+?)/i)), "regex:ite_load_full_build").set("siteAreaHa", ha(first(p.text, /(\d+\s*acres?)/i)), "regex:acres_v1").set("buildingSqm", sqm(first(p.text, /Technical Space:\s*([\d,]+\s*m2)/i)), "regex:technical_space"); | |
| 175 | 175 | c.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 176 | 176 | out.push(c.done(key("kaodata", campus), p.url, 0.8)); |
| 177 | 177 | for (const m of p.text.matchAll(/\b(K[A-Z]{2,4}-\d{2})\s*\((\d[\d.]*)\s*MW\s+([A-Za-z ]+?)\)/g)) { |
@@ -191,7 +191,7 @@ facilityParser("pulsant_facility_v1", (p) => { | ||
| 191 | 191 | if (!m) return []; |
| 192 | 192 | const r = new Rec(); |
| 193 | 193 | r.set("name", `Pulsant ${h1}`, "selector:h1").set("code", m[2], "selector:h1").set("city", m[1], "selector:h1").set("countryIso2", "GB", "default:operator"); |
| 194 | − r.set("itCapacityMw", mw(first(p.text, /Total IT power\s*\n?\s*(\d[\d.,]*\s*MW)/i)) ?? itMw(p.text), "regex:total_it_power_row").set("buildingSqm", sqm(first(p.text, /Total building size\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:building_size_row"); | |
| 194 | + r.set("itCapacityMw", mw(first(p.text, /Total IT power\s*\n?\s*(\d[\d.,]*\+?\s*MW\+?)/i)) ?? itMw(p.text), "regex:total_it_power_row").set("buildingSqm", sqm(first(p.text, /Total building size\s*\n?\s*([\d,.]+\s*m²)/i)), "regex:building_size_row"); | |
| 195 | 195 | const ren = first(p.text, /Renewable energy procurement\s*\n?\s*(\d+\s*%)/i); if (ren) r.set("renewableClaim", `${ren} renewable energy procurement`, "regex:renewable_row"); |
| 196 | 196 | const cool = first(p.text, /Cooling:\s*([^\n]{2,30})/i); if (cool) r.set("coolingType", `Cooling ${cool}`, "regex:cooling_row"); |
| 197 | 197 | r.set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
@@ -218,7 +218,7 @@ facilityParser("virtus_facility_v1", (p) => { | ||
| 218 | 218 | r.set("countryIso2", c, "url:segment>country"); |
| 219 | 219 | const idx = code ? p.text.search(new RegExp(`\\b${code}\\b`)) : -1; |
| 220 | 220 | const seq = idx >= 0 ? p.text.slice(idx, idx + 3000) : p.text.slice(0, 3000); |
| 221 | − if (code) r.set("itCapacityMw", mw(first(seq, /(\d[\d.,]*\s*MW) of IT load/i)) ?? itMw(seq), "regex:mw_of_it_load").set("buildingSqm", sqm(first(seq, /([\d,.]+\s*m2) (?:of )?net technical/i)), "regex:net_technical_sqm"); | |
| 221 | + if (code) r.set("itCapacityMw", mw(first(seq, /(\d[\d.,]*\+?\s*MW\+?) of IT load/i)) ?? itMw(seq), "regex:mw_of_it_load").set("buildingSqm", sqm(first(seq, /([\d,.]+\s*m2) (?:of )?net technical/i)), "regex:net_technical_sqm"); | |
| 222 | 222 | const st = statusOf(seq, 2500); if (st) r.set("status", st, "regex:status_v1"); |
| 223 | 223 | r.set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 224 | 224 | return [r.done(key("virtus", code ?? slug), p.url, code ? 0.85 : 0.7)]; |
@@ -263,7 +263,7 @@ facilityParser("adc_facility_v1", (p) => { | ||
| 263 | 263 | const city = m[2]!.trim(); |
| 264 | 264 | r.set("name", `Africa Data Centres ${m[1]} ${city}`, "regex:code_city").set("code", m[1], "regex:code_city").set("city", city, "regex:code_city"); |
| 265 | 265 | const c = countryOf(p.text.slice(0, 4000), city); if (c) r.set("countryIso2", c.iso, c.method); |
| 266 | − r.set("itCapacityMw", mw(first(p.text, /(\d[\d.]*\s*MW) of (?:IT\s+)?power/i)) ?? itMw(p.text), "regex:mw_of_power").set("plannedPowerMw", mw(first(p.text, /IT load of (\d[\d.]*\s*MW) upon completion/i) ?? first(p.text, /(?:developed|expanded) to [\d,]+ square met(?:re|er)s and (\d[\d.]*\s*MW)/i)), "regex:upon_completion"); | |
| 266 | + r.set("itCapacityMw", mw(first(p.text, /(\d[\d.]*\+?\s*MW\+?) of (?:IT\s+)?power/i)) ?? itMw(p.text), "regex:mw_of_power").set("plannedPowerMw", mw(first(p.text, /IT load of (\d[\d.]*\+?\s*MW\+?) upon completion/i) ?? first(p.text, /(?:developed|expanded) to [\d,]+ square met(?:re|er)s and (\d[\d.]*\+?\s*MW\+?)/i)), "regex:upon_completion"); | |
| 267 | 267 | r.set("buildingSqm", sqm(first(p.text, /([\d,]+\s*(?:m2|m²|square met(?:re|er)s)) of (?:IT space|secured rack space)/i)), "regex:it_space_sqm").set("tier", tier(p.text), "regex:tier_v1").set("certifications", certifications(p.text), "regex:certifications_v1").set("description", describe(p), "meta:description"); |
| 268 | 268 | return [r.done(key("adc", m[1]), p.url)]; |
| 269 | 269 | }); |
modified
apps/worker/src/connectors/operators2/shared.ts
+29 −8
@@ -59,20 +59,23 @@ export function mw(s: string | null | undefined): number | null { | ||
| 59 | 59 | return v; |
| 60 | 60 | } |
| 61 | 61 | |
| 62 | −/** MW explicitly labelled as IT / critical load for the facility described by `text`. */ | |
| 62 | +/** | |
| 63 | + * MW explicitly labelled as IT / critical load for the facility described by `text`. | |
| 64 | + * "At least" figures keep their raw shape — "10+MW", "20MW+", "150+ MW" — and `mw()` / `parseMw` read the stated figure. | |
| 65 | + */ | |
| 63 | 66 | export function itMw(text: string): number | null { |
| 64 | 67 | const pats = [ |
| 65 | − /(\d[\d.,]*\+?\s*MWs?)\s*(?:of\s+)?(?:total\s+)?(?:critical\s+)?(?:IT|critical)\s+(?:capacity|load|power)\b/i, | |
| 66 | − /(?:IT|critical)\s+(?:load|capacity|power)(?:\s+capacity)?(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW)/i, | |
| 67 | − /(\d[\d.,]*\+?\s*MWs?)\s+of\s+(?:critical\s+)?(?:IT|critical)\s+(?:load|power|capacity)/i, | |
| 68 | − /(\d[\d.,]*\+?\s*MWs?)\s+(?:critical|IT)\b/i, | |
| 68 | + /(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:critical\s+)?(?:IT|critical)\s+(?:capacity|load|power)\b/i, | |
| 69 | + /(?:IT|critical)\s+(?:load|capacity|power)(?:\s+capacity)?(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)/i, | |
| 70 | + /(\d[\d.,]*\+?\s*MWs?\+?)\s+of\s+(?:critical\s+)?(?:IT|critical)\s+(?:load|power|capacity)/i, | |
| 71 | + /(\d[\d.,]*\+?\s*MWs?\+?)\s+(?:critical|IT)\b/i, | |
| 69 | 72 | ]; |
| 70 | 73 | for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } } |
| 71 | 74 | return null; |
| 72 | 75 | } |
| 73 | 76 | /** MW labelled as (total) capacity / power without an IT qualifier. */ |
| 74 | 77 | export function capacityMw(text: string): number | null { |
| 75 | − const pats = [/(\d[\d.,]*\+?\s*MWs?)\s*(?:of\s+)?(?:total\s+)?(?:capacity|power)\b/i, /(?:total\s+)?(?:capacity|power)(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW)\b/i]; | |
| 78 | + const pats = [/(\d[\d.,]*\+?\s*MWs?\+?)\s*(?:of\s+)?(?:total\s+)?(?:capacity|power)\b/i, /(?:total\s+)?(?:capacity|power)(?:\s+of)?\s*[:\-–]?\s*(?:up to\s+)?(\d[\d.,]*\+?\s*MW\+?)(?![\w])/i]; | |
| 76 | 79 | for (const re of pats) { const m = text.match(re); if (m?.[1]) { const v = mw(m[1]); if (v) return v; } } |
| 77 | 80 | return null; |
| 78 | 81 | } |
@@ -176,6 +179,18 @@ export function key(op: string, ...parts: Array<string | null | undefined>): str | ||
| 176 | 179 | export function ldWithAddress(html: string, type?: string): Array<Record<string, unknown> & { address: Record<string, unknown>; geo?: Record<string, unknown> }> { |
| 177 | 180 | return jsonLd(html, type).filter((o) => o.address && typeof o.address === "object") as Array<Record<string, unknown> & { address: Record<string, unknown>; geo?: Record<string, unknown> }>; |
| 178 | 181 | } |
| 182 | +/** Leaf of the JSON-LD BreadcrumbList ("Richmond, VA") — the page's own location line, more reliable than a facility-name h1. */ | |
| 183 | +export function breadcrumbLeaf(html: string): string | null { | |
| 184 | + for (const o of jsonLd(html, "BreadcrumbList")) { | |
| 185 | + const items = o.itemListElement; | |
| 186 | + if (!Array.isArray(items) || !items.length) continue; | |
| 187 | + const last = items[items.length - 1] as Record<string, unknown> | undefined; | |
| 188 | + const item = last?.item as Record<string, unknown> | string | undefined; | |
| 189 | + const name = cleanText(String(last?.name ?? (typeof item === "object" ? item?.name : "") ?? "")); | |
| 190 | + if (name && !/^home$/i.test(name)) return name; | |
| 191 | + } | |
| 192 | + return null; | |
| 193 | +} | |
| 179 | 194 | export function ldCountry(a: Record<string, unknown>): string | null { |
| 180 | 195 | const c = a.addressCountry; |
| 181 | 196 | if (!c) return null; |
@@ -192,9 +207,15 @@ export function applyLdGeo(r: Rec, o: Record<string, unknown>, precision = "exac | ||
| 192 | 207 | if (Number.isFinite(lat) && Number.isFinite(lng) && (lat !== 0 || lng !== 0)) r.set("lat", lat, method).set("lng", lng, method).set("geoPrecision", precision, method); |
| 193 | 208 | } |
| 194 | 209 | |
| 210 | +/** | |
| 211 | + * Version shared by every operators2 parser — part of the effective extractor version (docs/CONNECTORS.md § 6). | |
| 212 | + * v2 (2026-09-12): "MW+" / "+MW" stat figures are read (itMw, capacityMw, stat regexes); EdgeConneX city from the breadcrumb. | |
| 213 | + */ | |
| 214 | +export const OPERATORS2_PARSER_VERSION = "v2"; | |
| 215 | + | |
| 195 | 216 | /** Register a facility parser with the standard shape. */ |
| 196 | −export function facilityParser(name: string, parse: (p: Page, doc: RawDocument) => ExtractedRecord[]): Parser { | |
| 197 | − const parser: Parser = { name, version: "v1", pageTypes: ["facility_page"], parse: (doc) => (isHtml(doc) ? parse(page(doc), doc) : []) }; | |
| 217 | +export function facilityParser(name: string, parse: (p: Page, doc: RawDocument) => ExtractedRecord[], version = OPERATORS2_PARSER_VERSION): Parser { | |
| 218 | + const parser: Parser = { name, version, pageTypes: ["facility_page"], parse: (doc) => (isHtml(doc) ? parse(page(doc), doc) : []) }; | |
| 198 | 219 | registerParser(parser); |
| 199 | 220 | return parser; |
| 200 | 221 | } |
modified
docs/CONNECTORS.md
+6 −6
@@ -82,8 +82,8 @@ export interface Parser { | ||
| 82 | 82 | A parser is a pure function of the archived document: **no network, no geocoding, no guessed figures** — it surfaces what the page publishes, with a per-field `methods` map for provenance (`{ itCapacityMw: "regex:it_mw_v1", lat: "json-ld:GeoCoordinates" }`). Parsers live in `apps/worker/src/connectors/<group>/<file>.ts` and are registered by the group's `index.ts`: |
| 83 | 83 | |
| 84 | 84 | - `operators1/` (Equinix, Digital Realty, NTT, CyrusOne, QTS, Vantage, STACK, CoreSite, Switch): each file exports a `Parser` object; `operators1/index.ts` `register()` calls `registerParser` for each. Helpers in `operators1/shared.ts` (`parseNaAddress`, `parseEuAddress`, `parseCityLine`, `mwAfter`, `mwWithContext`, `areaSqm`, `certificationsFrom`, `record(kind, key, url, data, methods, certainty)` which drops empty fields and sets `pageType: facility_page`). |
| 85 | −- `operators2/` (DataBank, EdgeConneX, TierPoint, NEXTDC, AirTrunk, atNorth, CloudHQ, …, grouped in `americas.ts` / `emea.ts` / `apac.ts`): parsers are declared with **`facilityParser(name, (page, doc) => ExtractedRecord[])`** from `operators2/shared.ts`, which registers on import and hands you a `Page` (`$`, `html`, visible `text` with nav/header/footer removed, `title`, `h1`, `desc`, `url`). Build records with the **`Rec`** builder: `new Rec().set(field, value, method)` ignores null/empty/NaN and records the method; `.done(key, url, certainty = 0.85, kind = "facility")`. Other helpers: `itMw`, `capacityMw`, `mw` (handles `1.125 MW` decimals and `MWs`), `sqm`, `ha`, `pue`, `tier`, `certifications`, `usAddress`, `ldWithAddress` / `applyLdAddress` / `applyLdGeo` (JSON-LD), `countryOf` / `countryFromCity` (deterministic metro → ISO table, not geocoding), `statusOf`, `key(op, ...parts)`. `operators2/index.ts` `register()` only checks that every name in `AMERICAS_PARSERS` / `EMEA_PARSERS` / `APAC_PARSERS` is registered. | |
| 86 | −- `news/` (`news_article_v1`, `news_planning_pdf_v1`, `news_edgar_fts_v1` + the `news_edgar_fts` implementation): article → `news_event` (+ `project` when `qualifiesAsProject`) via `articleContent(doc)` → `extractAnnouncement(title, text)` (`extract-project.ts`, unit-tested in `extract-project.test.ts`). Params: `keepText` (public-sector sources only), `minProjectMw` (5), `minInvestmentUsd` (50 000 000), `minAcres` (100), `lenient`. | |
| 85 | +- `operators2/` (DataBank, EdgeConneX, TierPoint, NEXTDC, AirTrunk, atNorth, CloudHQ, …, grouped in `americas.ts` / `emea.ts` / `apac.ts`): parsers are declared with **`facilityParser(name, (page, doc) => ExtractedRecord[], version = OPERATORS2_PARSER_VERSION)`** from `operators2/shared.ts`, which registers on import and hands you a `Page` (`$`, `html`, visible `text` with nav/header/footer removed, `title`, `h1`, `desc`, `url`). Build records with the **`Rec`** builder: `new Rec().set(field, value, method)` ignores null/empty/NaN and records the method; `.done(key, url, certainty = 0.85, kind = "facility")`. Other helpers: `itMw`, `capacityMw`, `mw` (handles `1.125 MW` decimals, `MWs`, and "at least" figures `10+MW` / `20MW+` — pass the raw string, `parseMw` reads the stated figure), `sqm`, `ha`, `pue`, `tier`, `certifications`, `usAddress`, `ldWithAddress` / `applyLdAddress` / `applyLdGeo` / `breadcrumbLeaf` (JSON-LD), `countryOf` / `countryFromCity` (deterministic metro → ISO table, not geocoding), `statusOf`, `key(op, ...parts)`. `operators2/index.ts` `register()` only checks that every name in `AMERICAS_PARSERS` / `EMEA_PARSERS` / `APAC_PARSERS` is registered. A stat regex that reads a figure before its label must accept the `+` suffix: `/(\d[\d.,]*\+?\s*MW\+?)\s*\n?\s*IT Capacity/`. | |
| 86 | +- `news/` (`news_article_v1`, `news_planning_pdf_v1`, `news_edgar_fts_v1` + the `news_edgar_fts` implementation): article → `news_event` (+ `project` when `qualifiesAsProject`) via `articleContent(doc)` → `extractAnnouncement(title, text)` (`extract-project.ts`, unit-tested in `extract-project.test.ts`). `articleContent` uses `mainText` (`packages/connectors/src/extract.ts`): the largest `<article>` / `<main>` block after removing nav / header / footer / aside / related / recommended / trending / popular / sidebar / comments blocks, cut at the first trailing teaser heading ("More in …", "Related articles", "Tags"). Operators are looked up in the title plus the first `OPERATOR_SCOPE_CHARS` (4 000) characters only; money figures in a company-background sentence (`MONEY_PORTFOLIO_RE`: "investment volume", "revenue", "assets under management"…) are never the project's investment. Params: `keepText` (public-sector sources only), `minProjectMw` (5), `minInvestmentUsd` (50 000 000), `minAcres` (100), `lenient`. | |
| 87 | 87 | - `cloud/` (AWS, Azure, GCP, Oracle, IBM, Alibaba, Tencent, Meta regions) and `datasets/` (PeeringDB, Wikidata, World Bank, OSM Overpass): mostly `implementation`s. |
| 88 | 88 | |
| 89 | 89 | `apps/worker/src/connectors/index.ts` `registerAllConnectors()` imports every `<group>/index.ts` found on disk and calls its `register()`; a group that throws is reported and the others still load. Make `register()` idempotent (guard with `listParsers()` or a module flag) — the worker, the CLI, `try-connector` and the fixture tests may all call it. Add a new group by creating `apps/worker/src/connectors/<group>/index.ts` with `export function register(): void`. |
@@ -151,8 +151,8 @@ Rules for the body: keep it byte-for-byte (no reformatting, no CRLF normalisatio | ||
| 151 | 151 | "itCapacityMw": 0.675, "totalPowerMw": null, // null = absent or null; numbers are compared exactly |
| 152 | 152 | "mentions.countriesIso2": ["US"], // dotted paths reach nested fields |
| 153 | 153 | "_todo": { // CORRECT values the parser does not produce yet → it.fails |
| 154 | − "city": "Richmond", | |
| 155 | − "_reason": "h1 fallback copies 'Richmond Data Center' into city" | |
| 154 | + "totalPowerMw": 36, | |
| 155 | + "_reason": "the figure precedes its label ('36 MW total power'); the stat regex only reads label-first" | |
| 156 | 156 | } |
| 157 | 157 | } |
| 158 | 158 | ], |
@@ -176,7 +176,7 @@ pnpm --filter @dci/worker test # everythi | ||
| 176 | 176 | pnpm --filter @dci/worker exec vitest run src/connectors/fixtures.test.ts -t databank # one connector |
| 177 | 177 | ``` |
| 178 | 178 | |
| 179 | −Golden coverage on 2026-09-12: 21 fixtures over 18 connectors (equinix, digitalrealty, stack, databank ×2, ntt, edgeconnex ×2, cyrusone, qts ×2, vantage, coresite, airtrunk, nextdc ×2, atnorth, cloudhq, datacenterfrontier, datacenterdynamics, loudoun-county). No planning PDF fixture yet: the archive holds no document with a PDF content type. | |
| 179 | +Golden coverage on 2026-09-12: 21 fixtures over 18 connectors (equinix, digitalrealty, stack, databank ×2, ntt, edgeconnex ×2, cyrusone, qts ×2, vantage, coresite, airtrunk, nextdc ×2, atnorth, cloudhq, datacenterfrontier, datacenterdynamics, loudoun-county), no open `_todo` (the nine gaps recorded on 2026-09-12 — EdgeConneX city from the h1, NEXTDC `20MW+`, QTS figure-first capacity and template-copied map pin, DCD teaser operator and portfolio money — were fixed the same day and promoted). No planning PDF fixture yet: the archive holds no document with a PDF content type. | |
| 180 | 180 | |
| 181 | 181 | ## 4. Live dry-run |
| 182 | 182 | |
@@ -200,7 +200,7 @@ Two different things share the word: | ||
| 200 | 200 | The **effective extractor version** stored on every document (`documents.extractor_version`) is computed by `effectiveExtractorVersion(cfg, connector)` in `apps/worker/src/configs.ts`: |
| 201 | 201 | |
| 202 | 202 | ``` |
| 203 | −<cfg.parserVersion>+<sha256("news_article_v1@news_v2|databank_facility_v1@v1|impl:peeringdb@v1")[0:10]> | |
| 203 | +<cfg.parserVersion>+<sha256("news_article_v1@news_v3|databank_facility_v1@v2|impl:peeringdb@v1")[0:10]> | |
| 204 | 204 | ``` |
| 205 | 205 | |
| 206 | 206 | i.e. the YAML `parserVersion` plus a hash of every referenced `Parser.name@Parser.version` (and `impl:<name>@<connector.parserVersion>`). Consequences: |
modified
packages/connectors/src/extract.test.ts
+24 −1
@@ -1,6 +1,6 @@ | ||
| 1 | 1 | import { describe, expect, it } from "vitest"; |
| 2 | 2 | import { load } from "cheerio"; |
| 3 | −import { applyTransform, embeddedJson, evalRule, extractAddress, extractGeo, jsonLd, jsonPath, pdfText, publishedDate, walkJson } from "./extract.js"; | |
| 3 | +import { applyTransform, cutTeasers, embeddedJson, evalRule, extractAddress, extractGeo, jsonLd, jsonPath, mainText, pdfText, publishedDate, walkJson } from "./extract.js"; | |
| 4 | 4 | |
| 5 | 5 | const HTML = `<html><head><title>Page</title> |
| 6 | 6 | <meta property="og:title" content="OG Title"><meta name="description" content="Desc"> |
@@ -90,6 +90,29 @@ describe("geo / address / date", () => { | ||
| 90 | 90 | }); |
| 91 | 91 | }); |
| 92 | 92 | |
| 93 | +describe("mainText", () => { | |
| 94 | + const story = `<p>${"The developer secured planning permission for a 12MW data center in Bucharest. ".repeat(6)}</p>`; | |
| 95 | + it("keeps the largest <article>, drops related / recommended / trending / sidebar / aside / footer blocks and cuts trailing teasers", () => { | |
| 96 | + const html = `<html><body><nav>Home</nav><main> | |
| 97 | +<article class="card"><a>Microsoft expands plans for La Porte campus</a></article> | |
| 98 | +<article itemprop="mainEntity"><h1>S+B Gruppe to build data center in Bucharest</h1>${story} | |
| 99 | +<div class="block-auto_featured_content"><h2>More in Construction & Site Selection</h2><a>Ignis to build DayOne's 300MW data center</a></div></article> | |
| 100 | +<div class="related-articles"><a>Google onboarding case study</a></div><div id="sidebar-popular"><a>Equinix opens LD14</a></div> | |
| 101 | +<section class="recommended-for-you"><a>Vantage raises $2bn</a></section><div class="trending-now"><a>Digital Realty results</a></div></main> | |
| 102 | +<aside><a>AWS buys site</a></aside><footer>Amazon Web Services</footer></body></html>`; | |
| 103 | + const t = mainText(html); | |
| 104 | + expect(t).toContain("S+B Gruppe to build data center in Bucharest"); | |
| 105 | + expect(t).toContain("planning permission"); | |
| 106 | + for (const leak of ["Microsoft", "Ignis", "More in", "Google", "Equinix", "Vantage", "Digital Realty", "AWS", "Amazon"]) expect(t, leak).not.toContain(leak); | |
| 107 | + }); | |
| 108 | + it("cutTeasers only cuts after the story has started and only on a heading line", () => { | |
| 109 | + expect(cutTeasers("Related\nshort lead")).toBe("Related\nshort lead"); | |
| 110 | + const body = "x".repeat(400); | |
| 111 | + expect(cutTeasers(`${body}\n Tags \n Bucharest\n Comments`)).toBe(body); | |
| 112 | + expect(cutTeasers(`${body}\nMore in the article follows here with a long sentence about the project itself.`)).toContain("More in the article"); | |
| 113 | + }); | |
| 114 | +}); | |
| 115 | + | |
| 93 | 116 | describe("pdfText limits", () => { |
| 94 | 117 | it("rejects oversized PDFs before parsing", async () => { |
| 95 | 118 | await expect(pdfText(Buffer.alloc(16 * 1024 * 1024), 60, { maxBytes: 15 * 1024 * 1024 })).rejects.toThrow(/too large/); |
modified
packages/connectors/src/extract.ts
+31 −5
@@ -126,16 +126,42 @@ export function metaDescription(html: string): string | null { | ||
| 126 | 126 | return cleanText($('meta[name="description"]').attr("content") ?? $('meta[property="og:description"]').attr("content") ?? null); |
| 127 | 127 | } |
| 128 | 128 | |
| 129 | −/** Main text of an article-like page (heuristic: largest <article>/<main>/content block). */ | |
| 129 | +/** Blocks that are never the story: chrome, related / recommended / trending teasers, sidebars, comments. */ | |
| 130 | +export const NON_CONTENT_SELECTOR = "script,style,noscript,nav,header,footer,aside,form,iframe,svg,[role=navigation],[role=complementary],[class*=related],[class*=recommend],[class*=trending],[class*=popular],[class*=sidebar],[class*=read-more],[class*=readmore],[class*=more-stories],[class*=also-like],[class*=comments],[id*=related],[id*=comments]"; | |
| 131 | +/** Heading that opens a trailing teaser section inside the story element itself ("More in Construction & Site Selection", "Related articles", "Tags"). */ | |
| 132 | +export const TEASER_HEADING_RE = /^\s*(?:More (?:in|from|on|like this)\b.*|Related(?: (?:articles?|stories|news|content|posts?|coverage|reading|links?))?|Recommended(?: for you| reading| stories| articles)?|You (?:may|might) also like|Most (?:read|popular|viewed)|Trending(?: now| stories)?|Popular (?:now|posts|stories|articles)|Latest (?:news|stories|posts|articles)|Read (?:more|next|also)|See also|Further reading|Editor'?s picks|Sponsored(?: content)?|Tags|Comments|Leave a (?:comment|reply))\s*:?\s*$/i; | |
| 133 | +/** Minimum story length before a teaser heading is allowed to cut the text (a short lead titled "Related" is not a teaser section). */ | |
| 134 | +const TEASER_CUT_MIN_CHARS = 300; | |
| 135 | + | |
| 136 | +/** Drop everything from the first teaser heading onwards (teasers are the last thing in a story element). */ | |
| 137 | +export function cutTeasers(text: string): string { | |
| 138 | + const lines = text.split("\n"); | |
| 139 | + let consumed = 0; | |
| 140 | + for (let i = 0; i < lines.length; i++) { | |
| 141 | + const line = lines[i]!; | |
| 142 | + if (consumed >= TEASER_CUT_MIN_CHARS && line.trim().length <= 60 && TEASER_HEADING_RE.test(line)) return lines.slice(0, i).join("\n").trimEnd(); | |
| 143 | + consumed += line.length; | |
| 144 | + } | |
| 145 | + return text; | |
| 146 | +} | |
| 147 | + | |
| 148 | +/** | |
| 149 | + * Main text of an article-like page: the largest <article> / <main> / content block, after removing navigation, footer, | |
| 150 | + * aside, related / recommended / trending / popular / sidebar blocks, and cut at the first trailing teaser heading — so a | |
| 151 | + * "More in …" list of other headlines never leaks into the story (and never names its operator). | |
| 152 | + */ | |
| 130 | 153 | export function mainText(html: string): string { |
| 131 | 154 | const $ = load(html); |
| 132 | − $("script,style,noscript,nav,header,footer,aside,form,iframe,svg").remove(); | |
| 155 | + $(NON_CONTENT_SELECTOR).remove(); | |
| 133 | 156 | const candidates = ["article", "main", '[role="main"]', ".article-body", ".press-release", ".entry-content", ".post-content", ".content", "#content", "body"]; |
| 134 | 157 | for (const c of candidates) { |
| 135 | − const el = $(c).first(); | |
| 136 | − if (el.length) { const t = htmlToText(el.html() ?? ""); if (t.length > 300) return t; } | |
| 158 | + // several matches (the story plus teaser <article class="card"> cards): the largest one is the story | |
| 159 | + const texts = $(c).toArray().map((el) => htmlToText($(el).html() ?? "")); | |
| 160 | + if (!texts.length) continue; | |
| 161 | + const t = cutTeasers(texts.sort((a, b) => b.length - a.length)[0]!); | |
| 162 | + if (t.length > 300) return t; | |
| 137 | 163 | } |
| 138 | − return htmlToText($("body").html() ?? html); | |
| 164 | + return cutTeasers(htmlToText($("body").html() ?? html)); | |
| 139 | 165 | } |
| 140 | 166 | |
| 141 | 167 | /** Publication date from meta tags / JSON-LD / <time>. */ |
| 142 | 168 | |