"use client"; import Link from "next/link"; import { useMemo, useState, type ReactNode } from "react"; import { Badge, Button, ScopeBadge, SemanticsLabel, SkeletonRows, Table, TBody, Td, Th, THead, Tr } from "@/components/ui"; import { cn } from "@/components/ui/cn"; import { DASH, fmtInt, fmtMw, fmtUsd, fmtValue } from "@/lib/format"; import { entityHref } from "@/lib/routes"; import type { TraceResult } from "@/lib/admin-api"; import { ErrorNote, ExtLink, fmtBytes, JsonView, KV, LevelBadge, Mono, RefreshButton, Stat } from "./primitives"; import { AdminPageHeader } from "./shell"; import { useAdminQuery } from "./use-admin-query"; /* ------------------------------------------------------------------------------------------ EXTRACTION DEBUGGER (§94–95): the pipeline for ONE document, stage by stage, without publishing. SOURCE DOCUMENT → RAW FETCH → PARSED TEXT → STRUCTURED DATA → EXTRACTED ENTITIES → EXTRACTED CLAIMS → NORMALIZED VALUES → MATCH CANDIDATES → RECONCILIATION → RESULTING DATABASE CHANGES. Data comes from GET /api/admin/documents/:id/trace (worker dry run); nothing is invented — when the worker is unreachable the ladder renders its error state. ------------------------------------------------------------------------------------------ */ interface Highlight { start: number; end: number; kind: "mw" | "usd"; raw: string; } /** Same regexes as apps/worker/src/trace.ts highlightFigures — MW / GW and money figures. */ export function highlightFigures(text: string): Highlight[] { const out: Highlight[] = []; for (const m of text.matchAll(/(\d{1,3}(?:[.,]\d{3})+|\d+(?:[.,]\d+)?)\+?\s?(gigawatts?|gw|megawatts?|mw)\b/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "mw", raw: m[0] }); for (const m of text.matchAll(/(?:US\$|USD|\$|€|£|A\$|C\$|S\$)\s?\d[\d.,]*\s*(?:trillion|billion|million|bn|m\b|b\b|k\b)?/gi)) out.push({ start: m.index ?? 0, end: (m.index ?? 0) + m[0].length, kind: "usd", raw: m[0] }); return out.sort((a, b) => a.start - b.start); } /** announcement.figures (connector-defined) → highlights when the entries carry offsets. */ function figuresFromAnnouncement(a: Record | null): Highlight[] { const figs = a?.figures; if (!Array.isArray(figs)) return []; const out: Highlight[] = []; for (const f of figs as Array>) { const start = typeof f.start === "number" ? f.start : typeof f.offset === "number" ? f.offset : null; const end = typeof f.end === "number" ? f.end : start !== null && typeof f.raw === "string" ? start + f.raw.length : null; if (start === null || end === null || end <= start) continue; const unit = String(f.unit ?? f.kind ?? "").toLowerCase(); out.push({ start, end, kind: unit.includes("usd") || unit.includes("$") || unit === "money" ? "usd" : "mw", raw: typeof f.raw === "string" ? f.raw : String(f.value ?? "") }); } return out.sort((a, b) => a.start - b.start); } /** Text with figure spans highlighted (mw = accent, usd = permitting tone) and an optional evidence window emphasised. */ export function HighlightedText({ text, highlights, focus, className, maxHeight = 320 }: { text: string; highlights: Highlight[]; focus?: { start: number; end: number } | null; className?: string; maxHeight?: number | null }) { const parts = useMemo(() => { const nodes: ReactNode[] = []; let i = 0; const marks = [...highlights].filter((h) => h.start >= 0 && h.end <= text.length && h.end > h.start); // merge focus as a soft background for (const [k, h] of marks.entries()) { if (h.start < i) continue; if (h.start > i) nodes.push({text.slice(i, h.start)}); nodes.push( {text.slice(h.start, h.end)} , ); i = h.end; } if (i < text.length) nodes.push({text.slice(i)}); return nodes; }, [text, highlights]); if (!text) return

No parsed text.

; return (
{focus ? ( <> {text.slice(0, focus.start)} {text.slice(focus.start, focus.end)} {text.slice(focus.end)} ) : ( parts )}
); } const STAGES = [ { id: "document", n: 1, label: "Source document" }, { id: "fetch", n: 2, label: "Raw fetch" }, { id: "text", n: 3, label: "Parsed text" }, { id: "records", n: 4, label: "Structured data" }, { id: "entities", n: 5, label: "Extracted entities" }, { id: "claims", n: 6, label: "Extracted claims" }, { id: "normalized", n: 7, label: "Normalized values" }, { id: "matches", n: 8, label: "Match candidates" }, { id: "reconciliation", n: 9, label: "Reconciliation (dry run)" }, { id: "changes", n: 10, label: "Resulting database changes" }, ] as const; function Stage({ n, label, count, tone, children, defaultOpen = true, aside }: { n: number; label: string; count?: ReactNode; tone?: "ok" | "warn" | "danger" | "muted"; children: ReactNode; defaultOpen?: boolean; aside?: ReactNode }) { const [open, setOpen] = useState(defaultOpen); return (
  • {n}
    {open &&
    {children}
    }
  • ); } function num(v: unknown): number | null { return typeof v === "number" && Number.isFinite(v) ? v : null; } function str(v: unknown): string | null { return typeof v === "string" && v ? v : null; } const MW_FIELDS = ["itCapacityMw", "totalPowerMw", "plannedPowerMw", "utilityCapacityMw", "gridConnectionMw", "ultimateCampusMw", "plannedMw", "it_capacity_mw", "total_power_mw", "planned_power_mw", "planned_mw"]; const DATE_FIELDS = ["openedOn", "announcedOn", "constructionStartedOn", "expectedOpening", "approvedOn", "permitFiledOn", "opened_on", "announced_on", "expected_opening"]; /** KV of the fields that matter for a normalized entity (MW columns, status, dates, geo, ids). */ function NormalizedEntity({ e }: { e: Record }) { const geo = (e.geo as Record | undefined) ?? null; const items: Array<{ k: ReactNode; v: ReactNode }> = []; for (const f of MW_FIELDS) if (num(e[f]) !== null) items.push({ k: {f}, v: {fmtMw(num(e[f]))} }); if (num(e.investmentUsd ?? e.investment_usd) !== null) items.push({ k: investmentUsd, v: {fmtUsd(num(e.investmentUsd ?? e.investment_usd))} }); if (str(e.status)) items.push({ k: "status", v: {String(e.status)} }); if (str(e.facilityType ?? e.facility_type)) items.push({ k: "type", v: String(e.facilityType ?? e.facility_type) }); for (const f of DATE_FIELDS) if (str(e[f])) items.push({ k: {f}, v: {String(e[f])} }); if (geo && num(geo.lat) !== null && num(geo.lng) !== null) items.push({ k: "geo", v: {num(geo.lat)!.toFixed(5)}, {num(geo.lng)!.toFixed(5)} {geo.precision ? `(${String(geo.precision)})` : ""} }); else if (num(e.lat) !== null && num(e.lng) !== null) items.push({ k: "geo", v: {num(e.lat)!.toFixed(5)}, {num(e.lng)!.toFixed(5)} }); for (const f of ["city", "regionName", "countryIso2", "country_iso2", "address", "operatorName", "operator", "metro"]) if (str(e[f])) items.push({ k: f, v: String(e[f]) }); const aiEv = str(e.aiEvidence ?? e.ai_evidence); if (aiEv) items.push({ k: "aiEvidence", v: {aiEv} }); if (str(e.projectClass ?? e.project_class)) items.push({ k: "projectClass", v: {String(e.projectClass ?? e.project_class)} }); const ids = (e.externalIds as Record | undefined) ?? null; if (ids && Object.keys(ids).length) items.push({ k: "externalIds", v: {Object.entries(ids).map(([k, v]) => `${k}=${String(v)}`).join(" · ")} }); return (
    {String(e.type ?? e.kind ?? "entity")} {str(e.name) ?? DASH} {str(e.key) && {String(e.key)}}
    {items.length ? :

    No capacity, status, date or geo field on this entity.

    }
    ); } function DebuggerHeader({ id, doc, fixture }: { id: string; doc: Record | null; fixture: boolean }) { const url = doc && typeof doc.url === "string" ? doc.url : null; const connectorId = doc && typeof doc.connectorId === "string" ? doc.connectorId : null; return ( Documents {connectorId && ( <> / {connectorId} )} / extraction debugger } title={url ? {url.replace(/^https?:\/\//, "")} : {id}} description={fixture ? "Sample trace for layout only — no worker call was made." : "Dry run of the full pipeline for this document: nothing below is written to the database."} /> ); } export function ExtractionDebugger({ id, fixture }: { id: string; fixture?: TraceResult | null }) { const [live, setLive] = useState(false); const q = useAdminQuery(fixture ? null : `documents/${encodeURIComponent(id)}/trace`, { query: live ? { live: 1 } : undefined, timeoutMs: 120_000 }); const t = fixture ?? q.data; const excerpt = t?.text.excerpt ?? ""; const highlights = useMemo(() => { const fromAnn = figuresFromAnnouncement(t?.announcement ?? null); return fromAnn.length ? fromAnn : highlightFigures(excerpt); }, [t, excerpt]); const [focusClaim, setFocusClaim] = useState(null); const unreachable = q.error && /worker unreachable|502|fetch failed/i.test(q.error); if (!fixture && q.error && !q.data) { return (
    ); start it or check DCI_WORKER_URL on the API.` : q.error} onRetry={() => void q.refresh()} />

    Nothing is shown from cache: a trace is recomputed on demand and never stored.

    ); } if (!t) { return (
    ); } const doc = t.document ?? {}; const rec = t.reconciliation; const entityByKey = new Map>(); for (const e of t.entities) if (str(e.key)) entityByKey.set(String(e.key), e); return (
    {fixture && ( Fixture — not real data )} {!fixture && {t.fetch?.source === "live" ? "live re-fetch" : "archived body"}} {t.error && {t.error}} {!fixture && ( )} {!fixture && void q.refresh()} busy={q.refreshing} updatedAt={q.updatedAt} label="Re-run" />}
    0 ? "warn" : undefined} /> c.evidence).length)} with evidence`} /> !["building", "facility", "campus"].includes(c.scope)).length)} note="stored, never summed" /> 0 ? "warn" : undefined} note={rec ? `${fmtInt(rec.projectsVetoed)} projects vetoed` : undefined} />
      open : undefined}> {str(doc.id) ?? id} }, { k: "Connector", v: str(doc.connectorId) ? {String(doc.connectorId)} : DASH }, { k: "URL", v: str(doc.url) ? : DASH }, { k: "Title", v: str(doc.title) ?? DASH }, { k: "Page type", v: str(doc.pageType) ? {String(doc.pageType)} : DASH }, { k: "Classifier", v: str(doc.classifier) ? {String(doc.classifier)} : DASH }, { k: "Content hash", v: str(doc.contentHash) ? {String(doc.contentHash).slice(0, 16)}… : DASH }, { k: "Extractor", v: str(doc.extractorVersion) ? {String(doc.extractorVersion)} : DASH }, { k: "Last extraction", v: doc.extractOk === false ? failed{str(doc.error) ? ` — ${String(doc.error)}` : ""} : doc.extractOk === true ? `${fmtInt(num(doc.extractCount) ?? 0)} entities` : DASH }, { k: "Quarantined", v: doc.quarantined ? yes : "no" }, ]} /> = 400 ? "danger" : undefined) : "muted"}> {t.fetch ? ( {t.fetch.source} }, { k: "HTTP", v: = 400 ? "text-danger" : undefined}>{t.fetch.status} }, { k: "Content type", v: t.fetch.contentType ?? DASH }, { k: "Bytes", v: fmtBytes(t.fetch.bytes) }, { k: "Fetcher", v: {t.fetch.fetcher} }, { k: "Storage key", v: t.fetch.storageKey ? {t.fetch.storageKey} : DASH }, ]} /> ) : (

      No body available — the document was never fetched or its archive is missing.

      )}
      {fmtInt(highlights.length)} figures highlighted}>
      {t.text.title && {t.text.title}} {t.text.classification && ( <> {t.text.classification.pageType} rule {t.text.classification.rule} {t.text.classification.eventType && {t.text.classification.eventType}} {t.text.classification.mw.length > 0 && MW seen: {t.text.classification.mw.join(", ")}} )}
      {t.text.length > excerpt.length &&

      Excerpt: first {fmtInt(excerpt.length)} of {fmtInt(t.text.length)} characters.

      } {t.announcement && (
      Announcement classification
      )}
      0}> {t.records.length === 0 ? (

      The parser produced no structured record from this body.

      ) : (
      {t.records.map((r, i) => (
      {r.kind} {r.key} {r.pageType && {r.pageType}} {r.certainty !== null && certainty {(r.certainty * 100).toFixed(0)}%}
      {Object.keys(r.methods).length > 0 && (

      methods: {Object.entries(r.methods).map(([k, v]) => `${k}=${v}`).join(" · ")}

      )}
      ))}
      )}
      0 ? "warn" : t.entities.length ? undefined : "muted"} defaultOpen={t.entities.length > 0}>

      Validation: {fmtInt(t.validation.total)} total · {fmtInt(t.validation.valid)} valid · {fmtInt(t.validation.rejected)} rejected

      {t.entities.length === 0 ? (

      No entity — a record needs an identifying name and enough fields to become one.

      ) : (
        {t.entities.map((e, i) => (
      • {String(e.type ?? e.kind ?? "entity")} {str(e.name) ?? DASH} {str(e.key) && {String(e.key)}} {str(e.status) && {String(e.status)}}
      • ))}
      )} {t.validation.issues.length > 0 && (
      Issues
        {t.validation.issues.map((is, i) => (
      • {String(is.level ?? is.severity ?? "issue")} {(is.field ?? is.path) && {String(is.field ?? is.path)}} {String(is.message ?? fmtValue(is))}
      • ))}
      )}
      0}> {t.claims.length === 0 ? (

      No numerical claim — either the text carries no MW / money figure or no supporting sentence was found (no sentence, no claim).

      ) : ( {t.claims.map((c, i) => { const ev = c.evidence; const evHi = ev ? highlightFigures(ev.text) : []; return ( setFocusClaim(i)} onMouseLeave={() => setFocusClaim(null)}> ); })}
      Entity Field Value Scope Semantics Evidence sentence
      {c.entityKey} {c.field} {c.unit === "MW" ? fmtMw(c.value) : fmtUsd(c.value)}
      {c.scopeReason}
      {c.semantics ? : {DASH}} {ev ? ( @{ev.start}–{ev.end} ) : ( no sentence (structured field) )}
      )}

      Only building / facility / campus scopes may populate a record’s capacity or investment column; portfolio, company, country, metro and unknown scopes are stored as claims and shown as evidence.

      0}> {t.entities.length === 0 ?

      Nothing to normalize.

      :
      {t.entities.map((e, i) => )}
      }
      0}> {t.matches.length === 0 ? (

      No facility resolution attempted (no facility entity, or the entity kind is not matched).

      ) : (
      {t.matches.map((m, i) => { const ent = entityByKey.get(m.entityKey); return (
      {str(ent?.name) ?? m.entityKey} {m.entityKey} {m.how} {m.matchedId && ( → {m.matchedId} )}
      {m.candidates.length === 0 ? (

      No candidate within the matching window — a real run would create a new facility.

      ) : ( {m.candidates.map((c) => ( ))}
      Candidate Operator City Distance Score Reasons
      {c.name}
      {c.id}
      {c.operatorName ?? DASH} {c.city ?? DASH} {c.distanceKm == null ? DASH : c.distanceKm < 1 ? `${Math.round(c.distanceKm * 1000)} m` : `${c.distanceKm.toFixed(1)} km`}
      = 0.92 ? "bg-ok" : c.score >= 0.6 ? "bg-warn" : "bg-ink-4")} style={{ width: `${Math.min(100, c.score * 100)}%` }} />
      {(c.score * 100).toFixed(0)}%
      {c.reasons.map((r) => ( {r} ))}
      )}
      ); })}

      ≥ 0.92 merges automatically, 0.60–0.92 goes to the duplicate workbench, below 0.60 creates a new record.

      )}
      0 ? "warn" : "ok") : "muted"}> {!rec ? (

      The dry-run ingest did not run (no valid entity, or the trace stopped earlier{t.error ? `: ${t.error}` : ""}).

      ) : (
      )}
      {!rec ? (

      No changes computed.

      ) : (
      {rec.refs.length > 0 && (
      Entities a real run would touch
        {rec.refs.map((r, i) => (
      • {r.type} {r.id}
      • ))}
      )} {rec.changes.length === 0 ?

      A real run would write nothing new for this document.

      : }

      Dry run: nothing above was written. Use “Reprocess” on the document to apply with the current parser.

      )}
    Worker log {fmtInt(t.logs.length)} lines
    {t.logs.length === 0 ? ( empty ) : ( t.logs.map((l, i) => (
    {l}
    )) )}
    ); }