spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import type { FetchLevel, NormalizedEntity, PageType, SourceKind, ValidationReport, Provenance } from "@dci/core";23/** A URL found by discovery, with a hint about what it is and how urgently it should be crawled. */4export interface DiscoveredUrl {5 url: string;6 pageType?: PageType;7 /** logical group in the connector schedule: "facility_pages", "newsroom", "regions", "dataset", … */8 group: string;9 priority?: number; // 0..100 (higher = sooner)10 lastmod?: string | null;11 discoveredFrom?: string | null;12 /** minimum fetch level (a connector may know a page needs rendering) */13 minLevel?: FetchLevel;14 meta?: Record<string, unknown>;15}1617export interface RawDocument {18 url: string;19 finalUrl: string;20 fetchedAt: string;21 status: number;22 contentType: string | null;23 body: Buffer;24 /** best-effort text body (utf-8) — for JSON/HTML/XML/markdown */25 text: string;26 headers: Record<string, string>;27 etag: string | null;28 lastModified: string | null;29 notModified: boolean;30 fetcher: "direct" | "firecrawl" | "scrapfly" | "cache";31 level: FetchLevel;32 durationMs: number;33 credits: number; // premium credits spent (Scrapfly/Firecrawl)34 /** Firecrawl markdown (when used) */35 markdown?: string | null;36 error?: { code: string; message: string } | null;37 /** hints from discovery */38 group?: string;39 pageType?: PageType;40 meta?: Record<string, unknown>;41}4243/** Extracted record = one candidate entity as found on one page, before normalization. */44export interface ExtractedRecord {45 kind: NormalizedEntity["entityType"];46 key: string; // connector-scoped stable key47 data: Record<string, unknown>;48 /** per-field extraction method for provenance: { it_capacity_mw: "regex:mw_v1", lat: "json-ld:geo" } */49 methods?: Record<string, string>;50 certainty?: number; // 0..1 overall extraction certainty51 url: string;52 pageType?: PageType;53}5455export interface FetchOptions {56 level?: FetchLevel;57 etag?: string | null;58 lastModified?: string | null;59 timeoutMs?: number;60 maxBytes?: number;61 accept?: string;62 headers?: Record<string, string>;63 renderJs?: boolean;64 country?: string;65 /** group in the connector schedule (used for rate limits / cost attribution) */66 group?: string;67 waitForSelector?: string;68}6970export interface Fetcher {71 readonly name: "direct" | "firecrawl" | "scrapfly";72 readonly level: FetchLevel;73 available(): boolean;74 fetch(url: string, options?: FetchOptions): Promise<RawDocument>;75}7677export interface ConnectorSchedule {78 [group: string]: string; // "daily" | "weekly" | "monthly" | "6h" | "12h" | "3h" | cron-like interval string79}8081export interface ConnectorContext {82 connectorId: string;83 sourceId: string;84 runId: string;85 /** fetch with the connector's rate limit, robots policy, cache and escalation policy applied */86 fetch(url: string, options?: FetchOptions): Promise<RawDocument>;87 /** previously stored state for this connector (cursor, seen keys, etc.) */88 getState<T = unknown>(key: string): Promise<T | null>;89 setState(key: string, value: unknown): Promise<void>;90 log(level: "debug" | "info" | "warn" | "error", msg: string, extra?: Record<string, unknown>): void;91 /** true when this URL was fetched before and unchanged since `lastFetched` (used to skip expensive work) */92 isKnownUnchanged(url: string, contentHash: string): Promise<boolean>;93 now(): string;94 /** default provenance skeleton for this connector (sourceId, connectorId, retrievedAt…) */95 provenance(url: string, extra?: Partial<Provenance>): Provenance;96 env: Record<string, string | undefined>;97 /** dry run: nothing persisted */98 dryRun: boolean;99}100101/**102 * A connector = one source, one set of discovery + extraction rules.103 * Implementations are mostly config-driven (YAML) with optional code-backed parsers.104 */105export interface Connector {106 id: string;107 sourceName: string;108 sourceDomain: string;109 sourceKind: SourceKind;110 type: "html" | "sitemap" | "rss" | "pdf" | "json" | "hybrid" | "api" | "dataset";111 parserVersion: string;112 schedule: ConnectorSchedule;113 license?: string | null;114 attribution?: string | null;115 priority: number; // 1 best … 5116 discover(ctx: ConnectorContext): Promise<DiscoveredUrl[]>;117 fetch(ctx: ConnectorContext, url: DiscoveredUrl): Promise<RawDocument>;118 extract(ctx: ConnectorContext, doc: RawDocument): Promise<ExtractedRecord[]>;119 normalize(ctx: ConnectorContext, records: ExtractedRecord[]): Promise<NormalizedEntity[]>;120 validate(ctx: ConnectorContext, entities: NormalizedEntity[]): Promise<ValidationReport>;121}122123/** Code-backed parser: turns a document into extracted records. Registered by name and referenced from YAML. */124export interface Parser {125 name: string; // e.g. "equinix_facility_v1"126 version: string;127 pageTypes?: PageType[];128 parse(doc: RawDocument, ctx: ConnectorContext, params?: Record<string, unknown>): Promise<ExtractedRecord[]> | ExtractedRecord[];129}130