import type { FetchLevel, NormalizedEntity, PageType, SourceKind, ValidationReport, Provenance } from "@dci/core"; /** A URL found by discovery, with a hint about what it is and how urgently it should be crawled. */ export interface DiscoveredUrl { url: string; pageType?: PageType; /** logical group in the connector schedule: "facility_pages", "newsroom", "regions", "dataset", … */ group: string; priority?: number; // 0..100 (higher = sooner) lastmod?: string | null; discoveredFrom?: string | null; /** minimum fetch level (a connector may know a page needs rendering) */ minLevel?: FetchLevel; meta?: Record; } export interface RawDocument { url: string; finalUrl: string; fetchedAt: string; status: number; contentType: string | null; body: Buffer; /** best-effort text body (utf-8) — for JSON/HTML/XML/markdown */ text: string; headers: Record; etag: string | null; lastModified: string | null; notModified: boolean; fetcher: "direct" | "firecrawl" | "scrapfly" | "cache"; level: FetchLevel; durationMs: number; credits: number; // premium credits spent (Scrapfly/Firecrawl) /** Firecrawl markdown (when used) */ markdown?: string | null; error?: { code: string; message: string } | null; /** hints from discovery */ group?: string; pageType?: PageType; meta?: Record; } /** Extracted record = one candidate entity as found on one page, before normalization. */ export interface ExtractedRecord { kind: NormalizedEntity["entityType"]; key: string; // connector-scoped stable key data: Record; /** per-field extraction method for provenance: { it_capacity_mw: "regex:mw_v1", lat: "json-ld:geo" } */ methods?: Record; certainty?: number; // 0..1 overall extraction certainty url: string; pageType?: PageType; } export interface FetchOptions { level?: FetchLevel; etag?: string | null; lastModified?: string | null; timeoutMs?: number; maxBytes?: number; accept?: string; headers?: Record; renderJs?: boolean; country?: string; /** group in the connector schedule (used for rate limits / cost attribution) */ group?: string; waitForSelector?: string; } export interface Fetcher { readonly name: "direct" | "firecrawl" | "scrapfly"; readonly level: FetchLevel; available(): boolean; fetch(url: string, options?: FetchOptions): Promise; } export interface ConnectorSchedule { [group: string]: string; // "daily" | "weekly" | "monthly" | "6h" | "12h" | "3h" | cron-like interval string } export interface ConnectorContext { connectorId: string; sourceId: string; runId: string; /** fetch with the connector's rate limit, robots policy, cache and escalation policy applied */ fetch(url: string, options?: FetchOptions): Promise; /** previously stored state for this connector (cursor, seen keys, etc.) */ getState(key: string): Promise; setState(key: string, value: unknown): Promise; log(level: "debug" | "info" | "warn" | "error", msg: string, extra?: Record): void; /** true when this URL was fetched before and unchanged since `lastFetched` (used to skip expensive work) */ isKnownUnchanged(url: string, contentHash: string): Promise; now(): string; /** default provenance skeleton for this connector (sourceId, connectorId, retrievedAt…) */ provenance(url: string, extra?: Partial): Provenance; env: Record; /** dry run: nothing persisted */ dryRun: boolean; } /** * A connector = one source, one set of discovery + extraction rules. * Implementations are mostly config-driven (YAML) with optional code-backed parsers. */ export interface Connector { id: string; sourceName: string; sourceDomain: string; sourceKind: SourceKind; type: "html" | "sitemap" | "rss" | "pdf" | "json" | "hybrid" | "api" | "dataset"; parserVersion: string; schedule: ConnectorSchedule; license?: string | null; attribution?: string | null; priority: number; // 1 best … 5 discover(ctx: ConnectorContext): Promise; fetch(ctx: ConnectorContext, url: DiscoveredUrl): Promise; extract(ctx: ConnectorContext, doc: RawDocument): Promise; normalize(ctx: ConnectorContext, records: ExtractedRecord[]): Promise; validate(ctx: ConnectorContext, entities: NormalizedEntity[]): Promise; } /** Code-backed parser: turns a document into extracted records. Registered by name and referenced from YAML. */ export interface Parser { name: string; // e.g. "equinix_facility_v1" version: string; pageTypes?: PageType[]; parse(doc: RawDocument, ctx: ConnectorContext, params?: Record): Promise | ExtractedRecord[]; }