SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
3 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%
4.8 KB · 130 lines typescript
Raw Blame History
1import type { FetchLevel, NormalizedEntity, PageType, SourceKind, ValidationReport, Provenance } from "@dci/core";23/** A URL found by discovery, with a hint about what it is and how urgently it should be crawled. */4export interface DiscoveredUrl {5  url: string;6  pageType?: PageType;7  /** logical group in the connector schedule: "facility_pages", "newsroom", "regions", "dataset", … */8  group: string;9  priority?: number; // 0..100 (higher = sooner)10  lastmod?: string | null;11  discoveredFrom?: string | null;12  /** minimum fetch level (a connector may know a page needs rendering) */13  minLevel?: FetchLevel;14  meta?: Record<string, unknown>;15}1617export interface RawDocument {18  url: string;19  finalUrl: string;20  fetchedAt: string;21  status: number;22  contentType: string | null;23  body: Buffer;24  /** best-effort text body (utf-8) — for JSON/HTML/XML/markdown */25  text: string;26  headers: Record<string, string>;27  etag: string | null;28  lastModified: string | null;29  notModified: boolean;30  fetcher: "direct" | "firecrawl" | "scrapfly" | "cache";31  level: FetchLevel;32  durationMs: number;33  credits: number; // premium credits spent (Scrapfly/Firecrawl)34  /** Firecrawl markdown (when used) */35  markdown?: string | null;36  error?: { code: string; message: string } | null;37  /** hints from discovery */38  group?: string;39  pageType?: PageType;40  meta?: Record<string, unknown>;41}4243/** Extracted record = one candidate entity as found on one page, before normalization. */44export interface ExtractedRecord {45  kind: NormalizedEntity["entityType"];46  key: string; // connector-scoped stable key47  data: Record<string, unknown>;48  /** per-field extraction method for provenance: { it_capacity_mw: "regex:mw_v1", lat: "json-ld:geo" } */49  methods?: Record<string, string>;50  certainty?: number; // 0..1 overall extraction certainty51  url: string;52  pageType?: PageType;53}5455export interface FetchOptions {56  level?: FetchLevel;57  etag?: string | null;58  lastModified?: string | null;59  timeoutMs?: number;60  maxBytes?: number;61  accept?: string;62  headers?: Record<string, string>;63  renderJs?: boolean;64  country?: string;65  /** group in the connector schedule (used for rate limits / cost attribution) */66  group?: string;67  waitForSelector?: string;68}6970export interface Fetcher {71  readonly name: "direct" | "firecrawl" | "scrapfly";72  readonly level: FetchLevel;73  available(): boolean;74  fetch(url: string, options?: FetchOptions): Promise<RawDocument>;75}7677export interface ConnectorSchedule {78  [group: string]: string; // "daily" | "weekly" | "monthly" | "6h" | "12h" | "3h" | cron-like interval string79}8081export interface ConnectorContext {82  connectorId: string;83  sourceId: string;84  runId: string;85  /** fetch with the connector's rate limit, robots policy, cache and escalation policy applied */86  fetch(url: string, options?: FetchOptions): Promise<RawDocument>;87  /** previously stored state for this connector (cursor, seen keys, etc.) */88  getState<T = unknown>(key: string): Promise<T | null>;89  setState(key: string, value: unknown): Promise<void>;90  log(level: "debug" | "info" | "warn" | "error", msg: string, extra?: Record<string, unknown>): void;91  /** true when this URL was fetched before and unchanged since `lastFetched` (used to skip expensive work) */92  isKnownUnchanged(url: string, contentHash: string): Promise<boolean>;93  now(): string;94  /** default provenance skeleton for this connector (sourceId, connectorId, retrievedAt…) */95  provenance(url: string, extra?: Partial<Provenance>): Provenance;96  env: Record<string, string | undefined>;97  /** dry run: nothing persisted */98  dryRun: boolean;99}100101/**102 * A connector = one source, one set of discovery + extraction rules.103 * Implementations are mostly config-driven (YAML) with optional code-backed parsers.104 */105export interface Connector {106  id: string;107  sourceName: string;108  sourceDomain: string;109  sourceKind: SourceKind;110  type: "html" | "sitemap" | "rss" | "pdf" | "json" | "hybrid" | "api" | "dataset";111  parserVersion: string;112  schedule: ConnectorSchedule;113  license?: string | null;114  attribution?: string | null;115  priority: number; // 1 best … 5116  discover(ctx: ConnectorContext): Promise<DiscoveredUrl[]>;117  fetch(ctx: ConnectorContext, url: DiscoveredUrl): Promise<RawDocument>;118  extract(ctx: ConnectorContext, doc: RawDocument): Promise<ExtractedRecord[]>;119  normalize(ctx: ConnectorContext, records: ExtractedRecord[]): Promise<NormalizedEntity[]>;120  validate(ctx: ConnectorContext, entities: NormalizedEntity[]): Promise<ValidationReport>;121}122123/** Code-backed parser: turns a document into extracted records. Registered by name and referenced from YAML. */124export interface Parser {125  name: string; // e.g. "equinix_facility_v1"126  version: string;127  pageTypes?: PageType[];128  parse(doc: RawDocument, ctx: ConnectorContext, params?: Record<string, unknown>): Promise<ExtractedRecord[]> | ExtractedRecord[];129}130