SPB Git forge
38commits 1branches 0releases
338.7 MBsize
maindefault branch
2 h agolast push
HTML 53.9% TypeScript 44.5% JavaScript 0.6% SQL 0.5%

Crawl reliability: standalone scheduler, run locks, starvation fix, Prometheus metrics, Redis budgets, premium error gate, discovery cap by priority; connector config fixes from audit

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Simon-Pierre Boucher committed 21 days ago (Sep 11, 2026) parent 0e30340

22 changed files +2,258 −312

modified apps/worker/README.md +31 −11
@@ -18,16 +18,19 @@ discover ─► registerDiscovered ─► dueDocuments ─► ctx.fetch ─► p
18 18
19 19 | file | role |
20 20 |---|---|
21 −| `env.ts` | typed `process.env` (`getEnv()`), `loadEnvFile()` via `node:process.loadEnvFile` for dev |
21 +| `env.ts` | typed `process.env` (`getEnv()`), `DCI_QUEUES` parsing, `loadEnvFile()` via `node:process.loadEnvFile` for dev |
22 +| `prom.ts` | in-process Prometheus registry (counters, gauges, histogram) and the worker's metric contract (`dci_crawl_*`, `dci_ingest_*`, `dci_queue_jobs`, `dci_worker_*`, `dci_scheduler_*`) |
23 +| `budget.ts` | `RedisBudgetStore`: shared daily premium budgets (`dci:budget:<provider>:<day>`) and per-connector daily caps (`dci:budget:connector:<id>:<day>`), installed into the fetchers with `setBudgetStore` |
24 +| `health-http.ts` | the `/healthz` + `/metrics` HTTP server shared by the worker and the standalone scheduler |
22 25 | `storage.ts` | S3/MinIO: `ensureBucket`, `putRaw` (key `raw/<connector>/<docId>/<hash>.<ext>.zst`, zstd → gzip fallback, never uploads the same hash twice), `getRaw`, `headRaw` |
23 26 | `configs.ts` | load YAML → `Connector` (registered `implementation` or `GenericConnector`), `syncConnectorsToDb()` upserts `connectors` + `sources` (id = `stableId("source", connectorId)`) |
24 −| `context.ts` | production `ConnectorContext`: robots.txt (+ crawl-delay), per-host token bucket, escalating fetch capped by `fetch.maxLevel` and `maxCreditsPerRun`, conditional GET validators, `connector_state`, `isKnownUnchanged`, capped run log → `connector_runs.log`, ClickHouse `crawl_log` row per fetch (never fatal), cost per fetcher |
27 +| `context.ts` | production `ConnectorContext`: robots.txt (+ crawl-delay), per-host token bucket, escalating fetch capped by `fetch.maxLevel`, `maxCreditsPerRun`, `maxCreditsPerDay` (Redis), the provider daily budgets, the discovery rule (sitemap/RSS never premium) and the error gate (a document failing twice is direct-only until every 4th attempt), conditional GET validators, `connector_state`, `isKnownUnchanged`, capped run log → `connector_runs.log`, Prometheus fetch metrics, ClickHouse `crawl_log` row per fetch (never fatal), cost per fetcher |
25 28 | `documents.ts` | URL registry: `registerDiscovered`, `dueDocuments`, `recordFetch` (adaptive `next_check`), `recordVersion` (line diff of text projections, significance), `updateExtraction`, stats |
26 29 | `scheduling.ts` | pure math: intervals, change-score EMA, backoff/quarantine, skip decision, health thresholds, budget level cap — unit-tested |
27 30 | `pipeline.ts` | `runConnector(id, opts)`: one run end to end, per-document isolation, inline `pLimit`, graceful abort, run status + connector health |
28 −| `scheduler.ts` | BullMQ queues `crawl` / `maintenance` (Redis prefix `dci` → keys `dci:crawl:*`), `enqueueRun()`, 60 s scheduler tick (per-group jobs, dedupe by jobId `<connector>__<group>`), repeatable maintenance jobs |
31 +| `scheduler.ts` | BullMQ queues `crawl` / `maintenance` (Redis prefix `dci` → keys `dci:crawl:*`), `enqueueRun()`, 60 s scheduler tick (per-group jobs, dedupe by jobId `<connector>__<group>`, no group job while a `full`/`discover` job is pending, discovery strictly on cadence), repeatable maintenance jobs, run-lock keys; runnable as the standalone scheduler process (`/healthz` + `/metrics`) |
29 32 | `maintenance.ts` | `doctor()` diagnostics → `system_alerts`, `cleanupMaintenance()` |
30 −| `main.ts` | worker process: register connector groups, sync configs, BullMQ Workers, heartbeat `dci:worker:status`, HTTP `/healthz` + `/metrics` |
33 +| `main.ts` | worker process: register connector groups, sync configs, BullMQ Workers for `DCI_QUEUES`, per-connector run lock (`dci:run-lock:<id>`), heartbeat `dci:worker:status`, HTTP `/healthz` + `/metrics`, graceful shutdown with deadline |
31 34 | `cli.ts` | `pnpm dci …` |
32 35 | `connectors/index.ts` | registers every `connectors/<group>/index.ts` (`register()`), tolerant of missing/broken groups |
33 36 | `ingest/`, `rankings.ts`, `metrics.ts`, `parsers/`, `connectors/<group>/` | owned by the ingest / connector agents |
@@ -51,8 +54,13 @@ discover ─► registerDiscovered ─► dueDocuments ─► ctx.fetch ─► p
51 54 `detectedChanges` from ingest; significance = max field-level significance, else 10/20/30 by diff ratio;
52 55 ClickHouse `page_changes` row.
53 56 * Premium fetchers (L3 Firecrawl, L4 Scrapfly) are used only when a page is due **and** direct fetch was
54 − blocked / JS-shell, only up to `fetch.maxLevel`, only while `fetch.maxCreditsPerRun` and the daily budgets
55 − (`DCI_*_DAILY_BUDGET`) allow. Robots disallow is a hard stop.
57 + blocked / JS-shell, only up to `fetch.maxLevel`, only while `fetch.maxCreditsPerRun`, the connector's
58 + `fetch.maxCreditsPerDay` and the provider daily budgets (`DCI_*_DAILY_BUDGET`, shared across workers through
59 + Redis) allow — never for sitemap / RSS discovery fetches, and only every 4th attempt for a document that already
60 + failed twice in a row. Robots disallow is a hard stop. See `docs/CRAWL-OPERATIONS.md` §4.
61 +* One run per connector at a time across all workers (Redis `dci:run-lock:<connector>`): a job that finds the
62 + lock held is delayed 60 s. Without it two overlapping runs fetch the same documents and both record a phantom
63 + "change" against a stale hash.
56 64
57 65 ## Runs and health
58 66
@@ -86,10 +94,16 @@ called with `dryRun: true`; discovered URLs are crawled virtually in priority or
86 94 ## Worker process
87 95
88 96 `pnpm --filter @dci/worker run dev` (or `pnpm dci worker`): registers connector groups, syncs configs,
89 −ensures the bucket and ClickHouse tables, starts BullMQ Workers (`DCI_CRAWL_CONCURRENCY`, default 3, and
90 −`DCI_MAINTENANCE_CONCURRENCY`), the scheduler loop (Redis lock `dci:scheduler:lock`), a 15 s heartbeat in
91 −`dci:worker:status` and HTTP on `WORKER_PORT` (default 8320): `GET /healthz` JSON, `GET /metrics` Prometheus
92 −text. SIGTERM/SIGINT: stop taking jobs, finish in-flight documents, mark runs `aborted`, exit.
97 +ensures the bucket and ClickHouse tables, installs the Redis budget store, starts BullMQ Workers for the queues in
98 +`DCI_QUEUES` (`crawl,maintenance`; `DCI_CRAWL_CONCURRENCY`, default 3, and `DCI_MAINTENANCE_CONCURRENCY`), the
99 +scheduler loop when the process consumes `crawl` (Redis lock `dci:scheduler:lock`; `DCI_SCHEDULER=0` to disable),
100 +a 15 s heartbeat in `dci:worker:status` and HTTP on `WORKER_PORT` (default 8320): `GET /healthz` JSON,
101 +`GET /metrics` Prometheus text (contract in `prom.ts`). SIGTERM/SIGINT: stop taking jobs, finish in-flight
102 +documents, mark runs `aborted`, exit — after `DCI_SHUTDOWN_TIMEOUT_MS` (50 s) in-flight runs are marked aborted in
103 +Postgres and the process exits anyway (keep it below compose's `stop_grace_period`).
104 +
105 +`pnpm --filter @dci/worker scheduler` (container role `scheduler`): only the scheduler loop plus `/healthz` +
106 +`/metrics` on `WORKER_PORT` (8321 in `compose.data.yml`), heartbeat `dci:scheduler:status`.
93 107
94 108 Maintenance schedulers (UTC): daily metrics 00:10, rankings 00:30, refresh-stats hourly, cleanup 01:00.
95 109
@@ -104,7 +118,12 @@ Maintenance schedulers (UTC): daily metrics 00:10, rankings 00:30, refresh-stats
104 118 | `SCRAPFLY_API_KEY` / `FIRECRAWL_API_KEY` | — | premium fetchers; absent = never escalate past L2 |
105 119 | `DCI_SCRAPFLY_DAILY_BUDGET` / `DCI_FIRECRAWL_DAILY_BUDGET` | 400 / 200 | daily credit caps |
106 120 | `DCI_CONFIG_DIR` | `<repo>/config/connectors` | YAML directory |
121 +| `DCI_QUEUES` | `crawl,maintenance` | queues consumed by this process (`maintenance` alone = the data node's `worker-maint`) |
122 +| `DCI_SCHEDULER` | on when `DCI_QUEUES` has `crawl` | `0` disables the embedded scheduler loop |
107 123 | `DCI_CRAWL_CONCURRENCY` / `DCI_MAINTENANCE_CONCURRENCY` | 3 / 1 | BullMQ concurrency |
124 +| `DCI_SHUTDOWN_TIMEOUT_MS` | 50000 | graceful shutdown deadline |
125 +| `DCI_BUDGET_REFRESH_MS` | 15000 | refresh of the shared Redis budget counters |
126 +| `DCI_ORPHAN_RUN_HOURS` | 6 | `running` runs older than this are marked `aborted` at start / cleanup |
108 127 | `DCI_SCHEDULER_INTERVAL_MS` | 60000 | scheduler tick |
109 128 | `WORKER_PORT` | 8320 | `/healthz`, `/metrics` |
110 129 | `DCI_LOG_LEVEL` | info | debug \| info \| warn \| error |
@@ -117,7 +136,8 @@ overriding existing variables).
117 136
118 137 `pnpm --filter @dci/worker test` — `scheduling.test.ts` covers interval resolution, EMA/scale, backoff and
119 138 quarantine, fetch bookkeeping, fingerprint noise handling, skip logic, significance, health, budget cap and
120 −`pLimit`.
139 +`pLimit`; `prom.test.ts` covers the Prometheus exposition format, `DCI_QUEUES` parsing, the budget key format,
140 +the discovery/error-gate cost rules, the scheduler's discovery cadence and the priority-aware discovery cap.
121 141
122 142 ## Smoke test
123 143
added apps/worker/src/budget.ts +114 −0
@@ -0,0 +1,114 @@
1 +/**
2 + * Shared premium-credit budgets in Redis, so that daily caps hold across worker processes and restarts.
3 + *
4 + * Keys (UTC day `YYYY-MM-DD`, values are floats written with INCRBYFLOAT, 3-day TTL):
5 + * dci:budget:<provider>:<day> credits spent today per provider (scrapfly | firecrawl)
6 + * dci:budget:connector:<connectorId>:<day> credits spent today per connector (enforces fetch.maxCreditsPerDay)
7 + *
8 + * The API (`/api/admin/ops`) can read them with plain GET / MGET; `budgetKey()` / `connectorBudgetKey()` build the
9 + * names. `RedisBudgetStore` implements the connectors package's pluggable `BudgetStore`: `spend()` is fire-and-forget
10 + * (never blocks a fetch, never throws), `used()` returns the last value read by the periodic refresh (every
11 + * DCI_BUDGET_REFRESH_MS) — the in-process counter of the fetchers guarantees the value never lags below what this
12 + * process spent itself.
13 + */
14 +import type { Redis } from "ioredis";
15 +import { budgetDay, setBudgetStore, type BudgetStore, type PremiumProvider } from "@dci/connectors";
16 +
17 +export const BUDGET_KEY_PREFIX = "dci:budget";
18 +export const BUDGET_TTL_SECONDS = 3 * 86_400;
19 +export const PROVIDERS: readonly PremiumProvider[] = ["scrapfly", "firecrawl"];
20 +
21 +export function budgetKey(provider: PremiumProvider, day = budgetDay()): string { return `${BUDGET_KEY_PREFIX}:${provider}:${day}`; }
22 +export function connectorBudgetKey(connectorId: string, day = budgetDay()): string { return `${BUDGET_KEY_PREFIX}:connector:${connectorId}:${day}`; }
23 +
24 +async function incr(redis: Redis, key: string, credits: number): Promise<number> {
25 + const res = (await redis.multi().incrbyfloat(key, credits).expire(key, BUDGET_TTL_SECONDS).exec()) ?? [];
26 + const first = res[0];
27 + if (!first) throw new Error("empty MULTI reply");
28 + if (first[0]) throw first[0];
29 + return Number(first[1]) || 0;
30 +}
31 +
32 +export class RedisBudgetStore implements BudgetStore {
33 + private readonly cache = new Map<string, number>();
34 + private readonly connectorCache = new Map<string, { value: number; at: number }>();
35 + private timer: NodeJS.Timeout | null = null;
36 + private lastError: string | null = null;
37 +
38 + constructor(private readonly redis: Redis, private readonly refreshMs = 15_000, private readonly log: (msg: string) => void = (m) => console.error(`[budget] ${m}`)) {}
39 +
40 + used(provider: PremiumProvider, day: string): number | null {
41 + return this.cache.get(budgetKey(provider, day)) ?? null;
42 + }
43 +
44 + spend(provider: PremiumProvider, day: string, credits: number): void {
45 + if (!(credits > 0)) return;
46 + const key = budgetKey(provider, day);
47 + this.cache.set(key, (this.cache.get(key) ?? 0) + credits); // optimistic, corrected by the next refresh
48 + incr(this.redis, key, credits).then((v) => this.cache.set(key, v), (e: Error) => this.fail(`spend ${key}: ${e.message}`));
49 + }
50 +
51 + /** Re-read today's provider counters (also called by start()). Never throws. */
52 + async refresh(): Promise<void> {
53 + const day = budgetDay();
54 + const keys = PROVIDERS.map((p) => budgetKey(p, day));
55 + try {
56 + const vals = await this.redis.mget(...keys);
57 + keys.forEach((k, i) => this.cache.set(k, Number(vals[i] ?? 0) || 0));
58 + if (this.lastError) { this.log("redis budget counters reachable again"); this.lastError = null; }
59 + } catch (e) { this.fail(`refresh: ${(e as Error).message}`); }
60 + }
61 +
62 + /** Credits a connector spent today (cached ≤ 10 s; 0 when Redis is unreachable — the global caps still apply). */
63 + async connectorUsed(connectorId: string, day = budgetDay()): Promise<number> {
64 + const key = connectorBudgetKey(connectorId, day);
65 + const c = this.connectorCache.get(key);
66 + if (c && Date.now() - c.at < 10_000) return c.value;
67 + try {
68 + const v = Number((await this.redis.get(key)) ?? 0) || 0;
69 + this.connectorCache.set(key, { value: v, at: Date.now() });
70 + return v;
71 + } catch (e) { this.fail(`connector ${connectorId}: ${(e as Error).message}`); return c?.value ?? 0; }
72 + }
73 +
74 + /** Account premium credits to a connector (fire-and-forget). */
75 + spendConnector(connectorId: string, credits: number, day = budgetDay()): void {
76 + if (!(credits > 0)) return;
77 + const key = connectorBudgetKey(connectorId, day);
78 + const c = this.connectorCache.get(key);
79 + this.connectorCache.set(key, { value: (c?.value ?? 0) + credits, at: c?.at ?? 0 });
80 + incr(this.redis, key, credits).then((v) => this.connectorCache.set(key, { value: v, at: Date.now() }), (e: Error) => this.fail(`spend connector ${key}: ${e.message}`));
81 + }
82 +
83 + /** Snapshot for /healthz: today's shared usage per provider. */
84 + snapshot(): Record<PremiumProvider, number> {
85 + const day = budgetDay();
86 + return Object.fromEntries(PROVIDERS.map((p) => [p, this.cache.get(budgetKey(p, day)) ?? 0])) as Record<PremiumProvider, number>;
87 + }
88 +
89 + async start(): Promise<this> {
90 + await this.refresh();
91 + this.timer = setInterval(() => void this.refresh(), this.refreshMs);
92 + this.timer.unref();
93 + return this;
94 + }
95 +
96 + stop(): void { if (this.timer) clearInterval(this.timer); this.timer = null; }
97 +
98 + private fail(msg: string): void {
99 + // log once per outage, not once per fetch
100 + if (this.lastError !== msg.split(":")[0]) this.log(`${msg} (budgets fall back to in-process counters)`);
101 + this.lastError = msg.split(":")[0] ?? msg;
102 + }
103 +}
104 +
105 +let installed: RedisBudgetStore | null = null;
106 +/** Create, start and register the Redis store with the fetchers. Idempotent. */
107 +export async function installBudgetStore(redis: Redis, refreshMs?: number): Promise<RedisBudgetStore> {
108 + if (installed) return installed;
109 + installed = await new RedisBudgetStore(redis, refreshMs).start();
110 + setBudgetStore(installed);
111 + return installed;
112 +}
113 +export function budgetStore(): RedisBudgetStore | null { return installed; }
114 +export function uninstallBudgetStore(): void { installed?.stop(); installed = null; setBudgetStore(null); }
modified apps/worker/src/connectors/datasets/osm-overpass.ts +8 −2
@@ -89,9 +89,9 @@ export class OsmOverpassConnector extends DatasetConnector {
89 89 last = doc;
90 90 const shed = doc.error?.code === "timeout" || doc.error?.code === "connection" || doc.status === 429 || doc.status === 504 || doc.status === 503 || doc.status === 502;
91 91 const why = doc.error ? `${doc.error.code}: ${doc.error.message}` : `HTTP ${doc.status}`;
92 − if (!shed || attempt === maxAttempts - 1) { ctx.log("warn", `Overpass tile ${String(u.meta?.tile)} failed at ${ep} (${why}); giving up for this run — cursor not advanced`); break; }
92 + if (!shed || attempt === maxAttempts - 1) { ctx.log("warn", `Overpass tile ${tileLabel(u)} failed at ${ep} (${why}); giving up for this run — cursor not advanced`); break; }
93 93 const wait = backoff * (attempt + 1);
94 − ctx.log("warn", `Overpass tile ${String(u.meta?.tile)}: ${why} at ${ep}; backing off ${wait} ms then trying ${eps[(attempt + 1) % eps.length]}`);
94 + ctx.log("warn", `Overpass tile ${tileLabel(u)}: ${why} at ${ep}; backing off ${wait} ms then trying ${eps[(attempt + 1) % eps.length]}`);
95 95 await sleep(wait);
96 96 }
97 97 return last!;
@@ -172,3 +172,9 @@ export class OsmOverpassConnector extends DatasetConnector {
172 172 return out;
173 173 }
174 174 }
175 +
176 +/** Tile label for logs — `meta` is not persisted by the document registry, so fall back to the bbox in the URL. */
177 +function tileLabel(u: DiscoveredUrl): string {
178 + if (u.meta?.tile) return String(u.meta.tile);
179 + try { return new URL(u.url).searchParams.get("data")?.match(/\(([-\d.,]+)\)/)?.[1] ?? "?"; } catch { return "?"; }
180 +}
modified apps/worker/src/context.ts +52 −14
@@ -1,7 +1,16 @@
1 1 /**
2 2 * Production ConnectorContext: robots.txt, per-host rate limit + crawl-delay, escalating fetch bounded by the
3 − * connector's levels and premium credit budget, connector_state persistence, content-hash lookups, structured
4 − * logging mirrored into connector_runs.log, and a ClickHouse crawl_log row per fetch (never fatal).
3 + * connector's levels and premium credit budgets (per run, per connector per day, per provider per day),
4 + * connector_state persistence, content-hash lookups, structured logging mirrored into connector_runs.log,
5 + * Prometheus fetch metrics and a ClickHouse crawl_log row per fetch (never fatal).
6 + *
7 + * Cost-control rules enforced here (see docs/CRAWL-OPERATIONS.md):
8 + * - discovery fetches (group `sitemap` / `rss`) are direct-only: L1 → L2, never Firecrawl / Scrapfly;
9 + * - premium levels stop once `fetch.maxCreditsPerRun` is spent in this run, once `fetch.maxCreditsPerDay` is
10 + * spent by this connector today (Redis `dci:budget:connector:<id>:<day>`), or once a provider's daily budget
11 + * (`DCI_*_DAILY_BUDGET`, shared via Redis `dci:budget:<provider>:<day>`) is exhausted;
12 + * - a document that failed twice in a row is fetched direct-only until it either succeeds or reaches every 4th
13 + * attempt (`premiumAllowedAfterErrors`), so a hard 403 does not burn credits on every backoff cycle.
5 14 */
6 15 import type { ConnectorContext, FetchOptions, RawDocument } from "@dci/connectors";
7 16 import { fetchWithEscalation, isAllowedByRobots, acquire, configureHost, failDoc, BROWSER_UA, scrapflyBudget, firecrawlBudget } from "@dci/connectors";
@@ -9,10 +18,12 @@ import type { FetchLevel, Provenance } from "@dci/core";
9 18 import { SOURCE_KIND_BASE } from "@dci/core";
10 19 import { getDb, connectorState, connectorRuns, sql, eq, and } from "@dci/db";
11 20 import { chInsert } from "@dci/db/clickhouse";
21 +import { budgetStore } from "./budget.js";
12 22 import type { LoadedConnector } from "./configs.js";
13 23 import { getEnv } from "./env.js";
14 24 import { documentIdFor, storedHashFor } from "./documents.js";
15 −import { maxLevelForBudget } from "./scheduling.js";
25 +import { fetchOutcome, metrics } from "./prom.js";
26 +import { isDiscoveryGroup, maxLevelForBudget, premiumAllowedAfterErrors } from "./scheduling.js";
16 27
17 28 export type LogLevel = "debug" | "info" | "warn" | "error";
18 29 export interface LogLine { t: string; level: LogLevel; msg: string }
@@ -21,6 +32,9 @@ export const RUN_LOG_CAP = 500;
21 32
22 33 export interface FetcherCost { count: number; credits: number; ms: number; errors: number }
23 34
35 +/** Per-URL hints the pipeline injects before calling connector.fetch (connectors call ctx.fetch without them). */
36 +export interface FetchHint { errorCount: number }
37 +
24 38 export interface RunContext extends ConnectorContext {
25 39 readonly loaded: LoadedConnector;
26 40 /** premium credits spent in this run */
@@ -29,6 +43,7 @@ export interface RunContext extends ConnectorContext {
29 43 notModifiedCount: number;
30 44 fetchErrorCount: number;
31 45 robotsBlocked: number;
46 + /** fetches whose level was capped by a budget rule (run / connector-day / provider-day / error gate) */
32 47 budgetCapped: number;
33 48 costByFetcher: Record<string, FetcherCost>;
34 49 logs: LogLine[];
@@ -36,8 +51,10 @@ export interface RunContext extends ConnectorContext {
36 51 flushLog(): Promise<void>;
37 52 /** connector_state snapshot written during a dry run (kept in memory only) */
38 53 readonly dryState: Map<string, unknown>;
39 − /** conditional-GET validators injected by the pipeline per URL (connectors call ctx.fetch without them) */
54 + /** conditional-GET validators injected by the pipeline per URL */
40 55 readonly validators: Map<string, { etag: string | null; lastModified: string | null }>;
56 + /** document scheduling hints injected by the pipeline per URL (error gate) */
57 + readonly hints: Map<string, FetchHint>;
41 58 }
42 59
43 60 export interface CreateContextOptions {
@@ -59,9 +76,13 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt
59 76 configureHost(host, cfg.fetch.rpm, cfg.fetch.concurrency);
60 77 const dryState = new Map<string, unknown>();
61 78 const validators = new Map<string, { etag: string | null; lastModified: string | null }>();
79 + const hints = new Map<string, FetchHint>();
62 80 const logs: LogLine[] = [];
63 81 let dirtyLog = false;
64 82 let flushing: Promise<void> | null = null;
83 + // credits this run added to the connector's daily counter (the Redis read may lag a few seconds)
84 + let dayCreditsThisRun = 0;
85 + let dayCapLogged = false;
65 86
66 87 const ctx: RunContext = {
67 88 loaded,
@@ -80,6 +101,7 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt
80 101 logs,
81 102 dryState,
82 103 validators,
104 + hints,
83 105 now: () => new Date().toISOString(),
84 106
85 107 log(level, msg, extra) {
@@ -143,18 +165,29 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt
143 165 ctx.robotsBlocked++;
144 166 ctx.log("warn", `robots.txt disallows ${url}`);
145 167 doc = failDoc(url, 1, "direct", "robots_disallow", "disallowed by robots.txt", started);
146 − account(doc, o.group);
168 + account(doc, o.group, started);
147 169 return doc;
148 170 }
149 171 }
150 172
151 − // premium levels only while the per-run budget allows, and only as high as the connector permits
152 − const budgetMax = maxLevelForBudget(cfg.fetch.maxLevel, ctx.credits, cfg.fetch.maxCreditsPerRun);
153 − if (budgetMax < cfg.fetch.maxLevel) ctx.budgetCapped++;
173 + // ---- how high may this fetch escalate?
174 + let budgetMax = maxLevelForBudget(cfg.fetch.maxLevel, ctx.credits, cfg.fetch.maxCreditsPerRun);
175 + let capReason: string | null = budgetMax < cfg.fetch.maxLevel ? "run budget" : null;
176 + if (budgetMax > 2 && isDiscoveryGroup(o.group)) { budgetMax = 2; capReason = "discovery is direct-only"; }
177 + if (budgetMax > 2 && cfg.fetch.maxCreditsPerDay !== undefined) {
178 + const usedToday = ((await budgetStore()?.connectorUsed(cfg.id)) ?? 0) + (budgetStore() ? 0 : dayCreditsThisRun);
179 + if (usedToday >= cfg.fetch.maxCreditsPerDay) {
180 + budgetMax = 2; capReason = `connector daily budget (${usedToday}/${cfg.fetch.maxCreditsPerDay})`;
181 + if (!dayCapLogged) { ctx.log("warn", `daily premium budget for ${cfg.id} exhausted (${usedToday}/${cfg.fetch.maxCreditsPerDay} credits) — direct fetches only until tomorrow (UTC)`); dayCapLogged = true; }
182 + }
183 + }
184 + const hint = hints.get(url);
185 + if (budgetMax > 2 && hint && !premiumAllowedAfterErrors(hint.errorCount)) { budgetMax = 2; capReason = `error gate (${hint.errorCount} consecutive failures)`; }
186 + if (budgetMax > 2 && scrapflyBudget.remaining() <= 0 && (budgetMax < 3 || firecrawlBudget.remaining() <= 0)) { budgetMax = 2; capReason = "provider daily budgets exhausted"; }
187 + if (capReason) ctx.budgetCapped++;
188 +
154 189 let level = Math.max(o.level ?? 1, cfg.fetch.level) as FetchLevel;
155 − if (level > budgetMax) { ctx.log("debug", `level ${level} requested for ${url} but budget caps at L${budgetMax}`); level = budgetMax; }
156 − // Scrapfly / Firecrawl daily budgets: when exhausted the fetcher reports unavailable and escalation stops there.
157 − if (budgetMax >= 4 && scrapflyBudget.remaining() <= 0 && budgetMax >= 3 && firecrawlBudget.remaining() <= 0) ctx.log("debug", "premium daily budgets exhausted — direct fetch only");
190 + if (level > budgetMax) { ctx.log("debug", `level ${level} requested for ${url} but capped at L${budgetMax} (${capReason ?? "connector maxLevel"})`); level = budgetMax; }
158 191
159 192 const delayMs = crawlDelay != null && Number.isFinite(crawlDelay) ? Math.min(30_000, Math.max(0, crawlDelay * 1000)) : 0;
160 193 const v = validators.get(url);
@@ -177,19 +210,24 @@ export function createRunContext(loaded: LoadedConnector, opts: CreateContextOpt
177 210 } catch (e) {
178 211 doc = failDoc(url, level, "direct", "fetch_threw", (e as Error).message, started);
179 212 } finally { release(); }
180 − account(doc, o.group);
213 + if (doc.credits > 0) { dayCreditsThisRun += doc.credits; budgetStore()?.spendConnector(cfg.id, doc.credits); }
214 + account(doc, o.group, started);
181 215 ctx.log("debug", `fetch ${url} → ${doc.error?.code ?? doc.status}${doc.notModified ? " (304)" : ""} via ${doc.fetcher} L${doc.level} ${doc.durationMs}ms ${doc.body.length}b${doc.credits ? ` ${doc.credits}cr` : ""}`);
182 216 return doc;
183 217 },
184 218 };
185 219
186 − function account(doc: RawDocument, group: string | undefined): void {
220 + function account(doc: RawDocument, group: string | undefined, started: number): void {
187 221 ctx.fetchCount++;
188 222 ctx.credits += doc.credits;
189 223 if (doc.notModified) ctx.notModifiedCount++;
190 224 if (doc.error) ctx.fetchErrorCount++;
191 225 const c = (ctx.costByFetcher[doc.fetcher] ??= { count: 0, credits: 0, ms: 0, errors: 0 });
192 226 c.count++; c.credits += doc.credits; c.ms += doc.durationMs; if (doc.error) c.errors++;
227 + // Prometheus (contract: dci_crawl_fetches_total{connector,level,outcome}, dci_crawl_fetch_duration_seconds, dci_crawl_credits_total{provider})
228 + metrics.fetches.inc({ connector: cfg.id, level: `L${doc.level}`, outcome: fetchOutcome(doc) });
229 + metrics.fetchDuration.observe(Math.max(0, Date.now() - started) / 1000);
230 + if (doc.credits > 0 && (doc.fetcher === "scrapfly" || doc.fetcher === "firecrawl")) metrics.credits.inc({ provider: doc.fetcher }, doc.credits);
193 231 let documentId = "";
194 232 try { documentId = documentIdFor(doc.url); } catch { /* invalid url */ }
195 233 void chInsert("crawl_log", [
@@ -221,7 +259,7 @@ function safeJson(v: unknown): string {
221 259 try { return JSON.stringify(v); } catch { return "[unserializable]"; }
222 260 }
223 261
224 −/** Snapshot of premium budgets for status endpoints / doctor. */
262 +/** Snapshot of premium budgets for status endpoints / doctor (shared Redis view when installed, else this process). */
225 263 export function budgetSnapshot(): { scrapfly: { day: string; used: number; limit: number }; firecrawl: { day: string; used: number; limit: number } } {
226 264 return { scrapfly: scrapflyBudget.snapshot(), firecrawl: firecrawlBudget.snapshot() };
227 265 }
modified apps/worker/src/env.ts +34 −2
@@ -8,6 +8,9 @@ import { fileURLToPath } from "node:url";
8 8 import process from "node:process";
9 9 import { hostname as osHostname } from "node:os";
10 10
11 +export type QueueName = "crawl" | "maintenance";
12 +export const ALL_QUEUES: readonly QueueName[] = ["crawl", "maintenance"];
13 +
11 14 export interface WorkerEnv {
12 15 databaseUrl: string;
13 16 redisUrl: string;
@@ -29,6 +32,16 @@ export interface WorkerEnv {
29 32 firecrawlDailyBudget: number;
30 33 /** minutes between scheduler ticks */
31 34 schedulerIntervalMs: number;
35 + /** queues this process consumes (DCI_QUEUES=crawl,maintenance) */
36 + queues: QueueName[];
37 + /** run the scheduler loop inside the worker (DCI_SCHEDULER=0 to rely on the dedicated scheduler service) */
38 + embeddedScheduler: boolean;
39 + /** hard deadline for a graceful shutdown before the process exits on its own (must stay < compose stop_grace_period) */
40 + shutdownTimeoutMs: number;
41 + /** how often the shared Redis budget counters are re-read */
42 + budgetRefreshMs: number;
43 + /** runs still `running` after this many hours are considered orphaned */
44 + orphanRunHours: number;
32 45 logLevel: "debug" | "info" | "warn" | "error";
33 46 hostname: string;
34 47 }
@@ -69,12 +82,20 @@ function num(v: string | undefined, fallback: number): number {
69 82 return Number.isFinite(n) && v !== undefined && v !== "" ? n : fallback;
70 83 }
71 84
85 +/** "crawl,maintenance" → ["crawl","maintenance"]; unknown names are ignored; empty/absent → both. */
86 +export function parseQueues(v: string | undefined): QueueName[] {
87 + const names = (v ?? "").split(",").map((s) => s.trim().toLowerCase()).filter(Boolean);
88 + const out = ALL_QUEUES.filter((q) => names.includes(q));
89 + return out.length ? [...out] : [...ALL_QUEUES];
90 +}
91 +
72 92 let cached: WorkerEnv | null = null;
73 93 export function getEnv(): WorkerEnv {
74 94 if (cached) return cached;
75 95 const e = process.env;
76 96 const root = findRepoRoot();
77 97 const level = (e.DCI_LOG_LEVEL ?? "info").toLowerCase();
98 + const queues = parseQueues(e.DCI_QUEUES);
78 99 cached = {
79 100 databaseUrl: e.DATABASE_URL ?? "postgres://dci:dci@127.0.0.1:5432/dci",
80 101 redisUrl: e.REDIS_URL ?? "redis://127.0.0.1:6379/0",
@@ -92,6 +113,12 @@ export function getEnv(): WorkerEnv {
92 113 scrapflyDailyBudget: num(e.DCI_SCRAPFLY_DAILY_BUDGET, 400),
93 114 firecrawlDailyBudget: num(e.DCI_FIRECRAWL_DAILY_BUDGET, 200),
94 115 schedulerIntervalMs: Math.max(5_000, num(e.DCI_SCHEDULER_INTERVAL_MS, 60_000)),
116 + queues,
117 + // a maintenance-only worker (data node) leaves scheduling to the dedicated scheduler service
118 + embeddedScheduler: e.DCI_SCHEDULER !== undefined ? e.DCI_SCHEDULER !== "0" && e.DCI_SCHEDULER.toLowerCase() !== "false" : queues.includes("crawl"),
119 + shutdownTimeoutMs: Math.max(5_000, num(e.DCI_SHUTDOWN_TIMEOUT_MS, 50_000)),
120 + budgetRefreshMs: Math.max(1_000, num(e.DCI_BUDGET_REFRESH_MS, 15_000)),
121 + orphanRunHours: Math.max(0.25, num(e.DCI_ORPHAN_RUN_HOURS, 6)),
95 122 logLevel: level === "debug" || level === "warn" || level === "error" ? level : "info",
96 123 hostname: e.HOSTNAME ?? osHostname(),
97 124 };
@@ -101,7 +128,7 @@ export function getEnv(): WorkerEnv {
101 128 /** Human-readable list of the variables we read (for `dci doctor` / README). */
102 129 export const ENV_DOC: Array<[name: string, def: string, doc: string]> = [
103 130 ["DATABASE_URL", "postgres://dci:dci@127.0.0.1:5432/dci", "Postgres (source of truth)"],
104 − ["REDIS_URL", "redis://127.0.0.1:6379/0", "Redis for BullMQ queues, locks, heartbeat"],
131 + ["REDIS_URL", "redis://127.0.0.1:6379/0", "Redis for BullMQ queues, locks, heartbeat, shared budgets"],
105 132 ["CLICKHOUSE_URL", "http://127.0.0.1:8123", "ClickHouse HTTP endpoint (analytics; optional)"],
106 133 ["CLICKHOUSE_DB", "dci", "ClickHouse database"],
107 134 ["S3_ENDPOINT", "http://127.0.0.1:9000", "S3/MinIO endpoint for raw bodies"],
@@ -109,11 +136,16 @@ export const ENV_DOC: Array<[name: string, def: string, doc: string]> = [
109 136 ["S3_ACCESS_KEY / S3_SECRET_KEY", "dci / —", "S3 credentials"],
110 137 ["S3_REGION", "us-east-1", "S3 region (any value for MinIO)"],
111 138 ["SCRAPFLY_API_KEY / FIRECRAWL_API_KEY", "—", "premium fetchers (L4 / L3); absent = never escalate"],
112 − ["DCI_SCRAPFLY_DAILY_BUDGET / DCI_FIRECRAWL_DAILY_BUDGET", "400 / 200", "daily premium credit caps"],
139 + ["DCI_SCRAPFLY_DAILY_BUDGET / DCI_FIRECRAWL_DAILY_BUDGET", "400 / 200", "daily premium credit caps (shared across workers via Redis dci:budget:<provider>:<day>)"],
113 140 ["DCI_CONFIG_DIR", "<repo>/config/connectors", "connector YAML directory"],
141 + ["DCI_QUEUES", "crawl,maintenance", "queues consumed by this process"],
142 + ["DCI_SCHEDULER", "1 when DCI_QUEUES includes crawl", "0 = do not run the scheduler loop in this worker"],
114 143 ["DCI_CRAWL_CONCURRENCY", "3", "BullMQ crawl jobs processed in parallel by this worker"],
115 144 ["DCI_MAINTENANCE_CONCURRENCY", "1", "maintenance jobs in parallel"],
116 145 ["DCI_SCHEDULER_INTERVAL_MS", "60000", "scheduler tick"],
146 + ["DCI_SHUTDOWN_TIMEOUT_MS", "50000", "graceful shutdown deadline (keep below compose stop_grace_period)"],
147 + ["DCI_BUDGET_REFRESH_MS", "15000", "refresh interval of the shared Redis budget counters"],
148 + ["DCI_ORPHAN_RUN_HOURS", "6", "runs still `running` after this are marked aborted at start / cleanup"],
117 149 ["WORKER_PORT", "8320", "/healthz and /metrics HTTP port"],
118 150 ["DCI_LOG_LEVEL", "info", "debug | info | warn | error"],
119 151 ["DCI_USER_AGENT", "DataCenterIndexBot/0.1 (…)", "bot identity for L1 fetches"],
added apps/worker/src/health-http.ts +48 −0
@@ -0,0 +1,48 @@
1 +/**
2 + * Minimal HTTP server shared by the worker (`main.ts`) and the standalone scheduler (`scheduler.ts`):
3 + * GET /healthz → JSON from `health()`; 503 while `ready()` is false (shutting down / not started)
4 + * GET /metrics → Prometheus text from prom.ts after `beforeScrape()` refreshed the gauges
5 + * Never throws into the caller; errors become 500 JSON responses.
6 + */
7 +import { createServer, type Server } from "node:http";
8 +import { renderMetrics } from "./prom.js";
9 +
10 +export interface HealthHttpOptions {
11 + port: number;
12 + role: string;
13 + health: () => Promise<Record<string, unknown>>;
14 + ready: () => boolean;
15 + /** refresh snapshot gauges (queue depth, budgets…) right before rendering */
16 + beforeScrape?: () => Promise<void>;
17 + log?: (msg: string) => void;
18 +}
19 +
20 +export function startHealthHttp(o: HealthHttpOptions): Server {
21 + const log = o.log ?? ((m: string) => console.error(`[${o.role}] ${m}`));
22 + const server = createServer(async (req, res) => {
23 + try {
24 + const url = new URL(req.url ?? "/", "http://localhost");
25 + if (url.pathname === "/healthz" || url.pathname === "/") {
26 + const body = await o.health();
27 + const ok = o.ready();
28 + res.writeHead(ok ? 200 : 503, { "content-type": "application/json", "cache-control": "no-store" });
29 + res.end(JSON.stringify({ ok, role: o.role, ...body }, null, 2));
30 + return;
31 + }
32 + if (url.pathname === "/metrics") {
33 + if (o.beforeScrape) await o.beforeScrape().catch((e: Error) => log(`metrics refresh failed: ${e.message}`));
34 + res.writeHead(200, { "content-type": "text/plain; version=0.0.4; charset=utf-8", "cache-control": "no-store" });
35 + res.end(renderMetrics());
36 + return;
37 + }
38 + res.writeHead(404, { "content-type": "application/json" });
39 + res.end(JSON.stringify({ error: "not found" }));
40 + } catch (e) {
41 + res.writeHead(500, { "content-type": "application/json" });
42 + res.end(JSON.stringify({ error: (e as Error).message }));
43 + }
44 + });
45 + server.on("error", (e) => log(`http error: ${e.message}`));
46 + server.listen(o.port, "0.0.0.0", () => log(`http :${o.port} (/healthz, /metrics)`));
47 + return server;
48 +}
modified apps/worker/src/main.ts +134 −99
@@ -1,20 +1,31 @@
1 1 /**
2 2 * Worker process: registers code-backed connectors, syncs YAML configs into Postgres, ensures the raw bucket and
3 − * ClickHouse tables, then runs BullMQ Workers for the `crawl` and `maintenance` queues plus the scheduler loop.
4 − * Heartbeat in Redis (dci:worker:status) and a tiny HTTP server (/healthz JSON, /metrics Prometheus text).
5 − * SIGTERM/SIGINT: stop taking jobs, let in-flight documents finish, mark runs aborted, exit.
3 + * ClickHouse tables, installs the shared Redis budget store, then runs BullMQ Workers for the queues listed in
4 + * DCI_QUEUES (`crawl`, `maintenance`; default both) plus the scheduler loop (unless DCI_SCHEDULER=0 or the process
5 + * consumes no crawl queue). Heartbeat in Redis (dci:worker:status) and HTTP /healthz (JSON) + /metrics (Prometheus,
6 + * prom.ts contract) on WORKER_PORT.
7 + *
8 + * One run per connector at a time across all workers: the crawl processor takes `dci:run-lock:<connector>`; a job
9 + * that finds the lock held is moved back to delayed (BullMQ DelayedError) instead of running concurrently.
10 + *
11 + * SIGTERM/SIGINT: stop taking jobs, let in-flight documents finish (runs end as `aborted`), exit. If that takes
12 + * longer than DCI_SHUTDOWN_TIMEOUT_MS (default 50 s, below compose's stop_grace_period), in-flight runs are marked
13 + * aborted in Postgres and the process exits anyway.
6 14 */
7 −import { createServer, type Server } from "node:http";
8 −import { Worker, type Job } from "bullmq";
15 +import { DelayedError, Worker, type Job } from "bullmq";
9 16 import { ensureClickHouse } from "@dci/db/clickhouse";
17 +import { closeDb } from "@dci/db";
18 +import { installBudgetStore, budgetStore, uninstallBudgetStore } from "./budget.js";
10 19 import { registerAllConnectors } from "./connectors/index.js";
11 20 import { loadAllConnectors, syncConnectorsToDb } from "./configs.js";
12 21 import { budgetSnapshot, creditsToday } from "./context.js";
13 −import { getEnv, loadEnvFile } from "./env.js";
22 +import { getEnv, loadEnvFile, type QueueName } from "./env.js";
23 +import { startHealthHttp } from "./health-http.js";
14 24 import { computeDailyMetrics, refreshStats } from "./metrics.js";
15 −import { activeRunIds, isAbortRequested, requestAbort, runConnector, type RunResult } from "./pipeline.js";
25 +import { abortActiveRunsInDb, activeRunIds, isAbortRequested, requestAbort, runConnector, type RunResult } from "./pipeline.js";
26 +import { metrics } from "./prom.js";
16 27 import { computeRankings } from "./rankings.js";
17 −import { CRAWL_QUEUE, MAINTENANCE_QUEUE, QUEUE_PREFIX, WORKER_STATUS_KEY, closeScheduler, getRedis, queueSnapshot, startSchedulerLoop, type CrawlJobData, type MaintenanceJobData } from "./scheduler.js";
28 +import { CRAWL_QUEUE, MAINTENANCE_QUEUE, QUEUE_PREFIX, RUN_LOCK_TTL_SECONDS, WORKER_STATUS_KEY, closeScheduler, getRedis, publishQueueMetrics, queueSnapshot, runLockKey, startSchedulerLoop, type CrawlJobData, type MaintenanceJobData } from "./scheduler.js";
18 29 import { ensureBucket } from "./storage.js";
19 30 import { cleanupMaintenance, reconcileOrphanRuns } from "./maintenance.js";
20 31
@@ -23,17 +34,21 @@ export interface WorkerStatus {
23 34 host: string;
24 35 pid: number;
25 36 version: string;
37 + queues: QueueName[];
38 + embeddedScheduler: boolean;
26 39 runningJobs: Array<{ id: string; name: string; connectorId?: string; task?: string; group?: string; since: string }>;
27 40 activeRuns: string[];
28 41 creditsToday: number;
29 42 budgets: ReturnType<typeof budgetSnapshot>;
43 + /** shared (Redis) usage today per provider, when the store is installed */
44 + sharedBudgets: Record<string, number> | null;
30 45 counters: Record<string, number>;
31 46 lastScheduler: { at: string; enqueued: number; checked: number } | null;
32 47 shuttingDown: boolean;
33 48 }
34 49
35 −const VERSION = "0.1.0";
36 −const counters: Record<string, number> = { jobs_completed: 0, jobs_failed: 0, runs_ok: 0, runs_partial: 0, runs_failed: 0, runs_aborted: 0, documents_fetched: 0, documents_changed: 0, documents_failed: 0, entities_valid: 0, credits_spent: 0, events: 0 };
50 +const VERSION = "0.2.0";
51 +const counters: Record<string, number> = { jobs_completed: 0, jobs_failed: 0, jobs_deferred_lock: 0, runs_ok: 0, runs_partial: 0, runs_failed: 0, runs_aborted: 0, documents_fetched: 0, documents_changed: 0, documents_failed: 0, entities_valid: 0, credits_spent: 0, events: 0 };
37 52 const running = new Map<string, WorkerStatus["runningJobs"][number]>();
38 53 let lastScheduler: WorkerStatus["lastScheduler"] = null;
39 54 let shuttingDown = false;
@@ -50,52 +65,22 @@ function accountRun(r: RunResult): void {
50 65 }
51 66
52 67 async function status(): Promise<WorkerStatus> {
53 − return { startedAt, host: getEnv().hostname, pid: process.pid, version: VERSION, runningJobs: [...running.values()], activeRuns: activeRunIds(), creditsToday: await creditsToday().catch(() => -1), budgets: budgetSnapshot(), counters, lastScheduler, shuttingDown };
54 −}
55 −
56 −function prometheus(s: WorkerStatus, queues: Awaited<ReturnType<typeof queueSnapshot>>): string {
57 − const lines: string[] = [];
58 − const g = (name: string, value: number, labels = "", help = "") => { if (help) lines.push(`# HELP ${name} ${help}`, `# TYPE ${name} gauge`); lines.push(`${name}${labels} ${Number.isFinite(value) ? value : 0}`); };
59 − g("dci_worker_up", s.shuttingDown ? 0 : 1, "", "1 while the worker accepts jobs");
60 − g("dci_worker_running_jobs", s.runningJobs.length, "", "jobs currently processed by this worker");
61 − g("dci_worker_credits_today", s.creditsToday, "", "premium credits recorded in connector_runs today");
62 − g("dci_budget_used", s.budgets.scrapfly.used, '{fetcher="scrapfly"}', "premium daily budget used (this process)");
63 − g("dci_budget_used", s.budgets.firecrawl.used, '{fetcher="firecrawl"}');
64 − g("dci_budget_limit", s.budgets.scrapfly.limit, '{fetcher="scrapfly"}');
65 − g("dci_budget_limit", s.budgets.firecrawl.limit, '{fetcher="firecrawl"}');
66 − for (const [k, v] of Object.entries(s.counters)) lines.push(`# TYPE dci_worker_${k}_total counter`, `dci_worker_${k}_total ${v}`);
67 − for (const q of queues) for (const k of ["waiting", "active", "delayed", "failed", "completed", "prioritized"] as const) g("dci_queue_jobs", q[k], `{queue="${q.name}",state="${k}"}`);
68 − g("dci_worker_uptime_seconds", (Date.now() - Date.parse(s.startedAt)) / 1000);
69 − return lines.join("\n") + "\n";
68 + const env = getEnv();
69 + return { startedAt, host: env.hostname, pid: process.pid, version: VERSION, queues: env.queues, embeddedScheduler: env.embeddedScheduler, runningJobs: [...running.values()], activeRuns: activeRunIds(), creditsToday: await creditsToday().catch(() => -1), budgets: budgetSnapshot(), sharedBudgets: budgetStore()?.snapshot() ?? null, counters, lastScheduler, shuttingDown };
70 70 }
71 71
72 −function startHttp(port: number): Server {
73 − const server = createServer(async (req, res) => {
74 − try {
75 − const url = new URL(req.url ?? "/", "http://localhost");
76 − if (url.pathname === "/healthz" || url.pathname === "/") {
77 − const s = await status();
78 − const queues = await queueSnapshot().catch(() => []);
79 − res.writeHead(shuttingDown ? 503 : 200, { "content-type": "application/json" });
80 − res.end(JSON.stringify({ ok: !shuttingDown, ...s, queues }, null, 2));
81 − return;
82 − }
83 − if (url.pathname === "/metrics") {
84 − const s = await status();
85 − const queues = await queueSnapshot().catch(() => []);
86 − res.writeHead(200, { "content-type": "text/plain; version=0.0.4; charset=utf-8" });
87 − res.end(prometheus(s, queues));
88 − return;
89 − }
90 − res.writeHead(404, { "content-type": "application/json" });
91 − res.end(JSON.stringify({ error: "not found" }));
92 − } catch (e) {
93 − res.writeHead(500, { "content-type": "application/json" });
94 − res.end(JSON.stringify({ error: (e as Error).message }));
95 − }
96 − });
97 − server.listen(port, "0.0.0.0", () => console.error(`[worker] http :${port} (/healthz, /metrics)`));
98 − return server;
72 +/** Refresh the snapshot gauges right before a Prometheus scrape. */
73 +async function refreshGauges(): Promise<void> {
74 + const env = getEnv();
75 + metrics.up.set(shuttingDown ? 0 : 1);
76 + metrics.runningJobs.set(running.size);
77 + metrics.uptime.set((Date.now() - Date.parse(startedAt)) / 1000);
78 + const b = budgetSnapshot();
79 + metrics.dailyBudget.set(env.scrapflyDailyBudget, { provider: "scrapfly" });
80 + metrics.dailyBudget.set(env.firecrawlDailyBudget, { provider: "firecrawl" });
81 + metrics.dailyUsed.set(b.scrapfly.used, { provider: "scrapfly" });
82 + metrics.dailyUsed.set(b.firecrawl.used, { provider: "firecrawl" });
83 + await publishQueueMetrics();
99 84 }
100 85
101 86 async function heartbeat(): Promise<void> {
@@ -116,80 +101,130 @@ export async function startWorker(opts: { withScheduler?: boolean } = {}): Promi
116 101 const loaded = loadAllConnectors();
117 102 const sync = await syncConnectorsToDb(loaded);
118 103 console.error(`[worker] synced ${sync.connectors} connectors / ${sync.sources} sources${sync.disabledInDb.length ? ` (disabled without YAML: ${sync.disabledInDb.join(", ")})` : ""}`);
119 − const orphans = await reconcileOrphanRuns(Number(process.env.DCI_ORPHAN_RUN_HOURS ?? 6)).catch((e: Error) => { console.error(`[worker] orphan reconciliation: ${e.message}`); return [] as string[]; });
104 + const orphans = await reconcileOrphanRuns(env.orphanRunHours).catch((e: Error) => { console.error(`[worker] orphan reconciliation: ${e.message}`); return [] as string[]; });
120 105 if (orphans.length) console.error(`[worker] aborted ${orphans.length} orphaned run(s): ${orphans.join(", ")}`);
121 106 await ensureBucket().catch((e: Error) => console.error(`[worker] storage: ${e.message}`));
122 107 await ensureClickHouse().catch((e: Error) => console.error(`[worker] clickhouse (optional): ${e.message}`));
123 108
124 109 const connection = getRedis();
125 − const crawlWorker = new Worker<CrawlJobData>(
126 − CRAWL_QUEUE,
127 − async (job: Job<CrawlJobData>) => {
128 − if (shuttingDown) throw new Error("worker shutting down");
129 − const d = job.data;
130 − running.set(String(job.id), { id: String(job.id), name: job.name, connectorId: d.connectorId, task: d.task, group: d.group, since: new Date().toISOString() });
131 − try {
132 − const r = await runConnector(d.connectorId, { task: d.task, group: d.group, limit: d.limit, force: d.force, urls: d.urls });
133 − accountRun(r);
134 − if (r.status === "failed") throw new Error(r.error ?? "run failed");
135 − return { runId: r.runId, status: r.status, stats: r.stats };
136 − } finally { running.delete(String(job.id)); }
137 − },
138 − { connection, prefix: QUEUE_PREFIX, concurrency: env.crawlConcurrency, lockDuration: 120_000, stalledInterval: 60_000, maxStalledCount: 1 },
139 − );
140 − const maintWorker = new Worker<MaintenanceJobData>(
141 − MAINTENANCE_QUEUE,
142 − async (job: Job<MaintenanceJobData>) => {
143 − running.set(String(job.id), { id: String(job.id), name: job.name, task: job.data.kind, since: new Date().toISOString() });
144 − try {
145 − switch (job.data.kind) {
146 − case "rankings": return await computeRankings();
147 − case "metrics": return await computeDailyMetrics();
148 − case "refresh-stats": return await refreshStats();
149 − case "cleanup": return await cleanupMaintenance();
150 − default: throw new Error(`unknown maintenance kind ${String((job.data as { kind?: string }).kind)}`);
110 + await installBudgetStore(connection, env.budgetRefreshMs).catch((e: Error) => console.error(`[worker] budget store: ${e.message} (in-process budgets only)`));
111 +
112 + const workers: Worker[] = [];
113 + if (env.queues.includes("crawl")) {
114 + const crawlWorker = new Worker<CrawlJobData>(
115 + CRAWL_QUEUE,
116 + async (job: Job<CrawlJobData>, token?: string) => {
117 + if (shuttingDown) throw new Error("worker shutting down");
118 + const d = job.data;
119 + // one run per connector at a time (across workers): otherwise two runs fetch the same documents and both
120 + // record a "change" against a stale content hash
121 + const lockKey = runLockKey(d.connectorId);
122 + const lockToken = `${env.hostname}:${process.pid}:${job.id}`;
123 + const got = await connection.set(lockKey, lockToken, "EX", RUN_LOCK_TTL_SECONDS, "NX");
124 + if (!got) {
125 + counters.jobs_deferred_lock!++;
126 + metrics.jobs.inc({ queue: CRAWL_QUEUE, result: "deferred" });
127 + const holder = await connection.get(lockKey).catch(() => null);
128 + console.error(`[worker] ${job.name} #${job.id}: ${d.connectorId} already running (${holder ?? "?"}) — retrying in 60 s`);
129 + await job.moveToDelayed(Date.now() + 60_000, token);
130 + throw new DelayedError();
131 + }
132 + running.set(String(job.id), { id: String(job.id), name: job.name, connectorId: d.connectorId, task: d.task, group: d.group, since: new Date().toISOString() });
133 + try {
134 + const r = await runConnector(d.connectorId, { task: d.task, group: d.group, limit: d.limit, force: d.force, urls: d.urls });
135 + accountRun(r);
136 + if (r.status === "failed") throw new Error(r.error ?? "run failed");
137 + return { runId: r.runId, status: r.status, stats: r.stats };
138 + } finally {
139 + running.delete(String(job.id));
140 + // release only our own lock (a crashed predecessor's lock expires via TTL)
141 + await connection.eval("if redis.call('get', KEYS[1]) == ARGV[1] then return redis.call('del', KEYS[1]) else return 0 end", 1, lockKey, lockToken).catch(() => undefined);
151 142 }
152 − } finally { running.delete(String(job.id)); }
153 − },
154 − { connection, prefix: QUEUE_PREFIX, concurrency: env.maintenanceConcurrency, lockDuration: 600_000 },
155 − );
156 − for (const w of [crawlWorker, maintWorker]) {
157 − w.on("completed", (job) => { counters.jobs_completed!++; console.error(`[worker] ${w.name} ${job.name} #${job.id} completed`); });
158 − w.on("failed", (job, err) => { counters.jobs_failed!++; console.error(`[worker] ${w.name} ${job?.name ?? "?"} #${job?.id ?? "?"} failed: ${err.message}`); });
143 + },
144 + { connection, prefix: QUEUE_PREFIX, concurrency: env.crawlConcurrency, lockDuration: 120_000, stalledInterval: 60_000, maxStalledCount: 1 },
145 + );
146 + workers.push(crawlWorker as Worker);
147 + }
148 + if (env.queues.includes("maintenance")) {
149 + const maintWorker = new Worker<MaintenanceJobData>(
150 + MAINTENANCE_QUEUE,
151 + async (job: Job<MaintenanceJobData>) => {
152 + running.set(String(job.id), { id: String(job.id), name: job.name, task: job.data.kind, since: new Date().toISOString() });
153 + try {
154 + switch (job.data.kind) {
155 + case "rankings": return await computeRankings();
156 + case "metrics": return await computeDailyMetrics();
157 + case "refresh-stats": return await refreshStats();
158 + case "cleanup": return await cleanupMaintenance();
159 + default: throw new Error(`unknown maintenance kind ${String((job.data as { kind?: string }).kind)}`);
160 + }
161 + } finally { running.delete(String(job.id)); }
162 + },
163 + { connection, prefix: QUEUE_PREFIX, concurrency: env.maintenanceConcurrency, lockDuration: 600_000, stalledInterval: 60_000, maxStalledCount: 1 },
164 + );
165 + workers.push(maintWorker as Worker);
166 + }
167 + for (const w of workers) {
168 + w.on("completed", (job) => { counters.jobs_completed!++; metrics.jobs.inc({ queue: w.name, result: "completed" }); console.error(`[worker] ${w.name} ${job.name} #${job.id} completed`); });
169 + w.on("failed", (job, err) => { if (err instanceof DelayedError) return; counters.jobs_failed!++; metrics.jobs.inc({ queue: w.name, result: "failed" }); console.error(`[worker] ${w.name} ${job?.name ?? "?"} #${job?.id ?? "?"} failed: ${err.message}`); });
170 + w.on("stalled", (jobId) => { metrics.jobs.inc({ queue: w.name, result: "stalled" }); console.error(`[worker] ${w.name} #${jobId} stalled — BullMQ will retry it once`); });
159 171 w.on("error", (err) => console.error(`[worker] ${w.name} error: ${err.message}`));
160 172 }
161 173
162 − const scheduler = opts.withScheduler === false ? null : startSchedulerLoop({
174 + const withScheduler = opts.withScheduler ?? env.embeddedScheduler;
175 + const scheduler = !withScheduler ? null : startSchedulerLoop({
163 176 onTick: (r) => { lastScheduler = { at: new Date().toISOString(), enqueued: r.enqueued.length, checked: r.checked }; if (r.enqueued.length) console.error(`[scheduler] enqueued ${r.enqueued.join(", ")}`); },
164 177 onError: (e) => console.error(`[scheduler] tick failed: ${e.message}`),
165 178 });
166 − const http = startHttp(env.workerPort);
179 + const http = startHealthHttp({ port: env.workerPort, role: "worker", health: async () => ({ ...(await status()), queueDepth: await queueSnapshot().catch(() => []) }), ready: () => !shuttingDown, beforeScrape: refreshGauges, log: (m) => console.error(`[worker] ${m}`) });
167 180 await heartbeat();
168 181 const hb = setInterval(() => void heartbeat(), 15_000);
169 − console.error(`[worker] ready · crawl concurrency ${env.crawlConcurrency} · ${loaded.length} connectors · host ${env.hostname}`);
182 + console.error(`[worker] ready · queues ${env.queues.join(",")} · crawl concurrency ${env.crawlConcurrency} · scheduler ${withScheduler ? "embedded" : "external"} · ${loaded.length} connectors · host ${env.hostname}`);
170 183
171 184 const stop = async () => {
172 185 if (shuttingDown) return;
173 186 shuttingDown = true;
174 187 requestAbort("SIGTERM");
175 − console.error("[worker] shutting down: finishing in-flight documents…");
188 + console.error(`[worker] shutting down: finishing in-flight documents (${running.size} job(s), deadline ${env.shutdownTimeoutMs} ms)…`);
176 189 scheduler?.stop();
177 190 clearInterval(hb);
178 − await Promise.allSettled([crawlWorker.close(), maintWorker.close()]);
179 − await heartbeat();
180 − http.close();
181 − await closeScheduler();
191 + const deadline = new Promise<"timeout">((resolve) => setTimeout(() => resolve("timeout"), env.shutdownTimeoutMs).unref());
192 + const closed = Promise.allSettled(workers.map((w) => w.close())).then(() => "closed" as const);
193 + const outcome = await Promise.race([closed, deadline]);
194 + if (outcome === "timeout") {
195 + const n = await Promise.race([abortActiveRunsInDb("shutdown deadline reached").catch(() => 0), new Promise<number>((r) => setTimeout(() => r(-1), 5_000).unref())]);
196 + console.error(`[worker] shutdown deadline reached — ${n < 0 ? "could not mark" : n} in-flight run(s) marked aborted, forcing exit`);
197 + void Promise.allSettled(workers.map((w) => w.close(true)));
198 + }
199 + // cleanup must never keep a stopping process alive: a query or a blocked Redis command still in flight
200 + // (postgres.js `end()` waits for active queries) is cut by the 5 s race — the caller exits right after.
201 + const cleanup = async () => {
202 + await heartbeat();
203 + http.close();
204 + uninstallBudgetStore();
205 + await closeScheduler();
206 + await closeDb().catch(() => undefined);
207 + };
208 + const done = await Promise.race([cleanup().then(() => "clean" as const), new Promise<"cut">((r) => setTimeout(() => r("cut"), 5_000).unref())]);
209 + console.error(`[worker] stopped (${done === "clean" ? "clean" : "cleanup cut short after 5 s"})`);
182 210 };
183 211 return { stop };
184 212 }
185 213
186 214 const isMain = process.argv[1] && /main\.(ts|js)$/.test(process.argv[1]);
187 215 if (isMain) {
216 + process.on("unhandledRejection", (e) => console.error(`[worker] unhandled rejection: ${(e as Error)?.stack ?? String(e)}`));
217 + process.on("uncaughtException", (e) => { console.error(`[worker] uncaught exception: ${e.stack ?? e.message}`); });
188 218 startWorker()
189 219 .then(({ stop }) => {
190 − const onSignal = (sig: string) => { console.error(`[worker] ${sig}`); void stop().then(() => process.exit(isAbortRequested() ? 0 : 0)); };
191 − process.on("SIGTERM", () => onSignal("SIGTERM"));
192 − process.on("SIGINT", () => onSignal("SIGINT"));
220 + const onSignal = (sig: string) => {
221 + console.error(`[worker] ${sig}`);
222 + // absolute ceiling: graceful deadline + 10 s of cleanup, still below compose's stop_grace_period (90 s)
223 + setTimeout(() => { console.error("[worker] hard exit (shutdown did not complete in time)"); process.exit(1); }, getEnv().shutdownTimeoutMs + 10_000).unref();
224 + void stop().then(() => process.exit(isAbortRequested() ? 0 : 0), (e: Error) => { console.error(`[worker] stop failed: ${e.message}`); process.exit(1); });
225 + };
226 + process.once("SIGTERM", () => onSignal("SIGTERM"));
227 + process.once("SIGINT", () => onSignal("SIGINT"));
193 228 })
194 229 .catch((e) => { console.error(`[worker] fatal: ${(e as Error).stack ?? (e as Error).message}`); process.exit(1); });
195 230 }
modified apps/worker/src/pipeline.ts +25 −2
@@ -10,12 +10,13 @@ import type { DiscoveredUrl, RawDocument } from "@dci/connectors";
10 10 import { entityIsValid, pageTitle } from "@dci/connectors";
11 11 import type { NormalizedEntity, ValidationIssue } from "@dci/core";
12 12 import { contentFingerprint, sha256, newId } from "@dci/core";
13 −import { getDb, connectorRuns, connectors as connectorsTable, eq } from "@dci/db";
13 +import { getDb, connectorRuns, connectors as connectorsTable, eq, inArray } from "@dci/db";
14 14 import { loadAllConnectors, requireConnector, type LoadedConnector } from "./configs.js";
15 15 import { createRunContext, type LogLevel, type LogLine, type RunContext } from "./context.js";
16 16 import { connectorDocStats, documentIdFor, dueDocuments, nextCheckByGroup, recordFetch, recordVersion, registerDiscovered, textProjection, updateExtraction, type DocumentRow } from "./documents.js";
17 17 import { ingestEntities } from "./ingest/index.js";
18 18 import type { IngestDocRef, IngestRun, IngestStats } from "./ingest/contract.js";
19 +import { metrics } from "./prom.js";
19 20 import { healthFrom, minIntervalMs, parseDiscoveredFrom, shouldSkipExtraction } from "./scheduling.js";
20 21 import { getRaw, putRaw } from "./storage.js";
21 22
@@ -93,6 +94,16 @@ export function abortReasonText(): string | null { return abortReason; }
93 94 export function activeRunIds(): string[] { return [...activeRuns]; }
94 95 /** For tests / long-lived processes that want to resume after a handled abort. */
95 96 export function clearAbort(): void { abortReason = null; }
97 +/**
98 + * Forced shutdown (graceful deadline reached): mark the runs still in flight as aborted in Postgres so they do not
99 + * linger as `running` until the orphan reconciliation. Best effort; returns the number of runs updated.
100 + */
101 +export async function abortActiveRunsInDb(reason: string): Promise<number> {
102 + const ids = [...activeRuns];
103 + if (!ids.length) return 0;
104 + await getDb().update(connectorRuns).set({ status: "aborted", finishedAt: new Date().toISOString(), error: `aborted: ${reason}`.slice(0, 2000) }).where(inArray(connectorRuns.id, ids));
105 + return ids.length;
106 +}
96 107
97 108 /* ---------- inline p-limit ---------- */
98 109 export function pLimit(concurrency: number): <T>(fn: () => Promise<T>) => Promise<T> {
@@ -149,11 +160,19 @@ async function extractAndIngest(ctx: RunContext, loaded: LoadedConnector, doc: D
149 160 for (const e of valid) if (result.samples.length < sampleCap) result.samples.push(e);
150 161 const pageType = (raw.meta?.pageType as string | undefined) ?? doc.pageType;
151 162 const classifier = (raw.meta?.classifier as string | undefined) ?? null;
163 + metrics.ingest.inc({ connector: loaded.cfg.id, result: "rejected" }, report.rejected);
152 164 if (!valid.length) return { ingest: null, valid: 0, rejected: report.rejected, classifier, pageType };
153 165 const run: IngestRun = { connectorId: loaded.cfg.id, sourceId: loaded.sourceId, runId: ctx.runId, sourceKind: loaded.cfg.kind, sourcePriority: loaded.cfg.priority, dryRun: Boolean(opts.dryRun) };
154 166 const ref: IngestDocRef = { documentId: doc.id, url: raw.finalUrl || doc.url, pageType, fetchedAt: raw.fetchedAt };
155 167 const ingest = await ingestEntities(run, valid, ref);
156 168 sumIngest(stats, ingest);
169 + if (!opts.dryRun) {
170 + metrics.ingest.inc({ connector: loaded.cfg.id, result: "created" }, ingest.created);
171 + metrics.ingest.inc({ connector: loaded.cfg.id, result: "updated" }, ingest.updated);
172 + metrics.ingest.inc({ connector: loaded.cfg.id, result: "merged" }, ingest.merged);
173 + metrics.ingest.inc({ connector: loaded.cfg.id, result: "unchanged" }, Math.max(0, valid.length - ingest.created - ingest.updated - ingest.merged));
174 + metrics.events.inc(undefined, ingest.events);
175 + }
157 176 return { ingest, valid: valid.length, rejected: report.rejected, classifier, pageType };
158 177 }
159 178
@@ -165,8 +184,10 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do
165 184 const force = Boolean(opts.force);
166 185 try {
167 186 if (!force && (doc.etag || doc.lastModified)) ctx.validators.set(doc.url, { etag: doc.etag, lastModified: doc.lastModified });
187 + ctx.hints.set(doc.url, { errorCount: doc.errorCount ?? 0 });
168 188 const raw = await connector.fetch(ctx, docToDiscovered(doc));
169 189 ctx.validators.delete(doc.url);
190 + ctx.hints.delete(doc.url);
170 191 stats.fetched++;
171 192
172 193 if (raw.error || raw.status >= 400) {
@@ -237,6 +258,7 @@ async function processDocument(ctx: RunContext, loaded: LoadedConnector, doc: Do
237 258 return { kind: changed ? "changed" : "fetched", ms: Date.now() - t0 };
238 259 } catch (e) {
239 260 ctx.validators.delete(doc.url);
261 + ctx.hints.delete(doc.url);
240 262 stats.failed++;
241 263 ctx.log("error", `${doc.url} unexpected: ${(e as Error).stack ?? (e as Error).message}`);
242 264 try {
@@ -350,7 +372,8 @@ export async function runConnector(connectorId: string, opts: RunOptions): Promi
350 372 else if (stats.failed > 0 || stats.rejected > 0) result.status = "partial";
351 373 else result.status = "ok";
352 374 result.error = fatal ? fatal.slice(0, 2000) : isAbortRequested() ? `aborted: ${abortReasonText()}` : null;
353 − ctx.log("info", `run ${result.status} in ${result.durationMs}ms · fetched=${stats.fetched} 304=${stats.notModified} changed=${stats.changed} unchanged=${stats.unchanged} failed=${stats.failed} entities=${stats.valid}/${stats.entities} created=${stats.created} updated=${stats.updated} events=${stats.events} credits=${stats.credits}`);
375 + if (!dryRun) metrics.runs.inc({ status: result.status });
376 + ctx.log("info", `run ${result.status} in ${result.durationMs}ms · fetched=${stats.fetched} 304=${stats.notModified} changed=${stats.changed} unchanged=${stats.unchanged} failed=${stats.failed} entities=${stats.valid}/${stats.entities} created=${stats.created} updated=${stats.updated} events=${stats.events} credits=${stats.credits}${ctx.budgetCapped ? ` budgetCapped=${ctx.budgetCapped}` : ""}`);
354 377
355 378 if (!dryRun) {
356 379 try {
added apps/worker/src/prom.test.ts +104 −0
@@ -0,0 +1,104 @@
1 +import { afterEach, describe, expect, it } from "vitest";
2 +import { capDiscovered } from "@dci/connectors";
3 +import type { DiscoveredUrl } from "@dci/connectors";
4 +import { fetchOutcome, metrics, renderMetrics, resetMetrics } from "./prom.js";
5 +import { parseQueues } from "./env.js";
6 +import { isDiscoveryGroup, premiumAllowedAfterErrors } from "./scheduling.js";
7 +import { discoveryDue } from "./scheduler.js";
8 +import { budgetKey, connectorBudgetKey } from "./budget.js";
9 +
10 +afterEach(() => resetMetrics());
11 +
12 +describe("prometheus registry", () => {
13 + it("renders counters, gauges and histograms in exposition format", () => {
14 + metrics.fetches.inc({ connector: "equinix", level: "L1", outcome: "ok" });
15 + metrics.fetches.inc({ connector: "equinix", level: "L1", outcome: "ok" });
16 + metrics.credits.inc({ provider: "scrapfly" }, 5.5);
17 + metrics.dailyBudget.set(400, { provider: "scrapfly" });
18 + metrics.fetchDuration.observe(0.3);
19 + metrics.fetchDuration.observe(7);
20 + metrics.events.inc(undefined, 3);
21 + const out = renderMetrics();
22 + expect(out).toContain('dci_crawl_fetches_total{connector="equinix",level="L1",outcome="ok"} 2');
23 + expect(out).toContain('dci_crawl_credits_total{provider="scrapfly"} 5.5');
24 + expect(out).toContain('dci_crawl_daily_budget{provider="scrapfly"} 400');
25 + expect(out).toContain('dci_crawl_fetch_duration_seconds_bucket{le="0.5"} 1');
26 + expect(out).toContain('dci_crawl_fetch_duration_seconds_bucket{le="+Inf"} 2');
27 + expect(out).toContain("dci_crawl_fetch_duration_seconds_count 2");
28 + expect(out).toContain("dci_events_total 3");
29 + expect(out).toContain("# TYPE dci_queue_jobs gauge");
30 + expect(out.endsWith("\n")).toBe(true);
31 + });
32 + it("escapes label values and ignores zero increments", () => {
33 + metrics.ingest.inc({ connector: 'a"b\\c', result: "created" }, 0);
34 + metrics.ingest.inc({ connector: 'a"b\\c', result: "created" }, 2);
35 + expect(renderMetrics()).toContain('dci_ingest_entities_total{connector="a\\"b\\\\c",result="created"} 2');
36 + });
37 + it("classifies fetch outcomes", () => {
38 + expect(fetchOutcome({ notModified: true, status: 304 })).toBe("not_modified");
39 + expect(fetchOutcome({ notModified: false, status: 200 })).toBe("ok");
40 + expect(fetchOutcome({ notModified: false, status: 403 })).toBe("blocked");
41 + expect(fetchOutcome({ notModified: false, status: 500 })).toBe("error");
42 + expect(fetchOutcome({ notModified: false, status: 0, error: { code: "robots_disallow" } })).toBe("blocked");
43 + expect(fetchOutcome({ notModified: false, status: 0, error: { code: "timeout" } })).toBe("error");
44 + });
45 +});
46 +
47 +describe("env / budgets", () => {
48 + it("parses DCI_QUEUES", () => {
49 + expect(parseQueues(undefined)).toEqual(["crawl", "maintenance"]);
50 + expect(parseQueues("maintenance")).toEqual(["maintenance"]);
51 + expect(parseQueues(" Crawl , bogus ")).toEqual(["crawl"]);
52 + expect(parseQueues("bogus")).toEqual(["crawl", "maintenance"]);
53 + });
54 + it("builds the Redis budget keys the API reads", () => {
55 + expect(budgetKey("scrapfly", "2026-09-11")).toBe("dci:budget:scrapfly:2026-09-11");
56 + expect(connectorBudgetKey("equinix", "2026-09-11")).toBe("dci:budget:connector:equinix:2026-09-11");
57 + });
58 +});
59 +
60 +describe("cost control rules", () => {
61 + it("discovery groups are direct-only", () => {
62 + expect(isDiscoveryGroup("sitemap")).toBe(true);
63 + expect(isDiscoveryGroup("rss")).toBe(true);
64 + expect(isDiscoveryGroup("facility_pages")).toBe(false);
65 + expect(isDiscoveryGroup(undefined)).toBe(false);
66 + });
67 + it("premium retries are gated after two consecutive failures", () => {
68 + expect(premiumAllowedAfterErrors(0)).toBe(true);
69 + expect(premiumAllowedAfterErrors(1)).toBe(true);
70 + expect(premiumAllowedAfterErrors(2)).toBe(false);
71 + expect(premiumAllowedAfterErrors(3)).toBe(false);
72 + expect(premiumAllowedAfterErrors(4)).toBe(true);
73 + expect(premiumAllowedAfterErrors(5)).toBe(false);
74 + expect(premiumAllowedAfterErrors(8)).toBe(true);
75 + });
76 +});
77 +
78 +describe("scheduler", () => {
79 + const now = new Date("2026-09-11T12:00:00Z");
80 + const H = 3_600_000;
81 + it("a never-discovered connector is due (full), one that was just discovered with zero docs is not", () => {
82 + expect(discoveryDue({ docCount: 0, lastDiscoverAt: null }, 6 * H, now)).toEqual({ due: true, task: "full" });
83 + expect(discoveryDue({ docCount: 0, lastDiscoverAt: new Date(now.getTime() - 5 * H).toISOString() }, 6 * H, now)).toEqual({ due: false, task: "full" });
84 + expect(discoveryDue({ docCount: 0, lastDiscoverAt: new Date(now.getTime() - 7 * H).toISOString() }, 6 * H, now)).toEqual({ due: true, task: "full" });
85 + });
86 + it("a populated connector re-discovers on cadence", () => {
87 + expect(discoveryDue({ docCount: 10, lastDiscoverAt: new Date(now.getTime() - 5 * H).toISOString() }, 6 * H, now)).toEqual({ due: false, task: "discover" });
88 + expect(discoveryDue({ docCount: 10, lastDiscoverAt: new Date(now.getTime() - 6 * H).toISOString() }, 6 * H, now)).toEqual({ due: true, task: "discover" });
89 + });
90 +});
91 +
92 +describe("discovery cap", () => {
93 + const u = (url: string, group: string, priority: number): DiscoveredUrl => ({ url, group, priority });
94 + it("keeps the highest-priority groups instead of the first N sitemap entries", () => {
95 + const urls = [u("/blog/1", "default", 30), u("/blog/2", "default", 30), u("/dc/a", "facility_pages", 60), u("/news/x", "newsroom", 50), u("/dc/b", "facility_pages", 60)];
96 + expect(capDiscovered(urls, 3).map((x) => x.url)).toEqual(["/dc/a", "/dc/b", "/news/x"]);
97 + expect(capDiscovered(urls, 10)).toHaveLength(5);
98 + expect(capDiscovered(urls, 0)).toHaveLength(5);
99 + });
100 + it("is stable for equal priorities", () => {
101 + const urls = [u("/a", "default", 30), u("/b", "default", 30), u("/c", "default", 30)];
102 + expect(capDiscovered(urls, 2).map((x) => x.url)).toEqual(["/a", "/b"]);
103 + });
104 +});
modified apps/worker/src/scheduler.ts +115 −17
@@ -5,34 +5,53 @@
5 5 *
6 6 * The scheduler loop (every DCI_SCHEDULER_INTERVAL_MS, default 60 s) enqueues one crawl job per connector group
7 7 * with due documents (jobId `<connector>__<group>` dedupes), a `discover` job when the connector's discovery
8 − * cadence is due, and keeps the repeatable maintenance jobs registered (daily metrics 00:10 UTC, rankings
9 − * 00:30 UTC, refresh-stats hourly, cleanup 01:00 UTC). connectors.enabled / paused are honored.
8 + * cadence is due (a `full` job for a connector that was never discovered), and keeps the repeatable maintenance
9 + * jobs registered (daily metrics 00:10 UTC, rankings 00:30 UTC, refresh-stats hourly, cleanup 01:00 UTC).
10 + * connectors.enabled / paused are honored. While a `full` / `discover` job of a connector is pending, no per-group
11 + * job is enqueued for it (the full job crawls everything it registers; two runs on one connector would double
12 + * fetches and record phantom "changes").
10 13 *
11 14 * BullMQ forbids ":" in queue names and custom job ids, hence the `dci` prefix + `__` separators.
15 + *
16 + * Run as a process (`dci-entrypoint scheduler`, `pnpm --filter @dci/worker scheduler`): the loop plus HTTP
17 + * /healthz + /metrics on WORKER_PORT. The crawl worker also runs the loop unless DCI_SCHEDULER=0; the Redis lock
18 + * `dci:scheduler:lock` makes concurrent loops harmless.
12 19 */
13 20 import { Queue, type JobsOptions } from "bullmq";
14 21 import { Redis } from "ioredis";
15 22 import { intervalMs } from "@dci/connectors";
16 −import { getDb, connectors as connectorsTable, sql } from "@dci/db";
17 −import { getEnv } from "./env.js";
23 +import { getDb, closeDb, connectors as connectorsTable, sql } from "@dci/db";
24 +import { getEnv, loadEnvFile } from "./env.js";
18 25 import { nextCheckByGroup } from "./documents.js";
26 +import { startHealthHttp } from "./health-http.js";
19 27 import type { RunTask } from "./pipeline.js";
28 +import { metrics } from "./prom.js";
20 29 import { minIntervalMs } from "./scheduling.js";
21 30
22 31 export const QUEUE_PREFIX = "dci";
23 32 export const CRAWL_QUEUE = "crawl";
24 33 export const MAINTENANCE_QUEUE = "maintenance";
25 34 export const WORKER_STATUS_KEY = "dci:worker:status";
35 +export const SCHEDULER_STATUS_KEY = "dci:scheduler:status";
26 36 const SCHEDULER_LOCK_KEY = "dci:scheduler:lock";
37 +/** Redis lock held by a worker while it runs a connector — one run per connector at a time across all workers. */
38 +export const RUN_LOCK_PREFIX = "dci:run-lock";
39 +export const RUN_LOCK_TTL_SECONDS = 4 * 3600;
27 40
28 41 export interface CrawlJobData { connectorId: string; task: RunTask; group?: string; limit?: number; force?: boolean; urls?: string[]; requestedBy?: string }
29 42 export type MaintenanceKind = "rankings" | "metrics" | "refresh-stats" | "cleanup";
30 43 export interface MaintenanceJobData { kind: MaintenanceKind; requestedBy?: string }
31 44
32 45 let _redis: Redis | null = null;
33 −/** Shared ioredis connection for BullMQ (maxRetriesPerRequest must be null for BullMQ). */
46 +/** Shared ioredis connection for BullMQ (maxRetriesPerRequest must be null for BullMQ). Reconnects on its own. */
34 47 export function getRedis(): Redis {
35 − if (!_redis) _redis = new Redis(getEnv().redisUrl, { maxRetriesPerRequest: null, enableReadyCheck: false, lazyConnect: false });
48 + if (!_redis) {
49 + _redis = new Redis(getEnv().redisUrl, { maxRetriesPerRequest: null, enableReadyCheck: false, lazyConnect: false, retryStrategy: (times) => Math.min(30_000, 200 * 2 ** Math.min(times, 8)) });
50 + let down = false;
51 + // an 'error' event without listener would be printed as an unhandled error by ioredis on every retry
52 + _redis.on("error", (e) => { if (!down) console.error(`[redis] ${e.message} — reconnecting`); down = true; });
53 + _redis.on("ready", () => { if (down) console.error("[redis] connection restored"); down = false; });
54 + }
36 55 return _redis;
37 56 }
38 57
@@ -51,6 +70,9 @@ export function crawlJobId(connectorId: string, task: RunTask, group?: string):
51 70 const clean = (s: string) => s.replace(/[^a-zA-Z0-9_-]/g, "-");
52 71 return task === "discover" ? `${clean(connectorId)}__discover` : `${clean(connectorId)}__${clean(group ?? "all")}`;
53 72 }
73 +export function runLockKey(connectorId: string): string { return `${RUN_LOCK_PREFIX}:${connectorId}`; }
74 +
75 +const PENDING_STATES = new Set(["waiting", "active", "delayed", "prioritized", "waiting-children"]);
54 76
55 77 /** Enqueue a crawl job (used by scheduler loop, CLI and API). Returns the job id; duplicates are coalesced. */
56 78 export async function enqueueRun(data: CrawlJobData, opts: { priority?: number; jobId?: string; delayMs?: number } = {}): Promise<{ jobId: string; queued: boolean }> {
@@ -59,7 +81,7 @@ export async function enqueueRun(data: CrawlJobData, opts: { priority?: number;
59 81 const existing = await q.getJob(jobId);
60 82 if (existing) {
61 83 const state = await existing.getState();
62 − if (state === "waiting" || state === "active" || state === "delayed" || state === "prioritized" || state === "waiting-children") return { jobId, queued: false };
84 + if (PENDING_STATES.has(state)) return { jobId, queued: false };
63 85 await existing.remove().catch(() => undefined);
64 86 }
65 87 const jobOpts: JobsOptions = { jobId, priority: opts.priority ?? 10 };
@@ -68,6 +90,16 @@ export async function enqueueRun(data: CrawlJobData, opts: { priority?: number;
68 90 return { jobId, queued: true };
69 91 }
70 92
93 +/** True when a whole-connector job (`full` = `<id>__all`, or `<id>__discover`) is pending for this connector. */
94 +async function wholeConnectorJobPending(connectorId: string): Promise<boolean> {
95 + const q = crawlQueue();
96 + for (const id of [crawlJobId(connectorId, "full"), crawlJobId(connectorId, "discover")]) {
97 + const job = await q.getJob(id);
98 + if (job && PENDING_STATES.has(await job.getState())) return true;
99 + }
100 + return false;
101 +}
102 +
71 103 export async function enqueueMaintenance(kind: MaintenanceKind, requestedBy = "cli"): Promise<string> {
72 104 const jobId = `manual__${kind}__${Date.now().toString(36)}`;
73 105 await maintenanceQueue().add(kind, { kind, requestedBy }, { jobId });
@@ -103,27 +135,39 @@ export function discoveryIntervalMs(schedule: Record<string, string>): number {
103 135 return Math.max(6 * 3_600_000, minIntervalMs(schedule));
104 136 }
105 137
106 −/** One scheduler pass. Uses a short Redis lock so several worker processes do not double-enqueue. */
138 +/**
139 + * Is a discovery pass due? `lastDiscoverAt` is authoritative: a connector whose discovery legitimately yields zero
140 + * URLs (blocked feed, filter matching nothing) must wait for the next cadence like any other, not run every tick.
141 + */
142 +export function discoveryDue(c: { docCount: number; lastDiscoverAt: string | null }, discEveryMs: number, now: Date): { due: boolean; task: RunTask } {
143 + const task: RunTask = c.docCount === 0 ? "full" : "discover";
144 + const last = c.lastDiscoverAt ? Date.parse(c.lastDiscoverAt) : NaN;
145 + if (!Number.isFinite(last)) return { due: true, task };
146 + return { due: now.getTime() - last >= discEveryMs, task };
147 +}
148 +
149 +/** One scheduler pass. Uses a short Redis lock so several processes do not double-enqueue. */
107 150 export async function schedulerTick(opts: { force?: boolean; now?: Date } = {}): Promise<TickResult> {
108 151 const res: TickResult = { checked: 0, enqueued: [], skipped: [], lockHeldElsewhere: false };
109 152 const redis = getRedis();
110 153 const now = opts.now ?? new Date();
111 154 const token = `${process.pid}:${now.getTime()}`;
112 155 const got = await redis.set(SCHEDULER_LOCK_KEY, token, "EX", 55, "NX");
113 − if (!got && !opts.force) { res.lockHeldElsewhere = true; return res; }
156 + if (!got && !opts.force) { res.lockHeldElsewhere = true; metrics.schedulerTicks.inc({ result: "lock_held" }); return res; }
114 157 try {
115 158 const rows = await schedulableConnectors();
116 159 for (const c of rows) {
117 160 res.checked++;
118 161 if (!c.enabled || c.paused) { res.skipped.push(`${c.id}:${!c.enabled ? "disabled" : "paused"}`); continue; }
119 162 // discovery (or first ever run) when the discovery cadence elapsed
120 − const discEvery = discoveryIntervalMs(c.schedule);
121 − const lastDisc = c.lastDiscoverAt ? Date.parse(c.lastDiscoverAt) : 0;
122 − if (c.docCount === 0 || now.getTime() - lastDisc >= discEvery) {
123 − const r = await enqueueRun({ connectorId: c.id, task: c.docCount === 0 ? "full" : "discover", requestedBy: "scheduler" }, { priority: c.docCount === 0 ? 5 : 20 });
163 + const disc = discoveryDue(c, discoveryIntervalMs(c.schedule), now);
164 + if (disc.due) {
165 + const r = await enqueueRun({ connectorId: c.id, task: disc.task, requestedBy: "scheduler" }, { priority: disc.task === "full" ? 5 : 20 });
124 166 if (r.queued) res.enqueued.push(r.jobId);
125 167 }
126 − // per-group crawl jobs for groups with due documents
168 + // per-group crawl jobs for groups with due documents — unless a whole-connector job is pending
169 + if (c.docCount === 0) continue;
170 + if (await wholeConnectorJobPending(c.id)) { res.skipped.push(`${c.id}:full-pending`); continue; }
127 171 const groups = await nextCheckByGroup(c.id);
128 172 for (const g of groups) {
129 173 if (g.due <= 0) continue;
@@ -132,9 +176,15 @@ export async function schedulerTick(opts: { force?: boolean; now?: Date } = {}):
132 176 }
133 177 }
134 178 await ensureMaintenanceSchedulers();
179 + metrics.schedulerTicks.inc({ result: "ok" });
180 + metrics.schedulerEnqueued.inc(undefined, res.enqueued.length);
181 + metrics.schedulerLastTick.set(Date.now() / 1000);
182 + } catch (e) {
183 + metrics.schedulerTicks.inc({ result: "error" });
184 + throw e;
135 185 } finally {
136 − const cur = await redis.get(SCHEDULER_LOCK_KEY);
137 − if (cur === token) await redis.del(SCHEDULER_LOCK_KEY);
186 + const cur = await redis.get(SCHEDULER_LOCK_KEY).catch(() => null);
187 + if (cur === token) await redis.del(SCHEDULER_LOCK_KEY).catch(() => undefined);
138 188 }
139 189 return res;
140 190 }
@@ -154,15 +204,24 @@ export function startSchedulerLoop(opts: { intervalMs?: number; onTick?: (r: Tic
154 204 }
155 205
156 206 export interface QueueSnapshot { name: string; waiting: number; active: number; delayed: number; failed: number; completed: number; prioritized: number }
207 +export const QUEUE_STATES = ["waiting", "active", "delayed", "failed", "completed", "prioritized"] as const;
157 208 export async function queueSnapshot(): Promise<QueueSnapshot[]> {
158 209 const out: QueueSnapshot[] = [];
159 210 for (const q of [crawlQueue(), maintenanceQueue()]) {
160 − const c = await q.getJobCounts("waiting", "active", "delayed", "failed", "completed", "prioritized");
211 + const c = await q.getJobCounts(...QUEUE_STATES);
161 212 out.push({ name: q.name, waiting: c.waiting ?? 0, active: c.active ?? 0, delayed: c.delayed ?? 0, failed: c.failed ?? 0, completed: c.completed ?? 0, prioritized: c.prioritized ?? 0 });
162 213 }
163 214 return out;
164 215 }
165 216
217 +/** Publish queue depths into the `dci_queue_jobs{queue,state}` gauge (called before every /metrics scrape). */
218 +export async function publishQueueMetrics(): Promise<QueueSnapshot[]> {
219 + const snap = await queueSnapshot();
220 + metrics.queueJobs.reset();
221 + for (const q of snap) for (const s of QUEUE_STATES) metrics.queueJobs.set(q[s], { queue: q.name, state: s });
222 + return snap;
223 +}
224 +
166 225 export async function pauseConnector(id: string, paused: boolean): Promise<void> {
167 226 await getDb().update(connectorsTable).set({ paused, updatedAt: new Date().toISOString() }).where(sql`${connectorsTable.id} = ${id}`);
168 227 }
@@ -173,3 +232,42 @@ export async function closeScheduler(): Promise<void> {
173 232 _crawlQueue = _maintQueue = null;
174 233 if (_redis) { await _redis.quit().catch(() => undefined); _redis = null; }
175 234 }
235 +
236 +/* ---------- standalone scheduler process ---------- */
237 +export async function startSchedulerProcess(): Promise<{ stop: () => Promise<void> }> {
238 + loadEnvFile();
239 + const env = getEnv();
240 + const startedAt = new Date().toISOString();
241 + let last: { at: string; checked: number; enqueued: string[]; skipped: number; lockHeldElsewhere: boolean; error: string | null } | null = null;
242 + let ticks = 0;
243 + let stopping = false;
244 + const loop = startSchedulerLoop({
245 + onTick: (r) => { ticks++; last = { at: new Date().toISOString(), checked: r.checked, enqueued: r.enqueued, skipped: r.skipped.length, lockHeldElsewhere: r.lockHeldElsewhere, error: null }; if (r.enqueued.length) console.error(`[scheduler] enqueued ${r.enqueued.join(", ")}`); },
246 + onError: (e) => { last = { at: new Date().toISOString(), checked: 0, enqueued: [], skipped: 0, lockHeldElsewhere: false, error: e.message }; console.error(`[scheduler] tick failed: ${e.message}`); },
247 + });
248 + const health = async () => ({ startedAt, pid: process.pid, host: env.hostname, intervalMs: env.schedulerIntervalMs, ticks, lastTick: last, queues: await queueSnapshot().catch((e: Error) => ({ error: e.message })) });
249 + const http = startHealthHttp({ port: env.workerPort, role: "scheduler", health, ready: () => !stopping && (last === null || last.error === null || Date.now() - Date.parse(last.at) < 3 * env.schedulerIntervalMs), beforeScrape: async () => { metrics.up.set(stopping ? 0 : 1); metrics.uptime.set((Date.now() - Date.parse(startedAt)) / 1000); await publishQueueMetrics(); } });
250 + const hb = setInterval(() => void getRedis().set(SCHEDULER_STATUS_KEY, JSON.stringify({ at: new Date().toISOString(), pid: process.pid, host: env.hostname, ticks, last }), "EX", 180).catch(() => undefined), 15_000);
251 + console.error(`[scheduler] loop every ${env.schedulerIntervalMs} ms · host ${env.hostname}`);
252 + return {
253 + stop: async () => {
254 + stopping = true;
255 + loop.stop();
256 + clearInterval(hb);
257 + http.close();
258 + await closeScheduler();
259 + await closeDb().catch(() => undefined);
260 + },
261 + };
262 +}
263 +
264 +const isMain = process.argv[1] && /scheduler\.(ts|js)$/.test(process.argv[1]);
265 +if (isMain) {
266 + startSchedulerProcess()
267 + .then(({ stop }) => {
268 + const onSignal = (sig: string) => { console.error(`[scheduler] ${sig}`); void stop().then(() => process.exit(0)); setTimeout(() => process.exit(0), 10_000).unref(); };
269 + process.on("SIGTERM", () => onSignal("SIGTERM"));
270 + process.on("SIGINT", () => onSignal("SIGINT"));
271 + })
272 + .catch((e) => { console.error(`[scheduler] fatal: ${(e as Error).stack ?? (e as Error).message}`); process.exit(1); });
273 +}
modified apps/worker/src/scheduling.ts +14 −0
@@ -141,3 +141,17 @@ export function maxLevelForBudget(cfgMaxLevel: number, creditsSpent: number, max
141 141 if (lvl <= 2) return lvl;
142 142 return creditsSpent >= maxCreditsPerRun ? 2 : lvl;
143 143 }
144 +
145 +/** Discovery fetches (sitemaps, RSS feeds) never use premium fetchers: direct transport only (L1, L2 = browser identity). */
146 +export const DISCOVERY_GROUPS: ReadonlySet<string> = new Set(["sitemap", "rss"]);
147 +export function isDiscoveryGroup(group: string | undefined): boolean { return group !== undefined && DISCOVERY_GROUPS.has(group); }
148 +
149 +/**
150 + * Premium escalation for a document that keeps failing: after two consecutive failed attempts (which may already
151 + * have cost credits), only every 4th attempt may escalate again — the error backoff (×2^(n−1), cap ×8) spaces those
152 + * out to at most one premium retry per ~week for a daily group. A successful fetch resets `errorCount` to 0.
153 + */
154 +export function premiumAllowedAfterErrors(errorCount: number): boolean {
155 + if (!Number.isFinite(errorCount) || errorCount < 2) return true;
156 + return errorCount % 4 === 0;
157 +}
modified config/connectors/datacenterdynamics.yaml +2 −0
@@ -13,6 +13,8 @@ coverage: { global: true }
13 13 fetch:
14 14 level: 1
15 15 maxLevel: 4
16 + renderJs: false # static article HTML; ASP only (≈1 credit/page instead of ≈6)
17 + maxCreditsPerDay: 60 # shared Redis cap across workers
16 18 rpm: 6
17 19 concurrency: 1
18 20 respectRobots: true
modified config/connectors/datacentremagazine.yaml +2 −0
@@ -13,6 +13,8 @@ coverage: { global: true }
13 13 fetch:
14 14 level: 1
15 15 maxLevel: 4
16 + renderJs: false # static article HTML; ASP only (≈1 credit/page instead of ≈6)
17 + maxCreditsPerDay: 60 # shared Redis cap across workers
16 18 rpm: 6
17 19 concurrency: 1
18 20 respectRobots: true
modified config/connectors/equinix.yaml +1 −0
@@ -32,6 +32,7 @@ discovery:
32 32 rss: ["https://newsroom.equinix.com/press-releases-global?pagetemplate=rss"] # fetcher heuristic fixed 2026-09-11: challenge-platform beacon on content-rich pages no longer escalates
33 33 include:
34 34 - "^https://www\\.equinix\\.com/data-centers/"
35 + - "^https://newsroom\\.equinix\\.com/"
35 36 exclude:
36 37 - "\\?"
37 38 - "/data-centers/(colocation|design|data-center-excellence|services|solutions)"
modified config/connectors/intelligentdatacentres.yaml +1 −0
@@ -12,6 +12,7 @@ coverage: { global: true }
12 12 fetch:
13 13 level: 1
14 14 maxLevel: 2
15 + userAgent: browser # RSS answers 403 to the bot UA (audit 2026-09-11)
15 16 rpm: 10
16 17 concurrency: 1
17 18 respectRobots: true
modified deploy/bin/remote/backup-run.sh +3 −0
@@ -50,6 +50,9 @@ if ! dc exec -T clickhouse clickhouse-client --user dci --password "$CH_PW" \
50 50 --query "BACKUP DATABASE $CH_DB TO File('/backups/dci-ch-$STAMP.zip') SETTINGS compression_method='zstd'" >/dev/null; then
51 51 say " clickhouse backup failed (non-fatal)"
52 52 fi
53 +# the zip is written by the clickhouse container user (uid 101, mode 640) → make it readable for the offsite rsync
54 +sudo chown -R "$(id -u):$(id -g)" "$BACKUP_DIR/clickhouse" 2>/dev/null || true
55 +chmod -R u+rwX,go+rX "$BACKUP_DIR/clickhouse" 2>/dev/null || true
53 56
54 57 # 4. MinIO raw archive mirror
55 58 say "mc mirror dci-raw → minio/dci-raw"
modified deploy/compose.data.yml +10 −5
@@ -6,7 +6,7 @@
6 6 # docker compose -f deploy/compose.data.yml --env-file deploy/.env.data run --rm migrate
7 7 # docker compose -f deploy/compose.data.yml --env-file deploy/.env.data run --rm cli run <connector> --dry-run --limit 5
8 8 #
9 −# Profiles: (none) = production set · ops = migrate/cli one-shots · maintenance = worker-maint (DCI_QUEUES=maintenance)
9 +# Profiles: (none) = production set (incl. scheduler + worker-maint, DCI_QUEUES=maintenance) · ops = migrate/cli one-shots
10 10 name: dci
11 11
12 12 x-logging: &logging
@@ -123,6 +123,8 @@ services:
123 123 cpus: 4
124 124
125 125 # Scheduler: enqueues connector runs (BullMQ) — co-located with Redis. The crawl workers run on BHS64b.
126 + # /healthz + /metrics on 8321 (container network only; scraped by Prometheus as scheduler:8321). The image's
127 + # HEALTHCHECK probes http://127.0.0.1:$WORKER_PORT/healthz.
126 128 scheduler:
127 129 <<: [*worker-image, *restart]
128 130 command: ["scheduler"]
@@ -132,18 +134,21 @@ services:
132 134 DCI_SCHEDULER_INTERVAL_MS: ${DCI_SCHEDULER_INTERVAL_MS:-60000}
133 135 WORKER_PORT: "8321"
134 136 healthcheck:
135 − disable: true
137 + test: ["CMD", "node", "-e", "fetch('http://127.0.0.1:8321/healthz').then(r=>process.exit(r.ok?0:1),()=>process.exit(1))"]
138 + interval: 30s
139 + timeout: 5s
140 + retries: 3
141 + start_period: 40s
136 142 depends_on:
137 143 postgres: { condition: service_healthy }
138 144 redis: { condition: service_healthy }
139 145 mem_limit: 2g
140 146 cpus: 2
141 147
142 − # Optional light worker for the maintenance queue (rankings, metrics, re-checks) — opt-in profile
143 − # because it requires the worker to honour DCI_QUEUES (otherwise it would also crawl from this node).
148 + # Light worker for the maintenance queue only (rankings, metrics, refresh-stats, cleanup) — the worker honours
149 + # DCI_QUEUES, so this node never crawls. Part of the default set since 2026-09-11 (was profile `maintenance`).
144 150 worker-maint:
145 151 <<: [*worker-image, *restart]
146 − profiles: ["maintenance"]
147 152 command: ["worker"]
148 153 logging: *logging
149 154 environment:
modified deploy/monitoring/alerts.yml +32 −1
@@ -28,11 +28,42 @@ groups:
28 28 annotations: { summary: "Jobs are waiting but no fetch completed in 30 min" }
29 29
30 30 - alert: PremiumBudgetNearlyExhausted
31 − expr: sum by (provider) (increase(dci_crawl_credits_total[24h])) / on (provider) sum by (provider) (dci_crawl_daily_budget) > 0.9
31 + # shared (Redis) usage today vs. the configured daily cap — survives worker restarts, unlike increase()
32 + expr: max by (provider) (dci_crawl_daily_credits_used) / on (provider) max by (provider) (dci_crawl_daily_budget) > 0.9
32 33 for: 5m
33 34 labels: { severity: warning }
34 35 annotations: { summary: "{{ $labels.provider }} premium credits above 90 % of the daily budget" }
35 36
37 + - alert: SchedulerStale
38 + expr: (time() - max(dci_scheduler_last_tick_timestamp_seconds{role="scheduler"})) > 600
39 + for: 5m
40 + labels: { severity: serious }
41 + annotations: { summary: "No successful scheduler pass in 10 min (dedicated scheduler)" }
42 +
43 + - alert: WorkerDown
44 + expr: max(dci_worker_up{instance_name="worker"}) == 0 or absent(dci_worker_up{instance_name="worker"})
45 + for: 5m
46 + labels: { severity: serious }
47 + annotations: { summary: "The crawl worker on BHS64b is not accepting jobs" }
48 +
49 + - alert: CrawlRunsFailing
50 + expr: sum(increase(dci_crawl_runs_total{status="failed"}[1h])) > 3
51 + for: 10m
52 + labels: { severity: warning }
53 + annotations: { summary: "More than 3 connector runs failed in the last hour" }
54 +
55 + - alert: FetchErrorRateHigh
56 + expr: sum(rate(dci_crawl_fetches_total{outcome=~"error|blocked"}[30m])) / clamp_min(sum(rate(dci_crawl_fetches_total[30m])), 0.001) > 0.3
57 + for: 30m
58 + labels: { severity: warning }
59 + annotations: { summary: "More than 30 % of fetches end in error/blocked over 30 min" }
60 +
61 + - alert: QueueBacklog
62 + expr: sum(dci_queue_jobs{queue="crawl",state=~"waiting|prioritized"}) > 50
63 + for: 30m
64 + labels: { severity: warning }
65 + annotations: { summary: "Crawl queue backlog above 50 jobs for 30 min — add a worker (profile scale) or raise DCI_CRAWL_CONCURRENCY" }
66 +
36 67 - alert: DiskAlmostFull
37 68 expr: (node_filesystem_avail_bytes{mountpoint="/",fstype!~"tmpfs|overlay"} / node_filesystem_size_bytes{mountpoint="/",fstype!~"tmpfs|overlay"}) < 0.10
38 69 for: 15m
modified deploy/monitoring/grafana/dashboards/dci-overview.json +1333 −155
@@ -1,177 +1,1355 @@
1 1 {
2 2 "uid": "dci-overview",
3 3 "title": "DataCenterIndex — overview",
4 − "tags": ["dci"],
4 + "tags": [
5 + "dci"
6 + ],
5 7 "timezone": "browser",
6 8 "schemaVersion": 39,
7 9 "version": 1,
8 10 "editable": false,
9 11 "graphTooltip": 1,
10 12 "refresh": "30s",
11 − "time": { "from": "now-6h", "to": "now" },
12 − "templating": { "list": [] },
13 − "annotations": { "list": [] },
13 + "time": {
14 + "from": "now-6h",
15 + "to": "now"
16 + },
17 + "templating": {
18 + "list": []
19 + },
20 + "annotations": {
21 + "list": []
22 + },
14 23 "panels": [
15 − { "type": "row", "title": "Crawl", "gridPos": { "h": 1, "w": 24, "x": 0, "y": 0 }, "id": 100, "collapsed": false },
16 −
17 − { "id": 1, "type": "stat", "title": "Fetches / min", "gridPos": { "h": 4, "w": 6, "x": 0, "y": 1 },
18 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
19 − "targets": [{ "refId": "A", "expr": "sum(rate(dci_crawl_fetches_total[5m])) * 60" }],
20 − "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "none", "graphMode": "area", "textMode": "value" },
21 − "fieldConfig": { "defaults": { "unit": "short", "decimals": 1, "color": { "mode": "fixed", "fixedColor": "#3987e5" } }, "overrides": [] } },
22 −
23 − { "id": 2, "type": "stat", "title": "Premium credits, last 24 h", "gridPos": { "h": 4, "w": 6, "x": 6, "y": 1 },
24 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
25 − "targets": [{ "refId": "A", "expr": "sum(increase(dci_crawl_credits_total[24h])) by (provider)", "legendFormat": "{{provider}}" }],
26 − "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "none", "graphMode": "none", "textMode": "value_and_name" },
27 − "fieldConfig": { "defaults": { "unit": "short", "decimals": 0, "color": { "mode": "fixed", "fixedColor": "#d95926" } }, "overrides": [] } },
28 −
29 − { "id": 3, "type": "stat", "title": "Jobs waiting", "gridPos": { "h": 4, "w": 6, "x": 12, "y": 1 },
30 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
31 − "targets": [{ "refId": "A", "expr": "sum(dci_queue_jobs{state=\"waiting\"})" }],
32 − "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "none", "graphMode": "area", "textMode": "value" },
33 − "fieldConfig": { "defaults": { "unit": "short", "decimals": 0, "color": { "mode": "fixed", "fixedColor": "#199e70" } }, "overrides": [] } },
34 −
35 − { "id": 4, "type": "stat", "title": "API p95 latency", "gridPos": { "h": 4, "w": 6, "x": 18, "y": 1 },
36 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
37 − "targets": [{ "refId": "A", "expr": "histogram_quantile(0.95, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))" }],
38 − "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "none", "graphMode": "area", "textMode": "value" },
39 − "fieldConfig": { "defaults": { "unit": "s", "decimals": 3, "color": { "mode": "fixed", "fixedColor": "#3987e5" } }, "overrides": [] } },
40 −
41 − { "id": 5, "type": "timeseries", "title": "Crawl rate by outcome (fetches / s)", "gridPos": { "h": 8, "w": 12, "x": 0, "y": 5 },
42 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
43 − "targets": [{ "refId": "A", "expr": "sum(rate(dci_crawl_fetches_total[5m])) by (outcome)", "legendFormat": "{{outcome}}" }],
44 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
45 − "fieldConfig": { "defaults": { "unit": "reqps", "custom": { "lineWidth": 2, "fillOpacity": 0, "pointSize": 4, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
24 + {
25 + "type": "row",
26 + "title": "Crawl",
27 + "gridPos": {
28 + "h": 1,
29 + "w": 24,
30 + "x": 0,
31 + "y": 0
32 + },
33 + "id": 100,
34 + "collapsed": false
35 + },
36 + {
37 + "id": 1,
38 + "type": "stat",
39 + "title": "Fetches / min",
40 + "gridPos": {
41 + "h": 4,
42 + "w": 6,
43 + "x": 0,
44 + "y": 1
45 + },
46 + "datasource": {
47 + "type": "prometheus",
48 + "uid": "dci-prom"
49 + },
50 + "targets": [
51 + {
52 + "refId": "A",
53 + "expr": "sum(rate(dci_crawl_fetches_total[5m])) * 60"
54 + }
55 + ],
56 + "options": {
57 + "reduceOptions": {
58 + "calcs": [
59 + "lastNotNull"
60 + ]
61 + },
62 + "colorMode": "none",
63 + "graphMode": "area",
64 + "textMode": "value"
65 + },
66 + "fieldConfig": {
67 + "defaults": {
68 + "unit": "short",
69 + "decimals": 1,
70 + "color": {
71 + "mode": "fixed",
72 + "fixedColor": "#3987e5"
73 + }
74 + },
75 + "overrides": []
76 + }
77 + },
78 + {
79 + "id": 2,
80 + "type": "stat",
81 + "title": "Premium credits, last 24 h",
82 + "gridPos": {
83 + "h": 4,
84 + "w": 6,
85 + "x": 6,
86 + "y": 1
87 + },
88 + "datasource": {
89 + "type": "prometheus",
90 + "uid": "dci-prom"
91 + },
92 + "targets": [
93 + {
94 + "refId": "A",
95 + "expr": "sum(increase(dci_crawl_credits_total[24h])) by (provider)",
96 + "legendFormat": "{{provider}}"
97 + }
98 + ],
99 + "options": {
100 + "reduceOptions": {
101 + "calcs": [
102 + "lastNotNull"
103 + ]
104 + },
105 + "colorMode": "none",
106 + "graphMode": "none",
107 + "textMode": "value_and_name"
108 + },
109 + "fieldConfig": {
110 + "defaults": {
111 + "unit": "short",
112 + "decimals": 0,
113 + "color": {
114 + "mode": "fixed",
115 + "fixedColor": "#d95926"
116 + }
117 + },
118 + "overrides": []
119 + }
120 + },
121 + {
122 + "id": 3,
123 + "type": "stat",
124 + "title": "Jobs waiting",
125 + "gridPos": {
126 + "h": 4,
127 + "w": 6,
128 + "x": 12,
129 + "y": 1
130 + },
131 + "datasource": {
132 + "type": "prometheus",
133 + "uid": "dci-prom"
134 + },
135 + "targets": [
136 + {
137 + "refId": "A",
138 + "expr": "sum(dci_queue_jobs{state=\"waiting\"})"
139 + }
140 + ],
141 + "options": {
142 + "reduceOptions": {
143 + "calcs": [
144 + "lastNotNull"
145 + ]
146 + },
147 + "colorMode": "none",
148 + "graphMode": "area",
149 + "textMode": "value"
150 + },
151 + "fieldConfig": {
152 + "defaults": {
153 + "unit": "short",
154 + "decimals": 0,
155 + "color": {
156 + "mode": "fixed",
157 + "fixedColor": "#199e70"
158 + }
159 + },
160 + "overrides": []
161 + }
162 + },
163 + {
164 + "id": 4,
165 + "type": "stat",
166 + "title": "API p95 latency",
167 + "gridPos": {
168 + "h": 4,
169 + "w": 6,
170 + "x": 18,
171 + "y": 1
172 + },
173 + "datasource": {
174 + "type": "prometheus",
175 + "uid": "dci-prom"
176 + },
177 + "targets": [
178 + {
179 + "refId": "A",
180 + "expr": "histogram_quantile(0.95, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))"
181 + }
182 + ],
183 + "options": {
184 + "reduceOptions": {
185 + "calcs": [
186 + "lastNotNull"
187 + ]
188 + },
189 + "colorMode": "none",
190 + "graphMode": "area",
191 + "textMode": "value"
192 + },
193 + "fieldConfig": {
194 + "defaults": {
195 + "unit": "s",
196 + "decimals": 3,
197 + "color": {
198 + "mode": "fixed",
199 + "fixedColor": "#3987e5"
200 + }
201 + },
202 + "overrides": []
203 + }
204 + },
205 + {
206 + "id": 5,
207 + "type": "timeseries",
208 + "title": "Crawl rate by outcome (fetches / s)",
209 + "gridPos": {
210 + "h": 8,
211 + "w": 12,
212 + "x": 0,
213 + "y": 5
214 + },
215 + "datasource": {
216 + "type": "prometheus",
217 + "uid": "dci-prom"
218 + },
219 + "targets": [
220 + {
221 + "refId": "A",
222 + "expr": "sum(rate(dci_crawl_fetches_total[5m])) by (outcome)",
223 + "legendFormat": "{{outcome}}"
224 + }
225 + ],
226 + "options": {
227 + "legend": {
228 + "displayMode": "list",
229 + "placement": "bottom",
230 + "showLegend": true
231 + },
232 + "tooltip": {
233 + "mode": "multi",
234 + "sort": "desc"
235 + }
236 + },
237 + "fieldConfig": {
238 + "defaults": {
239 + "unit": "reqps",
240 + "custom": {
241 + "lineWidth": 2,
242 + "fillOpacity": 0,
243 + "pointSize": 4,
244 + "showPoints": "never",
245 + "spanNulls": true,
246 + "axisSoftMin": 0
247 + }
248 + },
46 249 "overrides": [
47 − { "matcher": { "id": "byName", "options": "ok" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
48 − { "matcher": { "id": "byName", "options": "not_modified" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#199e70" } }] },
49 − { "matcher": { "id": "byName", "options": "error" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] },
50 − { "matcher": { "id": "byName", "options": "blocked" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#c98500" } }] },
51 − { "matcher": { "id": "byName", "options": "skipped" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d55181" } }] }
52 − ] } },
53 −
54 − { "id": 6, "type": "timeseries", "title": "Premium credits per hour, by provider", "gridPos": { "h": 8, "w": 12, "x": 12, "y": 5 },
55 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
56 − "targets": [{ "refId": "A", "expr": "sum(increase(dci_crawl_credits_total[1h])) by (provider)", "legendFormat": "{{provider}}" }],
57 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
58 − "fieldConfig": { "defaults": { "unit": "short", "custom": { "drawStyle": "bars", "lineWidth": 1, "fillOpacity": 70, "barAlignment": 0, "stacking": { "mode": "normal" }, "axisSoftMin": 0 } },
250 + {
251 + "matcher": {
252 + "id": "byName",
253 + "options": "ok"
254 + },
255 + "properties": [
256 + {
257 + "id": "color",
258 + "value": {
259 + "mode": "fixed",
260 + "fixedColor": "#3987e5"
261 + }
262 + }
263 + ]
264 + },
265 + {
266 + "matcher": {
267 + "id": "byName",
268 + "options": "not_modified"
269 + },
270 + "properties": [
271 + {
272 + "id": "color",
273 + "value": {
274 + "mode": "fixed",
275 + "fixedColor": "#199e70"
276 + }
277 + }
278 + ]
279 + },
280 + {
281 + "matcher": {
282 + "id": "byName",
283 + "options": "error"
284 + },
285 + "properties": [
286 + {
287 + "id": "color",
288 + "value": {
289 + "mode": "fixed",
290 + "fixedColor": "#d95926"
291 + }
292 + }
293 + ]
294 + },
295 + {
296 + "matcher": {
297 + "id": "byName",
298 + "options": "blocked"
299 + },
300 + "properties": [
301 + {
302 + "id": "color",
303 + "value": {
304 + "mode": "fixed",
305 + "fixedColor": "#c98500"
306 + }
307 + }
308 + ]
309 + },
310 + {
311 + "matcher": {
312 + "id": "byName",
313 + "options": "skipped"
314 + },
315 + "properties": [
316 + {
317 + "id": "color",
318 + "value": {
319 + "mode": "fixed",
320 + "fixedColor": "#d55181"
321 + }
322 + }
323 + ]
324 + }
325 + ]
326 + }
327 + },
328 + {
329 + "id": 6,
330 + "type": "timeseries",
331 + "title": "Premium credits per hour, by provider",
332 + "gridPos": {
333 + "h": 8,
334 + "w": 12,
335 + "x": 12,
336 + "y": 5
337 + },
338 + "datasource": {
339 + "type": "prometheus",
340 + "uid": "dci-prom"
341 + },
342 + "targets": [
343 + {
344 + "refId": "A",
345 + "expr": "sum(increase(dci_crawl_credits_total[1h])) by (provider)",
346 + "legendFormat": "{{provider}}"
347 + }
348 + ],
349 + "options": {
350 + "legend": {
351 + "displayMode": "list",
352 + "placement": "bottom",
353 + "showLegend": true
354 + },
355 + "tooltip": {
356 + "mode": "multi",
357 + "sort": "desc"
358 + }
359 + },
360 + "fieldConfig": {
361 + "defaults": {
362 + "unit": "short",
363 + "custom": {
364 + "drawStyle": "bars",
365 + "lineWidth": 1,
366 + "fillOpacity": 70,
367 + "barAlignment": 0,
368 + "stacking": {
369 + "mode": "normal"
370 + },
371 + "axisSoftMin": 0
372 + }
373 + },
59 374 "overrides": [
60 − { "matcher": { "id": "byName", "options": "scrapfly" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] },
61 − { "matcher": { "id": "byName", "options": "firecrawl" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] }
62 − ] } },
63 −
64 − { "id": 7, "type": "timeseries", "title": "Queue depth (jobs) by queue and state", "gridPos": { "h": 8, "w": 12, "x": 0, "y": 13 },
65 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
66 − "targets": [{ "refId": "A", "expr": "sum(dci_queue_jobs) by (queue, state)", "legendFormat": "{{queue}} · {{state}}" }],
67 − "options": { "legend": { "displayMode": "table", "placement": "right", "showLegend": true, "calcs": ["lastNotNull"] }, "tooltip": { "mode": "multi", "sort": "desc" } },
68 − "fieldConfig": { "defaults": { "unit": "short", "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
375 + {
376 + "matcher": {
377 + "id": "byName",
378 + "options": "scrapfly"
379 + },
380 + "properties": [
381 + {
382 + "id": "color",
383 + "value": {
384 + "mode": "fixed",
385 + "fixedColor": "#d95926"
386 + }
387 + }
388 + ]
389 + },
390 + {
391 + "matcher": {
392 + "id": "byName",
393 + "options": "firecrawl"
394 + },
395 + "properties": [
396 + {
397 + "id": "color",
398 + "value": {
399 + "mode": "fixed",
400 + "fixedColor": "#3987e5"
401 + }
402 + }
403 + ]
404 + }
405 + ]
406 + }
407 + },
408 + {
409 + "id": 7,
410 + "type": "timeseries",
411 + "title": "Queue depth (jobs) by queue and state",
412 + "gridPos": {
413 + "h": 8,
414 + "w": 12,
415 + "x": 0,
416 + "y": 13
417 + },
418 + "datasource": {
419 + "type": "prometheus",
420 + "uid": "dci-prom"
421 + },
422 + "targets": [
423 + {
424 + "refId": "A",
425 + "expr": "sum(dci_queue_jobs) by (queue, state)",
426 + "legendFormat": "{{queue}} · {{state}}"
427 + }
428 + ],
429 + "options": {
430 + "legend": {
431 + "displayMode": "table",
432 + "placement": "right",
433 + "showLegend": true,
434 + "calcs": [
435 + "lastNotNull"
436 + ]
437 + },
438 + "tooltip": {
439 + "mode": "multi",
440 + "sort": "desc"
441 + }
442 + },
443 + "fieldConfig": {
444 + "defaults": {
445 + "unit": "short",
446 + "custom": {
447 + "lineWidth": 2,
448 + "fillOpacity": 0,
449 + "showPoints": "never",
450 + "spanNulls": true,
451 + "axisSoftMin": 0
452 + }
453 + },
69 454 "overrides": [
70 − { "matcher": { "id": "byRegexp", "options": ".*waiting" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
71 − { "matcher": { "id": "byRegexp", "options": ".*active" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#199e70" } }] },
72 − { "matcher": { "id": "byRegexp", "options": ".*delayed" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#c98500" } }] },
73 − { "matcher": { "id": "byRegexp", "options": ".*failed" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
74 − ] } },
75 −
76 − { "id": 8, "type": "timeseries", "title": "Fetch duration p50 / p95 (s)", "gridPos": { "h": 8, "w": 12, "x": 12, "y": 13 },
77 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
78 − "targets": [
79 − { "refId": "A", "expr": "histogram_quantile(0.50, sum(rate(dci_crawl_fetch_duration_seconds_bucket[5m])) by (le))", "legendFormat": "p50" },
80 − { "refId": "B", "expr": "histogram_quantile(0.95, sum(rate(dci_crawl_fetch_duration_seconds_bucket[5m])) by (le))", "legendFormat": "p95" }
81 − ],
82 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
83 − "fieldConfig": { "defaults": { "unit": "s", "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
455 + {
456 + "matcher": {
457 + "id": "byRegexp",
458 + "options": ".*waiting"
459 + },
460 + "properties": [
461 + {
462 + "id": "color",
463 + "value": {
464 + "mode": "fixed",
465 + "fixedColor": "#3987e5"
466 + }
467 + }
468 + ]
469 + },
470 + {
471 + "matcher": {
472 + "id": "byRegexp",
473 + "options": ".*active"
474 + },
475 + "properties": [
476 + {
477 + "id": "color",
478 + "value": {
479 + "mode": "fixed",
480 + "fixedColor": "#199e70"
481 + }
482 + }
483 + ]
484 + },
485 + {
486 + "matcher": {
487 + "id": "byRegexp",
488 + "options": ".*delayed"
489 + },
490 + "properties": [
491 + {
492 + "id": "color",
493 + "value": {
494 + "mode": "fixed",
495 + "fixedColor": "#c98500"
496 + }
497 + }
498 + ]
499 + },
500 + {
501 + "matcher": {
502 + "id": "byRegexp",
503 + "options": ".*failed"
504 + },
505 + "properties": [
506 + {
507 + "id": "color",
508 + "value": {
509 + "mode": "fixed",
510 + "fixedColor": "#d95926"
511 + }
512 + }
513 + ]
514 + }
515 + ]
516 + }
517 + },
518 + {
519 + "id": 8,
520 + "type": "timeseries",
521 + "title": "Fetch duration p50 / p95 (s)",
522 + "gridPos": {
523 + "h": 8,
524 + "w": 12,
525 + "x": 12,
526 + "y": 13
527 + },
528 + "datasource": {
529 + "type": "prometheus",
530 + "uid": "dci-prom"
531 + },
532 + "targets": [
533 + {
534 + "refId": "A",
535 + "expr": "histogram_quantile(0.50, sum(rate(dci_crawl_fetch_duration_seconds_bucket[5m])) by (le))",
536 + "legendFormat": "p50"
537 + },
538 + {
539 + "refId": "B",
540 + "expr": "histogram_quantile(0.95, sum(rate(dci_crawl_fetch_duration_seconds_bucket[5m])) by (le))",
541 + "legendFormat": "p95"
542 + }
543 + ],
544 + "options": {
545 + "legend": {
546 + "displayMode": "list",
547 + "placement": "bottom",
548 + "showLegend": true
549 + },
550 + "tooltip": {
551 + "mode": "multi",
552 + "sort": "desc"
553 + }
554 + },
555 + "fieldConfig": {
556 + "defaults": {
557 + "unit": "s",
558 + "custom": {
559 + "lineWidth": 2,
560 + "fillOpacity": 0,
561 + "showPoints": "never",
562 + "spanNulls": true,
563 + "axisSoftMin": 0
564 + }
565 + },
84 566 "overrides": [
85 − { "matcher": { "id": "byName", "options": "p50" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
86 − { "matcher": { "id": "byName", "options": "p95" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
87 − ] } },
88 −
89 − { "type": "row", "title": "API", "gridPos": { "h": 1, "w": 24, "x": 0, "y": 21 }, "id": 101, "collapsed": false },
90 −
91 − { "id": 9, "type": "timeseries", "title": "API latency p50 / p95 / p99 (s)", "gridPos": { "h": 8, "w": 12, "x": 0, "y": 22 },
92 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
93 − "targets": [
94 − { "refId": "A", "expr": "histogram_quantile(0.50, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))", "legendFormat": "p50" },
95 − { "refId": "B", "expr": "histogram_quantile(0.95, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))", "legendFormat": "p95" },
96 − { "refId": "C", "expr": "histogram_quantile(0.99, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))", "legendFormat": "p99" }
97 − ],
98 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
99 − "fieldConfig": { "defaults": { "unit": "s", "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
567 + {
568 + "matcher": {
569 + "id": "byName",
570 + "options": "p50"
571 + },
572 + "properties": [
573 + {
574 + "id": "color",
575 + "value": {
576 + "mode": "fixed",
577 + "fixedColor": "#3987e5"
578 + }
579 + }
580 + ]
581 + },
582 + {
583 + "matcher": {
584 + "id": "byName",
585 + "options": "p95"
586 + },
587 + "properties": [
588 + {
589 + "id": "color",
590 + "value": {
591 + "mode": "fixed",
592 + "fixedColor": "#d95926"
593 + }
594 + }
595 + ]
596 + }
597 + ]
598 + }
599 + },
600 + {
601 + "id": 17,
602 + "type": "timeseries",
603 + "title": "Premium budget used today (shared, Redis) vs daily cap",
604 + "datasource": {
605 + "type": "prometheus",
606 + "uid": "dci-prom"
607 + },
608 + "gridPos": {
609 + "h": 8,
610 + "w": 12,
611 + "x": 0,
612 + "y": 21
613 + },
614 + "targets": [
615 + {
616 + "refId": "A",
617 + "expr": "max by (provider) (dci_crawl_daily_credits_used)",
618 + "legendFormat": "{{provider}} used"
619 + },
620 + {
621 + "refId": "B",
622 + "expr": "max by (provider) (dci_crawl_daily_budget)",
623 + "legendFormat": "{{provider}} cap"
624 + }
625 + ],
626 + "fieldConfig": {
627 + "defaults": {
628 + "unit": "short",
629 + "color": {
630 + "mode": "palette-classic"
631 + }
632 + },
633 + "overrides": []
634 + },
635 + "options": {
636 + "legend": {
637 + "displayMode": "list",
638 + "placement": "bottom"
639 + },
640 + "tooltip": {
641 + "mode": "multi"
642 + }
643 + }
644 + },
645 + {
646 + "id": 18,
647 + "type": "timeseries",
648 + "title": "Ingest results / min and connector runs / h",
649 + "datasource": {
650 + "type": "prometheus",
651 + "uid": "dci-prom"
652 + },
653 + "gridPos": {
654 + "h": 8,
655 + "w": 12,
656 + "x": 12,
657 + "y": 21
658 + },
659 + "targets": [
660 + {
661 + "refId": "A",
662 + "expr": "sum by (result) (rate(dci_ingest_entities_total[5m])) * 60",
663 + "legendFormat": "{{result}}"
664 + },
665 + {
666 + "refId": "B",
667 + "expr": "sum by (status) (increase(dci_crawl_runs_total[1h]))",
668 + "legendFormat": "runs {{status}} /h"
669 + }
670 + ],
671 + "fieldConfig": {
672 + "defaults": {
673 + "unit": "short",
674 + "color": {
675 + "mode": "palette-classic"
676 + }
677 + },
678 + "overrides": []
679 + },
680 + "options": {
681 + "legend": {
682 + "displayMode": "list",
683 + "placement": "bottom"
684 + },
685 + "tooltip": {
686 + "mode": "multi"
687 + }
688 + }
689 + },
690 + {
691 + "type": "row",
692 + "title": "API",
693 + "gridPos": {
694 + "h": 1,
695 + "w": 24,
696 + "x": 0,
697 + "y": 29
698 + },
699 + "id": 101,
700 + "collapsed": false
701 + },
702 + {
703 + "id": 9,
704 + "type": "timeseries",
705 + "title": "API latency p50 / p95 / p99 (s)",
706 + "gridPos": {
707 + "h": 8,
708 + "w": 12,
709 + "x": 0,
710 + "y": 30
711 + },
712 + "datasource": {
713 + "type": "prometheus",
714 + "uid": "dci-prom"
715 + },
716 + "targets": [
717 + {
718 + "refId": "A",
719 + "expr": "histogram_quantile(0.50, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))",
720 + "legendFormat": "p50"
721 + },
722 + {
723 + "refId": "B",
724 + "expr": "histogram_quantile(0.95, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))",
725 + "legendFormat": "p95"
726 + },
727 + {
728 + "refId": "C",
729 + "expr": "histogram_quantile(0.99, sum(rate(dci_api_request_duration_seconds_bucket[5m])) by (le))",
730 + "legendFormat": "p99"
731 + }
732 + ],
733 + "options": {
734 + "legend": {
735 + "displayMode": "list",
736 + "placement": "bottom",
737 + "showLegend": true
738 + },
739 + "tooltip": {
740 + "mode": "multi",
741 + "sort": "desc"
742 + }
743 + },
744 + "fieldConfig": {
745 + "defaults": {
746 + "unit": "s",
747 + "custom": {
748 + "lineWidth": 2,
749 + "fillOpacity": 0,
750 + "showPoints": "never",
751 + "spanNulls": true,
752 + "axisSoftMin": 0
753 + }
754 + },
100 755 "overrides": [
101 − { "matcher": { "id": "byName", "options": "p50" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
102 − { "matcher": { "id": "byName", "options": "p95" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] },
103 − { "matcher": { "id": "byName", "options": "p99" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#199e70" } }] }
104 − ] } },
105 −
106 − { "id": 10, "type": "timeseries", "title": "API requests / s by status class", "gridPos": { "h": 8, "w": 12, "x": 12, "y": 22 },
107 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
108 − "targets": [{ "refId": "A", "expr": "sum by (class) (label_replace(rate(dci_api_requests_total[5m]), \"class\", \"$1xx\", \"status\", \"(.).*\"))", "legendFormat": "{{class}}" }],
109 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
110 − "fieldConfig": { "defaults": { "unit": "reqps", "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
756 + {
757 + "matcher": {
758 + "id": "byName",
759 + "options": "p50"
760 + },
761 + "properties": [
762 + {
763 + "id": "color",
764 + "value": {
765 + "mode": "fixed",
766 + "fixedColor": "#3987e5"
767 + }
768 + }
769 + ]
770 + },
771 + {
772 + "matcher": {
773 + "id": "byName",
774 + "options": "p95"
775 + },
776 + "properties": [
777 + {
778 + "id": "color",
779 + "value": {
780 + "mode": "fixed",
781 + "fixedColor": "#d95926"
782 + }
783 + }
784 + ]
785 + },
786 + {
787 + "matcher": {
788 + "id": "byName",
789 + "options": "p99"
790 + },
791 + "properties": [
792 + {
793 + "id": "color",
794 + "value": {
795 + "mode": "fixed",
796 + "fixedColor": "#199e70"
797 + }
798 + }
799 + ]
800 + }
801 + ]
802 + }
803 + },
804 + {
805 + "id": 10,
806 + "type": "timeseries",
807 + "title": "API requests / s by status class",
808 + "gridPos": {
809 + "h": 8,
810 + "w": 12,
811 + "x": 12,
812 + "y": 30
813 + },
814 + "datasource": {
815 + "type": "prometheus",
816 + "uid": "dci-prom"
817 + },
818 + "targets": [
819 + {
820 + "refId": "A",
821 + "expr": "sum by (class) (label_replace(rate(dci_api_requests_total[5m]), \"class\", \"$1xx\", \"status\", \"(.).*\"))",
822 + "legendFormat": "{{class}}"
823 + }
824 + ],
825 + "options": {
826 + "legend": {
827 + "displayMode": "list",
828 + "placement": "bottom",
829 + "showLegend": true
830 + },
831 + "tooltip": {
832 + "mode": "multi",
833 + "sort": "desc"
834 + }
835 + },
836 + "fieldConfig": {
837 + "defaults": {
838 + "unit": "reqps",
839 + "custom": {
840 + "lineWidth": 2,
841 + "fillOpacity": 0,
842 + "showPoints": "never",
843 + "spanNulls": true,
844 + "axisSoftMin": 0
845 + }
846 + },
111 847 "overrides": [
112 − { "matcher": { "id": "byName", "options": "2xx" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
113 − { "matcher": { "id": "byName", "options": "3xx" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#199e70" } }] },
114 − { "matcher": { "id": "byName", "options": "4xx" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#c98500" } }] },
115 − { "matcher": { "id": "byName", "options": "5xx" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
116 − ] } },
117 −
118 − { "type": "row", "title": "Hosts", "gridPos": { "h": 1, "w": 24, "x": 0, "y": 30 }, "id": 102, "collapsed": false },
119 −
120 − { "id": 11, "type": "timeseries", "title": "CPU busy (%)", "gridPos": { "h": 8, "w": 8, "x": 0, "y": 31 },
121 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
122 − "targets": [{ "refId": "A", "expr": "100 * (1 - avg(rate(node_cpu_seconds_total{mode=\"idle\"}[5m])) by (node))", "legendFormat": "{{node}}" }],
123 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
124 − "fieldConfig": { "defaults": { "unit": "percent", "min": 0, "max": 100, "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true } },
848 + {
849 + "matcher": {
850 + "id": "byName",
851 + "options": "2xx"
852 + },
853 + "properties": [
854 + {
855 + "id": "color",
856 + "value": {
857 + "mode": "fixed",
858 + "fixedColor": "#3987e5"
859 + }
860 + }
861 + ]
862 + },
863 + {
864 + "matcher": {
865 + "id": "byName",
866 + "options": "3xx"
867 + },
868 + "properties": [
869 + {
870 + "id": "color",
871 + "value": {
872 + "mode": "fixed",
873 + "fixedColor": "#199e70"
874 + }
875 + }
876 + ]
877 + },
878 + {
879 + "matcher": {
880 + "id": "byName",
881 + "options": "4xx"
882 + },
883 + "properties": [
884 + {
885 + "id": "color",
886 + "value": {
887 + "mode": "fixed",
888 + "fixedColor": "#c98500"
889 + }
890 + }
891 + ]
892 + },
893 + {
894 + "matcher": {
895 + "id": "byName",
896 + "options": "5xx"
897 + },
898 + "properties": [
899 + {
900 + "id": "color",
901 + "value": {
902 + "mode": "fixed",
903 + "fixedColor": "#d95926"
904 + }
905 + }
906 + ]
907 + }
908 + ]
909 + }
910 + },
911 + {
912 + "type": "row",
913 + "title": "Hosts",
914 + "gridPos": {
915 + "h": 1,
916 + "w": 24,
917 + "x": 0,
918 + "y": 38
919 + },
920 + "id": 102,
921 + "collapsed": false
922 + },
923 + {
924 + "id": 11,
925 + "type": "timeseries",
926 + "title": "CPU busy (%)",
927 + "gridPos": {
928 + "h": 8,
929 + "w": 8,
930 + "x": 0,
931 + "y": 39
932 + },
933 + "datasource": {
934 + "type": "prometheus",
935 + "uid": "dci-prom"
936 + },
937 + "targets": [
938 + {
939 + "refId": "A",
940 + "expr": "100 * (1 - avg(rate(node_cpu_seconds_total{mode=\"idle\"}[5m])) by (node))",
941 + "legendFormat": "{{node}}"
942 + }
943 + ],
944 + "options": {
945 + "legend": {
946 + "displayMode": "list",
947 + "placement": "bottom",
948 + "showLegend": true
949 + },
950 + "tooltip": {
951 + "mode": "multi",
952 + "sort": "desc"
953 + }
954 + },
955 + "fieldConfig": {
956 + "defaults": {
957 + "unit": "percent",
958 + "min": 0,
959 + "max": 100,
960 + "custom": {
961 + "lineWidth": 2,
962 + "fillOpacity": 0,
963 + "showPoints": "never",
964 + "spanNulls": true
965 + }
966 + },
125 967 "overrides": [
126 − { "matcher": { "id": "byName", "options": "data" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
127 − { "matcher": { "id": "byName", "options": "crawl" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
128 − ] } },
129 −
130 − { "id": 12, "type": "timeseries", "title": "Memory used (%)", "gridPos": { "h": 8, "w": 8, "x": 8, "y": 31 },
131 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
132 − "targets": [{ "refId": "A", "expr": "100 * (1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)", "legendFormat": "{{node}}" }],
133 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
134 − "fieldConfig": { "defaults": { "unit": "percent", "min": 0, "max": 100, "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true } },
968 + {
969 + "matcher": {
970 + "id": "byName",
971 + "options": "data"
972 + },
973 + "properties": [
974 + {
975 + "id": "color",
976 + "value": {
977 + "mode": "fixed",
978 + "fixedColor": "#3987e5"
979 + }
980 + }
981 + ]
982 + },
983 + {
984 + "matcher": {
985 + "id": "byName",
986 + "options": "crawl"
987 + },
988 + "properties": [
989 + {
990 + "id": "color",
991 + "value": {
992 + "mode": "fixed",
993 + "fixedColor": "#d95926"
994 + }
995 + }
996 + ]
997 + }
998 + ]
999 + }
1000 + },
1001 + {
1002 + "id": 12,
1003 + "type": "timeseries",
1004 + "title": "Memory used (%)",
1005 + "gridPos": {
1006 + "h": 8,
1007 + "w": 8,
1008 + "x": 8,
1009 + "y": 39
1010 + },
1011 + "datasource": {
1012 + "type": "prometheus",
1013 + "uid": "dci-prom"
1014 + },
1015 + "targets": [
1016 + {
1017 + "refId": "A",
1018 + "expr": "100 * (1 - node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes)",
1019 + "legendFormat": "{{node}}"
1020 + }
1021 + ],
1022 + "options": {
1023 + "legend": {
1024 + "displayMode": "list",
1025 + "placement": "bottom",
1026 + "showLegend": true
1027 + },
1028 + "tooltip": {
1029 + "mode": "multi",
1030 + "sort": "desc"
1031 + }
1032 + },
1033 + "fieldConfig": {
1034 + "defaults": {
1035 + "unit": "percent",
1036 + "min": 0,
1037 + "max": 100,
1038 + "custom": {
1039 + "lineWidth": 2,
1040 + "fillOpacity": 0,
1041 + "showPoints": "never",
1042 + "spanNulls": true
1043 + }
1044 + },
135 1045 "overrides": [
136 − { "matcher": { "id": "byName", "options": "data" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
137 − { "matcher": { "id": "byName", "options": "crawl" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
138 − ] } },
139 −
140 − { "id": 13, "type": "timeseries", "title": "Root filesystem used (%)", "gridPos": { "h": 8, "w": 8, "x": 16, "y": 31 },
141 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
142 − "targets": [{ "refId": "A", "expr": "100 * (1 - node_filesystem_avail_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"} / node_filesystem_size_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"})", "legendFormat": "{{node}}" }],
143 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
144 − "fieldConfig": { "defaults": { "unit": "percent", "min": 0, "max": 100, "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true } },
1046 + {
1047 + "matcher": {
1048 + "id": "byName",
1049 + "options": "data"
1050 + },
1051 + "properties": [
1052 + {
1053 + "id": "color",
1054 + "value": {
1055 + "mode": "fixed",
1056 + "fixedColor": "#3987e5"
1057 + }
1058 + }
1059 + ]
1060 + },
1061 + {
1062 + "matcher": {
1063 + "id": "byName",
1064 + "options": "crawl"
1065 + },
1066 + "properties": [
1067 + {
1068 + "id": "color",
1069 + "value": {
1070 + "mode": "fixed",
1071 + "fixedColor": "#d95926"
1072 + }
1073 + }
1074 + ]
1075 + }
1076 + ]
1077 + }
1078 + },
1079 + {
1080 + "id": 13,
1081 + "type": "timeseries",
1082 + "title": "Root filesystem used (%)",
1083 + "gridPos": {
1084 + "h": 8,
1085 + "w": 8,
1086 + "x": 16,
1087 + "y": 39
1088 + },
1089 + "datasource": {
1090 + "type": "prometheus",
1091 + "uid": "dci-prom"
1092 + },
1093 + "targets": [
1094 + {
1095 + "refId": "A",
1096 + "expr": "100 * (1 - node_filesystem_avail_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"} / node_filesystem_size_bytes{mountpoint=\"/\",fstype!~\"tmpfs|overlay\"})",
1097 + "legendFormat": "{{node}}"
1098 + }
1099 + ],
1100 + "options": {
1101 + "legend": {
1102 + "displayMode": "list",
1103 + "placement": "bottom",
1104 + "showLegend": true
1105 + },
1106 + "tooltip": {
1107 + "mode": "multi",
1108 + "sort": "desc"
1109 + }
1110 + },
1111 + "fieldConfig": {
1112 + "defaults": {
1113 + "unit": "percent",
1114 + "min": 0,
1115 + "max": 100,
1116 + "custom": {
1117 + "lineWidth": 2,
1118 + "fillOpacity": 0,
1119 + "showPoints": "never",
1120 + "spanNulls": true
1121 + }
1122 + },
145 1123 "overrides": [
146 − { "matcher": { "id": "byName", "options": "data" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
147 − { "matcher": { "id": "byName", "options": "crawl" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
148 − ] } },
149 −
150 − { "type": "row", "title": "Containers & stores (data node)", "gridPos": { "h": 1, "w": 24, "x": 0, "y": 39 }, "id": 103, "collapsed": false },
151 −
152 − { "id": 14, "type": "timeseries", "title": "Container memory (working set)", "gridPos": { "h": 8, "w": 12, "x": 0, "y": 40 },
153 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
154 − "targets": [{ "refId": "A", "expr": "sum(container_memory_working_set_bytes{container_label_com_docker_compose_service!=\"\"}) by (container_label_com_docker_compose_service)", "legendFormat": "{{container_label_com_docker_compose_service}}" }],
155 − "options": { "legend": { "displayMode": "table", "placement": "right", "showLegend": true, "calcs": ["lastNotNull"], "sortBy": "Last *", "sortDesc": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
156 − "fieldConfig": { "defaults": { "unit": "bytes", "color": { "mode": "palette-classic" }, "custom": { "lineWidth": 1, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } }, "overrides": [] } },
157 −
158 − { "id": 15, "type": "timeseries", "title": "Postgres connections / Redis clients", "gridPos": { "h": 8, "w": 6, "x": 12, "y": 40 },
159 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
160 − "targets": [
161 − { "refId": "A", "expr": "sum(pg_stat_activity_count)", "legendFormat": "postgres connections" },
162 − { "refId": "B", "expr": "redis_connected_clients", "legendFormat": "redis clients" }
163 − ],
164 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": true }, "tooltip": { "mode": "multi", "sort": "desc" } },
165 − "fieldConfig": { "defaults": { "unit": "short", "custom": { "lineWidth": 2, "fillOpacity": 0, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } },
1124 + {
1125 + "matcher": {
1126 + "id": "byName",
1127 + "options": "data"
1128 + },
1129 + "properties": [
1130 + {
1131 + "id": "color",
1132 + "value": {
1133 + "mode": "fixed",
1134 + "fixedColor": "#3987e5"
1135 + }
1136 + }
1137 + ]
1138 + },
1139 + {
1140 + "matcher": {
1141 + "id": "byName",
1142 + "options": "crawl"
1143 + },
1144 + "properties": [
1145 + {
1146 + "id": "color",
1147 + "value": {
1148 + "mode": "fixed",
1149 + "fixedColor": "#d95926"
1150 + }
1151 + }
1152 + ]
1153 + }
1154 + ]
1155 + }
1156 + },
1157 + {
1158 + "type": "row",
1159 + "title": "Containers & stores (data node)",
1160 + "gridPos": {
1161 + "h": 1,
1162 + "w": 24,
1163 + "x": 0,
1164 + "y": 47
1165 + },
1166 + "id": 103,
1167 + "collapsed": false
1168 + },
1169 + {
1170 + "id": 14,
1171 + "type": "timeseries",
1172 + "title": "Container memory (working set)",
1173 + "gridPos": {
1174 + "h": 8,
1175 + "w": 12,
1176 + "x": 0,
1177 + "y": 48
1178 + },
1179 + "datasource": {
1180 + "type": "prometheus",
1181 + "uid": "dci-prom"
1182 + },
1183 + "targets": [
1184 + {
1185 + "refId": "A",
1186 + "expr": "sum(container_memory_working_set_bytes{container_label_com_docker_compose_service!=\"\"}) by (container_label_com_docker_compose_service)",
1187 + "legendFormat": "{{container_label_com_docker_compose_service}}"
1188 + }
1189 + ],
1190 + "options": {
1191 + "legend": {
1192 + "displayMode": "table",
1193 + "placement": "right",
1194 + "showLegend": true,
1195 + "calcs": [
1196 + "lastNotNull"
1197 + ],
1198 + "sortBy": "Last *",
1199 + "sortDesc": true
1200 + },
1201 + "tooltip": {
1202 + "mode": "multi",
1203 + "sort": "desc"
1204 + }
1205 + },
1206 + "fieldConfig": {
1207 + "defaults": {
1208 + "unit": "bytes",
1209 + "color": {
1210 + "mode": "palette-classic"
1211 + },
1212 + "custom": {
1213 + "lineWidth": 1,
1214 + "fillOpacity": 0,
1215 + "showPoints": "never",
1216 + "spanNulls": true,
1217 + "axisSoftMin": 0
1218 + }
1219 + },
1220 + "overrides": []
1221 + }
1222 + },
1223 + {
1224 + "id": 15,
1225 + "type": "timeseries",
1226 + "title": "Postgres connections / Redis clients",
1227 + "gridPos": {
1228 + "h": 8,
1229 + "w": 6,
1230 + "x": 12,
1231 + "y": 48
1232 + },
1233 + "datasource": {
1234 + "type": "prometheus",
1235 + "uid": "dci-prom"
1236 + },
1237 + "targets": [
1238 + {
1239 + "refId": "A",
1240 + "expr": "sum(pg_stat_activity_count)",
1241 + "legendFormat": "postgres connections"
1242 + },
1243 + {
1244 + "refId": "B",
1245 + "expr": "redis_connected_clients",
1246 + "legendFormat": "redis clients"
1247 + }
1248 + ],
1249 + "options": {
1250 + "legend": {
1251 + "displayMode": "list",
1252 + "placement": "bottom",
1253 + "showLegend": true
1254 + },
1255 + "tooltip": {
1256 + "mode": "multi",
1257 + "sort": "desc"
1258 + }
1259 + },
1260 + "fieldConfig": {
1261 + "defaults": {
1262 + "unit": "short",
1263 + "custom": {
1264 + "lineWidth": 2,
1265 + "fillOpacity": 0,
1266 + "showPoints": "never",
1267 + "spanNulls": true,
1268 + "axisSoftMin": 0
1269 + }
1270 + },
166 1271 "overrides": [
167 − { "matcher": { "id": "byName", "options": "postgres connections" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#3987e5" } }] },
168 − { "matcher": { "id": "byName", "options": "redis clients" }, "properties": [{ "id": "color", "value": { "mode": "fixed", "fixedColor": "#d95926" } }] }
169 − ] } },
170 −
171 − { "id": 16, "type": "timeseries", "title": "Postgres database size", "gridPos": { "h": 8, "w": 6, "x": 18, "y": 40 },
172 − "datasource": { "type": "prometheus", "uid": "dci-prom" },
173 − "targets": [{ "refId": "A", "expr": "pg_database_size_bytes{datname=\"dci\"}", "legendFormat": "dci" }],
174 − "options": { "legend": { "displayMode": "list", "placement": "bottom", "showLegend": false }, "tooltip": { "mode": "single" } },
175 − "fieldConfig": { "defaults": { "unit": "bytes", "color": { "mode": "fixed", "fixedColor": "#3987e5" }, "custom": { "lineWidth": 2, "fillOpacity": 10, "showPoints": "never", "spanNulls": true, "axisSoftMin": 0 } }, "overrides": [] } }
1272 + {
1273 + "matcher": {
1274 + "id": "byName",
1275 + "options": "postgres connections"
1276 + },
1277 + "properties": [
1278 + {
1279 + "id": "color",
1280 + "value": {
1281 + "mode": "fixed",
1282 + "fixedColor": "#3987e5"
1283 + }
1284 + }
1285 + ]
1286 + },
1287 + {
1288 + "matcher": {
1289 + "id": "byName",
1290 + "options": "redis clients"
1291 + },
1292 + "properties": [
1293 + {
1294 + "id": "color",
1295 + "value": {
1296 + "mode": "fixed",
1297 + "fixedColor": "#d95926"
1298 + }
1299 + }
1300 + ]
1301 + }
1302 + ]
1303 + }
1304 + },
1305 + {
1306 + "id": 16,
1307 + "type": "timeseries",
1308 + "title": "Postgres database size",
1309 + "gridPos": {
1310 + "h": 8,
1311 + "w": 6,
1312 + "x": 18,
1313 + "y": 48
1314 + },
1315 + "datasource": {
1316 + "type": "prometheus",
1317 + "uid": "dci-prom"
1318 + },
1319 + "targets": [
1320 + {
1321 + "refId": "A",
1322 + "expr": "pg_database_size_bytes{datname=\"dci\"}",
1323 + "legendFormat": "dci"
1324 + }
1325 + ],
1326 + "options": {
1327 + "legend": {
1328 + "displayMode": "list",
1329 + "placement": "bottom",
1330 + "showLegend": false
1331 + },
1332 + "tooltip": {
1333 + "mode": "single"
1334 + }
1335 + },
1336 + "fieldConfig": {
1337 + "defaults": {
1338 + "unit": "bytes",
1339 + "color": {
1340 + "mode": "fixed",
1341 + "fixedColor": "#3987e5"
1342 + },
1343 + "custom": {
1344 + "lineWidth": 2,
1345 + "fillOpacity": 10,
1346 + "showPoints": "never",
1347 + "spanNulls": true,
1348 + "axisSoftMin": 0
1349 + }
1350 + },
1351 + "overrides": []
1352 + }
1353 + }
176 1354 ]
177 1355 }
modified deploy/monitoring/prometheus.yml +12 −1
@@ -19,7 +19,11 @@ scrape_configs:
19 19 metrics_path: /api/metrics
20 20 static_configs: [{ targets: ["api:8311"], labels: { node: data, role: api } }]
21 21
22 − # Crawl workers (BHS64b) — /metrics on the worker health port (published on 10.68.0.2)
22 + # Crawl workers (BHS64b) — /metrics on the worker health port (published on 10.68.0.2); the maintenance-only
23 + # worker on this node is scraped over the compose network. Contract: dci_crawl_fetches_total{connector,level,outcome},
24 + # dci_crawl_credits_total{provider}, dci_crawl_daily_budget / dci_crawl_daily_credits_used{provider},
25 + # dci_crawl_fetch_duration_seconds, dci_ingest_entities_total{connector,result}, dci_events_total,
26 + # dci_queue_jobs{queue,state}, dci_worker_running_jobs, dci_worker_up (apps/worker/src/prom.ts).
23 27 - job_name: dci-worker
24 28 metrics_path: /metrics
25 29 static_configs:
@@ -27,6 +31,13 @@ scrape_configs:
27 31 labels: { node: crawl, role: worker, instance_name: worker }
28 32 - targets: ["10.68.0.2:8322"]
29 33 labels: { node: crawl, role: worker, instance_name: worker-b }
34 + - targets: ["worker-maint:8320"]
35 + labels: { node: data, role: worker, instance_name: worker-maint }
36 +
37 + # Scheduler (data node) — dci_scheduler_ticks_total, dci_scheduler_last_tick_timestamp_seconds, dci_queue_jobs
38 + - job_name: dci-scheduler
39 + metrics_path: /metrics
40 + static_configs: [{ targets: ["scheduler:8321"], labels: { node: data, role: scheduler } }]
30 41
31 42 # Edge Caddy
32 43 - job_name: dci-edge
added docs/CRAWL-OPERATIONS.md +163 −0
@@ -0,0 +1,163 @@
1 +# Crawl operations — DataCenterIndex.io
2 +
3 +Runbook for the crawl runtime (`apps/worker`) in production: the crawl worker on **BHS64b**
4 +(`compose.crawl.yml`, service `worker`, `/healthz` + `/metrics` on `10.68.0.2:8320`), the dedicated
5 +**scheduler** and the maintenance-only worker on **BHS128** (`compose.data.yml`, services `scheduler` :8321 and
6 +`worker-maint`), Postgres/Redis/ClickHouse/MinIO on BHS128. Deployment itself is in `docs/DEPLOY.md`; the
7 +runtime internals in `apps/worker/README.md`.
8 +
9 +Shell shortcuts used below:
10 +
11 +```bash
12 +alias dcd='ssh BHS128 "cd /srv/dci/app && docker compose -f deploy/compose.data.yml --env-file deploy/.env.data"' # data node
13 +alias dcc='ssh BHS64b "cd /srv/dci/app && docker compose -f deploy/compose.crawl.yml --env-file deploy/.env.crawl"' # crawl node
14 +alias dpsql='ssh BHS128 "cd /srv/dci/app && docker compose -f deploy/compose.data.yml --env-file deploy/.env.data exec -T postgres psql -U dci -d dci"'
15 +```
16 +
17 +## 1. Daily checks (5 minutes)
18 +
19 +| # | Check | Command / where | Expected |
20 +|---|---|---|---|
21 +| 1 | Containers | `deploy/bin/status.sh` | everything `Up … (healthy)`; **scheduler healthy** (it was in a restart loop before 2026-09-11) |
22 +| 2 | Worker liveness | `ssh BHS64b curl -s 10.68.0.2:8320/healthz \| jq '{ok,runningJobs,creditsToday,sharedBudgets,lastScheduler,queueDepth}'` | `ok: true`, `lastScheduler.at` < 2 min old |
23 +| 3 | Scheduler liveness | `dcd exec -T scheduler node -e "fetch('http://127.0.0.1:8321/healthz').then(r=>r.text()).then(console.log)"` | `ok: true`, `lastTick.error: null` |
24 +| 4 | Queue depth | Grafana → *Queue depth*, or `dci_queue_jobs{queue="crawl",state="waiting"}` | near 0 between ticks; a backlog > 50 for 30 min fires `QueueBacklog` |
25 +| 5 | Premium credits | Grafana → *Premium budget used today*; `ssh BHS128 … redis-cli … MGET dci:budget:scrapfly:$(date -u +%F) dci:budget:firecrawl:$(date -u +%F)` | well below `DCI_*_DAILY_BUDGET` (400 / 200); know **which connector** spent them (§4) |
26 +| 6 | Runs last 24 h | `dpsql -c "select status, count(*) from connector_runs where started_at > now() - interval '24 hours' group by 1"` | mostly `ok`; investigate `failed` / `aborted` |
27 +| 7 | Failing / degraded connectors | `dpsql -c "select id, health, last_status, left(last_error,80) from connectors where enabled and not paused and (health <> 'ok' or last_status in ('failed','partial'))"` | empty or known |
28 +| 8 | Doctor | `dcc run --rm cli doctor` | all critical checks pass; warn lines are triage input |
29 +| 9 | Alerts | Grafana → Alerting | none firing (`TargetDown`, `CrawlStalled`, `SchedulerStale`, `WorkerDown`, `PremiumBudgetNearlyExhausted`, `FetchErrorRateHigh`, `CrawlRunsFailing`, `QueueBacklog`, disk/memory) |
30 +| 10 | Backups | `ssh BHS128 systemctl list-timers dci-backup.timer`; `ssh BHS64b du -sh /srv/dci-backups/*` | timer waiting for 03:30, mirror grew last night |
31 +
32 +Weekly: `dpsql -c "select count(*) from documents where quarantined"` (should grow slowly, not jump), `select count(*)
33 +from entity_matches where status='pending'`, disk on both nodes (`status.sh`).
34 +
35 +## 2. Reading the metrics
36 +
37 +Both processes export the same registry (`apps/worker/src/prom.ts`); Prometheus scrapes `dci-worker` (BHS64b
38 +:8320/:8322 and `worker-maint:8320`) and `dci-scheduler` (`scheduler:8321`).
39 +
40 +| Metric | Meaning | Typical query |
41 +|---|---|---|
42 +| `dci_crawl_fetches_total{connector,level,outcome}` | one sample per fetch that returned (outcome `ok`, `not_modified`, `error`, `blocked`; level = final level `L1`–`L4`) | `sum by (outcome) (rate(…[5m]))`; per connector error ratio `sum by (connector) (rate(…{outcome=~"error\|blocked"}[1h])) / sum by (connector) (rate(…[1h]))` |
43 +| `dci_crawl_fetch_duration_seconds` (histogram) | wall time of one fetch incl. escalation | `histogram_quantile(0.95, sum(rate(…_bucket[5m])) by (le))` |
44 +| `dci_crawl_credits_total{provider}` | premium credits spent since process start | `sum by (provider) (increase(…[1h]))` |
45 +| `dci_crawl_daily_budget{provider}` / `dci_crawl_daily_credits_used{provider}` | daily cap and **shared** usage today (Redis, survives restarts) | `used / budget` → the `PremiumBudgetNearlyExhausted` rule |
46 +| `dci_ingest_entities_total{connector,result}` | entities handed to ingest: `created`, `updated`, `merged`, `unchanged`, `rejected` | `sum by (result) (rate(…[5m]))` |
47 +| `dci_events_total` | change events emitted by ingest | `increase(…[24h])` |
48 +| `dci_crawl_runs_total{status}` | runs finished by status (`ok`, `partial`, `failed`, `aborted`) | `CrawlRunsFailing` |
49 +| `dci_worker_jobs_total{queue,result}` | BullMQ jobs: `completed`, `failed`, `stalled`, `deferred` (run lock held by another worker) | many `deferred` = two workers keep colliding on the same connectors |
50 +| `dci_queue_jobs{queue,state}` | snapshot of queue depth per state at scrape time | `state="waiting"` / `"active"` / `"failed"` |
51 +| `dci_worker_running_jobs`, `dci_worker_up`, `dci_worker_uptime_seconds` | per process | `WorkerDown` |
52 +| `dci_scheduler_ticks_total{result}`, `dci_scheduler_enqueued_total`, `dci_scheduler_last_tick_timestamp_seconds` | scheduler health | `time() - max(last_tick) > 600` → `SchedulerStale` |
53 +
54 +Where the counters live outside Prometheus: `connector_runs.stats` (per run), `connectors.stats` (per connector,
55 +refreshed after each run), ClickHouse `crawl_log` (one row per fetch, 400 days) and `page_changes`.
56 +
57 +## 3. Triaging a failing or degraded connector
58 +
59 +`connectors.health` is `degraded` when ≥ 20 % of a run's fetches failed, `failing` at ≥ 50 % or when the run
60 +threw. Start from the last run:
61 +
62 +```bash
63 +dpsql -c "select id, task, status, started_at, stats->>'fetched' f, stats->>'failed' fl, stats->>'credits' cr, left(error,120) from connector_runs where connector_id='<id>' order by started_at desc limit 5"
64 +dpsql -c "select l->>'t' t, l->>'level' lvl, l->>'msg' from connector_runs r, jsonb_array_elements(r.log) l where r.id='<run_id>' and l->>'level' in ('warn','error') limit 40"
65 +dpsql -c "select split_part(error,':',1) code, count(*) from documents where connector_id='<id>' and error_count>0 group by 1 order by 2 desc"
66 +dcc run --rm cli docs <id> --due --limit 20 # what is about to be fetched
67 +dcc run --rm cli run <id> --dry-run --limit 3 --verbose # live reproduction from the crawl IP, nothing persisted
68 +```
69 +
70 +Common signatures:
71 +
72 +| Symptom | Cause | Action |
73 +|---|---|---|
74 +| `HTTP 403` on every page, credits > 0 | bot wall; Scrapfly ASP also refused | after 2 consecutive failures the runtime stops escalating (`premiumAllowedAfterErrors`), so the cost stops by itself. Fix the connector: `fetch.userAgent: browser`, `renderJs: false` (cheaper), or find a feed / sitemap that is open |
75 +| `rss … → 403` at discovery, `discovered 0 urls` | feed blocked for the bot identity (discovery is direct-only, never premium) | connector: `fetch.userAgent: browser` or another feed URL. The scheduler retries discovery on its cadence (≥ 6 h), never every tick |
76 +| `discovered 0 urls` with a working feed (`filter: 0/20 items match`) | connector filter too narrow | connector code / YAML; not a runtime issue |
77 +| `robots.txt disallows` | legal stop — leave it | remove the path from discovery |
78 +| `fetch_failed`, `timeout`, `Headers Timeout` | upstream slow (Overpass, big PDFs) | raise `fetch.timeoutMs` for the connector or lower `fetch.concurrency`; documents back off ×2^(n−1) automatically |
79 +| `partial` with validation errors only | extractor problem (`implausible MW`, `missing name`) | parser bug: file/line in the run log; data is **not** ingested for rejected entities |
80 +| every refetch `CHANGED` with `diff_summary` null, versions 2 = fetch_count | two runs on the same connector overlapped (fixed 2026-09-11 by the per-connector run lock and the scheduler skipping group jobs while a `full` job is pending) | should not recur; if it does, check `dci_worker_jobs_total{result="deferred"}` and `dci:run-lock:*` in Redis |
81 +| every refetch `CHANGED` with a small `ratio` (< 0.1) and sidebars in `added/removed` | page embeds "related articles" / ads: the fingerprint moves although the article did not | connector: extract main content only (`each`/`match` on the article container); the `change_frequency_score` will keep re-checking such pages twice as often until fixed |
82 +| connector `never_run`, `enabled = f` | disabled in YAML | intentional (17 operator connectors are parked) |
83 +
84 +Pause / resume without redeploying: `dcc run --rm cli pause <id>` / `resume <id>` (sets `connectors.paused`).
85 +Force one run now: `dcc run --rm cli enqueue <id> --task crawl --group newsroom` (a pending job with the same id is
86 +coalesced; a run already in progress makes the job wait 60 s on the run lock).
87 +
88 +## 4. Budgets and cost control
89 +
90 +Four independent caps, all enforced in `apps/worker/src/context.ts` before a fetch is allowed to escalate:
91 +
92 +1. **Per run** — `fetch.maxCreditsPerRun` (YAML, default 200): once spent, the rest of the run is direct-only.
93 +2. **Per connector per day** — `fetch.maxCreditsPerDay` (YAML, optional): Redis `dci:budget:connector:<id>:<day>`
94 + (UTC day, INCRBYFLOAT, 3-day TTL), shared by every worker. A warning is logged once per run when reached.
95 +3. **Per provider per day** — `DCI_SCRAPFLY_DAILY_BUDGET` / `DCI_FIRECRAWL_DAILY_BUDGET` (400 / 200): Redis
96 + `dci:budget:<provider>:<day>` shared by every worker (`RedisBudgetStore`, refreshed every
97 + `DCI_BUDGET_REFRESH_MS` = 15 s; the in-process counter never lets a worker undercount its own spend). When Redis
98 + is unreachable the in-process counter alone applies and one log line says so.
99 +4. **Never premium** on discovery fetches (groups `sitemap`, `rss`: L1 → L2 only), never on documents that are not
100 + due (`dueDocuments` filters on `next_check`; only `--force` / `--url` runs bypass it), and only every 4th attempt
101 + for a document that already failed twice in a row.
102 +
103 +Reading the counters (the API `/api/admin/ops` reads the same keys):
104 +
105 +```bash
106 +ssh BHS128 'cd /srv/dci/app && docker compose -f deploy/compose.data.yml --env-file deploy/.env.data exec -T redis sh -c "redis-cli -a \"\$REDIS_PASSWORD\" --no-auth-warning --scan --pattern \"dci:budget:*\" | sort | while read k; do echo \"\$k \$(redis-cli -a \"\$REDIS_PASSWORD\" --no-auth-warning get \$k)\"; done"'
107 +dpsql -c "select connector_id, sum((stats->>'credits')::float) credits, sum((stats->>'fetched')::int) fetched from connector_runs where started_at >= date_trunc('day', now() at time zone 'utc') and (stats->>'credits')::float > 0 group by 1 order by 2 desc"
108 +```
109 +
110 +Rules of thumb: Scrapfly with `asp + render_js` costs ~5–6 credits per page, `renderJs: false` ~1–2; a news
111 +connector behind a bot wall should have `maxCreditsPerRun ≤ 40` and a `maxCreditsPerDay`. A connector that
112 +needs premium for **every** page is a connector to rethink, not a budget to raise.
113 +
114 +Emergency stop: `dcc run --rm cli pause <id>`; or set `DCI_SCRAPFLY_DAILY_BUDGET=0` in `deploy/.env.crawl` and
115 +`deploy/bin/deploy.sh crawl --no-build` (the fetcher reports itself unavailable, escalation stops at L2).
116 +
117 +## 5. Scheduling model (what the scheduler does every 60 s)
118 +
119 +* Per enabled, unpaused connector: a `full` job if it was never discovered, a `discover` job when the discovery
120 + cadence (`schedule.discovery` → `schedule.sitemap` → shortest group interval, minimum 6 h) has elapsed since
121 + `connector_state.lastDiscoverAt` — a discovery that legitimately finds nothing waits like any other.
122 +* Per group with due documents: one `crawl` job `<connector>__<group>`, **unless** a `full`/`discover` job of that
123 + connector is pending. Job ids dedupe; `enqueueRun` re-adds only when the previous job finished.
124 +* Workers take `dci:run-lock:<connector>` (TTL 4 h) before running; a colliding job is moved to delayed for 60 s
125 + (`dci_worker_jobs_total{result="deferred"}`).
126 +* Maintenance schedulers (UTC): metrics 00:10, rankings 00:30, refresh-stats hourly, cleanup 01:00 (also
127 + reconciles orphaned runs). Orphans (`status = running` older than `DCI_ORPHAN_RUN_HOURS` = 6) are also aborted
128 + at every worker start.
129 +* Priorities: manual enqueue 1, `full` 5, group crawl 10, `discover` 20 (lower = sooner).
130 +
131 +## 6. Resilience behaviour
132 +
133 +| Event | What happens |
134 +|---|---|
135 +| `docker compose stop worker` (SIGTERM) | scheduler loop stops, no new jobs are taken, in-flight documents finish, runs end `aborted` (their remaining documents stay due), heartbeat says `shuttingDown`. Deadline `DCI_SHUTDOWN_TIMEOUT_MS` (50 s) < `stop_grace_period` (90 s): past it, in-flight runs are marked `aborted` in Postgres and the process exits; BullMQ re-queues the active jobs (stalled → retried once). |
136 +| Worker killed hard (OOM, host reboot) | the run lock expires (4 h TTL) or is bypassed by the orphan reconciliation at next start (`aborted`, 6 h); BullMQ's stalled check (60 s, `maxStalledCount: 1`) moves the job back to waiting once, then fails it. |
137 +| Redis unreachable | ioredis reconnects with backoff (one log line per outage); BullMQ workers resume; the scheduler tick fails and retries next interval (`SchedulerStale` after 10 min); budgets fall back to in-process counters. |
138 +| Postgres unreachable | per-document errors are caught (run ends `partial`/`failed`), the run row update is retried at the end; the worker keeps running. |
139 +| ClickHouse down | `crawl_log` / `page_changes` / observations inserts are best effort (`DCI_CLICKHOUSE_OPTIONAL=1`): a warning per failed insert, never a failed run. |
140 +| MinIO down | archive failures are logged per document; the document is still processed (no version diff without the old body). |
141 +| Scheduler container down | the crawl worker's embedded loop keeps scheduling (`DCI_SCHEDULER` defaults to on for a crawl worker); the Redis lock `dci:scheduler:lock` prevents double enqueues when both run. |
142 +
143 +## 7. Scaling
144 +
145 +* **More throughput on BHS64b**: raise `DCI_CRAWL_CONCURRENCY` (jobs in parallel; each job honours its
146 + connector's `fetch.concurrency` and `rpm`), or start the second worker: `dcc --profile scale up -d worker-b`
147 + (metrics on :8322, already a Prometheus target). Two workers never run the same connector at once (run lock).
148 +* **Watch**: `QueueBacklog`, `dci_worker_running_jobs` at the concurrency ceiling for long stretches, p95 fetch
149 + duration, memory of the worker container (`mem_limit` 12 g).
150 +* **Per-host politeness is per process**: the token bucket (`rpm`, `concurrency`) is in-process, so two workers
151 + double the pressure on a host — keep `rpm` conservative when scaling out.
152 +* **More connectors**: cost is dominated by discovery on the first run (`full`); `discovery.maxUrlsPerRun` now
153 + keeps the highest-priority groups (facility pages, seeds) rather than the first N sitemap entries.
154 +* **Maintenance jobs** run on the data node (`worker-maint`, `DCI_MAINTENANCE_CONCURRENCY` 2) and on the crawl
155 + worker; heavy rankings/metrics can be moved off the crawl node with `DCI_QUEUES=crawl` in `.env.crawl`.
156 +
157 +## 8. Backups (data node)
158 +
159 +Nightly `dci-backup.timer` 03:30 → `deploy/bin/remote/backup-run.sh`; on demand `deploy/bin/backup.sh`
160 +(≈ 1 min). Off-node mirror: `ubuntu@BHS64b:/srv/dci-backups` (rsync over dci0). Verify: `ssh BHS64b ls -la
161 +/srv/dci-backups/pg`. Known defect (2026-09-11): the ClickHouse `BACKUP DATABASE` zip is written by the container
162 +user (uid 101, mode 640) and is skipped by the rsync (`Permission denied`, exit 23) until `backup-run.sh` chowns it
163 +— Postgres and config backups are unaffected.
modified packages/connectors/src/discovery.ts +20 −3
@@ -83,10 +83,25 @@ export function classifyUrl(cfg: ConnectorConfig, url: string, discoveredFrom: s
83 83 return { url, group: "default", priority: 30, discoveredFrom };
84 84 }
85 85
86 +/**
87 + * Cap a classified URL set to `max`: every candidate is classified first, then the most valuable ones are kept —
88 + * higher priority first (seeds 80, facility groups typically 60, newsroom 50–70, `default` 30), ties by discovery
89 + * order. Truncating a sitemap by document order instead would silently drop facility pages that happen to be listed
90 + * after thousands of blog posts.
91 + */
92 +export function capDiscovered(urls: Iterable<DiscoveredUrl>, max: number): DiscoveredUrl[] {
93 + const all = [...urls];
94 + if (!Number.isFinite(max) || max <= 0 || all.length <= max) return all;
95 + const rank = (u: DiscoveredUrl) => (u.priority ?? 50) + (u.group && u.group !== "default" ? 0.5 : 0);
96 + return all.map((u, i) => ({ u, i })).sort((a, b) => rank(b.u) - rank(a.u) || a.i - b.i).slice(0, max).map((x) => x.u);
97 +}
98 +
86 99 /** Generic discovery: sitemaps (auto or listed), RSS feeds, seeds, and link-following from index pages. */
87 100 export async function discoverGeneric(cfg: ConnectorConfig, ctx: ConnectorContext): Promise<DiscoveredUrl[]> {
88 101 const found = new Map<string, DiscoveredUrl>();
89 − const add = (u: DiscoveredUrl | null) => { if (u && !found.has(u.url) && found.size < cfg.discovery.maxUrlsPerRun) found.set(u.url, u); };
102 + // classify everything, cap at the end (capDiscovered); the hard ceiling only bounds memory on pathological sitemaps
103 + const hardCap = Math.max(cfg.discovery.maxUrlsPerRun * 10, 50_000);
104 + const add = (u: DiscoveredUrl | null) => { if (u && !found.has(u.url) && found.size < hardCap) found.set(u.url, u); };
90 105 const origin = `https://${cfg.domain.replace(/^https?:\/\//, "").replace(/\/$/, "")}`;
91 106
92 107 // seeds
@@ -133,7 +148,7 @@ export async function discoverGeneric(cfg: ConnectorConfig, ctx: ConnectorContex
133 148 if (cfg.discovery.pagination) for (const ip of indexPages) for (let n = cfg.discovery.pagination.start; n <= cfg.discovery.pagination.max; n++) pages.push(ip.url.replace(/\/$/, "") + cfg.discovery.pagination.template.replace("{n}", String(n)));
134 149 let stale = 0;
135 150 for (const p of pages) {
136 − if (found.size >= cfg.discovery.maxUrlsPerRun) break;
151 + if (found.size >= hardCap) break;
137 152 const doc = await ctx.fetch(p, { group: "index" });
138 153 if (doc.error || doc.status !== 200) { if (++stale >= 3) break; continue; }
139 154 const before = found.size;
@@ -141,7 +156,9 @@ export async function discoverGeneric(cfg: ConnectorConfig, ctx: ConnectorContex
141 156 if (found.size === before) { if (++stale >= 3) break; } else stale = 0;
142 157 }
143 158 }
144 − return [...found.values()];
159 + const kept = capDiscovered(found.values(), cfg.discovery.maxUrlsPerRun);
160 + if (kept.length < found.size) ctx.log("warn", `discovery capped at ${kept.length}/${found.size} urls (discovery.maxUrlsPerRun) — highest-priority groups kept`);
161 + return kept;
145 162 }
146 163
147 164 /** Robots-declared sitemaps are useful even when discovery.sitemap is false. */
148 165