import { afterEach, beforeEach, describe, expect, it } from "vitest"; import type { FetchLevel } from "@dci/core"; import type { Fetcher, FetchOptions, RawDocument } from "./types.js"; import { fetchWithEscalation, fetchers } from "./fetchers.js"; /** * fetchWithEscalation with the four level fetchers stubbed (no network). Each stub records the levels it was asked * for and returns a scripted document; the real registry entries are restored after every test. */ function stub(level: FetchLevel, script: (opts: FetchOptions) => Partial | { unavailable: true }, calls: FetchLevel[]): Fetcher { return { name: level >= 4 ? "scrapfly" : level === 3 ? "firecrawl" : "direct", level, available() { const r = script({}); return !("unavailable" in r); }, async fetch(url, opts = {}) { calls.push(level); const r = script(opts) as Partial; const body = Buffer.from(r.text ?? "", "utf8"); return { url, finalUrl: url, fetchedAt: "2026-09-12T00:00:00Z", status: 200, contentType: "text/html", body, text: "", headers: {}, etag: null, lastModified: null, notModified: false, fetcher: this.name, level, durationMs: 1, credits: level === 3 ? 1 : level === 4 ? 6 : 0, ...r }; }, }; } const RICH = "" + "

Real facility content, 12 MW, Ashburn, colocation.

".repeat(40) + ""; const SHELL = '
'; const saved: Record = { ...fetchers }; let calls: FetchLevel[] = []; beforeEach(() => { calls.length = 0; }); afterEach(() => { for (const l of [1, 2, 3, 4] as FetchLevel[]) fetchers[l] = saved[l]; }); describe("fetchWithEscalation", () => { it("returns the first good direct document without touching premium levels", async () => { fetchers[1] = stub(1, () => ({ text: RICH }), calls); fetchers[2] = stub(2, () => ({ text: RICH }), calls); fetchers[3] = stub(3, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 4 }); expect(d.level).toBe(1); expect(calls).toEqual([1]); expect(d.meta?.attempts).toEqual(["L1:200"]); }); it("escalates past a 403 and a JS shell, then stops at the first real page", async () => { fetchers[1] = stub(1, () => ({ status: 403, text: "" }), calls); fetchers[2] = stub(2, () => ({ text: SHELL }), calls); fetchers[3] = stub(3, () => ({ text: RICH }), calls); fetchers[4] = stub(4, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 4 }); expect(calls).toEqual([1, 2, 3]); expect(d.level).toBe(3); expect(d.meta?.attempts).toEqual(["L1:403", "L2:200", "L3:200"]); }); it("HTTP 429 is never escalated and exposes retryAfterMs from Retry-After", async () => { fetchers[1] = stub(1, () => ({ status: 429, headers: { "retry-after": "90" }, text: "" }), calls); fetchers[2] = stub(2, () => ({ text: RICH }), calls); fetchers[3] = stub(3, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 4 }); expect(calls).toEqual([1]); expect(d.status).toBe(429); expect(d.meta).toMatchObject({ rateLimited: true, retryAfterMs: 90_000, attempts: ["L1:429"] }); expect(d.credits).toBe(0); }); it("a 429 without Retry-After still stops escalation (no retryAfterMs key)", async () => { fetchers[1] = stub(1, () => ({ status: 429, text: "" }), calls); fetchers[2] = stub(2, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(calls).toEqual([1]); expect(d.meta?.rateLimited).toBe(true); expect(d.meta).not.toHaveProperty("retryAfterMs"); }); it("creditsLeft: skips premium levels whose expected cost exceeds the remaining run budget", async () => { fetchers[1] = stub(1, () => ({ status: 403, text: "" }), calls); fetchers[2] = stub(2, () => ({ status: 403, text: "" }), calls); fetchers[3] = stub(3, () => ({ text: SHELL, credits: 1 }), calls); // Firecrawl returns a shell → would escalate to L4 fetchers[4] = stub(4, () => ({ text: RICH, credits: 6 }), calls); // 5 credits left: L3 (1) is affordable, L4 (6) is not → never spends on both const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 4, creditsLeft: 5 }); expect(calls).toEqual([1, 2, 3]); expect(d.level).toBe(3); // L3 spent 1 of the 5 credits; the skipped L4 (cost 6 > 4 left) is visible in the attempt trace expect(d.meta?.attempts).toEqual(["L1:403", "L2:403", "L3:200", "L4:over_budget(6>4)"]); // 1 credit left and Scrapfly ASP-only (cost 1): after L3 spends it, L4 is over budget calls.length = 0; const d2 = await fetchWithEscalation("https://x.com/a", { maxLevel: 4, creditsLeft: 1, renderJs: false }); expect(calls).toEqual([1, 2, 3]); expect(d2.level).toBe(3); // 0 credits left: no premium at all, best direct attempt is returned calls.length = 0; const d3 = await fetchWithEscalation("https://x.com/a", { maxLevel: 4, creditsLeft: 0 }); expect(calls).toEqual([1, 2]); expect(d3.status).toBe(403); expect(d3.meta?.attempts).toEqual(["L1:403", "L2:403", "L3:over_budget(1>0)", "L4:over_budget(6>0)"]); // no creditsLeft given → legacy behaviour, both premium levels reachable calls.length = 0; const d4 = await fetchWithEscalation("https://x.com/a", { maxLevel: 4 }); expect(calls).toEqual([1, 2, 3, 4]); expect(d4.level).toBe(4); }); it("unavailable premium fetchers are skipped and recorded", async () => { fetchers[1] = stub(1, () => ({ status: 403, text: "" }), calls); fetchers[2] = stub(2, () => ({ status: 403, text: "" }), calls); fetchers[3] = stub(3, () => ({ unavailable: true }), calls); fetchers[4] = stub(4, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 4 }); expect(calls).toEqual([1, 2, 4]); expect(d.meta?.attempts).toEqual(["L1:403", "L2:403", "L3:unavailable", "L4:200"]); }); it("hard failures (ssrf, dns, too_large, 404, 410) are returned immediately", async () => { for (const [code, status] of [["ssrf_blocked", 0], ["dns", 0], ["too_large", 200]] as const) { calls.length = 0; fetchers[1] = stub(1, () => ({ status, text: "", error: { code, message: code } }), calls); fetchers[2] = stub(2, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(calls, code).toEqual([1]); expect(d.error?.code).toBe(code); } for (const status of [404, 410]) { calls.length = 0; fetchers[1] = stub(1, () => ({ status, text: "" }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(calls).toEqual([1]); expect(d.status).toBe(status); } }); it("propagates noarchive from headers or meta robots on whichever level served the page", async () => { fetchers[1] = stub(1, () => ({ status: 403, text: "" }), calls); fetchers[2] = stub(2, () => ({ text: RICH, headers: { "x-robots-tag": "noarchive" } }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(d.level).toBe(2); expect(d.meta).toMatchObject({ noarchive: true, robotsDirective: "header:noarchive" }); calls.length = 0; fetchers[1] = stub(1, () => ({ text: '' + RICH + "" }), calls); const d2 = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(d2.meta).toMatchObject({ noarchive: true, robotsDirective: "meta:noindex" }); fetchers[1] = stub(1, () => ({ text: RICH }), calls); const clean = await fetchWithEscalation("https://x.com/a", { maxLevel: 2 }); expect(clean.meta?.noarchive).toBeUndefined(); }); it("304 is returned as-is", async () => { fetchers[1] = stub(1, () => ({ status: 304, notModified: true, text: "" }), calls); fetchers[2] = stub(2, () => ({ text: RICH }), calls); const d = await fetchWithEscalation("https://x.com/a", { maxLevel: 2, etag: '"e"' }); expect(d.notModified).toBe(true); expect(calls).toEqual([1]); }); });