import { createHash } from "node:crypto"; export function sha256(input: string | Uint8Array): string { return createHash("sha256").update(input).digest("hex"); } /** Canonical URL for fingerprinting: lowercase host, no fragment, sorted query, trailing slash trimmed, tracking params dropped. */ export function canonicalUrl(raw: string): string { const u = new URL(raw); u.hash = ""; u.hostname = u.hostname.toLowerCase(); if ((u.protocol === "https:" && u.port === "443") || (u.protocol === "http:" && u.port === "80")) u.port = ""; const drop = /^(utm_|fbclid|gclid|mc_cid|mc_eid|ref$|source$|_ga$|_gl$)/i; const params = [...u.searchParams.entries()].filter(([k]) => !drop.test(k)).sort(([a], [b]) => a.localeCompare(b)); u.search = ""; for (const [k, v] of params) u.searchParams.append(k, v); let s = u.toString(); if (u.pathname !== "/" && s.endsWith("/") && !u.search) s = s.slice(0, -1); return s; } export function urlFingerprint(raw: string): string { return sha256(canonicalUrl(raw)).slice(0, 32); } /** * Content fingerprint that ignores volatile noise: whitespace runs, HTML comments, script/style bodies, * nonces, CSRF tokens, timestamps in attributes. Two fetches of an unchanged page should hash identically. */ export function contentFingerprint(body: string, contentType: string | null | undefined): string { let s = body; if (!contentType || /html/i.test(contentType)) { s = s .replace(//g, "") .replace(/]*>[\s\S]*?<\/script>/gi, "") .replace(/]*>[\s\S]*?<\/style>/gi, "") .replace(/]*>[\s\S]*?<\/noscript>/gi, "") .replace(/\s(nonce|data-nonce|csrf-token|data-csrf|data-timestamp|data-build|data-reactroot)="[^"]*"/gi, "") .replace(/]+(csrf|nonce|build|generated)[^>]*>/gi, "") .replace(/]+type="hidden"[^>]*>/gi, ""); } s = s.replace(/\s+/g, " ").trim(); return sha256(s); } /** Plain-text projection of HTML for diffing and regex extraction. */ export function htmlToText(html: string): string { return html .replace(/]*>[\s\S]*?<\/script>/gi, " ") .replace(/]*>[\s\S]*?<\/style>/gi, " ") .replace(//g, " ") .replace(/<\/(p|div|li|tr|h[1-6]|section|article|br|table|ul|ol|dd|dt)>/gi, "\n") .replace(//gi, "\n") .replace(/<[^>]+>/g, " ") .replace(/ /g, " ") .replace(/&/g, "&") .replace(/</g, "<") .replace(/>/g, ">") .replace(/"/g, '"') .replace(/'|'/g, "'") .replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d))) .replace(/[ \t]+/g, " ") .replace(/\n\s*\n+/g, "\n") .trim(); }