spb/datacenterindex
Public
HTML 53.9%
TypeScript 44.5%
JavaScript 0.6%
SQL 0.5%
1import { createHash } from "node:crypto";23export function sha256(input: string | Uint8Array): string {4 return createHash("sha256").update(input).digest("hex");5}67/** Canonical URL for fingerprinting: lowercase host, no fragment, sorted query, trailing slash trimmed, tracking params dropped. */8export function canonicalUrl(raw: string): string {9 const u = new URL(raw);10 u.hash = "";11 u.hostname = u.hostname.toLowerCase();12 if ((u.protocol === "https:" && u.port === "443") || (u.protocol === "http:" && u.port === "80")) u.port = "";13 const drop = /^(utm_|fbclid|gclid|mc_cid|mc_eid|ref$|source$|_ga$|_gl$)/i;14 const params = [...u.searchParams.entries()].filter(([k]) => !drop.test(k)).sort(([a], [b]) => a.localeCompare(b));15 u.search = "";16 for (const [k, v] of params) u.searchParams.append(k, v);17 let s = u.toString();18 if (u.pathname !== "/" && s.endsWith("/") && !u.search) s = s.slice(0, -1);19 return s;20}2122export function urlFingerprint(raw: string): string {23 return sha256(canonicalUrl(raw)).slice(0, 32);24}2526/**27 * Content fingerprint that ignores volatile noise: whitespace runs, HTML comments, script/style bodies,28 * nonces, CSRF tokens, timestamps in attributes. Two fetches of an unchanged page should hash identically.29 */30export function contentFingerprint(body: string, contentType: string | null | undefined): string {31 let s = body;32 if (!contentType || /html/i.test(contentType)) {33 s = s34 .replace(/<!--[\s\S]*?-->/g, "")35 .replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, "")36 .replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, "")37 .replace(/<noscript\b[^>]*>[\s\S]*?<\/noscript>/gi, "")38 .replace(/\s(nonce|data-nonce|csrf-token|data-csrf|data-timestamp|data-build|data-reactroot)="[^"]*"/gi, "")39 .replace(/<meta[^>]+(csrf|nonce|build|generated)[^>]*>/gi, "")40 .replace(/<input[^>]+type="hidden"[^>]*>/gi, "");41 }42 s = s.replace(/\s+/g, " ").trim();43 return sha256(s);44}4546/** Plain-text projection of HTML for diffing and regex extraction. */47export function htmlToText(html: string): string {48 return html49 .replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, " ")50 .replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, " ")51 .replace(/<!--[\s\S]*?-->/g, " ")52 .replace(/<\/(p|div|li|tr|h[1-6]|section|article|br|table|ul|ol|dd|dt)>/gi, "\n")53 .replace(/<br\s*\/?>/gi, "\n")54 .replace(/<[^>]+>/g, " ")55 .replace(/ /g, " ")56 .replace(/&/g, "&")57 .replace(/</g, "<")58 .replace(/>/g, ">")59 .replace(/"/g, '"')60 .replace(/'|'/g, "'")61 .replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d)))62 .replace(/[ \t]+/g, " ")63 .replace(/\n\s*\n+/g, "\n")64 .trim();65}66