SPB Git forge

spb/trouve-ka

Public ★ Pinned

Trouve-KA — moteur de recherche web indépendant, Québec-first. Crawler distribué, index OpenSearch, ranking bilingue, galerie d'images. En prod : www.trouve-ka.com

40commits 1branches 0releases
9.2 MBsize
maindefault branch
23 days agolast push
Python 51.9% TypeScript 32.7% JavaScript 4.8% CSS 4.1% Shell 2.9% HTML 1.8% SQL 1.5%
4.6 KB · 129 lines python
Raw Blame History
1# Trouve-KA — enfilement d'URLs depuis un sitemap2# Author: Simon-Pierre Boucher3# Contact: contact@spboucher.ai45"""Enfile dans le frontier les URLs d'un sitemap (urlset ou sitemapindex).67Usage : .venv/bin/python scripts/bootstrap-seeds/enqueue-sitemap.py <sitemap_url> [...]8        [--priority 0.9] [--limit 5000] [--no-follow-index]910- Suit les index de sitemaps (1 niveau), dans l'ordre du fichier.11- Trie les URLs par profondeur de chemin (pages de haut niveau d'abord)12  avant de tronquer à --limit, pour respecter le cap par domaine avec13  les pages les plus importantes.14- N'enfile que les URLs appartenant au domaine du sitemap (garde-fou).15- Le cap dur de 5000 URLs connues/domaine reste appliqué par la base16  (enqueue_urls_bulk).17"""1819import argparse20import asyncio21import gzip22import re23import sys24import urllib.request2526from trouveka.config import get_settings27from trouveka.database import Database28from trouveka.logging import get_logger29from trouveka.shared import canonicalize_url, extract_domain3031log = get_logger("crawler.enqueue_sitemap")3233LOC_RE = re.compile(r"<loc>\s*([^<\s]+)\s*</loc>")34CHUNK = 1000353637def fetch(url: str, user_agent: str, timeout: float = 30.0) -> str:38    req = urllib.request.Request(url, headers={"User-Agent": user_agent})39    with urllib.request.urlopen(req, timeout=timeout) as resp:40        body = resp.read(20_000_000)41        if resp.headers.get("Content-Encoding") == "gzip" or url.endswith(".gz"):42            try:43                body = gzip.decompress(body)44            except OSError:45                pass46        return body.decode("utf-8", errors="replace")474849def collect_locs(sitemap_url: str, user_agent: str, *, follow_index: bool) -> list[str]:50    body = fetch(sitemap_url, user_agent)51    locs = LOC_RE.findall(body)52    if "<sitemapindex" in body:53        if not follow_index:54            log.info("index ignoré (--no-follow-index)", extra={"ctx": {"url": sitemap_url}})55            return []56        urls: list[str] = []57        for child in locs:58            try:59                child_body = fetch(child, user_agent)60            except Exception as exc:  # noqa: BLE001 — un enfant cassé ne bloque pas le reste61                log.info("sitemap enfant illisible", extra={"ctx": {"url": child, "err": str(exc)}})62                continue63            if "<sitemapindex" in child_body:64                continue  # un seul niveau d'indirection65            urls.extend(LOC_RE.findall(child_body))66        return urls67    return locs686970async def enqueue(sitemap_url: str, *, priority: float, limit: int, follow_index: bool) -> int:71    settings = get_settings()72    sitemap_domain = extract_domain(sitemap_url)73    if not sitemap_domain:74        log.info("URL de sitemap invalide", extra={"ctx": {"url": sitemap_url}})75        return 07677    raw = collect_locs(sitemap_url, settings.crawler_user_agent, follow_index=follow_index)78    items: list[tuple[str, str, float]] = []79    seen: set[str] = set()80    for loc in raw:81        url = canonicalize_url(loc)82        if not url or url in seen:83            continue84        domain = extract_domain(url)85        if domain != sitemap_domain:86            continue  # garde-fou : on reste sur le domaine du sitemap87        seen.add(url)88        items.append((url, domain, priority))8990    # Pages de haut niveau d'abord (moins de segments de chemin), ordre du91    # sitemap préservé à profondeur égale, puis troncature au budget.92    items.sort(key=lambda t: len([s for s in t[0].split("/", 3)[-1].split("/") if s]))93    items = items[:limit]9495    db = Database(settings.database_url, pool_min=1, pool_max=3)96    await db.connect()97    added = 098    try:99        for i in range(0, len(items), CHUNK):100            _, n = await db.enqueue_urls_bulk(items[i : i + CHUNK], depth=1, source_url_id=None)101            added += n102    finally:103        await db.close()104    log.info(105        "sitemap enfilé",106        extra={"ctx": {"sitemap": sitemap_url, "locs": len(raw), "candidates": len(items), "added": added}},107    )108    return added109110111def main() -> None:112    ap = argparse.ArgumentParser(description=__doc__)113    ap.add_argument("sitemaps", nargs="+", help="URL(s) de sitemap.xml ou d'index")114    ap.add_argument("--priority", type=float, default=0.9)115    ap.add_argument("--limit", type=int, default=5000, help="max d'URLs candidates par sitemap")116    ap.add_argument("--no-follow-index", action="store_true")117    args = ap.parse_args()118119    total = 0120    for sm in args.sitemaps:121        total += asyncio.run(122            enqueue(sm, priority=args.priority, limit=args.limit, follow_index=not args.no_follow_index)123        )124    print(f"total ajouté au frontier : {total}", file=sys.stderr)125126127if __name__ == "__main__":128    main()129