|
1 |
+# Trouve-KA — enfilement d'URLs depuis un sitemap |
|
2 |
+# Author: Simon-Pierre Boucher |
|
3 |
+# Contact: contact@spboucher.ai |
|
4 |
+ |
|
5 |
+"""Enfile dans le frontier les URLs d'un sitemap (urlset ou sitemapindex). |
|
6 |
+ |
|
7 |
+Usage : .venv/bin/python scripts/bootstrap-seeds/enqueue-sitemap.py <sitemap_url> [...] |
|
8 |
+ [--priority 0.9] [--limit 5000] [--no-follow-index] |
|
9 |
+ |
|
10 |
+- Suit les index de sitemaps (1 niveau), dans l'ordre du fichier. |
|
11 |
+- Trie les URLs par profondeur de chemin (pages de haut niveau d'abord) |
|
12 |
+ avant de tronquer à --limit, pour respecter le cap par domaine avec |
|
13 |
+ les pages les plus importantes. |
|
14 |
+- N'enfile que les URLs appartenant au domaine du sitemap (garde-fou). |
|
15 |
+- Le cap dur de 5000 URLs connues/domaine reste appliqué par la base |
|
16 |
+ (enqueue_urls_bulk). |
|
17 |
+""" |
|
18 |
+ |
|
19 |
+import argparse |
|
20 |
+import asyncio |
|
21 |
+import gzip |
|
22 |
+import re |
|
23 |
+import sys |
|
24 |
+import urllib.request |
|
25 |
+ |
|
26 |
+from trouveka.config import get_settings |
|
27 |
+from trouveka.database import Database |
|
28 |
+from trouveka.logging import get_logger |
|
29 |
+from trouveka.shared import canonicalize_url, extract_domain |
|
30 |
+ |
|
31 |
+log = get_logger("crawler.enqueue_sitemap") |
|
32 |
+ |
|
33 |
+LOC_RE = re.compile(r"<loc>\s*([^<\s]+)\s*</loc>") |
|
34 |
+CHUNK = 1000 |
|
35 |
+ |
|
36 |
+ |
|
37 |
+def fetch(url: str, user_agent: str, timeout: float = 30.0) -> str: |
|
38 |
+ req = urllib.request.Request(url, headers={"User-Agent": user_agent}) |
|
39 |
+ with urllib.request.urlopen(req, timeout=timeout) as resp: |
|
40 |
+ body = resp.read(20_000_000) |
|
41 |
+ if resp.headers.get("Content-Encoding") == "gzip" or url.endswith(".gz"): |
|
42 |
+ try: |
|
43 |
+ body = gzip.decompress(body) |
|
44 |
+ except OSError: |
|
45 |
+ pass |
|
46 |
+ return body.decode("utf-8", errors="replace") |
|
47 |
+ |
|
48 |
+ |
|
49 |
+def collect_locs(sitemap_url: str, user_agent: str, *, follow_index: bool) -> list[str]: |
|
50 |
+ body = fetch(sitemap_url, user_agent) |
|
51 |
+ locs = LOC_RE.findall(body) |
|
52 |
+ if "<sitemapindex" in body: |
|
53 |
+ if not follow_index: |
|
54 |
+ log.info("index ignoré (--no-follow-index)", extra={"ctx": {"url": sitemap_url}}) |
|
55 |
+ return [] |
|
56 |
+ urls: list[str] = [] |
|
57 |
+ for child in locs: |
|
58 |
+ try: |
|
59 |
+ child_body = fetch(child, user_agent) |
|
60 |
+ except Exception as exc: # noqa: BLE001 — un enfant cassé ne bloque pas le reste |
|
61 |
+ log.info("sitemap enfant illisible", extra={"ctx": {"url": child, "err": str(exc)}}) |
|
62 |
+ continue |
|
63 |
+ if "<sitemapindex" in child_body: |
|
64 |
+ continue # un seul niveau d'indirection |
|
65 |
+ urls.extend(LOC_RE.findall(child_body)) |
|
66 |
+ return urls |
|
67 |
+ return locs |
|
68 |
+ |
|
69 |
+ |
|
70 |
+async def enqueue(sitemap_url: str, *, priority: float, limit: int, follow_index: bool) -> int: |
|
71 |
+ settings = get_settings() |
|
72 |
+ sitemap_domain = extract_domain(sitemap_url) |
|
73 |
+ if not sitemap_domain: |
|
74 |
+ log.info("URL de sitemap invalide", extra={"ctx": {"url": sitemap_url}}) |
|
75 |
+ return 0 |
|
76 |
+ |
|
77 |
+ raw = collect_locs(sitemap_url, settings.crawler_user_agent, follow_index=follow_index) |
|
78 |
+ items: list[tuple[str, str, float]] = [] |
|
79 |
+ seen: set[str] = set() |
|
80 |
+ for loc in raw: |
|
81 |
+ url = canonicalize_url(loc) |
|
82 |
+ if not url or url in seen: |
|
83 |
+ continue |
|
84 |
+ domain = extract_domain(url) |
|
85 |
+ if domain != sitemap_domain: |
|
86 |
+ continue # garde-fou : on reste sur le domaine du sitemap |
|
87 |
+ seen.add(url) |
|
88 |
+ items.append((url, domain, priority)) |
|
89 |
+ |
|
90 |
+ # Pages de haut niveau d'abord (moins de segments de chemin), ordre du |
|
91 |
+ # sitemap préservé à profondeur égale, puis troncature au budget. |
|
92 |
+ items.sort(key=lambda t: len([s for s in t[0].split("/", 3)[-1].split("/") if s])) |
|
93 |
+ items = items[:limit] |
|
94 |
+ |
|
95 |
+ db = Database(settings.database_url, pool_min=1, pool_max=3) |
|
96 |
+ await db.connect() |
|
97 |
+ added = 0 |
|
98 |
+ try: |
|
99 |
+ for i in range(0, len(items), CHUNK): |
|
100 |
+ _, n = await db.enqueue_urls_bulk(items[i : i + CHUNK], depth=1, source_url_id=None) |
|
101 |
+ added += n |
|
102 |
+ finally: |
|
103 |
+ await db.close() |
|
104 |
+ log.info( |
|
105 |
+ "sitemap enfilé", |
|
106 |
+ extra={"ctx": {"sitemap": sitemap_url, "locs": len(raw), "candidates": len(items), "added": added}}, |
|
107 |
+ ) |
|
108 |
+ return added |
|
109 |
+ |
|
110 |
+ |
|
111 |
+def main() -> None: |
|
112 |
+ ap = argparse.ArgumentParser(description=__doc__) |
|
113 |
+ ap.add_argument("sitemaps", nargs="+", help="URL(s) de sitemap.xml ou d'index") |
|
114 |
+ ap.add_argument("--priority", type=float, default=0.9) |
|
115 |
+ ap.add_argument("--limit", type=int, default=5000, help="max d'URLs candidates par sitemap") |
|
116 |
+ ap.add_argument("--no-follow-index", action="store_true") |
|
117 |
+ args = ap.parse_args() |
|
118 |
+ |
|
119 |
+ total = 0 |
|
120 |
+ for sm in args.sitemaps: |
|
121 |
+ total += asyncio.run( |
|
122 |
+ enqueue(sm, priority=args.priority, limit=args.limit, follow_index=not args.no_follow_index) |
|
123 |
+ ) |
|
124 |
+ print(f"total ajouté au frontier : {total}", file=sys.stderr) |
|
125 |
+ |
|
126 |
+ |
|
127 |
+if __name__ == "__main__": |
|
128 |
+ main() |