pivot Groupe KA : le moteur n'indexe plus que les 14 sites KA
- allowlist KA_DOMAINS + is_ka_domain (exclusions.py), garde-fous frontier (db.py) et crawler (worker.py) : aucun fetch/enqueue hors périmètre - politesse réduite 0,25 s sur nos propres domaines, score Québec plancher 1.0 - nouvel index tk-ka-pages + nouvelle base trouveka_ka (ancien corpus web ouvert intact : rollback via KA_ONLY=false + .env.bak-avant-pivot-ka) - ingestion par sitemaps : scripts/ka-sitemaps/ingest.py (streaming, priorité pages/villes > fiches), process PM2 tk-sitemaps (cron 4 h 45) - /api/search : nouveau paramètre site= (filtre par domaine KA) - catégories verticales KA (logement, emplois, restaurants…) dans l'indexeur
10 changed files +307 −91
modified
apps/api/main.py
+2 −1
@@ -144,6 +144,7 @@ async def api_search( | ||
| 144 | 144 | language: str | None = Query(None, pattern="^(fr|en)$"), |
| 145 | 145 | location: str | None = None, |
| 146 | 146 | category: str | None = Query(None, max_length=40), |
| 147 | + site: str | None = Query(None, max_length=300), | |
| 147 | 148 | quebec_only: bool = False, |
| 148 | 149 | freshness: str | None = Query(None, pattern="^(day|week|month|year)$"), |
| 149 | 150 | images: bool = False, |
@@ -152,7 +153,7 @@ async def api_search( | ||
| 152 | 153 | query_text = f"{q} {location}" if location else q |
| 153 | 154 | body, analysis = build_search_body( |
| 154 | 155 | query_text, page=page, limit=limit, language=language, |
| 155 | − category=category, quebec_only=quebec_only, freshness=freshness, | |
| 156 | + category=category, site=site, quebec_only=quebec_only, freshness=freshness, | |
| 156 | 157 | images_only=images, |
| 157 | 158 | ) |
| 158 | 159 | |
modified
packages/config/__init__.py
+5 −0
@@ -31,6 +31,11 @@ class Settings(BaseSettings): | ||
| 31 | 31 | max_per_host_concurrency: int = 2 |
| 32 | 32 | default_host_delay: float = 2.0 # secondes entre deux requêtes vers un même hôte |
| 33 | 33 | |
| 34 | + # Pivot Groupe KA (2026-08-23) : le moteur n'indexe que les sites KA. | |
| 35 | + # ka_only=False restaure le comportement « web québécois ouvert » (rollback). | |
| 36 | + ka_only: bool = True | |
| 37 | + ka_host_delay: float = 0.25 # politesse réduite : ce sont nos propres serveurs | |
| 38 | + | |
| 34 | 39 | # Crawler — limites de sécurité (chaque réponse a des limites) |
| 35 | 40 | max_response_bytes: int = 3_000_000 |
| 36 | 41 | max_redirects: int = 5 |
modified
packages/database/db.py
+12 −0
@@ -14,12 +14,21 @@ from typing import Any | ||
| 14 | 14 | |
| 15 | 15 | import asyncpg |
| 16 | 16 | |
| 17 | +from trouveka.config import get_settings | |
| 18 | +from trouveka.shared import is_in_scope | |
| 19 | + | |
| 17 | 20 | |
| 18 | 21 | def _dsn(url: str) -> str: |
| 19 | 22 | # asyncpg accepte postgresql:// mais pas postgresql+asyncpg:// |
| 20 | 23 | return url.replace("postgresql+asyncpg://", "postgresql://") |
| 21 | 24 | |
| 22 | 25 | |
| 26 | +def _domain_in_scope(domain: str) -> bool: | |
| 27 | + """Garde-fou du pivot KA-only (2026-08-23) : dernier rempart avant le | |
| 28 | + frontier — même une soumission/un seed accidentel hors Groupe KA est refusé.""" | |
| 29 | + return not get_settings().ka_only or is_in_scope(domain) | |
| 30 | + | |
| 31 | + | |
| 23 | 32 | class Database: |
| 24 | 33 | def __init__(self, database_url: str, *, pool_min: int = 2, pool_max: int = 10): |
| 25 | 34 | self._url = _dsn(database_url) |
@@ -135,6 +144,8 @@ class Database: | ||
| 135 | 144 | max_urls_per_domain: int = 5000, |
| 136 | 145 | ) -> int | None: |
| 137 | 146 | """Ajoute une URL au frontier si inconnue. Retourne url_id si ajoutée, None sinon.""" |
| 147 | + if not _domain_in_scope(domain): | |
| 148 | + return None | |
| 138 | 149 | async with self.pool.acquire() as conn: |
| 139 | 150 | async with conn.transaction(): |
| 140 | 151 | domain_id = await conn.fetchval( |
@@ -199,6 +210,7 @@ class Database: | ||
| 199 | 210 | Ici : 1 transaction, 1 verrou par domaine, ordre de verrouillage stable |
| 200 | 211 | (tri par domaine puis par id) pour éviter les deadlocks croisés. |
| 201 | 212 | """ |
| 213 | + items = [t for t in items if _domain_in_scope(t[1])] | |
| 202 | 214 | if not items: |
| 203 | 215 | return {}, 0 |
| 204 | 216 | domains = sorted({d for _, d, _ in items}) |
modified
packages/shared/__init__.py
+4 −1
@@ -7,7 +7,7 @@ | ||
| 7 | 7 | from .urls import canonicalize_url, display_url, extract_domain, is_http_url |
| 8 | 8 | from .ssrf import is_safe_url, is_safe_ip |
| 9 | 9 | from .hashing import content_hash, text_fingerprint |
| 10 | −from .exclusions import EXCLUDED_DOMAINS, is_excluded_domain | |
| 10 | +from .exclusions import EXCLUDED_DOMAINS, KA_DOMAINS, is_excluded_domain, is_in_scope, is_ka_domain | |
| 11 | 11 | |
| 12 | 12 | __all__ = [ |
| 13 | 13 | "canonicalize_url", |
@@ -15,7 +15,10 @@ __all__ = [ | ||
| 15 | 15 | "extract_domain", |
| 16 | 16 | "is_http_url", |
| 17 | 17 | "EXCLUDED_DOMAINS", |
| 18 | + "KA_DOMAINS", | |
| 18 | 19 | "is_excluded_domain", |
| 20 | + "is_in_scope", | |
| 21 | + "is_ka_domain", | |
| 19 | 22 | "is_safe_url", |
| 20 | 23 | "is_safe_ip", |
| 21 | 24 | "content_hash", |
modified
packages/shared/exclusions.py
+31 −0
@@ -48,3 +48,34 @@ def is_excluded_domain(domain: str) -> bool: | ||
| 48 | 48 | return True |
| 49 | 49 | _, _, d = d.partition(".") |
| 50 | 50 | return False |
| 51 | + | |
| 52 | + | |
| 53 | +# --------------------------------------------------------------------------- | |
| 54 | +# Pivot 2026-08-23 — Trouve·Ka devient le moteur de recherche du GROUPE KA. | |
| 55 | +# Le périmètre de crawl est une ALLOWLIST stricte : seuls les sites de | |
| 56 | +# l'écosystème KA sont découverts, crawlés et indexés. | |
| 57 | +# | |
| 58 | +# Exclus volontairement : api-ka.com (API pure, rien d'indexable), | |
| 59 | +# forma-ka.com (aucune couche SEO/sitemap à ce jour), | |
| 60 | +# administration-ka.com (console privée). | |
| 61 | +KA_DOMAINS: frozenset[str] = frozenset({ | |
| 62 | + "groupe-ka.com", "trouve-ka.com", | |
| 63 | + "lou-ka.com", "immo-ka.com", "vrai-prix.com", "toit-ka.com", "valoplex.com", | |
| 64 | + "auto-ka.com", "fabri-ka.com", "food-ka.com", "resto-ka.com", | |
| 65 | + "sorti-ka.com", "crea-ka.com", "job-ka.com", | |
| 66 | +}) | |
| 67 | + | |
| 68 | + | |
| 69 | +def is_ka_domain(domain: str) -> bool: | |
| 70 | + """Le domaine (ou l'un de ses parents) appartient-il au Groupe KA ?""" | |
| 71 | + d = domain.lower().rstrip(".") | |
| 72 | + while d: | |
| 73 | + if d in KA_DOMAINS: | |
| 74 | + return True | |
| 75 | + _, _, d = d.partition(".") | |
| 76 | + return False | |
| 77 | + | |
| 78 | + | |
| 79 | +def is_in_scope(domain: str) -> bool: | |
| 80 | + """Périmètre de crawl du moteur : domaines du Groupe KA uniquement.""" | |
| 81 | + return is_ka_domain(domain) | |
modified
scripts/bootstrap-seeds/seeds.txt
+8 −85
@@ -1,95 +1,18 @@ | ||
| 1 | −# Trouve-KA — seeds de démarrage (qualité > quantité, CLAUDE.md §8) | |
| 2 | −# Author: Simon-Pierre Boucher | |
| 3 | −# Contact: contact@spboucher.ai | |
| 4 | −# Nœuds fortement connectés du web québécois. Une URL par ligne, # = commentaire. | |
| 5 | − | |
| 6 | −# --- Gouvernement du Québec --- | |
| 7 | −https://www.quebec.ca/ | |
| 8 | −https://www.assnat.qc.ca/ | |
| 9 | −https://www.revenuquebec.ca/ | |
| 10 | −https://www.ramq.gouv.qc.ca/ | |
| 11 | −https://saaq.gouv.qc.ca/ | |
| 12 | −https://www.cnesst.gouv.qc.ca/ | |
| 13 | −https://www.hydroquebec.com/ | |
| 14 | −https://www.investquebec.com/ | |
| 15 | −https://www.transitionenergetique.gouv.qc.ca/ | |
| 16 | − | |
| 17 | −# --- Municipalités --- | |
| 18 | −https://montreal.ca/ | |
| 19 | −https://www.ville.quebec.qc.ca/ | |
| 20 | −https://www.laval.ca/ | |
| 21 | −https://www.gatineau.ca/ | |
| 22 | −https://www.sherbrooke.ca/ | |
| 23 | −https://www.longueuil.quebec/ | |
| 24 | −https://www.trois-rivieres.ca/ | |
| 25 | −https://ville.saguenay.ca/ | |
| 26 | −https://www.levis.ca/ | |
| 27 | −https://www.terrebonne.ca/ | |
| 28 | −https://www.drummondville.ca/ | |
| 29 | −https://www.rimouski.ca/ | |
| 30 | − | |
| 31 | −# --- Universités et cégeps --- | |
| 32 | −https://www.ulaval.ca/ | |
| 33 | −https://www.umontreal.ca/ | |
| 34 | −https://www.mcgill.ca/ | |
| 35 | −https://uqam.ca/ | |
| 36 | −https://www.usherbrooke.ca/ | |
| 37 | −https://www.concordia.ca/ | |
| 38 | −https://www.polymtl.ca/ | |
| 39 | −https://www.etsmtl.ca/ | |
| 40 | −https://www.hec.ca/ | |
| 41 | −https://www.uqac.ca/ | |
| 42 | −https://www.uqtr.ca/ | |
| 43 | −https://www.uqar.ca/ | |
| 44 | −https://uqo.ca/ | |
| 45 | −https://www.inrs.ca/ | |
| 46 | −https://www.cegepsquebec.ca/ | |
| 47 | − | |
| 48 | −# --- Médias --- | |
| 49 | −https://www.lapresse.ca/ | |
| 50 | −https://www.ledevoir.com/ | |
| 51 | −https://www.journaldemontreal.com/ | |
| 52 | −https://www.journaldequebec.com/ | |
| 53 | −https://ici.radio-canada.ca/ | |
| 54 | −https://www.tvanouvelles.ca/ | |
| 55 | −https://www.lesoleil.com/ | |
| 56 | −https://www.ledroit.com/ | |
| 57 | −https://www.latribune.ca/ | |
| 58 | −https://www.lenouvelliste.ca/ | |
| 59 | −https://www.noovo.info/ | |
| 60 | −https://www.lactualite.com/ | |
| 61 | − | |
| 62 | −# --- Affaires, annuaires, associations --- | |
| 63 | −https://www.registreentreprises.gouv.qc.ca/ | |
| 64 | −https://www.desjardins.com/ | |
| 65 | −https://www.fccq.ca/ | |
| 66 | −https://www.ccmm.ca/ | |
| 67 | −https://www.pagesjaunes.ca/ | |
| 68 | −https://quebec.craigslist.org/ | |
| 69 | −https://www.cqcd.org/ | |
| 70 | −https://www.manufacturiersquebec.ca/ | |
| 71 | − | |
| 72 | −# --- Tourisme et culture --- | |
| 73 | −https://www.bonjourquebec.com/ | |
| 74 | −https://www.mtl.org/ | |
| 75 | −https://www.quebec-cite.com/ | |
| 76 | −https://www.tourisme-charlevoix.com/ | |
| 77 | −https://www.tourismegaspesie.com/ | |
| 78 | −https://www.sepaq.com/ | |
| 79 | −https://www.lavitrine.com/ | |
| 80 | −https://www.banq.qc.ca/ | |
| 81 | − | |
| 82 | −# --- Écosystème Groupe Ka --- | |
| 1 | +# Trouve-KA — seeds de crawl | |
| 2 | +# Pivot 2026-08-23 : moteur de recherche du GROUPE KA — les seeds sont les | |
| 3 | +# pages d'accueil des 14 sites de l'écosystème (l'alimentation principale | |
| 4 | +# passe par l'ingestion des sitemaps : scripts/ka-sitemaps/ingest.py). | |
| 83 | 5 | https://www.groupe-ka.com/ |
| 6 | +https://www.trouve-ka.com/ | |
| 84 | 7 | https://www.lou-ka.com/ |
| 85 | 8 | https://www.immo-ka.com/ |
| 86 | 9 | https://www.vrai-prix.com/ |
| 10 | +https://www.toit-ka.com/ | |
| 11 | +https://www.valoplex.com/ | |
| 87 | 12 | https://www.auto-ka.com/ |
| 88 | 13 | https://www.fabri-ka.com/ |
| 89 | 14 | https://www.food-ka.com/ |
| 90 | −https://www.trouve-ka.com/ | |
| 91 | −https://www.crea-ka.com/ | |
| 92 | 15 | https://www.resto-ka.com/ |
| 93 | 16 | https://www.sorti-ka.com/ |
| 17 | +https://www.crea-ka.com/ | |
| 94 | 18 | https://www.job-ka.com/ |
| 95 | −https://www.api-ka.com/ | |
added
scripts/ka-sitemaps/ingest.py
+180 −0
@@ -0,0 +1,180 @@ | ||
| 1 | +# Trouve-KA — ingestion des sitemaps du Groupe KA (pivot 2026-08-23) | |
| 2 | +# Author: Simon-Pierre Boucher | |
| 3 | +# Contact: contact@spboucher.ai | |
| 4 | + | |
| 5 | +"""Alimentation principale du moteur : enfile dans le frontier TOUTES les URLs | |
| 6 | +déclarées par les sitemaps des sites du Groupe KA. | |
| 7 | + | |
| 8 | +Usage : .venv/bin/python scripts/ka-sitemaps/ingest.py [--sites lou-ka.com,…] | |
| 9 | + [--max-per-site N] [--shallow-priority 0.9] [--deep-priority 0.5] | |
| 10 | + | |
| 11 | +- Découverte des sitemaps via robots.txt (lignes « Sitemap: »), repli sur | |
| 12 | + /sitemap.xml. Suit les index de sitemaps (1 niveau, le pattern seo.py KA). | |
| 13 | +- Traitement en STREAMING chunk par chunk (vrai-prix déclare ~3,75 M d'URLs | |
| 14 | + sur 84 chunks : jamais tout en mémoire). | |
| 15 | +- Priorité par URL : les pages peu profondes (accueil, villes, catégories) | |
| 16 | + passent avant les fiches — le moteur sert l'essentiel d'abord, la longue | |
| 17 | + traîne s'indexe en continu. | |
| 18 | +- Ré-exécutable à volonté (les URLs connues sont ignorées par la base) : | |
| 19 | + planifié quotidiennement via PM2 (process tk-sitemaps) pour ramasser les | |
| 20 | + nouvelles fiches publiées. | |
| 21 | +""" | |
| 22 | + | |
| 23 | +import argparse | |
| 24 | +import asyncio | |
| 25 | +import gzip | |
| 26 | +import re | |
| 27 | +import sys | |
| 28 | +import urllib.request | |
| 29 | + | |
| 30 | +from trouveka.config import get_settings | |
| 31 | +from trouveka.database import Database | |
| 32 | +from trouveka.logging import get_logger | |
| 33 | +from trouveka.shared import KA_DOMAINS, canonicalize_url, extract_domain, is_ka_domain | |
| 34 | + | |
| 35 | +log = get_logger("crawler.ingest_ka_sitemaps") | |
| 36 | + | |
| 37 | +LOC_RE = re.compile(r"<loc>\s*([^<\s]+)\s*</loc>") | |
| 38 | +SITEMAP_LINE_RE = re.compile(r"(?im)^\s*sitemap\s*:\s*(\S+)") | |
| 39 | +DB_CHUNK = 1000 | |
| 40 | + | |
| 41 | +# Domaines à ingérer (sous-ensemble de l'allowlist : forma-ka/api-ka exclus | |
| 42 | +# tant qu'ils n'exposent pas de sitemap réel). | |
| 43 | +INGEST_DOMAINS: tuple[str, ...] = tuple(sorted(KA_DOMAINS)) | |
| 44 | + | |
| 45 | + | |
| 46 | +def fetch(url: str, user_agent: str, timeout: float = 30.0) -> str: | |
| 47 | + req = urllib.request.Request(url, headers={"User-Agent": user_agent}) | |
| 48 | + with urllib.request.urlopen(req, timeout=timeout) as resp: | |
| 49 | + body = resp.read(50_000_000) | |
| 50 | + if resp.headers.get("Content-Encoding") == "gzip" or url.endswith(".gz"): | |
| 51 | + try: | |
| 52 | + body = gzip.decompress(body) | |
| 53 | + except OSError: | |
| 54 | + pass | |
| 55 | + return body.decode("utf-8", errors="replace") | |
| 56 | + | |
| 57 | + | |
| 58 | +def discover_sitemaps(domain: str, user_agent: str) -> list[str]: | |
| 59 | + """Sitemaps déclarés par robots.txt, sinon /sitemap.xml.""" | |
| 60 | + base = f"https://www.{domain}" | |
| 61 | + try: | |
| 62 | + robots = fetch(f"{base}/robots.txt", user_agent, timeout=15.0) | |
| 63 | + declared = [u for u in SITEMAP_LINE_RE.findall(robots) if is_ka_domain(extract_domain(u) or "")] | |
| 64 | + if declared: | |
| 65 | + return declared | |
| 66 | + except Exception as exc: # noqa: BLE001 — robots absent/cassé : on tente le chemin standard | |
| 67 | + log.info("robots.txt illisible", extra={"ctx": {"domain": domain, "err": str(exc)}}) | |
| 68 | + return [f"{base}/sitemap.xml"] | |
| 69 | + | |
| 70 | + | |
| 71 | +def path_depth(url: str) -> int: | |
| 72 | + return len([s for s in url.split("/", 3)[-1].split("/") if s]) | |
| 73 | + | |
| 74 | + | |
| 75 | +async def ingest_domain( | |
| 76 | + db: Database, | |
| 77 | + domain: str, | |
| 78 | + *, | |
| 79 | + user_agent: str, | |
| 80 | + max_urls_per_domain: int, | |
| 81 | + max_per_site: int | None, | |
| 82 | + shallow_priority: float, | |
| 83 | + deep_priority: float, | |
| 84 | +) -> tuple[int, int]: | |
| 85 | + """Ingestion streaming d'un domaine. Retourne (urls_vues, urls_ajoutées).""" | |
| 86 | + seen_total = 0 | |
| 87 | + added_total = 0 | |
| 88 | + | |
| 89 | + async def enqueue_batch(locs: list[str]) -> None: | |
| 90 | + nonlocal seen_total, added_total | |
| 91 | + items: list[tuple[str, str, float]] = [] | |
| 92 | + for loc in locs: | |
| 93 | + url = canonicalize_url(loc) | |
| 94 | + if not url: | |
| 95 | + continue | |
| 96 | + d = extract_domain(url) | |
| 97 | + if not d or not is_ka_domain(d): | |
| 98 | + continue # garde-fou : on reste dans le périmètre KA | |
| 99 | + prio = shallow_priority if path_depth(url) <= 2 else deep_priority | |
| 100 | + items.append((url, d, prio)) | |
| 101 | + seen_total += len(items) | |
| 102 | + for i in range(0, len(items), DB_CHUNK): | |
| 103 | + _, n = await db.enqueue_urls_bulk( | |
| 104 | + items[i : i + DB_CHUNK], depth=1, source_url_id=None, | |
| 105 | + max_urls_per_domain=max_urls_per_domain, | |
| 106 | + ) | |
| 107 | + added_total += n | |
| 108 | + | |
| 109 | + for sitemap_url in discover_sitemaps(domain, user_agent): | |
| 110 | + try: | |
| 111 | + body = fetch(sitemap_url, user_agent) | |
| 112 | + except Exception as exc: # noqa: BLE001 | |
| 113 | + log.info("sitemap illisible", extra={"ctx": {"url": sitemap_url, "err": str(exc)}}) | |
| 114 | + continue | |
| 115 | + if "<sitemapindex" in body: | |
| 116 | + children = LOC_RE.findall(body) | |
| 117 | + log.info("index de sitemaps", extra={"ctx": {"domain": domain, "chunks": len(children)}}) | |
| 118 | + for child in children: | |
| 119 | + if max_per_site and seen_total >= max_per_site: | |
| 120 | + break | |
| 121 | + try: | |
| 122 | + child_body = fetch(child, user_agent) | |
| 123 | + except Exception as exc: # noqa: BLE001 — un chunk cassé ne bloque pas le reste | |
| 124 | + log.info("chunk illisible", extra={"ctx": {"url": child, "err": str(exc)}}) | |
| 125 | + continue | |
| 126 | + if "<sitemapindex" in child_body: | |
| 127 | + continue # un seul niveau d'indirection (pattern seo.py) | |
| 128 | + await enqueue_batch(LOC_RE.findall(child_body)) | |
| 129 | + else: | |
| 130 | + await enqueue_batch(LOC_RE.findall(body)) | |
| 131 | + if max_per_site and seen_total >= max_per_site: | |
| 132 | + break | |
| 133 | + | |
| 134 | + log.info("domaine ingéré", extra={"ctx": {"domain": domain, "vues": seen_total, "ajoutées": added_total}}) | |
| 135 | + return seen_total, added_total | |
| 136 | + | |
| 137 | + | |
| 138 | +async def run(sites: list[str], max_per_site: int | None, shallow: float, deep: float) -> None: | |
| 139 | + settings = get_settings() | |
| 140 | + db = Database(settings.database_url, pool_min=1, pool_max=3) | |
| 141 | + await db.connect() | |
| 142 | + grand_seen = grand_added = 0 | |
| 143 | + try: | |
| 144 | + for domain in sites: | |
| 145 | + seen, added = await ingest_domain( | |
| 146 | + db, domain, | |
| 147 | + user_agent=settings.crawler_user_agent, | |
| 148 | + max_urls_per_domain=settings.max_urls_per_domain, | |
| 149 | + max_per_site=max_per_site, | |
| 150 | + shallow_priority=shallow, deep_priority=deep, | |
| 151 | + ) | |
| 152 | + grand_seen += seen | |
| 153 | + grand_added += added | |
| 154 | + print(f"{domain}: {seen} URLs vues, {added} ajoutées au frontier", file=sys.stderr) | |
| 155 | + finally: | |
| 156 | + await db.close() | |
| 157 | + print(f"TOTAL : {grand_seen} URLs vues, {grand_added} ajoutées au frontier", file=sys.stderr) | |
| 158 | + | |
| 159 | + | |
| 160 | +def main() -> None: | |
| 161 | + ap = argparse.ArgumentParser(description=__doc__) | |
| 162 | + ap.add_argument("--sites", help="domaines à ingérer, séparés par des virgules (défaut : tous)") | |
| 163 | + ap.add_argument("--max-per-site", type=int, default=None, help="plafond d'URLs par site (défaut : illimité)") | |
| 164 | + ap.add_argument("--shallow-priority", type=float, default=0.9, help="priorité des pages peu profondes") | |
| 165 | + ap.add_argument("--deep-priority", type=float, default=0.5, help="priorité des fiches profondes") | |
| 166 | + args = ap.parse_args() | |
| 167 | + | |
| 168 | + if args.sites: | |
| 169 | + sites = [s.strip().removeprefix("www.") for s in args.sites.split(",") if s.strip()] | |
| 170 | + unknown = [s for s in sites if s not in KA_DOMAINS] | |
| 171 | + if unknown: | |
| 172 | + ap.error(f"domaines hors périmètre KA : {', '.join(unknown)}") | |
| 173 | + else: | |
| 174 | + sites = list(INGEST_DOMAINS) | |
| 175 | + | |
| 176 | + asyncio.run(run(sites, args.max_per_site, args.shallow_priority, args.deep_priority)) | |
| 177 | + | |
| 178 | + | |
| 179 | +if __name__ == "__main__": | |
| 180 | + main() | |
modified
services/crawler/worker.py
+26 −4
@@ -28,7 +28,13 @@ from trouveka.logging import get_logger | ||
| 28 | 28 | from trouveka.parser import looks_like_garbage, parse_html |
| 29 | 29 | from trouveka.queue import Coordination |
| 30 | 30 | from trouveka.search_core import SearchCore |
| 31 | −from trouveka.shared import canonicalize_url, content_hash, extract_domain, is_excluded_domain | |
| 31 | +from trouveka.shared import ( | |
| 32 | + canonicalize_url, | |
| 33 | + content_hash, | |
| 34 | + extract_domain, | |
| 35 | + is_excluded_domain, | |
| 36 | + is_ka_domain, | |
| 37 | +) | |
| 32 | 38 | from trouveka.types import ErrorCode, Outcome |
| 33 | 39 | |
| 34 | 40 | from .fetcher import Fetcher, scheme_host |
@@ -208,6 +214,12 @@ class CrawlerWorker: | ||
| 208 | 214 | await self.db.release_item(url_id, status="blocked") |
| 209 | 215 | return |
| 210 | 216 | |
| 217 | + # Pivot KA-only : un reliquat de l'ancien frontier « web ouvert » est | |
| 218 | + # classé sans être crawlé (aucun fetch hors des sites du Groupe KA). | |
| 219 | + if self.s.ka_only and not is_ka_domain(domain): | |
| 220 | + await self.db.release_item(url_id, status="blocked") | |
| 221 | + return | |
| 222 | + | |
| 211 | 223 | host = scheme_host(url) |
| 212 | 224 | |
| 213 | 225 | # robots.txt d'abord (le fetch de robots ne compte pas dans la politesse) |
@@ -220,8 +232,10 @@ class CrawlerWorker: | ||
| 220 | 232 | await self.db.release_item(url_id, status="done", error_code=ErrorCode.ROBOTS_DENIED) |
| 221 | 233 | return |
| 222 | 234 | |
| 223 | − # Politesse par hôte, tous workers confondus | |
| 224 | − delay = max(robots_delay or 0, self.s.default_host_delay) | |
| 235 | + # Politesse par hôte, tous workers confondus. Nos propres sites (KA) | |
| 236 | + # tolèrent un rythme bien plus soutenu que le web ouvert. | |
| 237 | + base_delay = self.s.ka_host_delay if is_ka_domain(domain) else self.s.default_host_delay | |
| 238 | + delay = max(robots_delay or 0, base_delay) | |
| 225 | 239 | if not await self.coord.acquire_host_slot(domain, delay): |
| 226 | 240 | await self._defer(url_id, delay + 0.5) |
| 227 | 241 | return |
@@ -248,7 +262,8 @@ class CrawlerWorker: | ||
| 248 | 262 | final = canonicalize_url(result.final_url) or result.final_url |
| 249 | 263 | if final != url: |
| 250 | 264 | final_domain = extract_domain(final) |
| 251 | − if final_domain and not is_excluded_domain(final_domain) and not looks_like_trap(final): | |
| 265 | + in_scope = not self.s.ka_only or (final_domain and is_ka_domain(final_domain)) | |
| 266 | + if final_domain and in_scope and not is_excluded_domain(final_domain) and not looks_like_trap(final): | |
| 252 | 267 | await self.db.enqueue_url( |
| 253 | 268 | final, final_domain, priority=item["priority"], depth=item["depth"], |
| 254 | 269 | source_url_id=url_id, max_urls_per_domain=self.s.max_urls_per_domain, |
@@ -285,6 +300,11 @@ class CrawlerWorker: | ||
| 285 | 300 | |
| 286 | 301 | # --- score Québec (immédiat, déterministe) |
| 287 | 302 | signals = score_page(page, domain) |
| 303 | + # Pivot KA-only : nos propres pages sont pertinentes par définition — | |
| 304 | + # score plancher 1.0 (les lieux/organisations extraits restent utilisés | |
| 305 | + # par le boost de localité du ranking). | |
| 306 | + if is_ka_domain(domain): | |
| 307 | + signals.score = 1.0 | |
| 288 | 308 | |
| 289 | 309 | # --- détection de changement |
| 290 | 310 | chash = content_hash(page.title, page.body) |
@@ -440,6 +460,8 @@ class CrawlerWorker: | ||
| 440 | 460 | target_domain = extract_domain(link.url) |
| 441 | 461 | if not target_domain or is_excluded_domain(target_domain): |
| 442 | 462 | continue |
| 463 | + if self.s.ka_only and not is_ka_domain(target_domain): | |
| 464 | + continue # pivot KA-only : aucun lien sortant hors Groupe KA | |
| 443 | 465 | same_domain = target_domain == source_domain |
| 444 | 466 | target_row = await self._domain_row_by_name(target_domain) |
| 445 | 467 | is_new = target_row is None |
modified
services/indexer/build.py
+26 −0
@@ -22,9 +22,35 @@ _EDU_SUFFIXES = (".ulaval.ca", ".umontreal.ca", ".mcgill.ca", ".uqam.ca", ".ushe | ||
| 22 | 22 | ".concordia.ca", ".polymtl.ca", ".etsmtl.ca", ".hec.ca") |
| 23 | 23 | |
| 24 | 24 | |
| 25 | +# Verticales du Groupe KA (pivot 2026-08-23) — permet le filtre par catégorie | |
| 26 | +# métier dans la recherche interne (logement, emplois, restaurants…). | |
| 27 | +_KA_CATEGORIES = { | |
| 28 | + "groupe-ka.com": "portail", | |
| 29 | + "trouve-ka.com": "recherche", | |
| 30 | + "lou-ka.com": "logement", | |
| 31 | + "immo-ka.com": "immobilier", | |
| 32 | + "toit-ka.com": "immobilier", | |
| 33 | + "vrai-prix.com": "estimation", | |
| 34 | + "valoplex.com": "estimation", | |
| 35 | + "auto-ka.com": "vehicules", | |
| 36 | + "fabri-ka.com": "artisans", | |
| 37 | + "food-ka.com": "epicerie", | |
| 38 | + "resto-ka.com": "restaurants", | |
| 39 | + "sorti-ka.com": "evenements", | |
| 40 | + "crea-ka.com": "createurs", | |
| 41 | + "job-ka.com": "emplois", | |
| 42 | +} | |
| 43 | + | |
| 44 | + | |
| 25 | 45 | def categorize_domain(domain: str) -> list[str]: |
| 26 | 46 | d = domain.lower() |
| 27 | 47 | cats: list[str] = [] |
| 48 | + parts = d.split(".") | |
| 49 | + for i in range(len(parts) - 1): | |
| 50 | + ka_cat = _KA_CATEGORIES.get(".".join(parts[i:])) | |
| 51 | + if ka_cat: | |
| 52 | + cats.append(ka_cat) | |
| 53 | + break | |
| 28 | 54 | if d in _GOV_DOMAINS or any(d.endswith(s) for s in _GOV_SUFFIXES) or ".gouv." in d: |
| 29 | 55 | cats.append("government") |
| 30 | 56 | if d in _NEWS_DOMAINS: |
modified
services/ranking/query.py
+13 −0
@@ -60,6 +60,7 @@ def build_search_body( | ||
| 60 | 60 | limit: int = 10, |
| 61 | 61 | language: str | None = None, |
| 62 | 62 | category: str | None = None, |
| 63 | + site: str | None = None, | |
| 63 | 64 | quebec_only: bool = False, |
| 64 | 65 | freshness: str | None = None, |
| 65 | 66 | images_only: bool = False, |
@@ -93,6 +94,18 @@ def build_search_body( | ||
| 93 | 94 | filters.append({"term": {"language": language}}) |
| 94 | 95 | if category: |
| 95 | 96 | filters.append({"term": {"categories": category}}) |
| 97 | + if site: | |
| 98 | + # Filtre par site (pivot Groupe KA) : domaines séparés par des virgules, | |
| 99 | + # tolérant au préfixe www. (l'index stocke le domaine canonicalisé sans www). | |
| 100 | + site_domains: list[str] = [] | |
| 101 | + for s in site.split(","): | |
| 102 | + d = s.strip().lower() | |
| 103 | + if not d: | |
| 104 | + continue | |
| 105 | + bare = d[4:] if d.startswith("www.") else d | |
| 106 | + site_domains.extend((bare, f"www.{bare}")) | |
| 107 | + if site_domains: | |
| 108 | + filters.append({"terms": {"domain": sorted(set(site_domains))}}) | |
| 96 | 109 | if quebec_only: |
| 97 | 110 | filters.append({ |
| 98 | 111 | "bool": { |
| 99 | 112 | |