#!/usr/bin/env python3 # ----------------------------------------------------------------------------- # Groupe KA — Annuaire des déménageurs du Québec # build_demenageurs.py : balayage Serper Places sur ~200 villes des 17 régions # administratives, filtre « déménagement », dédoublonnage CID/téléphone, # sortie canonique data/demenageurs.json (consommée par Lou-Ka et Immo-Ka). # ----------------------------------------------------------------------------- from __future__ import annotations import json import os import re import sys import time import unicodedata from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path from threading import Lock from urllib.request import Request, urlopen API_KEY = os.environ.get("SERPER_API_KEY", "") OUT = Path(__file__).parent / "demenageurs.json" # Région administrative → villes balayées (couverture provinciale complète) REGIONS: dict[str, list[str]] = { "Bas-Saint-Laurent": [ "Rimouski", "Rivière-du-Loup", "Matane", "Mont-Joli", "Amqui", "La Pocatière", "Témiscouata-sur-le-Lac", "Trois-Pistoles", "Dégelis"], "Saguenay–Lac-Saint-Jean": [ "Saguenay", "Chicoutimi", "Jonquière", "La Baie", "Alma", "Dolbeau-Mistassini", "Roberval", "Saint-Félicien", "Normandin"], "Capitale-Nationale": [ "Québec", "L'Ancienne-Lorette", "Saint-Augustin-de-Desmaures", "Beaupré", "Baie-Saint-Paul", "La Malbaie", "Donnacona", "Pont-Rouge", "Saint-Raymond", "Sainte-Catherine-de-la-Jacques-Cartier", "Stoneham-et-Tewkesbury", "Boischatel", "Château-Richer"], "Mauricie": [ "Trois-Rivières", "Shawinigan", "La Tuque", "Louiseville", "Saint-Tite", "Saint-Étienne-des-Grès"], "Estrie": [ "Sherbrooke", "Magog", "Granby", "Cowansville", "Bromont", "Coaticook", "Lac-Mégantic", "Windsor", "East Angus", "Val-des-Sources", "Waterloo"], "Montréal": [ "Montréal", "Montréal-Nord", "Saint-Laurent Montréal", "LaSalle", "Anjou", "Ahuntsic", "Rosemont", "Hochelaga-Maisonneuve", "Le Plateau-Mont-Royal", "Verdun", "Lachine", "Pierrefonds", "Rivière-des-Prairies", "Côte-des-Neiges", "Westmount", "Côte-Saint-Luc", "Dollard-des-Ormeaux", "Pointe-Claire", "Dorval", "Kirkland", "Beaconsfield", "Sainte-Anne-de-Bellevue", "Outremont", "Mont-Royal"], "Outaouais": [ "Gatineau", "Hull Gatineau", "Aylmer Gatineau", "Buckingham Gatineau", "Cantley", "Chelsea", "Val-des-Monts", "Maniwaki", "Papineauville", "La Pêche"], "Abitibi-Témiscamingue": [ "Rouyn-Noranda", "Val-d'Or", "Amos", "La Sarre", "Ville-Marie QC", "Senneterre", "Malartic"], "Côte-Nord": [ "Baie-Comeau", "Sept-Îles", "Port-Cartier", "Forestville", "Havre-Saint-Pierre", "Fermont"], "Nord-du-Québec": [ "Chibougamau", "Lebel-sur-Quévillon", "Matagami", "Chapais"], "Gaspésie–Îles-de-la-Madeleine": [ "Gaspé", "Chandler", "New Richmond", "Bonaventure", "Carleton-sur-Mer", "Sainte-Anne-des-Monts", "Percé", "Les Îles-de-la-Madeleine"], "Chaudière-Appalaches": [ "Lévis", "Saint-Georges", "Thetford Mines", "Montmagny", "Sainte-Marie", "Beauceville", "Saint-Joseph-de-Beauce", "L'Islet", "Lac-Etchemin", "Saint-Anselme", "Saint-Henri"], "Laval": ["Laval", "Chomedey Laval", "Sainte-Rose Laval"], "Lanaudière": [ "Terrebonne", "Repentigny", "Mascouche", "Joliette", "L'Assomption", "Rawdon", "Saint-Lin-Laurentides", "Berthierville", "Lavaltrie", "Saint-Charles-Borromée", "Chertsey"], "Laurentides": [ "Saint-Jérôme", "Blainville", "Boisbriand", "Sainte-Thérèse", "Mirabel", "Saint-Eustache", "Deux-Montagnes", "Rosemère", "Bois-des-Filion", "Sainte-Adèle", "Sainte-Agathe-des-Monts", "Mont-Tremblant", "Mont-Laurier", "Prévost", "Saint-Sauveur", "Lachute", "Saint-Colomban"], "Montérégie": [ "Longueuil", "Brossard", "Saint-Lambert", "Boucherville", "Saint-Bruno-de-Montarville", "Saint-Hubert Longueuil", "Châteauguay", "Saint-Jean-sur-Richelieu", "Salaberry-de-Valleyfield", "Vaudreuil-Dorion", "Saint-Hyacinthe", "Sorel-Tracy", "Beloeil", "Mont-Saint-Hilaire", "Chambly", "La Prairie", "Candiac", "Sainte-Catherine QC", "Saint-Constant", "Mercier", "Beauharnois", "Varennes", "Sainte-Julie", "Contrecoeur", "Marieville", "Farnham", "Acton Vale", "Rigaud", "L'Île-Perrot", "Pincourt", "Saint-Rémi"], "Centre-du-Québec": [ "Drummondville", "Victoriaville", "Bécancour", "Nicolet", "Plessisville", "Princeville", "Kingsey Falls", "Warwick QC"], } # Grandes villes : pagination plus profonde + variante de requête BIG = {"Montréal", "Québec", "Laval", "Gatineau", "Longueuil", "Sherbrooke", "Trois-Rivières", "Saguenay", "Lévis", "Terrebonne", "Saint-Jérôme", "Brossard", "Repentigny", "Drummondville", "Saint-Jean-sur-Richelieu"} MOVER_RE = re.compile(r"demenag|moving|movers|transport et demenagement", re.I) _lock = Lock() _stats = {"queries": 0, "kept": 0, "raw": 0} def norm(s: str) -> str: return unicodedata.normalize("NFKD", s or "").encode("ascii", "ignore").decode() def slugify(s: str) -> str: s = norm(s).lower() s = re.sub(r"[^a-z0-9]+", "-", s).strip("-") return s or "x" def norm_phone(p: str | None) -> str: d = re.sub(r"\D", "", p or "") return d[-10:] if len(d) >= 10 else d def serper_places(q: str, page: int) -> list[dict]: body = json.dumps({"q": q, "gl": "ca", "hl": "fr", "page": page}).encode() req = Request("https://google.serper.dev/places", data=body, headers={"X-API-KEY": API_KEY, "Content-Type": "application/json"}) for attempt in range(3): try: with urlopen(req, timeout=30) as r: with _lock: _stats["queries"] += 1 return json.loads(r.read()).get("places", []) except Exception: time.sleep(2 * (attempt + 1)) return [] def is_mover(p: dict) -> bool: return bool(MOVER_RE.search(norm(p.get("category", ""))) or MOVER_RE.search(norm(p.get("title", "")))) def in_quebec(p: dict) -> bool: addr = p.get("address", "") return (", QC" in addr or " QC " in addr or addr.endswith(" QC") or not addr) # certaines fiches n'ont pas d'adresse : région du balayage CITY_RE = re.compile(r",\s*([^,]+?),\s*(?:QC|Québec)\b") def parse_city(addr: str) -> str: m = CITY_RE.search(addr or "") return m.group(1).strip() if m else "" def sweep_city(region: str, city: str) -> list[dict]: label = city.replace(" QC", "").replace(" Montréal", "").replace(" Gatineau", "").replace(" Laval", "").replace(" Longueuil", "") queries = [f"déménageur {city} QC"] max_pages = 5 if city in BIG else 3 if city in BIG: queries.append(f"entreprise de déménagement {city} QC") out = [] for q in queries: for page in range(1, max_pages + 1): places = serper_places(q, page) with _lock: _stats["raw"] += len(places) for p in places: if is_mover(p) and in_quebec(p): p["_region"] = region p["_query_city"] = label out.append(p) if len(places) < 10: break return out def main() -> None: if not API_KEY: sys.exit("SERPER_API_KEY manquante") jobs = [(r, c) for r, cities in REGIONS.items() for c in cities] # ville balayée → région (pour rattacher la ville réelle de l'adresse) city_region = {} for r, cities in REGIONS.items(): for c in cities: city_region[norm(c.split(" QC")[0]).lower()] = r raw: list[dict] = [] with ThreadPoolExecutor(max_workers=8) as ex: futs = {ex.submit(sweep_city, r, c): (r, c) for r, c in jobs} done = 0 for f in as_completed(futs): raw.extend(f.result()) done += 1 if done % 25 == 0: print(f" … {done}/{len(jobs)} villes, {len(raw)} fiches brutes, " f"{_stats['queries']} requêtes", flush=True) # Dédoublonnage : CID d'abord, sinon téléphone, sinon (nom, ville) by_key: dict[str, dict] = {} for p in raw: key = ("cid:" + p["cid"]) if p.get("cid") else "" if not key: ph = norm_phone(p.get("phoneNumber")) key = ("tel:" + ph) if ph else ("nc:" + slugify(p.get("title", "")) + ":" + slugify(parse_city(p.get("address", "")))) cur = by_key.get(key) if cur is None or (not cur.get("website") and p.get("website")): if cur: p.setdefault("_region", cur["_region"]) by_key[key] = p movers = [] seen_slug: set[str] = set() for p in by_key.values(): addr_city = parse_city(p.get("address", "")) region = city_region.get(norm(addr_city).lower(), p["_region"]) city = addr_city or p["_query_city"] base = slugify(f"{p.get('title','')}-{city}") slug = base i = 2 while slug in seen_slug: slug = f"{base}-{i}" i += 1 seen_slug.add(slug) movers.append({ "id": slug, "name": (p.get("title") or "").strip(), "city": city, "region": region, "address": p.get("address", ""), "phone": p.get("phoneNumber", ""), "website": p.get("website", ""), "rating": p.get("rating"), "reviews": p.get("ratingCount"), "lat": p.get("latitude"), "lng": p.get("longitude"), "category": p.get("category", ""), "cid": p.get("cid", ""), }) movers.sort(key=lambda m: (norm(m["region"]), norm(m["city"]), norm(m["name"]))) doc = { "generated": time.strftime("%Y-%m-%d"), "source": "Google Places via Serper — balayage " + str(len(jobs)) + " villes, 17 régions", "count": len(movers), "movers": movers, } OUT.write_text(json.dumps(doc, ensure_ascii=False, indent=1), encoding="utf-8") regions = {} for m in movers: regions[m["region"]] = regions.get(m["region"], 0) + 1 print(f"\n✅ {len(movers)} déménageurs uniques ({_stats['raw']} fiches brutes, " f"{_stats['queries']} requêtes Serper) → {OUT}") for r, n in sorted(regions.items(), key=lambda x: -x[1]): print(f" {r}: {n}") if __name__ == "__main__": main()