8 nouveaux ATS (Taleo, Njoyn, UltiPro, iCIMS, SuccessFactors, ADP WFN, Digital Recruiters, Workland) + 23 employeurs publics/parapublics et grands employeurs QC
- Secteur public : Santé Québec (emplois.sante.quebec), Emplois Santé Montréal (njoyn), CIUSSS de l Est-de-l Île-de-Montréal (taleo), CSSDM (workland), villes de Gatineau et Terrebonne, STL, RTC (njoyn via Scrapfly anti-Radware) - Grands employeurs : Bombardier, Cascades, Domtar (SuccessFactors sitemap RSS/urlset + microdonnées), Agnico Eagle, Bell Textron, Bayshore (Taleo REST + repli anglophone), Messer Canada (UltiPro), Lassonde, Mitsubishi HC Capital, MSI (ADP WFN + repli en_US), CAPREIT, McCarthy Tétrault, Omni, EDF Renouvelables (iCIMS + JSON-LD) - gen_connectors : EXTRA_ATTRS multi-attributs pour les nouveaux ATS - jobka/connectors/_jsonld.py : extraction schema.org/JobPosting partagée Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
34 changed files +2,093 −0
added
data/feeds-public.json
+368 −0
@@ -0,0 +1,368 @@ | ||
| 1 | +[ | |
| 2 | + { | |
| 3 | + "ats": "taleo", | |
| 4 | + "source_id": "agnico_eagle", | |
| 5 | + "employer": "Agnico Eagle Mines", | |
| 6 | + "host": "agnicoeagle", | |
| 7 | + "section": "2", | |
| 8 | + "url": "https://www.agnicoeagle.com", | |
| 9 | + "sectors": [ | |
| 10 | + "Mines", | |
| 11 | + "Ingénierie", | |
| 12 | + "Métiers spécialisés" | |
| 13 | + ], | |
| 14 | + "cities": [ | |
| 15 | + "Abitibi-Témiscamingue", | |
| 16 | + "Malartic", | |
| 17 | + "Rouyn-Noranda" | |
| 18 | + ] | |
| 19 | + }, | |
| 20 | + { | |
| 21 | + "ats": "taleo", | |
| 22 | + "source_id": "ciusss_est_montreal", | |
| 23 | + "employer": "CIUSSS de l'Est-de-l'Île-de-Montréal", | |
| 24 | + "host": "ciusssemtl", | |
| 25 | + "section": "cemtl", | |
| 26 | + "url": "https://ciusss-estmtl.gouv.qc.ca", | |
| 27 | + "sectors": [ | |
| 28 | + "Santé", | |
| 29 | + "Services sociaux", | |
| 30 | + "Secteur public" | |
| 31 | + ], | |
| 32 | + "cities": [ | |
| 33 | + "Montréal" | |
| 34 | + ] | |
| 35 | + }, | |
| 36 | + { | |
| 37 | + "ats": "taleo", | |
| 38 | + "source_id": "bayshore", | |
| 39 | + "employer": "Bayshore HealthCare", | |
| 40 | + "host": "bayshore", | |
| 41 | + "section": "bs_ex", | |
| 42 | + "url": "https://www.bayshore.ca", | |
| 43 | + "sectors": [ | |
| 44 | + "Santé", | |
| 45 | + "Soins à domicile" | |
| 46 | + ], | |
| 47 | + "cities": [ | |
| 48 | + "Montréal", | |
| 49 | + "Québec" | |
| 50 | + ] | |
| 51 | + }, | |
| 52 | + { | |
| 53 | + "ats": "taleo", | |
| 54 | + "source_id": "bell_textron", | |
| 55 | + "employer": "Bell Textron Canada", | |
| 56 | + "host": "textron", | |
| 57 | + "section": "bell", | |
| 58 | + "url": "https://www.bellflight.com", | |
| 59 | + "sectors": [ | |
| 60 | + "Aéronautique", | |
| 61 | + "Fabrication" | |
| 62 | + ], | |
| 63 | + "cities": [ | |
| 64 | + "Mirabel" | |
| 65 | + ] | |
| 66 | + }, | |
| 67 | + { | |
| 68 | + "ats": "njoyn", | |
| 69 | + "source_id": "sante_montreal", | |
| 70 | + "employer": "Santé Québec — Montréal", | |
| 71 | + "base": "https://emplois.santemontreal.qc.ca", | |
| 72 | + "cl": "CL3", | |
| 73 | + "clid": "54327", | |
| 74 | + "url": "https://santemontreal.ca", | |
| 75 | + "sectors": [ | |
| 76 | + "Santé", | |
| 77 | + "Services sociaux", | |
| 78 | + "Secteur public" | |
| 79 | + ], | |
| 80 | + "cities": [ | |
| 81 | + "Montréal" | |
| 82 | + ] | |
| 83 | + }, | |
| 84 | + { | |
| 85 | + "ats": "njoyn", | |
| 86 | + "source_id": "ville_gatineau", | |
| 87 | + "employer": "Ville de Gatineau", | |
| 88 | + "base": "https://gatineau.njoyn.com", | |
| 89 | + "cl": "CL2", | |
| 90 | + "clid": "27082", | |
| 91 | + "use_scrapfly": true, | |
| 92 | + "url": "https://www.gatineau.ca", | |
| 93 | + "sectors": [ | |
| 94 | + "Municipal", | |
| 95 | + "Secteur public" | |
| 96 | + ], | |
| 97 | + "cities": [ | |
| 98 | + "Gatineau" | |
| 99 | + ] | |
| 100 | + }, | |
| 101 | + { | |
| 102 | + "ats": "njoyn", | |
| 103 | + "source_id": "ville_terrebonne", | |
| 104 | + "employer": "Ville de Terrebonne", | |
| 105 | + "base": "https://clients.njoyn.com", | |
| 106 | + "cl": "cl4", | |
| 107 | + "clid": "71764", | |
| 108 | + "use_scrapfly": true, | |
| 109 | + "url": "https://www.ville.terrebonne.qc.ca", | |
| 110 | + "sectors": [ | |
| 111 | + "Municipal", | |
| 112 | + "Secteur public" | |
| 113 | + ], | |
| 114 | + "cities": [ | |
| 115 | + "Terrebonne" | |
| 116 | + ] | |
| 117 | + }, | |
| 118 | + { | |
| 119 | + "ats": "njoyn", | |
| 120 | + "source_id": "stl_laval", | |
| 121 | + "employer": "Société de transport de Laval", | |
| 122 | + "base": "https://lavaltransit.njoyn.com", | |
| 123 | + "cl": "CL2", | |
| 124 | + "clid": "60406", | |
| 125 | + "use_scrapfly": true, | |
| 126 | + "url": "https://www.stlaval.ca", | |
| 127 | + "sectors": [ | |
| 128 | + "Transport collectif", | |
| 129 | + "Secteur public" | |
| 130 | + ], | |
| 131 | + "cities": [ | |
| 132 | + "Laval" | |
| 133 | + ] | |
| 134 | + }, | |
| 135 | + { | |
| 136 | + "ats": "njoyn", | |
| 137 | + "source_id": "rtc_quebec", | |
| 138 | + "employer": "Réseau de transport de la Capitale", | |
| 139 | + "base": "https://rtc.njoyn.com", | |
| 140 | + "cl": "CGI", | |
| 141 | + "clid": "23009", | |
| 142 | + "use_scrapfly": true, | |
| 143 | + "url": "https://www.rtcquebec.ca", | |
| 144 | + "sectors": [ | |
| 145 | + "Transport collectif", | |
| 146 | + "Secteur public" | |
| 147 | + ], | |
| 148 | + "cities": [ | |
| 149 | + "Québec" | |
| 150 | + ] | |
| 151 | + }, | |
| 152 | + { | |
| 153 | + "ats": "ultipro", | |
| 154 | + "source_id": "messer_canada", | |
| 155 | + "employer": "Messer Canada", | |
| 156 | + "org": "MES1005MESR", | |
| 157 | + "board": "dbd63926-d756-486f-a254-b2a6ede1d26e", | |
| 158 | + "url": "https://www.messer-ca.com", | |
| 159 | + "sectors": [ | |
| 160 | + "Gaz industriels", | |
| 161 | + "Industrie" | |
| 162 | + ], | |
| 163 | + "cities": [ | |
| 164 | + "Montréal", | |
| 165 | + "Québec", | |
| 166 | + "Saint-Georges" | |
| 167 | + ] | |
| 168 | + }, | |
| 169 | + { | |
| 170 | + "ats": "icims", | |
| 171 | + "source_id": "capreit", | |
| 172 | + "employer": "CAPREIT", | |
| 173 | + "sub": "careers-capreit", | |
| 174 | + "url": "https://www.capreit.ca", | |
| 175 | + "sectors": [ | |
| 176 | + "Immobilier", | |
| 177 | + "Gestion immobilière" | |
| 178 | + ], | |
| 179 | + "cities": [ | |
| 180 | + "Montréal" | |
| 181 | + ] | |
| 182 | + }, | |
| 183 | + { | |
| 184 | + "ats": "icims", | |
| 185 | + "source_id": "mccarthy_tetrault", | |
| 186 | + "employer": "McCarthy Tétrault", | |
| 187 | + "sub": "careers-mccarthyca", | |
| 188 | + "url": "https://www.mccarthy.ca", | |
| 189 | + "sectors": [ | |
| 190 | + "Juridique", | |
| 191 | + "Services professionnels" | |
| 192 | + ], | |
| 193 | + "cities": [ | |
| 194 | + "Montréal", | |
| 195 | + "Québec" | |
| 196 | + ] | |
| 197 | + }, | |
| 198 | + { | |
| 199 | + "ats": "icims", | |
| 200 | + "source_id": "omni_hotels_horaire", | |
| 201 | + "employer": "Omni Hôtel Mont-Royal", | |
| 202 | + "sub": "externalhourly-omnihotels", | |
| 203 | + "url": "https://www.omnihotels.com/fr/hotels/montreal-mont-royal", | |
| 204 | + "sectors": [ | |
| 205 | + "Hôtellerie", | |
| 206 | + "Restauration" | |
| 207 | + ], | |
| 208 | + "cities": [ | |
| 209 | + "Montréal" | |
| 210 | + ] | |
| 211 | + }, | |
| 212 | + { | |
| 213 | + "ats": "icims", | |
| 214 | + "source_id": "omni_hotels_gestion", | |
| 215 | + "employer": "Omni Hôtel Mont-Royal", | |
| 216 | + "sub": "externalmanager-omnihotels", | |
| 217 | + "url": "https://www.omnihotels.com/fr/hotels/montreal-mont-royal", | |
| 218 | + "sectors": [ | |
| 219 | + "Hôtellerie", | |
| 220 | + "Gestion" | |
| 221 | + ], | |
| 222 | + "cities": [ | |
| 223 | + "Montréal" | |
| 224 | + ] | |
| 225 | + }, | |
| 226 | + { | |
| 227 | + "ats": "icims", | |
| 228 | + "source_id": "edf_renouvelables", | |
| 229 | + "employer": "EDF Renouvelables Canada", | |
| 230 | + "sub": "cafrench-edf-re", | |
| 231 | + "url": "https://www.edf-renouvelables.ca", | |
| 232 | + "sectors": [ | |
| 233 | + "Énergie", | |
| 234 | + "Ingénierie" | |
| 235 | + ], | |
| 236 | + "cities": [ | |
| 237 | + "Montréal" | |
| 238 | + ] | |
| 239 | + }, | |
| 240 | + { | |
| 241 | + "ats": "successfactors", | |
| 242 | + "source_id": "bombardier", | |
| 243 | + "employer": "Bombardier", | |
| 244 | + "base": "https://jobs.bombardier.com", | |
| 245 | + "url": "https://bombardier.com", | |
| 246 | + "sectors": [ | |
| 247 | + "Aéronautique", | |
| 248 | + "Ingénierie", | |
| 249 | + "Fabrication" | |
| 250 | + ], | |
| 251 | + "cities": [ | |
| 252 | + "Dorval", | |
| 253 | + "Mirabel", | |
| 254 | + "Montréal" | |
| 255 | + ] | |
| 256 | + }, | |
| 257 | + { | |
| 258 | + "ats": "successfactors", | |
| 259 | + "source_id": "domtar", | |
| 260 | + "employer": "Domtar", | |
| 261 | + "base": "https://jobs.domtar.com", | |
| 262 | + "url": "https://www.domtar.com", | |
| 263 | + "sectors": [ | |
| 264 | + "Pâtes et papiers", | |
| 265 | + "Foresterie", | |
| 266 | + "Fabrication" | |
| 267 | + ], | |
| 268 | + "cities": [ | |
| 269 | + "Montréal", | |
| 270 | + "Windsor", | |
| 271 | + "Lachine" | |
| 272 | + ] | |
| 273 | + }, | |
| 274 | + { | |
| 275 | + "ats": "successfactors", | |
| 276 | + "source_id": "cascades", | |
| 277 | + "employer": "Cascades", | |
| 278 | + "base": "https://jobs.cascades.com", | |
| 279 | + "url": "https://www.cascades.com", | |
| 280 | + "sectors": [ | |
| 281 | + "Pâtes et papiers", | |
| 282 | + "Emballage", | |
| 283 | + "Fabrication" | |
| 284 | + ], | |
| 285 | + "cities": [ | |
| 286 | + "Kingsey Falls", | |
| 287 | + "Drummondville", | |
| 288 | + "Candiac" | |
| 289 | + ] | |
| 290 | + }, | |
| 291 | + { | |
| 292 | + "ats": "adp", | |
| 293 | + "source_id": "lassonde", | |
| 294 | + "employer": "Industries Lassonde", | |
| 295 | + "cid": "f96ce972-7721-4c7b-99ea-9a970ddc824a", | |
| 296 | + "ccid": "9200604444876_2", | |
| 297 | + "url": "https://www.lassonde.com", | |
| 298 | + "sectors": [ | |
| 299 | + "Agroalimentaire", | |
| 300 | + "Fabrication" | |
| 301 | + ], | |
| 302 | + "cities": [ | |
| 303 | + "Rougemont", | |
| 304 | + "Montréal" | |
| 305 | + ] | |
| 306 | + }, | |
| 307 | + { | |
| 308 | + "ats": "adp", | |
| 309 | + "source_id": "mitsubishi_hc_capital", | |
| 310 | + "employer": "Mitsubishi HC Capital Canada", | |
| 311 | + "cid": "b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7", | |
| 312 | + "ccid": "9200144510729_2", | |
| 313 | + "url": "https://www.mhccna.com", | |
| 314 | + "sectors": [ | |
| 315 | + "Finance", | |
| 316 | + "Financement commercial" | |
| 317 | + ], | |
| 318 | + "cities": [ | |
| 319 | + "Laval", | |
| 320 | + "Trois-Rivières" | |
| 321 | + ] | |
| 322 | + }, | |
| 323 | + { | |
| 324 | + "ats": "adp", | |
| 325 | + "source_id": "msi_gestion", | |
| 326 | + "employer": "MSI Gestion immobilière", | |
| 327 | + "cid": "47d4e792-26e5-471c-92ac-73c60131bab8", | |
| 328 | + "ccid": "19000101_000001", | |
| 329 | + "url": "https://www.msimmobiliers.com", | |
| 330 | + "sectors": [ | |
| 331 | + "Immobilier", | |
| 332 | + "Gestion immobilière" | |
| 333 | + ], | |
| 334 | + "cities": [ | |
| 335 | + "Montréal" | |
| 336 | + ] | |
| 337 | + }, | |
| 338 | + { | |
| 339 | + "ats": "digitalrecruiters", | |
| 340 | + "source_id": "sante_quebec", | |
| 341 | + "employer": "Santé Québec", | |
| 342 | + "base": "https://emplois.sante.quebec", | |
| 343 | + "url": "https://sante.quebec", | |
| 344 | + "sectors": [ | |
| 345 | + "Santé", | |
| 346 | + "Secteur public", | |
| 347 | + "Administration" | |
| 348 | + ], | |
| 349 | + "cities": [ | |
| 350 | + "Montréal", | |
| 351 | + "Québec" | |
| 352 | + ] | |
| 353 | + }, | |
| 354 | + { | |
| 355 | + "ats": "workland", | |
| 356 | + "source_id": "cssdm", | |
| 357 | + "employer": "Centre de services scolaire de Montréal", | |
| 358 | + "list_url": "https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/", | |
| 359 | + "url": "https://www.cssdm.gouv.qc.ca", | |
| 360 | + "sectors": [ | |
| 361 | + "Éducation", | |
| 362 | + "Secteur public" | |
| 363 | + ], | |
| 364 | + "cities": [ | |
| 365 | + "Montréal" | |
| 366 | + ] | |
| 367 | + } | |
| 368 | +] | |
added
jobka/connectors/_jsonld.py
+160 −0
@@ -0,0 +1,160 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/_jsonld.py | |
| 6 | +# Rôle : Extraction du balisage schema.org JobPosting (JSON-LD) — partagé | |
| 7 | +# par les connecteurs dont la source publie des pages détail HTML | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Utilitaires JSON-LD (schema.org/JobPosting). | |
| 11 | + | |
| 12 | +Beaucoup de sites (iCIMS, Jobillico, Espresso-Jobs, Digital Recruiters, | |
| 13 | +Workland/Atlas…) embarquent un bloc ``<script type="application/ld+json">`` | |
| 14 | +conforme à schema.org sur la page détail de chaque offre. On l'extrait ici de | |
| 15 | +façon tolérante (JSON imparfait, @graph, listes) et on le convertit en champs | |
| 16 | +standard prêts à verser dans un ``JobPosting``. | |
| 17 | +""" | |
| 18 | +from __future__ import annotations | |
| 19 | + | |
| 20 | +import json | |
| 21 | +import re | |
| 22 | + | |
| 23 | +_SCRIPT_RE = re.compile( | |
| 24 | + r'<script[^>]*type=["\']application/ld\+json["\'][^>]*>(.*?)</script>', | |
| 25 | + re.S | re.I) | |
| 26 | + | |
| 27 | +_UNIT = {"hour": "hour", "hourly": "hour", "day": "day", "week": "week", | |
| 28 | + "month": "month", "year": "year", "annual": "year"} | |
| 29 | + | |
| 30 | + | |
| 31 | +def _iter_nodes(node): | |
| 32 | + """Itère tous les objets JSON-LD (racine, @graph, listes imbriquées).""" | |
| 33 | + if isinstance(node, list): | |
| 34 | + for item in node: | |
| 35 | + yield from _iter_nodes(item) | |
| 36 | + elif isinstance(node, dict): | |
| 37 | + yield node | |
| 38 | + yield from _iter_nodes(node.get("@graph") or []) | |
| 39 | + | |
| 40 | + | |
| 41 | +def extract_jobposting(html: str) -> dict | None: | |
| 42 | + """Retourne le premier objet @type=JobPosting trouvé dans la page.""" | |
| 43 | + for m in _SCRIPT_RE.finditer(html or ""): | |
| 44 | + raw = m.group(1).strip() | |
| 45 | + try: | |
| 46 | + data = json.loads(raw) | |
| 47 | + except ValueError: | |
| 48 | + # JSON avec contrôle non échappé (fréquent) : tentative de secours | |
| 49 | + try: | |
| 50 | + data = json.loads(re.sub(r"[\x00-\x1f]", " ", raw)) | |
| 51 | + except ValueError: | |
| 52 | + continue | |
| 53 | + for node in _iter_nodes(data): | |
| 54 | + t = node.get("@type") | |
| 55 | + types = t if isinstance(t, list) else [t] | |
| 56 | + if any(str(x).lower() == "jobposting" for x in types if x): | |
| 57 | + return node | |
| 58 | + return None | |
| 59 | + | |
| 60 | + | |
| 61 | +def _first(value): | |
| 62 | + if isinstance(value, list): | |
| 63 | + return value[0] if value else None | |
| 64 | + return value | |
| 65 | + | |
| 66 | + | |
| 67 | +def jobposting_fields(node: dict) -> dict: | |
| 68 | + """Aplati un JobPosting JSON-LD en champs standard Job·Ka (dict sparse).""" | |
| 69 | + out: dict = {} | |
| 70 | + if not node: | |
| 71 | + return out | |
| 72 | + out["title"] = node.get("title") or node.get("name") or "" | |
| 73 | + out["description_html"] = node.get("description") or "" | |
| 74 | + out["date_posted"] = node.get("datePosted") or None | |
| 75 | + out["date_deadline"] = node.get("validThrough") or None | |
| 76 | + et = node.get("employmentType") | |
| 77 | + out["employment_label"] = ", ".join(et) if isinstance(et, list) else (et or "") | |
| 78 | + | |
| 79 | + org = _first(node.get("hiringOrganization")) | |
| 80 | + if isinstance(org, dict): | |
| 81 | + out["employer"] = org.get("name") or "" | |
| 82 | + elif isinstance(org, str): | |
| 83 | + out["employer"] = org | |
| 84 | + | |
| 85 | + loc = _first(node.get("jobLocation")) | |
| 86 | + if isinstance(loc, dict): | |
| 87 | + addr = loc.get("address") or {} | |
| 88 | + if isinstance(addr, str): | |
| 89 | + out["location_label"] = addr | |
| 90 | + elif isinstance(addr, dict): | |
| 91 | + out["city"] = addr.get("addressLocality") or "" | |
| 92 | + out["region_code"] = addr.get("addressRegion") or "" | |
| 93 | + out["postal_code"] = addr.get("postalCode") or "" | |
| 94 | + out["address"] = addr.get("streetAddress") or "" | |
| 95 | + geo = loc.get("geo") or {} | |
| 96 | + if isinstance(geo, dict) and geo.get("latitude") is not None: | |
| 97 | + try: | |
| 98 | + out["lat"] = float(geo["latitude"]) | |
| 99 | + out["lng"] = float(geo["longitude"]) | |
| 100 | + except (TypeError, ValueError): | |
| 101 | + pass | |
| 102 | + | |
| 103 | + sal = node.get("baseSalary") | |
| 104 | + if isinstance(sal, dict): | |
| 105 | + val = sal.get("value") | |
| 106 | + unit = None | |
| 107 | + lo = hi = None | |
| 108 | + if isinstance(val, dict): | |
| 109 | + lo = val.get("minValue", val.get("value")) | |
| 110 | + hi = val.get("maxValue") | |
| 111 | + unit = val.get("unitText") | |
| 112 | + elif isinstance(val, (int, float)): | |
| 113 | + lo = val | |
| 114 | + if isinstance(lo, str): | |
| 115 | + try: | |
| 116 | + lo = float(lo.replace(",", ".")) | |
| 117 | + except ValueError: | |
| 118 | + lo = None | |
| 119 | + if isinstance(hi, str): | |
| 120 | + try: | |
| 121 | + hi = float(hi.replace(",", ".")) | |
| 122 | + except ValueError: | |
| 123 | + hi = None | |
| 124 | + if lo: | |
| 125 | + out["salary_min"] = float(lo) | |
| 126 | + out["salary_max"] = float(hi) if hi else None | |
| 127 | + out["salary_unit"] = _UNIT.get(str(unit or "").lower()) | |
| 128 | + return out | |
| 129 | + | |
| 130 | + | |
| 131 | +def apply_fields(job, fields: dict, *, override_employer: bool = False) -> None: | |
| 132 | + """Verse les champs extraits dans un JobPosting sans écraser l'existant.""" | |
| 133 | + from ..normalize import clean_html | |
| 134 | + if fields.get("description_html") and not job.description: | |
| 135 | + job.description = clean_html(fields["description_html"]) | |
| 136 | + if fields.get("title") and not job.title: | |
| 137 | + job.title = fields["title"] | |
| 138 | + if fields.get("employer") and (override_employer or not job.employer): | |
| 139 | + job.employer = fields["employer"] | |
| 140 | + if fields.get("date_posted") and not job.date_posted: | |
| 141 | + job.date_posted = fields["date_posted"] | |
| 142 | + if fields.get("date_deadline") and not job.date_deadline: | |
| 143 | + job.date_deadline = fields["date_deadline"] | |
| 144 | + if fields.get("city") and not job.city: | |
| 145 | + job.city = fields["city"] | |
| 146 | + if fields.get("postal_code") and not job.postal_code: | |
| 147 | + job.postal_code = fields["postal_code"] | |
| 148 | + if fields.get("address") and not job.address: | |
| 149 | + job.address = fields["address"] | |
| 150 | + if fields.get("location_label") and not job.location_label: | |
| 151 | + job.location_label = fields["location_label"] | |
| 152 | + if fields.get("salary_min") is not None and job.salary_min is None: | |
| 153 | + job.salary_min = fields["salary_min"] | |
| 154 | + job.salary_max = fields.get("salary_max") | |
| 155 | + job.salary_unit = fields.get("salary_unit") | |
| 156 | + if fields.get("employment_label"): | |
| 157 | + job.details.setdefault("employment_label", fields["employment_label"]) | |
| 158 | + if fields.get("lat") is not None and job.lat is None: | |
| 159 | + job.lat = fields["lat"] | |
| 160 | + job.lng = fields.get("lng") | |
added
jobka/connectors/adp.py
+160 −0
@@ -0,0 +1,160 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/adp.py | |
| 6 | +# Rôle : Classe de plateforme ADP Workforce Now (centre de carrières) — | |
| 7 | +# API JSON publique job-requisitions, un employeur = une sous-classe | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme ADP Workforce Now (workforcenow.adp.com). | |
| 11 | + | |
| 12 | +API JSON publique du centre de carrières : | |
| 13 | +- liste : GET /mascsr/default/careercenter/public/events/staffing/v1/ | |
| 14 | + job-requisitions?cid=<CID>&ccId=<CCID>&lang=<fr_CA>&$top=&$skip= | |
| 15 | + -> {jobRequisitions:[{itemID, requisitionTitle, postDate, | |
| 16 | + payGradeRange, workLevelCode, requisitionLocations, | |
| 17 | + customFieldGroup}]} | |
| 18 | +- détail : GET .../job-requisitions/<itemID>?cid=… -> requisitionDescription | |
| 19 | + (HTML). Visité avec cache BD + budget. | |
| 20 | +""" | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import os | |
| 24 | + | |
| 25 | +from ..schema import JobPosting, clean_html, is_quebec_location | |
| 26 | +from .base import BaseConnector | |
| 27 | + | |
| 28 | +PAGE_SIZE = 100 | |
| 29 | +MAX_DETAILS = int(os.environ.get("JOBKA_ADP_DETAIL_LIMIT", "80")) | |
| 30 | + | |
| 31 | +_API = ("https://workforcenow.adp.com/mascsr/default/careercenter/public/" | |
| 32 | + "events/staffing/v1/job-requisitions") | |
| 33 | + | |
| 34 | +_SALARY_UNIT = {"AN": "year", "ANNUEL": "year", "YR": "year", "YEAR": "year", | |
| 35 | + "HR": "hour", "HO": "hour", "HORAIRE": "hour", "HOUR": "hour"} | |
| 36 | + | |
| 37 | + | |
| 38 | +class ADPWorkforceNowConnector(BaseConnector): | |
| 39 | + """Base ADP WFN — sous-classes : définir source_id, EMPLOYER, CID, CCID.""" | |
| 40 | + | |
| 41 | + ats = "adp" | |
| 42 | + request_delay = 1.0 | |
| 43 | + | |
| 44 | + EMPLOYER = "" | |
| 45 | + CID = "" # GUID du centre de carrières | |
| 46 | + CCID = "19000101_000001" # identifiant du « career center » | |
| 47 | + LANG = "fr_CA" | |
| 48 | + quebec_only = True | |
| 49 | + max_pages = 10 | |
| 50 | + | |
| 51 | + def _params(self, extra: dict | None = None) -> dict: | |
| 52 | + p = {"cid": self.CID, "ccId": self.CCID, "lang": self.LANG, | |
| 53 | + "locale": self.LANG} | |
| 54 | + p.update(extra or {}) | |
| 55 | + return p | |
| 56 | + | |
| 57 | + @staticmethod | |
| 58 | + def _locations(req: dict) -> list[dict]: | |
| 59 | + out = [] | |
| 60 | + for loc in req.get("requisitionLocations") or []: | |
| 61 | + addr = loc.get("address") or {} | |
| 62 | + out.append({ | |
| 63 | + "city": addr.get("cityName") or "", | |
| 64 | + "prov": ((addr.get("countrySubdivisionLevel1") or {}) | |
| 65 | + .get("codeValue") or ""), | |
| 66 | + "postal": addr.get("postalCode") or "", | |
| 67 | + "label": ((loc.get("nameCode") or {}).get("shortName") | |
| 68 | + or "").strip(), | |
| 69 | + }) | |
| 70 | + return out | |
| 71 | + | |
| 72 | + def _keep(self, locations: list[dict]) -> bool: | |
| 73 | + if not self.quebec_only: | |
| 74 | + return True | |
| 75 | + return any(l["prov"].upper() == "QC" | |
| 76 | + or is_quebec_location(f"{l['label']} {l['city']}") | |
| 77 | + for l in locations) | |
| 78 | + | |
| 79 | + def _fetch_detail(self, item_id: str) -> dict: | |
| 80 | + data = self.get(f"{_API}/{item_id}", params=self._params(), | |
| 81 | + headers={"Accept": "application/json"}).json() | |
| 82 | + return {"description": clean_html( | |
| 83 | + data.get("requisitionDescription") or "")} | |
| 84 | + | |
| 85 | + def fetch(self) -> list[JobPosting]: | |
| 86 | + out: list[JobPosting] = [] | |
| 87 | + details_used = 0 | |
| 88 | + skip = 0 | |
| 89 | + for _ in range(self.max_pages): | |
| 90 | + data = self.get(_API, params=self._params( | |
| 91 | + {"$top": str(PAGE_SIZE), "$skip": str(skip)}), | |
| 92 | + headers={"Accept": "application/json"}).json() | |
| 93 | + reqs = data.get("jobRequisitions") or [] | |
| 94 | + if not reqs and skip == 0 and self.LANG != "en_US": | |
| 95 | + # certains centres de carrières ne répondent qu'en anglais | |
| 96 | + self.LANG = "en_US" | |
| 97 | + continue | |
| 98 | + if not reqs: | |
| 99 | + break | |
| 100 | + for r in reqs: | |
| 101 | + locations = self._locations(r) | |
| 102 | + if not self._keep(locations): | |
| 103 | + continue | |
| 104 | + iid = str(r.get("itemID") or "") | |
| 105 | + if not iid: | |
| 106 | + continue | |
| 107 | + qc = next((l for l in locations | |
| 108 | + if l["prov"].upper() == "QC" | |
| 109 | + or is_quebec_location(f"{l['label']} {l['city']}")), | |
| 110 | + locations[0] if locations else | |
| 111 | + {"city": "", "postal": "", "label": ""}) | |
| 112 | + job = JobPosting( | |
| 113 | + source=self.source_id, external_id=iid, | |
| 114 | + url=("https://workforcenow.adp.com/mascsr/default/mdf/" | |
| 115 | + f"recruitment/recruitment.html?cid={self.CID}" | |
| 116 | + f"&ccId={self.CCID}&lang={self.LANG}&jobId={iid}"), | |
| 117 | + employer=self.EMPLOYER, | |
| 118 | + title=r.get("requisitionTitle") or "", | |
| 119 | + city=qc["city"], postal_code=qc["postal"], | |
| 120 | + location_label=qc["label"], | |
| 121 | + date_posted=r.get("postDate") or None, | |
| 122 | + ats=self.ats, | |
| 123 | + ) | |
| 124 | + pay = r.get("payGradeRange") or {} | |
| 125 | + lo = ((pay.get("minimumRate") or {}).get("amountValue")) | |
| 126 | + hi = ((pay.get("maximumRate") or {}).get("amountValue")) | |
| 127 | + if lo: | |
| 128 | + unit = None | |
| 129 | + for c in ((r.get("customFieldGroup") or {}) | |
| 130 | + .get("codeFields") or []): | |
| 131 | + if ((c.get("nameCode") or {}) | |
| 132 | + .get("codeValue")) == "SalaryType": | |
| 133 | + unit = _SALARY_UNIT.get( | |
| 134 | + str(c.get("codeValue") or "").upper()) \ | |
| 135 | + or _SALARY_UNIT.get( | |
| 136 | + str(c.get("shortName") or "").upper()) | |
| 137 | + job.salary_min = float(lo) | |
| 138 | + job.salary_max = float(hi) if hi else None | |
| 139 | + job.salary_unit = unit or ("year" if float(lo) > 5000 | |
| 140 | + else "hour") | |
| 141 | + wl = (r.get("workLevelCode") or {}).get("shortName") | |
| 142 | + if wl: | |
| 143 | + job.details["employment_label"] = wl | |
| 144 | + key = (r.get("postDate") or "") + (job.title or "")[:40] | |
| 145 | + if details_used < MAX_DETAILS: | |
| 146 | + fresh = [False] | |
| 147 | + | |
| 148 | + def _fn(i=iid, fresh=fresh): | |
| 149 | + fresh[0] = True | |
| 150 | + return self._fetch_detail(i) | |
| 151 | + | |
| 152 | + d = self.detail(iid, key, _fn) | |
| 153 | + if fresh[0]: | |
| 154 | + details_used += 1 | |
| 155 | + job.description = d.get("description", "") | |
| 156 | + out.append(job) | |
| 157 | + if len(reqs) < PAGE_SIZE: | |
| 158 | + break | |
| 159 | + skip += PAGE_SIZE | |
| 160 | + return out | |
added
jobka/connectors/agnico_eagle.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/agnico_eagle.py | |
| 6 | +# Rôle : Connecteur Agnico Eagle Mines — taleo (host='agnicoeagle', section='2') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .taleo import TaleoConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class AgnicoEagleConnector(TaleoConnector): | |
| 14 | + source_id = 'agnico_eagle' | |
| 15 | + EMPLOYER = 'Agnico Eagle Mines' | |
| 16 | + HOST = 'agnicoeagle' | |
| 17 | + SECTION = '2' | |
added
jobka/connectors/bayshore.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/bayshore.py | |
| 6 | +# Rôle : Connecteur Bayshore HealthCare — taleo (host='bayshore', section='bs_ex') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .taleo import TaleoConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class BayshoreConnector(TaleoConnector): | |
| 14 | + source_id = 'bayshore' | |
| 15 | + EMPLOYER = 'Bayshore HealthCare' | |
| 16 | + HOST = 'bayshore' | |
| 17 | + SECTION = 'bs_ex' | |
added
jobka/connectors/bell_textron.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/bell_textron.py | |
| 6 | +# Rôle : Connecteur Bell Textron Canada — taleo (host='textron', section='bell') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .taleo import TaleoConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class BellTextronConnector(TaleoConnector): | |
| 14 | + source_id = 'bell_textron' | |
| 15 | + EMPLOYER = 'Bell Textron Canada' | |
| 16 | + HOST = 'textron' | |
| 17 | + SECTION = 'bell' | |
added
jobka/connectors/bombardier.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/bombardier.py | |
| 6 | +# Rôle : Connecteur Bombardier — successfactors (base='https://jobs.bombardier.com') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .successfactors import SuccessFactorsConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class BombardierConnector(SuccessFactorsConnector): | |
| 14 | + source_id = 'bombardier' | |
| 15 | + EMPLOYER = 'Bombardier' | |
| 16 | + BASE = 'https://jobs.bombardier.com' | |
added
jobka/connectors/capreit.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/capreit.py | |
| 6 | +# Rôle : Connecteur CAPREIT — icims (sub='careers-capreit') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .icims import ICIMSConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class CapreitConnector(ICIMSConnector): | |
| 14 | + source_id = 'capreit' | |
| 15 | + EMPLOYER = 'CAPREIT' | |
| 16 | + SUB = 'careers-capreit' | |
added
jobka/connectors/cascades.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/cascades.py | |
| 6 | +# Rôle : Connecteur Cascades — successfactors (base='https://jobs.cascades.com') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .successfactors import SuccessFactorsConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class CascadesConnector(SuccessFactorsConnector): | |
| 14 | + source_id = 'cascades' | |
| 15 | + EMPLOYER = 'Cascades' | |
| 16 | + BASE = 'https://jobs.cascades.com' | |
added
jobka/connectors/ciusss_est_montreal.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/ciusss_est_montreal.py | |
| 6 | +# Rôle : Connecteur CIUSSS de l'Est-de-l'Île-de-Montréal — taleo (host='ciusssemtl', section='cemtl') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .taleo import TaleoConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class CiusssEstMontrealConnector(TaleoConnector): | |
| 14 | + source_id = 'ciusss_est_montreal' | |
| 15 | + EMPLOYER = "CIUSSS de l'Est-de-l'Île-de-Montréal" | |
| 16 | + HOST = 'ciusssemtl' | |
| 17 | + SECTION = 'cemtl' | |
added
jobka/connectors/cssdm.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/cssdm.py | |
| 6 | +# Rôle : Connecteur Centre de services scolaire de Montréal — workland (list_url='https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .workland import WorklandConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class CssdmConnector(WorklandConnector): | |
| 14 | + source_id = 'cssdm' | |
| 15 | + EMPLOYER = 'Centre de services scolaire de Montréal' | |
| 16 | + DEFAULT_CITY = "Montréal" | |
| 17 | + LIST_URL = 'https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/' | |
added
jobka/connectors/digitalrecruiters.py
+81 −0
@@ -0,0 +1,81 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/digitalrecruiters.py | |
| 6 | +# Rôle : Classe de plateforme Cegid Digital Recruiters (sites carrières) — | |
| 7 | +# sitemap + JSON-LD des pages annonce (ex. Santé Québec) | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme Cegid Digital Recruiters. | |
| 11 | + | |
| 12 | +Les sites carrières Digital Recruiters (ex. emplois.sante.quebec) sont rendus | |
| 13 | +côté serveur : | |
| 14 | +- liste : GET <BASE>/<locale>/sitemap.xml -> URLs /annonce/<id>-<slug> | |
| 15 | +- détail : chaque page annonce embarque un JSON-LD schema.org/JobPosting | |
| 16 | + complet (description, lieu, salaire, dates) — cache BD + budget. | |
| 17 | +""" | |
| 18 | +from __future__ import annotations | |
| 19 | + | |
| 20 | +import os | |
| 21 | +import re | |
| 22 | + | |
| 23 | +from ..schema import JobPosting, is_quebec_location | |
| 24 | +from . import _jsonld | |
| 25 | +from .base import BaseConnector | |
| 26 | + | |
| 27 | +MAX_DETAILS = int(os.environ.get("JOBKA_DR_DETAIL_LIMIT", "120")) | |
| 28 | + | |
| 29 | +_AD_RE = re.compile(r"<loc>([^<]*/annonce/(\d+)[^<]*)</loc>", re.I) | |
| 30 | + | |
| 31 | + | |
| 32 | +class DigitalRecruitersConnector(BaseConnector): | |
| 33 | + """Base Digital Recruiters — sous-classes : définir source_id, EMPLOYER, | |
| 34 | + BASE (ex. https://emplois.sante.quebec) et LOCALE.""" | |
| 35 | + | |
| 36 | + ats = "digitalrecruiters" | |
| 37 | + request_delay = 0.8 | |
| 38 | + | |
| 39 | + EMPLOYER = "" | |
| 40 | + BASE = "" | |
| 41 | + LOCALE = "fr_CA" | |
| 42 | + quebec_only = True | |
| 43 | + | |
| 44 | + def _fetch_detail(self, url: str) -> dict: | |
| 45 | + html = self.get(url).text | |
| 46 | + node = _jsonld.extract_jobposting(html) | |
| 47 | + return _jsonld.jobposting_fields(node) if node else {} | |
| 48 | + | |
| 49 | + def fetch(self) -> list[JobPosting]: | |
| 50 | + xml = self.get(f"{self.BASE}/{self.LOCALE}/sitemap.xml").text | |
| 51 | + out: list[JobPosting] = [] | |
| 52 | + details_used = 0 | |
| 53 | + seen: set[str] = set() | |
| 54 | + for url, eid in _AD_RE.findall(xml): | |
| 55 | + if eid in seen: | |
| 56 | + continue | |
| 57 | + seen.add(eid) | |
| 58 | + job = JobPosting(source=self.source_id, external_id=eid, url=url, | |
| 59 | + employer=self.EMPLOYER, title="", ats=self.ats) | |
| 60 | + fields: dict = {} | |
| 61 | + if details_used < MAX_DETAILS: | |
| 62 | + fresh = [False] | |
| 63 | + | |
| 64 | + def _fn(u=url, fresh=fresh): | |
| 65 | + fresh[0] = True | |
| 66 | + return self._fetch_detail(u) | |
| 67 | + | |
| 68 | + fields = self.detail(eid, eid, _fn) | |
| 69 | + if fresh[0]: | |
| 70 | + details_used += 1 | |
| 71 | + if not fields: | |
| 72 | + continue | |
| 73 | + _jsonld.apply_fields(job, fields) | |
| 74 | + if self.quebec_only and job.city and not is_quebec_location( | |
| 75 | + f"{job.city} {fields.get('region_code', '')}"): | |
| 76 | + if (fields.get("region_code") or "").upper() != "QC": | |
| 77 | + continue | |
| 78 | + if not job.title: | |
| 79 | + continue | |
| 80 | + out.append(job) | |
| 81 | + return out | |
added
jobka/connectors/domtar.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/domtar.py | |
| 6 | +# Rôle : Connecteur Domtar — successfactors (base='https://jobs.domtar.com') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .successfactors import SuccessFactorsConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class DomtarConnector(SuccessFactorsConnector): | |
| 14 | + source_id = 'domtar' | |
| 15 | + EMPLOYER = 'Domtar' | |
| 16 | + BASE = 'https://jobs.domtar.com' | |
added
jobka/connectors/edf_renouvelables.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/edf_renouvelables.py | |
| 6 | +# Rôle : Connecteur EDF Renouvelables Canada — icims (sub='cafrench-edf-re') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .icims import ICIMSConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class EdfRenouvelablesConnector(ICIMSConnector): | |
| 14 | + source_id = 'edf_renouvelables' | |
| 15 | + EMPLOYER = 'EDF Renouvelables Canada' | |
| 16 | + SUB = 'cafrench-edf-re' | |
added
jobka/connectors/icims.py
+114 −0
@@ -0,0 +1,114 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/icims.py | |
| 6 | +# Rôle : Classe de plateforme iCIMS (portail carrières <org>.icims.com) | |
| 7 | +# — liste iframe paginée + JSON-LD des pages détail | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme iCIMS. | |
| 11 | + | |
| 12 | +- liste : GET https://<sub>.icims.com/jobs/search?ss=1&in_iframe=1&pr=<page> | |
| 13 | + (HTML léger, liens /jobs/<id>/<slug>/job) — pagination pr=0,1,2… | |
| 14 | +- détail : GET https://<sub>.icims.com/jobs/<id>/<slug>/job?in_iframe=1 | |
| 15 | + -> JSON-LD schema.org/JobPosting complet (description, lieu, dates) | |
| 16 | + | |
| 17 | +Le lieu n'apparaissant pas toujours sur la liste, le filtre Québec est | |
| 18 | +appliqué après lecture du détail (avec cache BD : chaque offre n'est visitée | |
| 19 | +qu'une fois). | |
| 20 | +""" | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import os | |
| 24 | +import re | |
| 25 | + | |
| 26 | +from ..schema import JobPosting, is_quebec_location | |
| 27 | +from . import _jsonld | |
| 28 | +from .base import BaseConnector | |
| 29 | + | |
| 30 | +MAX_DETAILS = int(os.environ.get("JOBKA_ICIMS_DETAIL_LIMIT", "150")) | |
| 31 | + | |
| 32 | +_LINK_RE = re.compile(r'href="https?://[^"]*?/jobs/(\d+)/([^/"]+)/job[^"]*"') | |
| 33 | + | |
| 34 | + | |
| 35 | +class ICIMSConnector(BaseConnector): | |
| 36 | + """Base iCIMS — sous-classes : définir source_id, EMPLOYER, SUB | |
| 37 | + (sous-domaine complet, ex. « careers-capreit »).""" | |
| 38 | + | |
| 39 | + ats = "icims" | |
| 40 | + request_delay = 1.0 | |
| 41 | + | |
| 42 | + EMPLOYER = "" | |
| 43 | + SUB = "" # <SUB>.icims.com | |
| 44 | + quebec_only = True | |
| 45 | + max_pages = 15 | |
| 46 | + | |
| 47 | + @property | |
| 48 | + def _base(self) -> str: | |
| 49 | + return f"https://{self.SUB}.icims.com" | |
| 50 | + | |
| 51 | + def _fetch_detail(self, jid: str, slug: str) -> dict: | |
| 52 | + html = self.get(f"{self._base}/jobs/{jid}/{slug}/job", | |
| 53 | + params={"in_iframe": "1"}).text | |
| 54 | + node = _jsonld.extract_jobposting(html) | |
| 55 | + return _jsonld.jobposting_fields(node) if node else {} | |
| 56 | + | |
| 57 | + def fetch(self) -> list[JobPosting]: | |
| 58 | + seen: dict[str, str] = {} | |
| 59 | + for page in range(self.max_pages): | |
| 60 | + html = self.get(f"{self._base}/jobs/search", | |
| 61 | + params={"ss": "1", "in_iframe": "1", | |
| 62 | + "pr": str(page)}).text | |
| 63 | + links = _LINK_RE.findall(html) | |
| 64 | + new = 0 | |
| 65 | + for jid, slug in links: | |
| 66 | + if jid not in seen: | |
| 67 | + seen[jid] = slug | |
| 68 | + new += 1 | |
| 69 | + if not links or new == 0: | |
| 70 | + break | |
| 71 | + | |
| 72 | + out: list[JobPosting] = [] | |
| 73 | + details_used = 0 | |
| 74 | + for jid, slug in seen.items(): | |
| 75 | + if details_used >= MAX_DETAILS: | |
| 76 | + # budget épuisé : ne servir que le cache (sans jamais y écrire | |
| 77 | + # un vide) — l'offre sera reprise à la prochaine synchronisation | |
| 78 | + from .. import db | |
| 79 | + if self._detail_con is None: | |
| 80 | + self._detail_con = db.connect() | |
| 81 | + cached = db.get_cached_detail(self._detail_con, self.source_id, | |
| 82 | + jid, jid) | |
| 83 | + if cached is None: | |
| 84 | + continue | |
| 85 | + fields = cached | |
| 86 | + else: | |
| 87 | + fresh = [False] | |
| 88 | + | |
| 89 | + def _fn(j=jid, s=slug, fresh=fresh): | |
| 90 | + fresh[0] = True | |
| 91 | + return self._fetch_detail(j, s) | |
| 92 | + | |
| 93 | + fields = self.detail(jid, jid, _fn) | |
| 94 | + if fresh[0]: | |
| 95 | + details_used += 1 | |
| 96 | + if not fields: | |
| 97 | + continue | |
| 98 | + place = " ".join(str(fields.get(k) or "") for k in | |
| 99 | + ("city", "region_code", "postal_code")) | |
| 100 | + if self.quebec_only and not ( | |
| 101 | + (fields.get("region_code") or "").upper() == "QC" | |
| 102 | + or is_quebec_location(place)): | |
| 103 | + continue | |
| 104 | + job = JobPosting( | |
| 105 | + source=self.source_id, external_id=jid, | |
| 106 | + url=f"{self._base}/jobs/{jid}/{slug}/job", | |
| 107 | + employer=self.EMPLOYER, | |
| 108 | + title="", ats=self.ats, | |
| 109 | + ) | |
| 110 | + _jsonld.apply_fields(job, fields) | |
| 111 | + if not job.title: | |
| 112 | + job.title = slug.replace("-", " ").strip().capitalize() | |
| 113 | + out.append(job) | |
| 114 | + return out | |
added
jobka/connectors/lassonde.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/lassonde.py | |
| 6 | +# Rôle : Connecteur Industries Lassonde — adp (cid='f96ce972-7721-4c7b-99ea-9a970ddc824a', ccid='9200604444876_2') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .adp import ADPWorkforceNowConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class LassondeConnector(ADPWorkforceNowConnector): | |
| 14 | + source_id = 'lassonde' | |
| 15 | + EMPLOYER = 'Industries Lassonde' | |
| 16 | + CID = 'f96ce972-7721-4c7b-99ea-9a970ddc824a' | |
| 17 | + CCID = '9200604444876_2' | |
added
jobka/connectors/mccarthy_tetrault.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/mccarthy_tetrault.py | |
| 6 | +# Rôle : Connecteur McCarthy Tétrault — icims (sub='careers-mccarthyca') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .icims import ICIMSConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class MccarthyTetraultConnector(ICIMSConnector): | |
| 14 | + source_id = 'mccarthy_tetrault' | |
| 15 | + EMPLOYER = 'McCarthy Tétrault' | |
| 16 | + SUB = 'careers-mccarthyca' | |
added
jobka/connectors/messer_canada.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/messer_canada.py | |
| 6 | +# Rôle : Connecteur Messer Canada — ultipro (org='MES1005MESR', board='dbd63926-d756-486f-a254-b2a6ede1d26e') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .ultipro import UltiProConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class MesserCanadaConnector(UltiProConnector): | |
| 14 | + source_id = 'messer_canada' | |
| 15 | + EMPLOYER = 'Messer Canada' | |
| 16 | + ORG = 'MES1005MESR' | |
| 17 | + BOARD = 'dbd63926-d756-486f-a254-b2a6ede1d26e' | |
added
jobka/connectors/mitsubishi_hc_capital.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/mitsubishi_hc_capital.py | |
| 6 | +# Rôle : Connecteur Mitsubishi HC Capital Canada — adp (cid='b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7', ccid='9200144510729_2') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .adp import ADPWorkforceNowConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class MitsubishiHcCapitalConnector(ADPWorkforceNowConnector): | |
| 14 | + source_id = 'mitsubishi_hc_capital' | |
| 15 | + EMPLOYER = 'Mitsubishi HC Capital Canada' | |
| 16 | + CID = 'b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7' | |
| 17 | + CCID = '9200144510729_2' | |
added
jobka/connectors/msi_gestion.py
+17 −0
@@ -0,0 +1,17 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/msi_gestion.py | |
| 6 | +# Rôle : Connecteur MSI Gestion immobilière — adp (cid='47d4e792-26e5-471c-92ac-73c60131bab8', ccid='19000101_000001') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .adp import ADPWorkforceNowConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class MsiGestionConnector(ADPWorkforceNowConnector): | |
| 14 | + source_id = 'msi_gestion' | |
| 15 | + EMPLOYER = 'MSI Gestion immobilière' | |
| 16 | + CID = '47d4e792-26e5-471c-92ac-73c60131bab8' | |
| 17 | + CCID = '19000101_000001' | |
added
jobka/connectors/njoyn.py
+182 −0
@@ -0,0 +1,182 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/njoyn.py | |
| 6 | +# Rôle : Classe de plateforme Njoyn (CGI) — très répandu au secteur public | |
| 7 | +# québécois (santé, villes, sociétés d'État). HTML serveur + jeton. | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme Njoyn (CGI). | |
| 11 | + | |
| 12 | +Les babillards Njoyn servent un HTML rendu côté serveur : | |
| 13 | + <BASE>/<CL>/xweb/xweb.asp?clid=<CLID>&page=joblisting&lang=<n> | |
| 14 | +Le serveur redirige en ajoutant un jeton de session (tbtoken) — il suffit de | |
| 15 | +suivre la redirection. Chaque offre est un <article> (habillage moderne) ou | |
| 16 | +une ligne de tableau (habillage classique) pointant vers | |
| 17 | +``page=jobdetails&jobid=J####-####``. | |
| 18 | + | |
| 19 | +⚠ Les domaines *.njoyn.com sont derrière Radware Bot Manager : pour ces | |
| 20 | +tenants, mettre ``USE_SCRAPFLY = True`` (contournement ASP). Les babillards | |
| 21 | +servis sur un domaine personnalisé (ex. emplois.santemontreal.qc.ca) se | |
| 22 | +consultent directement. | |
| 23 | +""" | |
| 24 | +from __future__ import annotations | |
| 25 | + | |
| 26 | +import hashlib | |
| 27 | +import html as _html | |
| 28 | +import os | |
| 29 | +import re | |
| 30 | + | |
| 31 | +from ..schema import JobPosting, clean_html | |
| 32 | +from .base import BaseConnector | |
| 33 | + | |
| 34 | +MAX_DETAILS = int(os.environ.get("JOBKA_NJOYN_DETAIL_LIMIT", "40")) | |
| 35 | + | |
| 36 | +_ARTICLE_RE = re.compile(r"<article[^>]*>(.*?)</article>", re.S | re.I) | |
| 37 | +_H4_RE = re.compile(r"<h4[^>]*>(.*?)</h4>", re.S | re.I) | |
| 38 | +_TIME_RE = re.compile(r"<time[^>]*>(.*?)</time>", re.S | re.I) | |
| 39 | +_JOBID_RE = re.compile(r"jobid=(J\d{4}-\d+)", re.I) | |
| 40 | +_FIELD_RE = re.compile( | |
| 41 | + r"<li><strong>([^<:]+?)(?: )?\s*:?\s*</strong>\s*(.*?)</li>", | |
| 42 | + re.S | re.I) | |
| 43 | +_APPLY_RE = re.compile( | |
| 44 | + r"<a[^>]*class='btn btn-primary'[^>]*href='([^']+)'", re.I) | |
| 45 | +_ROW_LINK_RE = re.compile( | |
| 46 | + r'''<a[^>]*href=["'][^"']*jobid=(J\d{4}-\d+)[^"']*["'][^>]*>(.*?)</a>''', | |
| 47 | + re.S | re.I) | |
| 48 | + | |
| 49 | + | |
| 50 | +def _txt(s: str) -> str: | |
| 51 | + return re.sub(r"\s+", " ", _html.unescape(re.sub(r"<[^>]+>", " ", s or ""))).strip() | |
| 52 | + | |
| 53 | + | |
| 54 | +class NjoynConnector(BaseConnector): | |
| 55 | + """Base Njoyn — sous-classes : définir source_id, EMPLOYER, BASE, CL, CLID.""" | |
| 56 | + | |
| 57 | + ats = "njoyn" | |
| 58 | + request_delay = 1.2 | |
| 59 | + | |
| 60 | + EMPLOYER = "" | |
| 61 | + BASE = "" # ex. https://emplois.santemontreal.qc.ca | |
| 62 | + CL = "CL3" # instance (CL2, CL3, CL4…) | |
| 63 | + CLID = "" # identifiant client Njoyn | |
| 64 | + LANG = "2" # 2 = français, 1 = anglais | |
| 65 | + USE_SCRAPFLY = False # requis pour les domaines *.njoyn.com (Radware) | |
| 66 | + | |
| 67 | + def _url(self, page: str, jobid: str = "") -> str: | |
| 68 | + u = (f"{self.BASE}/{self.CL}/xweb/xweb.asp?clid={self.CLID}" | |
| 69 | + f"&page={page}&lang={self.LANG}") | |
| 70 | + if jobid: | |
| 71 | + u += f"&jobid={jobid}" | |
| 72 | + return u | |
| 73 | + | |
| 74 | + def _html(self, url: str) -> str: | |
| 75 | + if self.USE_SCRAPFLY: | |
| 76 | + return self.get_scrapfly(url, render_js=False) | |
| 77 | + text = self.get(url).text | |
| 78 | + if "perfdrive" in text[:4000]: # mur anti-bot : bascule Scrapfly | |
| 79 | + return self.get_scrapfly(url, render_js=False) | |
| 80 | + return text | |
| 81 | + | |
| 82 | + def _fetch_detail(self, jobid: str) -> dict: | |
| 83 | + page = self._html(self._url("jobdetails", jobid)) | |
| 84 | + page = re.sub(r"<(noscript|nav|script|style)[^>]*>.*?</\1>", " ", page, | |
| 85 | + flags=re.S | re.I) | |
| 86 | + # habillage moderne : sections <div class="row"><h2>Titre</h2>corps</div> | |
| 87 | + sections = re.findall( | |
| 88 | + r'<div class="row">\s*<h2[^>]*>(.*?)</h2>(.*?)</div>', page, | |
| 89 | + re.S | re.I) | |
| 90 | + if sections: | |
| 91 | + body = "\n\n".join(f"{clean_html(h)}\n{clean_html(b)}" | |
| 92 | + for h, b in sections) | |
| 93 | + return {"description": body[:20000]} | |
| 94 | + # habillage classique : contenu principal après le <h1> | |
| 95 | + m = re.search(r"<h1[^>]*>.*?</h1>(.*?)(?:<form|<footer|</main)", page, | |
| 96 | + re.S | re.I) | |
| 97 | + body = m.group(1) if m else "" | |
| 98 | + return {"description": clean_html(body)[:20000]} | |
| 99 | + | |
| 100 | + def _parse_articles(self, page: str) -> list[dict]: | |
| 101 | + items = [] | |
| 102 | + for art in _ARTICLE_RE.findall(page): | |
| 103 | + m = _JOBID_RE.search(art) | |
| 104 | + if not m: | |
| 105 | + continue | |
| 106 | + fields = {_txt(k).lower(): _txt(v) for k, v in _FIELD_RE.findall(art)} | |
| 107 | + apply_m = _APPLY_RE.search(art) | |
| 108 | + h4 = _H4_RE.search(art) | |
| 109 | + t = _TIME_RE.search(art) | |
| 110 | + items.append({ | |
| 111 | + "jobid": m.group(1), | |
| 112 | + "title": _txt(h4.group(1)) if h4 else "", | |
| 113 | + "date": _txt(t.group(1)).lstrip("Le ").strip() if t else "", | |
| 114 | + "fields": fields, | |
| 115 | + "apply_url": apply_m.group(1) if apply_m else "", | |
| 116 | + }) | |
| 117 | + return items | |
| 118 | + | |
| 119 | + def _parse_rows(self, page: str) -> list[dict]: | |
| 120 | + """Habillage classique : simples liens de tableau vers jobdetails.""" | |
| 121 | + items, seen = [], set() | |
| 122 | + for jobid, label in _ROW_LINK_RE.findall(page): | |
| 123 | + title = _txt(label) | |
| 124 | + if not title or title.lower().startswith(("en savoir", "postuler", | |
| 125 | + "apply", "more")): | |
| 126 | + title = "" | |
| 127 | + if jobid in seen: | |
| 128 | + # garder le premier libellé non vide | |
| 129 | + if title: | |
| 130 | + for it in items: | |
| 131 | + if it["jobid"] == jobid and not it["title"]: | |
| 132 | + it["title"] = title | |
| 133 | + continue | |
| 134 | + seen.add(jobid) | |
| 135 | + items.append({"jobid": jobid, "title": title, "date": "", | |
| 136 | + "fields": {}, "apply_url": ""}) | |
| 137 | + return [it for it in items if it["title"]] | |
| 138 | + | |
| 139 | + def fetch(self) -> list[JobPosting]: | |
| 140 | + page = self._html(self._url("joblisting")) | |
| 141 | + items = self._parse_articles(page) or self._parse_rows(page) | |
| 142 | + out: list[JobPosting] = [] | |
| 143 | + details_used = 0 | |
| 144 | + for it in items: | |
| 145 | + f = it["fields"] | |
| 146 | + salary = f.get("salaire", "") | |
| 147 | + if re.match(r"^0\s*\$\s*-\s*0\s*\$", salary): | |
| 148 | + salary = "" | |
| 149 | + job = JobPosting( | |
| 150 | + source=self.source_id, external_id=it["jobid"], | |
| 151 | + url=self._url("jobdetails", it["jobid"]), | |
| 152 | + employer=f.get("établissement") or f.get("etablissement") | |
| 153 | + or self.EMPLOYER, | |
| 154 | + title=it["title"], | |
| 155 | + city=f.get("villes et arrondissements", "").split(",")[0], | |
| 156 | + location_label=f.get("villes et arrondissements", ""), | |
| 157 | + salary_label=salary, | |
| 158 | + date_posted=it["date"] or None, | |
| 159 | + ats=self.ats, | |
| 160 | + ) | |
| 161 | + if f.get("statut de l'employé") or f.get("statut de l'employe"): | |
| 162 | + job.details["employment_label"] = ( | |
| 163 | + f.get("statut de l'employé") or f.get("statut de l'employe")) | |
| 164 | + if f.get("niveau de scolarité"): | |
| 165 | + job.requirements["scolarite"] = f["niveau de scolarité"] | |
| 166 | + if it["apply_url"]: | |
| 167 | + job.details["apply_url"] = it["apply_url"] | |
| 168 | + key = hashlib.sha1( | |
| 169 | + f"{it['title']}|{it['date']}".encode()).hexdigest()[:12] | |
| 170 | + if details_used < MAX_DETAILS: | |
| 171 | + fresh = [False] | |
| 172 | + | |
| 173 | + def _fn(j=it["jobid"], fresh=fresh): | |
| 174 | + fresh[0] = True | |
| 175 | + return self._fetch_detail(j) | |
| 176 | + | |
| 177 | + d = self.detail(it["jobid"], key, _fn) | |
| 178 | + if fresh[0]: | |
| 179 | + details_used += 1 | |
| 180 | + job.description = d.get("description", "") | |
| 181 | + out.append(job) | |
| 182 | + return out | |
added
jobka/connectors/omni_hotels_gestion.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/omni_hotels_gestion.py | |
| 6 | +# Rôle : Connecteur Omni Hôtel Mont-Royal — icims (sub='externalmanager-omnihotels') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .icims import ICIMSConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class OmniHotelsGestionConnector(ICIMSConnector): | |
| 14 | + source_id = 'omni_hotels_gestion' | |
| 15 | + EMPLOYER = 'Omni Hôtel Mont-Royal' | |
| 16 | + SUB = 'externalmanager-omnihotels' | |
added
jobka/connectors/omni_hotels_horaire.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/omni_hotels_horaire.py | |
| 6 | +# Rôle : Connecteur Omni Hôtel Mont-Royal — icims (sub='externalhourly-omnihotels') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .icims import ICIMSConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class OmniHotelsHoraireConnector(ICIMSConnector): | |
| 14 | + source_id = 'omni_hotels_horaire' | |
| 15 | + EMPLOYER = 'Omni Hôtel Mont-Royal' | |
| 16 | + SUB = 'externalhourly-omnihotels' | |
added
jobka/connectors/rtc_quebec.py
+19 −0
@@ -0,0 +1,19 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/rtc_quebec.py | |
| 6 | +# Rôle : Connecteur Réseau de transport de la Capitale — njoyn (base='https://rtc.njoyn.com', cl='CGI') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .njoyn import NjoynConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class RtcQuebecConnector(NjoynConnector): | |
| 14 | + source_id = 'rtc_quebec' | |
| 15 | + EMPLOYER = 'Réseau de transport de la Capitale' | |
| 16 | + BASE = 'https://rtc.njoyn.com' | |
| 17 | + CL = 'CGI' | |
| 18 | + CLID = '23009' | |
| 19 | + USE_SCRAPFLY = True | |
added
jobka/connectors/sante_montreal.py
+18 −0
@@ -0,0 +1,18 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/sante_montreal.py | |
| 6 | +# Rôle : Connecteur Santé Québec — Montréal — njoyn (base='https://emplois.santemontreal.qc.ca', cl='CL3') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .njoyn import NjoynConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class SanteMontrealConnector(NjoynConnector): | |
| 14 | + source_id = 'sante_montreal' | |
| 15 | + EMPLOYER = 'Santé Québec — Montréal' | |
| 16 | + BASE = 'https://emplois.santemontreal.qc.ca' | |
| 17 | + CL = 'CL3' | |
| 18 | + CLID = '54327' | |
added
jobka/connectors/sante_quebec.py
+16 −0
@@ -0,0 +1,16 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/sante_quebec.py | |
| 6 | +# Rôle : Connecteur Santé Québec — digitalrecruiters (base='https://emplois.sante.quebec') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .digitalrecruiters import DigitalRecruitersConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class SanteQuebecConnector(DigitalRecruitersConnector): | |
| 14 | + source_id = 'sante_quebec' | |
| 15 | + EMPLOYER = 'Santé Québec' | |
| 16 | + BASE = 'https://emplois.sante.quebec' | |
added
jobka/connectors/stl_laval.py
+19 −0
@@ -0,0 +1,19 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/stl_laval.py | |
| 6 | +# Rôle : Connecteur Société de transport de Laval — njoyn (base='https://lavaltransit.njoyn.com', cl='CL2') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .njoyn import NjoynConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class StlLavalConnector(NjoynConnector): | |
| 14 | + source_id = 'stl_laval' | |
| 15 | + EMPLOYER = 'Société de transport de Laval' | |
| 16 | + BASE = 'https://lavaltransit.njoyn.com' | |
| 17 | + CL = 'CL2' | |
| 18 | + CLID = '60406' | |
| 19 | + USE_SCRAPFLY = True | |
added
jobka/connectors/successfactors.py
+161 −0
@@ -0,0 +1,161 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/successfactors.py | |
| 6 | +# Rôle : Classe de plateforme SAP SuccessFactors (sites carrières « Career | |
| 7 | +# Site Builder » / jobs2web) — sitemap RSS ou XML + microdonnées | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme SAP SuccessFactors (Career Site Builder, ex-jobs2web). | |
| 11 | + | |
| 12 | +Deux formats de « sitemap.xml » selon la génération du site : | |
| 13 | +- RSS 2.0 : <item> avec titre « Poste (VILLE, PROV, PAYS[, CP]) » ET la | |
| 14 | + description HTML complète — tout en une requête ; | |
| 15 | +- urlset : URLs /job/<Ville>-<titre>-<Prov>-<CP>/<id>/ + lastmod ; la | |
| 16 | + description est lue sur la page détail (microdonnées itemprop, avec cache | |
| 17 | + BD + budget). Le filtre Québec s'appuie sur la province/le code postal | |
| 18 | + présents dans le slug — aucun détail hors Québec n'est visité. | |
| 19 | +""" | |
| 20 | +from __future__ import annotations | |
| 21 | + | |
| 22 | +import html as _html | |
| 23 | +import os | |
| 24 | +import re | |
| 25 | +import urllib.parse | |
| 26 | + | |
| 27 | +from ..schema import JobPosting, clean_html, is_quebec_location | |
| 28 | +from .base import BaseConnector | |
| 29 | + | |
| 30 | +MAX_DETAILS = int(os.environ.get("JOBKA_SF_DETAIL_LIMIT", "150")) | |
| 31 | + | |
| 32 | +_ITEM_RE = re.compile(r"<item>(.*?)</item>", re.S | re.I) | |
| 33 | +_TAG_RE = {t: re.compile(rf"<{t}>(.*?)</{t}>", re.S | re.I) | |
| 34 | + for t in ("title", "link", "description", "pubDate")} | |
| 35 | +_LOC_RE = re.compile(r"<loc>([^<]+)</loc>\s*(?:<lastmod>([^<]+)</lastmod>)?", | |
| 36 | + re.I) | |
| 37 | +_RSS_TITLE_RE = re.compile(r"^(.*)\(([^,()]+),\s*([^,()]+?),\s*([A-Z]{2})" | |
| 38 | + r"(?:,\s*([^()]+))?\)\s*$", re.S) | |
| 39 | +_QC_PROV = {"QC", "QUÉBEC", "QUEBEC"} | |
| 40 | +_QC_SLUG_RE = re.compile(r"-(qu[ée]b|qc)-|-[ghj]\d[a-z][- ]?\d[a-z]\d/", re.I) | |
| 41 | +_ITEMPROP_RE = re.compile( | |
| 42 | + r'<(?:span|div|p)[^>]*itemprop="description"[^>]*>(.*?)' | |
| 43 | + r"</(?:span|div|p)>", re.S | re.I) | |
| 44 | +_META_RE = re.compile( | |
| 45 | + r'itemprop="(datePosted|addressLocality|postalCode|employmentType)"' | |
| 46 | + r'[^>]*content="([^"]*)"', re.I) | |
| 47 | + | |
| 48 | + | |
| 49 | +class SuccessFactorsConnector(BaseConnector): | |
| 50 | + """Base SuccessFactors CSB — sous-classes : définir source_id, EMPLOYER, | |
| 51 | + BASE (URL du site carrières, ex. https://jobs.bombardier.com).""" | |
| 52 | + | |
| 53 | + ats = "successfactors" | |
| 54 | + request_delay = 1.0 | |
| 55 | + | |
| 56 | + EMPLOYER = "" | |
| 57 | + BASE = "" | |
| 58 | + quebec_only = True | |
| 59 | + | |
| 60 | + def _cdata(self, s: str) -> str: | |
| 61 | + s = (s or "").strip() | |
| 62 | + if s.startswith("<![CDATA["): | |
| 63 | + s = s[9:] | |
| 64 | + if s.endswith("]]>"): | |
| 65 | + s = s[:-3] | |
| 66 | + return _html.unescape(s.strip()) | |
| 67 | + | |
| 68 | + # -- format RSS ------------------------------------------------------------ | |
| 69 | + def _fetch_rss(self, xml: str) -> list[JobPosting]: | |
| 70 | + out = [] | |
| 71 | + for raw in _ITEM_RE.findall(xml): | |
| 72 | + def g(tag: str, raw=raw) -> str: | |
| 73 | + m = _TAG_RE[tag].search(raw) | |
| 74 | + return self._cdata(m.group(1)) if m else "" | |
| 75 | + vals = {t: g(t) for t in _TAG_RE} | |
| 76 | + link = vals["link"] | |
| 77 | + m_id = re.search(r"/(\d+)/?$", link) | |
| 78 | + if not link or not m_id: | |
| 79 | + continue | |
| 80 | + title, city, prov = vals["title"], "", "" | |
| 81 | + m = _RSS_TITLE_RE.match(vals["title"]) | |
| 82 | + if m: | |
| 83 | + title = m.group(1).strip() | |
| 84 | + city, prov = m.group(2).strip(), m.group(3).strip() | |
| 85 | + if self.quebec_only and not ( | |
| 86 | + prov.upper() in _QC_PROV | |
| 87 | + or (city and is_quebec_location(city))): | |
| 88 | + continue | |
| 89 | + job = JobPosting( | |
| 90 | + source=self.source_id, external_id=m_id.group(1), | |
| 91 | + url=link, employer=self.EMPLOYER, title=title, | |
| 92 | + description=clean_html(self._cdata(vals["description"])), | |
| 93 | + city=city.title() if city.isupper() else city, | |
| 94 | + date_posted=vals["pubDate"] or None, | |
| 95 | + ats=self.ats, | |
| 96 | + ) | |
| 97 | + out.append(job) | |
| 98 | + return out | |
| 99 | + | |
| 100 | + # -- format urlset --------------------------------------------------------- | |
| 101 | + def _fetch_detail(self, url: str) -> dict: | |
| 102 | + html = self.get(url).text | |
| 103 | + d: dict = {} | |
| 104 | + chunks = _ITEMPROP_RE.findall(html) | |
| 105 | + if chunks: | |
| 106 | + d["description"] = clean_html("\n".join(chunks))[:20000] | |
| 107 | + for prop, content in _META_RE.findall(html): | |
| 108 | + d[prop] = content | |
| 109 | + return d | |
| 110 | + | |
| 111 | + def _fetch_urlset(self, xml: str) -> list[JobPosting]: | |
| 112 | + out = [] | |
| 113 | + details_used = 0 | |
| 114 | + for loc, lastmod in _LOC_RE.findall(xml): | |
| 115 | + loc = _html.unescape(loc.strip()) | |
| 116 | + m = re.search(r"/job/([^/]+)/(\d+)/?$", loc) | |
| 117 | + if not m: | |
| 118 | + continue | |
| 119 | + slug_raw, eid = m.group(1), m.group(2) | |
| 120 | + slug = urllib.parse.unquote(slug_raw) | |
| 121 | + if self.quebec_only and not _QC_SLUG_RE.search(slug + "/"): | |
| 122 | + continue | |
| 123 | + city = slug.split("-")[0] | |
| 124 | + postal = "" | |
| 125 | + m_cp = re.search(r"([GHJ]\d[A-Z])[- ]?(\d[A-Z]\d)", slug, re.I) | |
| 126 | + if m_cp: | |
| 127 | + postal = f"{m_cp.group(1)}{m_cp.group(2)}".upper() | |
| 128 | + job = JobPosting( | |
| 129 | + source=self.source_id, external_id=eid, url=loc, | |
| 130 | + employer=self.EMPLOYER, | |
| 131 | + title=" ".join(slug.split("-")[1:-3]) or slug, | |
| 132 | + city=city, postal_code=postal, | |
| 133 | + date_posted=lastmod or None, | |
| 134 | + ats=self.ats, | |
| 135 | + ) | |
| 136 | + if details_used < MAX_DETAILS: | |
| 137 | + fresh = [False] | |
| 138 | + | |
| 139 | + def _fn(u=loc, fresh=fresh): | |
| 140 | + fresh[0] = True | |
| 141 | + return self._fetch_detail(u) | |
| 142 | + | |
| 143 | + d = self.detail(eid, lastmod or eid, _fn) | |
| 144 | + if fresh[0]: | |
| 145 | + details_used += 1 | |
| 146 | + if d.get("description"): | |
| 147 | + job.description = d["description"] | |
| 148 | + if d.get("datePosted"): | |
| 149 | + job.date_posted = d["datePosted"] | |
| 150 | + if d.get("addressLocality"): | |
| 151 | + job.city = d["addressLocality"] | |
| 152 | + if d.get("employmentType"): | |
| 153 | + job.details["employment_label"] = d["employmentType"] | |
| 154 | + out.append(job) | |
| 155 | + return out | |
| 156 | + | |
| 157 | + def fetch(self) -> list[JobPosting]: | |
| 158 | + xml = self.get(f"{self.BASE}/sitemap.xml").text | |
| 159 | + if "<rss" in xml[:300].lower(): | |
| 160 | + return self._fetch_rss(xml) | |
| 161 | + return self._fetch_urlset(xml) | |
added
jobka/connectors/taleo.py
+185 −0
@@ -0,0 +1,185 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/taleo.py | |
| 6 | +# Rôle : Classe de plateforme Oracle Taleo Enterprise (careersection REST) | |
| 7 | +# — un employeur = une sous-classe (HOST, SECTION, EMPLOYER) | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme Oracle Taleo Enterprise. | |
| 11 | + | |
| 12 | +Les sections carrières Taleo exposent une API REST publique (celle que le | |
| 13 | +frontal jobsearch.ftl consomme lui-même) : | |
| 14 | +- amorce : GET https://<host>.taleo.net/careersection/<section>/jobsearch.ftl | |
| 15 | + (cookies de session + numéro de portail « portal=NNN » dans le HTML) | |
| 16 | +- liste : POST https://<host>.taleo.net/careersection/rest/jobboard/searchjobs | |
| 17 | + ?lang=fr&portal=<NNN> body JSON {pageNo, sorting…} | |
| 18 | + -> {requisitionList:[{jobId, contestNo, column:[...], | |
| 19 | + locationsColumns:[i]}], pagingData:{totalCount, pageSize}} | |
| 20 | +- détail : la page jobdetail.ftl?job=<contestNo> (description HTML rendue | |
| 21 | + côté serveur) — visitée avec cache BD et budget. | |
| 22 | + | |
| 23 | +Le contenu des colonnes dépend de la configuration du portail : on repère la | |
| 24 | +colonne des lieux via `locationsColumns` et on suppose la première colonne | |
| 25 | +comme titre (constaté sur tous les portails testés). | |
| 26 | +""" | |
| 27 | +from __future__ import annotations | |
| 28 | + | |
| 29 | +import hashlib | |
| 30 | +import json | |
| 31 | +import os | |
| 32 | +import re | |
| 33 | + | |
| 34 | +from ..schema import JobPosting, clean_html, is_quebec_location | |
| 35 | +from .base import BaseConnector | |
| 36 | + | |
| 37 | +MAX_DETAILS = int(os.environ.get("JOBKA_TALEO_DETAIL_LIMIT", "120")) | |
| 38 | + | |
| 39 | +_DESC_RE = re.compile( | |
| 40 | + r'requisitionDescriptionInterface[^>]*>', re.I) | |
| 41 | + | |
| 42 | + | |
| 43 | +class TaleoConnector(BaseConnector): | |
| 44 | + """Base Taleo — sous-classes : définir source_id, EMPLOYER, HOST, SECTION | |
| 45 | + (et PORTAL si le numéro n'apparaît pas dans le HTML d'amorce).""" | |
| 46 | + | |
| 47 | + ats = "taleo" | |
| 48 | + request_delay = 1.0 | |
| 49 | + | |
| 50 | + EMPLOYER = "" | |
| 51 | + HOST = "" # <host>.taleo.net | |
| 52 | + SECTION = "2" # /careersection/<section>/ | |
| 53 | + PORTAL = "" # numéro de portail (sinon extrait du HTML) | |
| 54 | + LANG = "fr" | |
| 55 | + quebec_only = True | |
| 56 | + max_pages = 40 | |
| 57 | + | |
| 58 | + @property | |
| 59 | + def _base(self) -> str: | |
| 60 | + return f"https://{self.HOST}.taleo.net/careersection" | |
| 61 | + | |
| 62 | + def _careers_url(self) -> str: | |
| 63 | + return f"{self._base}/{self.SECTION}/jobsearch.ftl?lang={self.LANG}" | |
| 64 | + | |
| 65 | + def _bootstrap(self) -> str: | |
| 66 | + """Cookies de session + numéro de portail.""" | |
| 67 | + html = self.get(self._careers_url()).text | |
| 68 | + if self.PORTAL: | |
| 69 | + return str(self.PORTAL) | |
| 70 | + m = re.search(r"portal=(\d+)", html) | |
| 71 | + if not m: | |
| 72 | + raise RuntimeError(f"taleo {self.HOST}: numéro de portail introuvable") | |
| 73 | + return m.group(1) | |
| 74 | + | |
| 75 | + @staticmethod | |
| 76 | + def _parse_locations(raw: str) -> list[str]: | |
| 77 | + """La colonne des lieux est un JSON sérialisé : '["Québec-Malartic"]'.""" | |
| 78 | + if not raw: | |
| 79 | + return [] | |
| 80 | + try: | |
| 81 | + val = json.loads(raw) | |
| 82 | + if isinstance(val, list): | |
| 83 | + return [str(v) for v in val] | |
| 84 | + except ValueError: | |
| 85 | + pass | |
| 86 | + return [raw] | |
| 87 | + | |
| 88 | + def _keep(self, locations: list[str]) -> bool: | |
| 89 | + if not self.quebec_only: | |
| 90 | + return True | |
| 91 | + return any(is_quebec_location(l.replace("-", " ")) for l in locations) | |
| 92 | + | |
| 93 | + def _fetch_detail(self, contest_no: str) -> dict: | |
| 94 | + html = self.get(f"{self._base}/{self.SECTION}/jobdetail.ftl", | |
| 95 | + params={"job": contest_no, "lang": self.LANG}).text | |
| 96 | + # description : contenu des blocs « requisitionDescriptionInterface » | |
| 97 | + # (rendus côté serveur), sinon la zone principale de la page | |
| 98 | + chunks = re.findall( | |
| 99 | + r'<(?:span|div)[^>]*id="requisitionDescriptionInterface[^"]*"[^>]*>' | |
| 100 | + r"(.*?)</(?:span|div)>", html, re.S) | |
| 101 | + text = clean_html("\n".join(c for c in chunks if len(c) > 40)) | |
| 102 | + if not text: | |
| 103 | + m = re.search(r'<div[^>]*class="[^"]*mastercontentpanel[^"]*"[^>]*>' | |
| 104 | + r"(.*?)<!--", html, re.S) | |
| 105 | + text = clean_html(m.group(1)) if m else "" | |
| 106 | + return {"description": text} | |
| 107 | + | |
| 108 | + def fetch(self) -> list[JobPosting]: | |
| 109 | + portal = self._bootstrap() | |
| 110 | + url = (f"{self._base}/rest/jobboard/searchjobs" | |
| 111 | + f"?lang={self.LANG}&portal={portal}") | |
| 112 | + body_base = { | |
| 113 | + "multilineEnabled": False, | |
| 114 | + "sortingSelection": {"sortBySelectionParam": "3", | |
| 115 | + "ascendingSortingOrder": "false"}, | |
| 116 | + "fieldData": {"fields": {"KEYWORD": "", "LOCATION": ""}, | |
| 117 | + "valid": True}, | |
| 118 | + "filterSelectionParam": {"searchFilterSelections": [ | |
| 119 | + {"id": "POSTING_DATE", "selectedValues": []}, | |
| 120 | + {"id": "LOCATION", "selectedValues": []}]}, | |
| 121 | + "advancedSearchFiltersSelectionParam": | |
| 122 | + {"searchFilterSelections": []}, | |
| 123 | + } | |
| 124 | + out: list[JobPosting] = [] | |
| 125 | + details_used = 0 | |
| 126 | + page = 1 | |
| 127 | + seen: set[str] = set() | |
| 128 | + while page <= self.max_pages: | |
| 129 | + data = self.post(url, json={**body_base, "pageNo": page}, | |
| 130 | + headers={"tz": "GMT-04:00"}).json() | |
| 131 | + reqs = data.get("requisitionList") or [] | |
| 132 | + if not reqs and page == 1 and self.LANG != "en": | |
| 133 | + # section carrières anglophone seulement : rebasculer en « en » | |
| 134 | + self.LANG = "en" | |
| 135 | + portal = self._bootstrap() | |
| 136 | + url = (f"{self._base}/rest/jobboard/searchjobs" | |
| 137 | + f"?lang={self.LANG}&portal={portal}") | |
| 138 | + continue | |
| 139 | + if not reqs: | |
| 140 | + break | |
| 141 | + for r in reqs: | |
| 142 | + cols = r.get("column") or [] | |
| 143 | + loc_idx = (r.get("locationsColumns") or [1])[0] | |
| 144 | + title = str(cols[0]) if cols else "" | |
| 145 | + locations = self._parse_locations( | |
| 146 | + str(cols[loc_idx])) if len(cols) > loc_idx else [] | |
| 147 | + date_raw = str(cols[-1]) if len(cols) > 2 else "" | |
| 148 | + contest = str(r.get("contestNo") or r.get("jobId") or "") | |
| 149 | + if not contest or contest in seen or not title: | |
| 150 | + continue | |
| 151 | + seen.add(contest) | |
| 152 | + if not self._keep(locations): | |
| 153 | + continue | |
| 154 | + job = JobPosting( | |
| 155 | + source=self.source_id, external_id=contest, | |
| 156 | + url=(f"{self._base}/{self.SECTION}/jobdetail.ftl" | |
| 157 | + f"?job={contest}&lang={self.LANG}"), | |
| 158 | + employer=self.EMPLOYER, | |
| 159 | + title=title, | |
| 160 | + location_label=locations[0].replace("-", ", ") | |
| 161 | + if locations else "", | |
| 162 | + date_posted=date_raw or None, | |
| 163 | + ats=self.ats, | |
| 164 | + ) | |
| 165 | + key = hashlib.sha1( | |
| 166 | + f"{title}|{date_raw}".encode()).hexdigest()[:12] | |
| 167 | + if details_used < MAX_DETAILS: | |
| 168 | + fresh = [False] | |
| 169 | + | |
| 170 | + def _fn(c=contest, fresh=fresh): | |
| 171 | + fresh[0] = True | |
| 172 | + return self._fetch_detail(c) | |
| 173 | + | |
| 174 | + d = self.detail(contest, key, _fn) | |
| 175 | + if fresh[0]: | |
| 176 | + details_used += 1 | |
| 177 | + job.description = d.get("description", "") | |
| 178 | + out.append(job) | |
| 179 | + paging = data.get("pagingData") or {} | |
| 180 | + total = int(paging.get("totalCount") or 0) | |
| 181 | + size = int(paging.get("pageSize") or len(reqs) or 25) | |
| 182 | + if total and page * size >= total: | |
| 183 | + break | |
| 184 | + page += 1 | |
| 185 | + return out | |
added
jobka/connectors/ultipro.py
+162 −0
@@ -0,0 +1,162 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/ultipro.py | |
| 6 | +# Rôle : Classe de plateforme UKG Pro Recruiting (UltiPro) — API JSON | |
| 7 | +# publique LoadSearchResults, un employeur = une sous-classe | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +"""Plateforme UKG Pro Recruiting (recruiting.ultipro.com). | |
| 11 | + | |
| 12 | +API JSON publique (celle du frontal) : | |
| 13 | +- liste : POST https://recruiting.ultipro.com/<ORG>/JobBoard/<BOARD>/ | |
| 14 | + JobBoardView/LoadSearchResults | |
| 15 | + body {opportunitySearch:{Top, Skip, OrderBy…}} | |
| 16 | + -> {opportunities:[{Id, Title, RequisitionNumber, BriefDescription, | |
| 17 | + FullTime, JobCategoryName, PostedDate, Locations:[{Address}]}], | |
| 18 | + totalCount} | |
| 19 | +- détail : POST .../JobBoardView/LoadOpportunity {opportunityId} -> Description | |
| 20 | + complète (HTML). Visité avec cache BD + budget. | |
| 21 | +""" | |
| 22 | +from __future__ import annotations | |
| 23 | + | |
| 24 | +import os | |
| 25 | +import re | |
| 26 | + | |
| 27 | +from ..schema import JobPosting, clean_html, is_quebec_location | |
| 28 | +from .base import BaseConnector | |
| 29 | + | |
| 30 | +PAGE_SIZE = 50 | |
| 31 | +MAX_DETAILS = int(os.environ.get("JOBKA_ULTIPRO_DETAIL_LIMIT", "80")) | |
| 32 | + | |
| 33 | + | |
| 34 | +class UltiProConnector(BaseConnector): | |
| 35 | + """Base UltiPro — sous-classes : définir source_id, EMPLOYER, ORG, BOARD.""" | |
| 36 | + | |
| 37 | + ats = "ultipro" | |
| 38 | + request_delay = 0.8 | |
| 39 | + | |
| 40 | + EMPLOYER = "" | |
| 41 | + ORG = "" # ex. MES1005MESR | |
| 42 | + BOARD = "" # GUID du JobBoard | |
| 43 | + quebec_only = True | |
| 44 | + max_pages = 20 | |
| 45 | + | |
| 46 | + @property | |
| 47 | + def _board_url(self) -> str: | |
| 48 | + return f"https://recruiting.ultipro.com/{self.ORG}/JobBoard/{self.BOARD}" | |
| 49 | + | |
| 50 | + @staticmethod | |
| 51 | + def _loc_fields(loc: dict) -> tuple[str, str, str, str]: | |
| 52 | + """(ville, province, code postal, libellé) d'un objet Location.""" | |
| 53 | + addr = loc.get("Address") or {} | |
| 54 | + city = addr.get("City") or "" | |
| 55 | + state = ((addr.get("State") or {}).get("Code") | |
| 56 | + if isinstance(addr.get("State"), dict) | |
| 57 | + else addr.get("State")) or "" | |
| 58 | + postal = addr.get("PostalCode") or "" | |
| 59 | + label = loc.get("LocalizedDescription") or city | |
| 60 | + return city, str(state), postal, label | |
| 61 | + | |
| 62 | + def _keep(self, locations: list[dict]) -> bool: | |
| 63 | + if not self.quebec_only: | |
| 64 | + return True | |
| 65 | + for loc in locations: | |
| 66 | + city, state, postal, label = self._loc_fields(loc) | |
| 67 | + if state.upper() == "QC" or re.match(r"^[GHJ]\d[A-Z]", postal.upper()): | |
| 68 | + return True | |
| 69 | + if is_quebec_location(f"{label} {city}"): | |
| 70 | + return True | |
| 71 | + return False | |
| 72 | + | |
| 73 | + _DESC_RE = re.compile(r'"Description"\s*:\s*"((?:[^"\\]|\\.)*)"') | |
| 74 | + | |
| 75 | + def _fetch_detail(self, opp_id: str) -> dict: | |
| 76 | + """La page OpportunityDetail embarque l'objet JSON de l'offre — | |
| 77 | + on en extrait la description complète (chaîne JSON échappée).""" | |
| 78 | + html = self.get(f"{self._board_url}/OpportunityDetail", | |
| 79 | + params={"opportunityId": opp_id}).text | |
| 80 | + best = "" | |
| 81 | + for m in self._DESC_RE.finditer(html): | |
| 82 | + try: | |
| 83 | + import json as _json | |
| 84 | + val = _json.loads(f'"{m.group(1)}"') | |
| 85 | + except ValueError: | |
| 86 | + continue | |
| 87 | + if len(val) > len(best): | |
| 88 | + best = val | |
| 89 | + return {"description": clean_html(best)} | |
| 90 | + | |
| 91 | + def fetch(self) -> list[JobPosting]: | |
| 92 | + url = f"{self._board_url}/JobBoardView/LoadSearchResults" | |
| 93 | + out: list[JobPosting] = [] | |
| 94 | + details_used = 0 | |
| 95 | + skip = 0 | |
| 96 | + for _ in range(self.max_pages): | |
| 97 | + body = { | |
| 98 | + "opportunitySearch": { | |
| 99 | + "Top": PAGE_SIZE, "Skip": skip, "QueryString": "", | |
| 100 | + "OrderBy": [{"Value": "postedDateDesc", | |
| 101 | + "PropertyName": "PostedDate", | |
| 102 | + "Ascending": False}], | |
| 103 | + "Filters": [], | |
| 104 | + }, | |
| 105 | + "matchCriteria": {"PreferredJobs": [], "Educations": [], | |
| 106 | + "LicenseAndCertifications": [], "Skills": [], | |
| 107 | + "hasNoLicenses": False, "SkippedSkills": []}, | |
| 108 | + } | |
| 109 | + data = self.post(url, json=body, | |
| 110 | + headers={"Accept": "application/json"}).json() | |
| 111 | + opps = data.get("opportunities") or [] | |
| 112 | + if not opps: | |
| 113 | + break | |
| 114 | + for o in opps: | |
| 115 | + locations = o.get("Locations") or [] | |
| 116 | + if not self._keep(locations): | |
| 117 | + continue | |
| 118 | + oid = str(o.get("Id") or "") | |
| 119 | + if not oid: | |
| 120 | + continue | |
| 121 | + city = label = postal = "" | |
| 122 | + for loc in locations: | |
| 123 | + c, s, p, lbl = self._loc_fields(loc) | |
| 124 | + if s.upper() == "QC" or is_quebec_location(f"{lbl} {c}"): | |
| 125 | + city, postal, label = c or lbl, p, lbl | |
| 126 | + break | |
| 127 | + job = JobPosting( | |
| 128 | + source=self.source_id, external_id=oid, | |
| 129 | + url=f"{self._board_url}/OpportunityDetail?opportunityId={oid}", | |
| 130 | + employer=self.EMPLOYER, | |
| 131 | + title=o.get("Title") or "", | |
| 132 | + city=city, postal_code=postal, location_label=label, | |
| 133 | + date_posted=o.get("PostedDate") or None, | |
| 134 | + ats=self.ats, | |
| 135 | + ) | |
| 136 | + job.details["requisition"] = o.get("RequisitionNumber") or "" | |
| 137 | + if o.get("JobCategoryName"): | |
| 138 | + job.details["team"] = o["JobCategoryName"] | |
| 139 | + if o.get("FullTime") is True: | |
| 140 | + job.details["employment_label"] = "Temps plein" | |
| 141 | + elif o.get("FullTime") is False: | |
| 142 | + job.details["employment_label"] = "Temps partiel" | |
| 143 | + key = (o.get("PostedDate") or "") + (o.get("Title") or "")[:40] | |
| 144 | + if details_used < MAX_DETAILS: | |
| 145 | + fresh = [False] | |
| 146 | + | |
| 147 | + def _fn(i=oid, fresh=fresh): | |
| 148 | + fresh[0] = True | |
| 149 | + return self._fetch_detail(i) | |
| 150 | + | |
| 151 | + d = self.detail(oid, key, _fn) | |
| 152 | + if fresh[0]: | |
| 153 | + details_used += 1 | |
| 154 | + job.description = d.get("description", "") | |
| 155 | + if not job.description: | |
| 156 | + job.description = clean_html(o.get("BriefDescription") or "") | |
| 157 | + out.append(job) | |
| 158 | + skip += PAGE_SIZE | |
| 159 | + total = int(data.get("totalCount") or 0) | |
| 160 | + if total and skip >= total: | |
| 161 | + break | |
| 162 | + return out | |
added
jobka/connectors/ville_gatineau.py
+19 −0
@@ -0,0 +1,19 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/ville_gatineau.py | |
| 6 | +# Rôle : Connecteur Ville de Gatineau — njoyn (base='https://gatineau.njoyn.com', cl='CL2') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .njoyn import NjoynConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class VilleGatineauConnector(NjoynConnector): | |
| 14 | + source_id = 'ville_gatineau' | |
| 15 | + EMPLOYER = 'Ville de Gatineau' | |
| 16 | + BASE = 'https://gatineau.njoyn.com' | |
| 17 | + CL = 'CL2' | |
| 18 | + CLID = '27082' | |
| 19 | + USE_SCRAPFLY = True | |
added
jobka/connectors/ville_terrebonne.py
+19 −0
@@ -0,0 +1,19 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/ville_terrebonne.py | |
| 6 | +# Rôle : Connecteur Ville de Terrebonne — njoyn (base='https://clients.njoyn.com', cl='cl4') | |
| 7 | +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18] | |
| 8 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 9 | +# ============================================================================= | |
| 10 | +from .njoyn import NjoynConnector | |
| 11 | + | |
| 12 | + | |
| 13 | +class VilleTerrebonneConnector(NjoynConnector): | |
| 14 | + source_id = 'ville_terrebonne' | |
| 15 | + EMPLOYER = 'Ville de Terrebonne' | |
| 16 | + BASE = 'https://clients.njoyn.com' | |
| 17 | + CL = 'cl4' | |
| 18 | + CLID = '71764' | |
| 19 | + USE_SCRAPFLY = True | |
added
jobka/connectors/workland.py
+90 −0
@@ -0,0 +1,90 @@ | ||
| 1 | +# ============================================================================= | |
| 2 | +# Job·Ka — Groupe KA | |
| 3 | +# Auteur : Simon-Pierre Boucher | |
| 4 | +# Contact : contact@spboucher.ai | |
| 5 | +# Fichier : jobka/connectors/workland.py | |
| 6 | +# Rôle : Classe de plateforme Workland / Atlas (ATS québécois utilisé par | |
| 7 | +# les centres de services scolaires, PME…) — liste rendue côté | |
| 8 | +# serveur sur la page carrières de l'employeur | |
| 9 | +# Créé : 2026-08-18 Modifié : 2026-08-18 | |
| 10 | +# ============================================================================= | |
| 11 | +"""Plateforme Workland (atlas.workland.com). | |
| 12 | + | |
| 13 | +Les pages atlas.workland.com sont une SPA Angular sans données côté serveur | |
| 14 | +(le bloc JSON-LD reste vide même après rendu) : on s'appuie donc sur la page | |
| 15 | +carrières de l'employeur (LIST_URL), rendue côté serveur, qui liste les liens | |
| 16 | +https://atlas.workland.com/work/<id>/<slug> — typiquement dans un tableau | |
| 17 | +« titre | date limite | lien » (ex. CSSDM). Le titre et la date limite | |
| 18 | +proviennent de la ligne du tableau ; à défaut, du slug. | |
| 19 | +""" | |
| 20 | +from __future__ import annotations | |
| 21 | + | |
| 22 | +import html as _html | |
| 23 | +import re | |
| 24 | + | |
| 25 | +from ..schema import JobPosting | |
| 26 | +from .base import BaseConnector | |
| 27 | + | |
| 28 | +_WORK_RE = re.compile( | |
| 29 | + r"https?://atlas\.workland\.com/work/(\d+)/([a-z0-9-]+)", re.I) | |
| 30 | +_ROW_RE = re.compile(r"<tr[^>]*>(.*?)</tr>", re.S | re.I) | |
| 31 | +_TD_RE = re.compile(r"<td[^>]*>(.*?)</td>", re.S | re.I) | |
| 32 | +_DATE_RE = re.compile(r"\d{1,2}(?:er)?\s+[a-zû]{3,9}\.?\s+\d{4}|\d{4}-\d{2}-\d{2}", | |
| 33 | + re.I) | |
| 34 | + | |
| 35 | + | |
| 36 | +def _txt(s: str) -> str: | |
| 37 | + return re.sub(r"\s+", " ", | |
| 38 | + _html.unescape(re.sub(r"<[^>]+>", " ", s or ""))).strip() | |
| 39 | + | |
| 40 | + | |
| 41 | +class WorklandConnector(BaseConnector): | |
| 42 | + """Base Workland — sous-classes : définir source_id, EMPLOYER, LIST_URL | |
| 43 | + (page carrières de l'employeur qui liste les liens atlas).""" | |
| 44 | + | |
| 45 | + ats = "workland" | |
| 46 | + request_delay = 1.0 | |
| 47 | + | |
| 48 | + EMPLOYER = "" | |
| 49 | + LIST_URL = "" | |
| 50 | + DEFAULT_CITY = "" # ville par défaut (ex. « Montréal » pour le CSSDM) | |
| 51 | + | |
| 52 | + def fetch(self) -> list[JobPosting]: | |
| 53 | + page = self.get(self.LIST_URL).text | |
| 54 | + out: list[JobPosting] = [] | |
| 55 | + seen: set[str] = set() | |
| 56 | + | |
| 57 | + # 1) lignes de tableau « titre | date limite | lien atlas » | |
| 58 | + for row in _ROW_RE.findall(page): | |
| 59 | + m = _WORK_RE.search(row) | |
| 60 | + if not m or m.group(1) in seen: | |
| 61 | + continue | |
| 62 | + tds = [_txt(td) for td in _TD_RE.findall(row)] | |
| 63 | + title = next((t for t in tds if t and not _DATE_RE.fullmatch(t)), "") | |
| 64 | + deadline = next((t for t in tds if t and _DATE_RE.fullmatch(t)), None) | |
| 65 | + if not title: | |
| 66 | + title = m.group(2).replace("-", " ").capitalize() | |
| 67 | + seen.add(m.group(1)) | |
| 68 | + out.append(JobPosting( | |
| 69 | + source=self.source_id, external_id=m.group(1), | |
| 70 | + url=f"https://atlas.workland.com/work/{m.group(1)}/{m.group(2)}", | |
| 71 | + employer=self.EMPLOYER, title=title, | |
| 72 | + city=self.DEFAULT_CITY, | |
| 73 | + date_deadline=deadline, | |
| 74 | + ats=self.ats, | |
| 75 | + )) | |
| 76 | + | |
| 77 | + # 2) liens atlas hors tableau (autres habillages) | |
| 78 | + for wid, slug in _WORK_RE.findall(page): | |
| 79 | + if wid in seen: | |
| 80 | + continue | |
| 81 | + seen.add(wid) | |
| 82 | + out.append(JobPosting( | |
| 83 | + source=self.source_id, external_id=wid, | |
| 84 | + url=f"https://atlas.workland.com/work/{wid}/{slug}", | |
| 85 | + employer=self.EMPLOYER, | |
| 86 | + title=slug.replace("-", " ").capitalize(), | |
| 87 | + city=self.DEFAULT_CITY, | |
| 88 | + ats=self.ats, | |
| 89 | + )) | |
| 90 | + return out | |
modified
scripts/gen_connectors.py
+39 −0
@@ -54,6 +54,20 @@ ORG_ATTR = {"lever": "ORG", "greenhouse": "BOARD", | ||
| 54 | 54 | "smartrecruiters": "COMPANY", "ashby": "ORG", "workable": "ORG", |
| 55 | 55 | "recruitee": "ORG", "breezy": "ORG", "bamboohr": "ORG"} |
| 56 | 56 | |
| 57 | +# ATS à attributs multiples (secteur public/parapublic et grands employeurs) : | |
| 58 | +# feed[clé] -> attribut de classe. Voir data/feeds-public.json. | |
| 59 | +EXTRA_ATTRS = { | |
| 60 | + "taleo": [("host", "HOST"), ("section", "SECTION"), ("portal", "PORTAL")], | |
| 61 | + "njoyn": [("base", "BASE"), ("cl", "CL"), ("clid", "CLID"), | |
| 62 | + ("use_scrapfly", "USE_SCRAPFLY")], | |
| 63 | + "ultipro": [("org", "ORG"), ("board", "BOARD")], | |
| 64 | + "icims": [("sub", "SUB")], | |
| 65 | + "successfactors": [("base", "BASE")], | |
| 66 | + "adp": [("cid", "CID"), ("ccid", "CCID")], | |
| 67 | + "digitalrecruiters": [("base", "BASE")], | |
| 68 | + "workland": [("list_url", "LIST_URL")], | |
| 69 | +} | |
| 70 | + | |
| 57 | 71 | CAREERS_URL = { |
| 58 | 72 | "workday": "https://{tenant}.{host}.myworkdayjobs.com/{site}", |
| 59 | 73 | "lever": "https://jobs.lever.co/{org}", |
@@ -64,8 +78,28 @@ CAREERS_URL = { | ||
| 64 | 78 | "recruitee": "https://{org}.recruitee.com", |
| 65 | 79 | "breezy": "https://{org}.breezy.hr", |
| 66 | 80 | "bamboohr": "https://{org}.bamboohr.com/careers", |
| 81 | + "taleo": "https://{host}.taleo.net/careersection/{section}/jobsearch.ftl?lang=fr", | |
| 82 | + "njoyn": "{base}/{cl}/xweb/xweb.asp?clid={clid}&page=joblisting&lang=2", | |
| 83 | + "ultipro": "https://recruiting.ultipro.com/{org}/JobBoard/{board}", | |
| 84 | + "icims": "https://{sub}.icims.com/jobs/search?ss=1", | |
| 85 | + "successfactors": "{base}", | |
| 86 | + "adp": ("https://workforcenow.adp.com/mascsr/default/mdf/recruitment/" | |
| 87 | + "recruitment.html?cid={cid}&ccId={ccid}&lang=fr_CA"), | |
| 88 | + "digitalrecruiters": "{base}", | |
| 89 | + "workland": "{list_url}", | |
| 67 | 90 | } |
| 68 | 91 | |
| 92 | +PLATFORMS.update({ | |
| 93 | + "taleo": ("taleo", "TaleoConnector"), | |
| 94 | + "njoyn": ("njoyn", "NjoynConnector"), | |
| 95 | + "ultipro": ("ultipro", "UltiProConnector"), | |
| 96 | + "icims": ("icims", "ICIMSConnector"), | |
| 97 | + "successfactors": ("successfactors", "SuccessFactorsConnector"), | |
| 98 | + "adp": ("adp", "ADPWorkforceNowConnector"), | |
| 99 | + "digitalrecruiters": ("digitalrecruiters", "DigitalRecruitersConnector"), | |
| 100 | + "workland": ("workland", "WorklandConnector"), | |
| 101 | +}) | |
| 102 | + | |
| 69 | 103 | |
| 70 | 104 | def slugify(s: str) -> str: |
| 71 | 105 | s = unicodedata.normalize("NFD", s) |
@@ -110,6 +144,11 @@ def gen_one(feed: dict) -> tuple[str, str] | None: | ||
| 110 | 144 | body = (f" TENANT = {feed['tenant']!r}\n" |
| 111 | 145 | f" HOST = {feed['host']!r}\n" |
| 112 | 146 | f" SITE = {feed['site']!r}\n") |
| 147 | + elif ats in EXTRA_ATTRS: | |
| 148 | + pairs = [(k, a) for k, a in EXTRA_ATTRS[ats] | |
| 149 | + if feed.get(k) not in (None, "")] | |
| 150 | + detail = ", ".join(f"{a.lower()}={feed[k]!r}" for k, a in pairs[:2]) | |
| 151 | + body = "".join(f" {a} = {feed[k]!r}\n" for k, a in pairs) | |
| 113 | 152 | else: |
| 114 | 153 | attr = ORG_ATTR[ats] |
| 115 | 154 | detail = f"org « {feed['org']} »" |
| 116 | 155 | |