SPB Git forge

spb/job-ka

Public
229commits 1branches 0releases
38.1 MBsize
maindefault branch
3 h agolast push
HTML 82.1% Python 14.6% TypeScript 1.9% CSS 1% JavaScript 0.5%

8 nouveaux ATS (Taleo, Njoyn, UltiPro, iCIMS, SuccessFactors, ADP WFN, Digital Recruiters, Workland) + 23 employeurs publics/parapublics et grands employeurs QC

- Secteur public : Santé Québec (emplois.sante.quebec), Emplois Santé Montréal
  (njoyn), CIUSSS de l Est-de-l Île-de-Montréal (taleo), CSSDM (workland),
  villes de Gatineau et Terrebonne, STL, RTC (njoyn via Scrapfly anti-Radware)
- Grands employeurs : Bombardier, Cascades, Domtar (SuccessFactors sitemap
  RSS/urlset + microdonnées), Agnico Eagle, Bell Textron, Bayshore (Taleo REST
  + repli anglophone), Messer Canada (UltiPro), Lassonde, Mitsubishi HC
  Capital, MSI (ADP WFN + repli en_US), CAPREIT, McCarthy Tétrault, Omni,
  EDF Renouvelables (iCIMS + JSON-LD)
- gen_connectors : EXTRA_ATTRS multi-attributs pour les nouveaux ATS
- jobka/connectors/_jsonld.py : extraction schema.org/JobPosting partagée

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Simon-Pierre Boucher committed 1 mo ago (Aug 18, 2026) parent 5a66f48

34 changed files +2,093 −0

added data/feeds-public.json +368 −0
@@ -0,0 +1,368 @@
1 +[
2 + {
3 + "ats": "taleo",
4 + "source_id": "agnico_eagle",
5 + "employer": "Agnico Eagle Mines",
6 + "host": "agnicoeagle",
7 + "section": "2",
8 + "url": "https://www.agnicoeagle.com",
9 + "sectors": [
10 + "Mines",
11 + "Ingénierie",
12 + "Métiers spécialisés"
13 + ],
14 + "cities": [
15 + "Abitibi-Témiscamingue",
16 + "Malartic",
17 + "Rouyn-Noranda"
18 + ]
19 + },
20 + {
21 + "ats": "taleo",
22 + "source_id": "ciusss_est_montreal",
23 + "employer": "CIUSSS de l'Est-de-l'Île-de-Montréal",
24 + "host": "ciusssemtl",
25 + "section": "cemtl",
26 + "url": "https://ciusss-estmtl.gouv.qc.ca",
27 + "sectors": [
28 + "Santé",
29 + "Services sociaux",
30 + "Secteur public"
31 + ],
32 + "cities": [
33 + "Montréal"
34 + ]
35 + },
36 + {
37 + "ats": "taleo",
38 + "source_id": "bayshore",
39 + "employer": "Bayshore HealthCare",
40 + "host": "bayshore",
41 + "section": "bs_ex",
42 + "url": "https://www.bayshore.ca",
43 + "sectors": [
44 + "Santé",
45 + "Soins à domicile"
46 + ],
47 + "cities": [
48 + "Montréal",
49 + "Québec"
50 + ]
51 + },
52 + {
53 + "ats": "taleo",
54 + "source_id": "bell_textron",
55 + "employer": "Bell Textron Canada",
56 + "host": "textron",
57 + "section": "bell",
58 + "url": "https://www.bellflight.com",
59 + "sectors": [
60 + "Aéronautique",
61 + "Fabrication"
62 + ],
63 + "cities": [
64 + "Mirabel"
65 + ]
66 + },
67 + {
68 + "ats": "njoyn",
69 + "source_id": "sante_montreal",
70 + "employer": "Santé Québec — Montréal",
71 + "base": "https://emplois.santemontreal.qc.ca",
72 + "cl": "CL3",
73 + "clid": "54327",
74 + "url": "https://santemontreal.ca",
75 + "sectors": [
76 + "Santé",
77 + "Services sociaux",
78 + "Secteur public"
79 + ],
80 + "cities": [
81 + "Montréal"
82 + ]
83 + },
84 + {
85 + "ats": "njoyn",
86 + "source_id": "ville_gatineau",
87 + "employer": "Ville de Gatineau",
88 + "base": "https://gatineau.njoyn.com",
89 + "cl": "CL2",
90 + "clid": "27082",
91 + "use_scrapfly": true,
92 + "url": "https://www.gatineau.ca",
93 + "sectors": [
94 + "Municipal",
95 + "Secteur public"
96 + ],
97 + "cities": [
98 + "Gatineau"
99 + ]
100 + },
101 + {
102 + "ats": "njoyn",
103 + "source_id": "ville_terrebonne",
104 + "employer": "Ville de Terrebonne",
105 + "base": "https://clients.njoyn.com",
106 + "cl": "cl4",
107 + "clid": "71764",
108 + "use_scrapfly": true,
109 + "url": "https://www.ville.terrebonne.qc.ca",
110 + "sectors": [
111 + "Municipal",
112 + "Secteur public"
113 + ],
114 + "cities": [
115 + "Terrebonne"
116 + ]
117 + },
118 + {
119 + "ats": "njoyn",
120 + "source_id": "stl_laval",
121 + "employer": "Société de transport de Laval",
122 + "base": "https://lavaltransit.njoyn.com",
123 + "cl": "CL2",
124 + "clid": "60406",
125 + "use_scrapfly": true,
126 + "url": "https://www.stlaval.ca",
127 + "sectors": [
128 + "Transport collectif",
129 + "Secteur public"
130 + ],
131 + "cities": [
132 + "Laval"
133 + ]
134 + },
135 + {
136 + "ats": "njoyn",
137 + "source_id": "rtc_quebec",
138 + "employer": "Réseau de transport de la Capitale",
139 + "base": "https://rtc.njoyn.com",
140 + "cl": "CGI",
141 + "clid": "23009",
142 + "use_scrapfly": true,
143 + "url": "https://www.rtcquebec.ca",
144 + "sectors": [
145 + "Transport collectif",
146 + "Secteur public"
147 + ],
148 + "cities": [
149 + "Québec"
150 + ]
151 + },
152 + {
153 + "ats": "ultipro",
154 + "source_id": "messer_canada",
155 + "employer": "Messer Canada",
156 + "org": "MES1005MESR",
157 + "board": "dbd63926-d756-486f-a254-b2a6ede1d26e",
158 + "url": "https://www.messer-ca.com",
159 + "sectors": [
160 + "Gaz industriels",
161 + "Industrie"
162 + ],
163 + "cities": [
164 + "Montréal",
165 + "Québec",
166 + "Saint-Georges"
167 + ]
168 + },
169 + {
170 + "ats": "icims",
171 + "source_id": "capreit",
172 + "employer": "CAPREIT",
173 + "sub": "careers-capreit",
174 + "url": "https://www.capreit.ca",
175 + "sectors": [
176 + "Immobilier",
177 + "Gestion immobilière"
178 + ],
179 + "cities": [
180 + "Montréal"
181 + ]
182 + },
183 + {
184 + "ats": "icims",
185 + "source_id": "mccarthy_tetrault",
186 + "employer": "McCarthy Tétrault",
187 + "sub": "careers-mccarthyca",
188 + "url": "https://www.mccarthy.ca",
189 + "sectors": [
190 + "Juridique",
191 + "Services professionnels"
192 + ],
193 + "cities": [
194 + "Montréal",
195 + "Québec"
196 + ]
197 + },
198 + {
199 + "ats": "icims",
200 + "source_id": "omni_hotels_horaire",
201 + "employer": "Omni Hôtel Mont-Royal",
202 + "sub": "externalhourly-omnihotels",
203 + "url": "https://www.omnihotels.com/fr/hotels/montreal-mont-royal",
204 + "sectors": [
205 + "Hôtellerie",
206 + "Restauration"
207 + ],
208 + "cities": [
209 + "Montréal"
210 + ]
211 + },
212 + {
213 + "ats": "icims",
214 + "source_id": "omni_hotels_gestion",
215 + "employer": "Omni Hôtel Mont-Royal",
216 + "sub": "externalmanager-omnihotels",
217 + "url": "https://www.omnihotels.com/fr/hotels/montreal-mont-royal",
218 + "sectors": [
219 + "Hôtellerie",
220 + "Gestion"
221 + ],
222 + "cities": [
223 + "Montréal"
224 + ]
225 + },
226 + {
227 + "ats": "icims",
228 + "source_id": "edf_renouvelables",
229 + "employer": "EDF Renouvelables Canada",
230 + "sub": "cafrench-edf-re",
231 + "url": "https://www.edf-renouvelables.ca",
232 + "sectors": [
233 + "Énergie",
234 + "Ingénierie"
235 + ],
236 + "cities": [
237 + "Montréal"
238 + ]
239 + },
240 + {
241 + "ats": "successfactors",
242 + "source_id": "bombardier",
243 + "employer": "Bombardier",
244 + "base": "https://jobs.bombardier.com",
245 + "url": "https://bombardier.com",
246 + "sectors": [
247 + "Aéronautique",
248 + "Ingénierie",
249 + "Fabrication"
250 + ],
251 + "cities": [
252 + "Dorval",
253 + "Mirabel",
254 + "Montréal"
255 + ]
256 + },
257 + {
258 + "ats": "successfactors",
259 + "source_id": "domtar",
260 + "employer": "Domtar",
261 + "base": "https://jobs.domtar.com",
262 + "url": "https://www.domtar.com",
263 + "sectors": [
264 + "Pâtes et papiers",
265 + "Foresterie",
266 + "Fabrication"
267 + ],
268 + "cities": [
269 + "Montréal",
270 + "Windsor",
271 + "Lachine"
272 + ]
273 + },
274 + {
275 + "ats": "successfactors",
276 + "source_id": "cascades",
277 + "employer": "Cascades",
278 + "base": "https://jobs.cascades.com",
279 + "url": "https://www.cascades.com",
280 + "sectors": [
281 + "Pâtes et papiers",
282 + "Emballage",
283 + "Fabrication"
284 + ],
285 + "cities": [
286 + "Kingsey Falls",
287 + "Drummondville",
288 + "Candiac"
289 + ]
290 + },
291 + {
292 + "ats": "adp",
293 + "source_id": "lassonde",
294 + "employer": "Industries Lassonde",
295 + "cid": "f96ce972-7721-4c7b-99ea-9a970ddc824a",
296 + "ccid": "9200604444876_2",
297 + "url": "https://www.lassonde.com",
298 + "sectors": [
299 + "Agroalimentaire",
300 + "Fabrication"
301 + ],
302 + "cities": [
303 + "Rougemont",
304 + "Montréal"
305 + ]
306 + },
307 + {
308 + "ats": "adp",
309 + "source_id": "mitsubishi_hc_capital",
310 + "employer": "Mitsubishi HC Capital Canada",
311 + "cid": "b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7",
312 + "ccid": "9200144510729_2",
313 + "url": "https://www.mhccna.com",
314 + "sectors": [
315 + "Finance",
316 + "Financement commercial"
317 + ],
318 + "cities": [
319 + "Laval",
320 + "Trois-Rivières"
321 + ]
322 + },
323 + {
324 + "ats": "adp",
325 + "source_id": "msi_gestion",
326 + "employer": "MSI Gestion immobilière",
327 + "cid": "47d4e792-26e5-471c-92ac-73c60131bab8",
328 + "ccid": "19000101_000001",
329 + "url": "https://www.msimmobiliers.com",
330 + "sectors": [
331 + "Immobilier",
332 + "Gestion immobilière"
333 + ],
334 + "cities": [
335 + "Montréal"
336 + ]
337 + },
338 + {
339 + "ats": "digitalrecruiters",
340 + "source_id": "sante_quebec",
341 + "employer": "Santé Québec",
342 + "base": "https://emplois.sante.quebec",
343 + "url": "https://sante.quebec",
344 + "sectors": [
345 + "Santé",
346 + "Secteur public",
347 + "Administration"
348 + ],
349 + "cities": [
350 + "Montréal",
351 + "Québec"
352 + ]
353 + },
354 + {
355 + "ats": "workland",
356 + "source_id": "cssdm",
357 + "employer": "Centre de services scolaire de Montréal",
358 + "list_url": "https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/",
359 + "url": "https://www.cssdm.gouv.qc.ca",
360 + "sectors": [
361 + "Éducation",
362 + "Secteur public"
363 + ],
364 + "cities": [
365 + "Montréal"
366 + ]
367 + }
368 +]
added jobka/connectors/_jsonld.py +160 −0
@@ -0,0 +1,160 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/_jsonld.py
6 +# Rôle : Extraction du balisage schema.org JobPosting (JSON-LD) — partagé
7 +# par les connecteurs dont la source publie des pages détail HTML
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Utilitaires JSON-LD (schema.org/JobPosting).
11 +
12 +Beaucoup de sites (iCIMS, Jobillico, Espresso-Jobs, Digital Recruiters,
13 +Workland/Atlas…) embarquent un bloc ``<script type="application/ld+json">``
14 +conforme à schema.org sur la page détail de chaque offre. On l'extrait ici de
15 +façon tolérante (JSON imparfait, @graph, listes) et on le convertit en champs
16 +standard prêts à verser dans un ``JobPosting``.
17 +"""
18 +from __future__ import annotations
19 +
20 +import json
21 +import re
22 +
23 +_SCRIPT_RE = re.compile(
24 + r'<script[^>]*type=["\']application/ld\+json["\'][^>]*>(.*?)</script>',
25 + re.S | re.I)
26 +
27 +_UNIT = {"hour": "hour", "hourly": "hour", "day": "day", "week": "week",
28 + "month": "month", "year": "year", "annual": "year"}
29 +
30 +
31 +def _iter_nodes(node):
32 + """Itère tous les objets JSON-LD (racine, @graph, listes imbriquées)."""
33 + if isinstance(node, list):
34 + for item in node:
35 + yield from _iter_nodes(item)
36 + elif isinstance(node, dict):
37 + yield node
38 + yield from _iter_nodes(node.get("@graph") or [])
39 +
40 +
41 +def extract_jobposting(html: str) -> dict | None:
42 + """Retourne le premier objet @type=JobPosting trouvé dans la page."""
43 + for m in _SCRIPT_RE.finditer(html or ""):
44 + raw = m.group(1).strip()
45 + try:
46 + data = json.loads(raw)
47 + except ValueError:
48 + # JSON avec contrôle non échappé (fréquent) : tentative de secours
49 + try:
50 + data = json.loads(re.sub(r"[\x00-\x1f]", " ", raw))
51 + except ValueError:
52 + continue
53 + for node in _iter_nodes(data):
54 + t = node.get("@type")
55 + types = t if isinstance(t, list) else [t]
56 + if any(str(x).lower() == "jobposting" for x in types if x):
57 + return node
58 + return None
59 +
60 +
61 +def _first(value):
62 + if isinstance(value, list):
63 + return value[0] if value else None
64 + return value
65 +
66 +
67 +def jobposting_fields(node: dict) -> dict:
68 + """Aplati un JobPosting JSON-LD en champs standard Job·Ka (dict sparse)."""
69 + out: dict = {}
70 + if not node:
71 + return out
72 + out["title"] = node.get("title") or node.get("name") or ""
73 + out["description_html"] = node.get("description") or ""
74 + out["date_posted"] = node.get("datePosted") or None
75 + out["date_deadline"] = node.get("validThrough") or None
76 + et = node.get("employmentType")
77 + out["employment_label"] = ", ".join(et) if isinstance(et, list) else (et or "")
78 +
79 + org = _first(node.get("hiringOrganization"))
80 + if isinstance(org, dict):
81 + out["employer"] = org.get("name") or ""
82 + elif isinstance(org, str):
83 + out["employer"] = org
84 +
85 + loc = _first(node.get("jobLocation"))
86 + if isinstance(loc, dict):
87 + addr = loc.get("address") or {}
88 + if isinstance(addr, str):
89 + out["location_label"] = addr
90 + elif isinstance(addr, dict):
91 + out["city"] = addr.get("addressLocality") or ""
92 + out["region_code"] = addr.get("addressRegion") or ""
93 + out["postal_code"] = addr.get("postalCode") or ""
94 + out["address"] = addr.get("streetAddress") or ""
95 + geo = loc.get("geo") or {}
96 + if isinstance(geo, dict) and geo.get("latitude") is not None:
97 + try:
98 + out["lat"] = float(geo["latitude"])
99 + out["lng"] = float(geo["longitude"])
100 + except (TypeError, ValueError):
101 + pass
102 +
103 + sal = node.get("baseSalary")
104 + if isinstance(sal, dict):
105 + val = sal.get("value")
106 + unit = None
107 + lo = hi = None
108 + if isinstance(val, dict):
109 + lo = val.get("minValue", val.get("value"))
110 + hi = val.get("maxValue")
111 + unit = val.get("unitText")
112 + elif isinstance(val, (int, float)):
113 + lo = val
114 + if isinstance(lo, str):
115 + try:
116 + lo = float(lo.replace(",", "."))
117 + except ValueError:
118 + lo = None
119 + if isinstance(hi, str):
120 + try:
121 + hi = float(hi.replace(",", "."))
122 + except ValueError:
123 + hi = None
124 + if lo:
125 + out["salary_min"] = float(lo)
126 + out["salary_max"] = float(hi) if hi else None
127 + out["salary_unit"] = _UNIT.get(str(unit or "").lower())
128 + return out
129 +
130 +
131 +def apply_fields(job, fields: dict, *, override_employer: bool = False) -> None:
132 + """Verse les champs extraits dans un JobPosting sans écraser l'existant."""
133 + from ..normalize import clean_html
134 + if fields.get("description_html") and not job.description:
135 + job.description = clean_html(fields["description_html"])
136 + if fields.get("title") and not job.title:
137 + job.title = fields["title"]
138 + if fields.get("employer") and (override_employer or not job.employer):
139 + job.employer = fields["employer"]
140 + if fields.get("date_posted") and not job.date_posted:
141 + job.date_posted = fields["date_posted"]
142 + if fields.get("date_deadline") and not job.date_deadline:
143 + job.date_deadline = fields["date_deadline"]
144 + if fields.get("city") and not job.city:
145 + job.city = fields["city"]
146 + if fields.get("postal_code") and not job.postal_code:
147 + job.postal_code = fields["postal_code"]
148 + if fields.get("address") and not job.address:
149 + job.address = fields["address"]
150 + if fields.get("location_label") and not job.location_label:
151 + job.location_label = fields["location_label"]
152 + if fields.get("salary_min") is not None and job.salary_min is None:
153 + job.salary_min = fields["salary_min"]
154 + job.salary_max = fields.get("salary_max")
155 + job.salary_unit = fields.get("salary_unit")
156 + if fields.get("employment_label"):
157 + job.details.setdefault("employment_label", fields["employment_label"])
158 + if fields.get("lat") is not None and job.lat is None:
159 + job.lat = fields["lat"]
160 + job.lng = fields.get("lng")
added jobka/connectors/adp.py +160 −0
@@ -0,0 +1,160 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/adp.py
6 +# Rôle : Classe de plateforme ADP Workforce Now (centre de carrières) —
7 +# API JSON publique job-requisitions, un employeur = une sous-classe
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme ADP Workforce Now (workforcenow.adp.com).
11 +
12 +API JSON publique du centre de carrières :
13 +- liste : GET /mascsr/default/careercenter/public/events/staffing/v1/
14 + job-requisitions?cid=<CID>&ccId=<CCID>&lang=<fr_CA>&$top=&$skip=
15 + -> {jobRequisitions:[{itemID, requisitionTitle, postDate,
16 + payGradeRange, workLevelCode, requisitionLocations,
17 + customFieldGroup}]}
18 +- détail : GET .../job-requisitions/<itemID>?cid=… -> requisitionDescription
19 + (HTML). Visité avec cache BD + budget.
20 +"""
21 +from __future__ import annotations
22 +
23 +import os
24 +
25 +from ..schema import JobPosting, clean_html, is_quebec_location
26 +from .base import BaseConnector
27 +
28 +PAGE_SIZE = 100
29 +MAX_DETAILS = int(os.environ.get("JOBKA_ADP_DETAIL_LIMIT", "80"))
30 +
31 +_API = ("https://workforcenow.adp.com/mascsr/default/careercenter/public/"
32 + "events/staffing/v1/job-requisitions")
33 +
34 +_SALARY_UNIT = {"AN": "year", "ANNUEL": "year", "YR": "year", "YEAR": "year",
35 + "HR": "hour", "HO": "hour", "HORAIRE": "hour", "HOUR": "hour"}
36 +
37 +
38 +class ADPWorkforceNowConnector(BaseConnector):
39 + """Base ADP WFN — sous-classes : définir source_id, EMPLOYER, CID, CCID."""
40 +
41 + ats = "adp"
42 + request_delay = 1.0
43 +
44 + EMPLOYER = ""
45 + CID = "" # GUID du centre de carrières
46 + CCID = "19000101_000001" # identifiant du « career center »
47 + LANG = "fr_CA"
48 + quebec_only = True
49 + max_pages = 10
50 +
51 + def _params(self, extra: dict | None = None) -> dict:
52 + p = {"cid": self.CID, "ccId": self.CCID, "lang": self.LANG,
53 + "locale": self.LANG}
54 + p.update(extra or {})
55 + return p
56 +
57 + @staticmethod
58 + def _locations(req: dict) -> list[dict]:
59 + out = []
60 + for loc in req.get("requisitionLocations") or []:
61 + addr = loc.get("address") or {}
62 + out.append({
63 + "city": addr.get("cityName") or "",
64 + "prov": ((addr.get("countrySubdivisionLevel1") or {})
65 + .get("codeValue") or ""),
66 + "postal": addr.get("postalCode") or "",
67 + "label": ((loc.get("nameCode") or {}).get("shortName")
68 + or "").strip(),
69 + })
70 + return out
71 +
72 + def _keep(self, locations: list[dict]) -> bool:
73 + if not self.quebec_only:
74 + return True
75 + return any(l["prov"].upper() == "QC"
76 + or is_quebec_location(f"{l['label']} {l['city']}")
77 + for l in locations)
78 +
79 + def _fetch_detail(self, item_id: str) -> dict:
80 + data = self.get(f"{_API}/{item_id}", params=self._params(),
81 + headers={"Accept": "application/json"}).json()
82 + return {"description": clean_html(
83 + data.get("requisitionDescription") or "")}
84 +
85 + def fetch(self) -> list[JobPosting]:
86 + out: list[JobPosting] = []
87 + details_used = 0
88 + skip = 0
89 + for _ in range(self.max_pages):
90 + data = self.get(_API, params=self._params(
91 + {"$top": str(PAGE_SIZE), "$skip": str(skip)}),
92 + headers={"Accept": "application/json"}).json()
93 + reqs = data.get("jobRequisitions") or []
94 + if not reqs and skip == 0 and self.LANG != "en_US":
95 + # certains centres de carrières ne répondent qu'en anglais
96 + self.LANG = "en_US"
97 + continue
98 + if not reqs:
99 + break
100 + for r in reqs:
101 + locations = self._locations(r)
102 + if not self._keep(locations):
103 + continue
104 + iid = str(r.get("itemID") or "")
105 + if not iid:
106 + continue
107 + qc = next((l for l in locations
108 + if l["prov"].upper() == "QC"
109 + or is_quebec_location(f"{l['label']} {l['city']}")),
110 + locations[0] if locations else
111 + {"city": "", "postal": "", "label": ""})
112 + job = JobPosting(
113 + source=self.source_id, external_id=iid,
114 + url=("https://workforcenow.adp.com/mascsr/default/mdf/"
115 + f"recruitment/recruitment.html?cid={self.CID}"
116 + f"&ccId={self.CCID}&lang={self.LANG}&jobId={iid}"),
117 + employer=self.EMPLOYER,
118 + title=r.get("requisitionTitle") or "",
119 + city=qc["city"], postal_code=qc["postal"],
120 + location_label=qc["label"],
121 + date_posted=r.get("postDate") or None,
122 + ats=self.ats,
123 + )
124 + pay = r.get("payGradeRange") or {}
125 + lo = ((pay.get("minimumRate") or {}).get("amountValue"))
126 + hi = ((pay.get("maximumRate") or {}).get("amountValue"))
127 + if lo:
128 + unit = None
129 + for c in ((r.get("customFieldGroup") or {})
130 + .get("codeFields") or []):
131 + if ((c.get("nameCode") or {})
132 + .get("codeValue")) == "SalaryType":
133 + unit = _SALARY_UNIT.get(
134 + str(c.get("codeValue") or "").upper()) \
135 + or _SALARY_UNIT.get(
136 + str(c.get("shortName") or "").upper())
137 + job.salary_min = float(lo)
138 + job.salary_max = float(hi) if hi else None
139 + job.salary_unit = unit or ("year" if float(lo) > 5000
140 + else "hour")
141 + wl = (r.get("workLevelCode") or {}).get("shortName")
142 + if wl:
143 + job.details["employment_label"] = wl
144 + key = (r.get("postDate") or "") + (job.title or "")[:40]
145 + if details_used < MAX_DETAILS:
146 + fresh = [False]
147 +
148 + def _fn(i=iid, fresh=fresh):
149 + fresh[0] = True
150 + return self._fetch_detail(i)
151 +
152 + d = self.detail(iid, key, _fn)
153 + if fresh[0]:
154 + details_used += 1
155 + job.description = d.get("description", "")
156 + out.append(job)
157 + if len(reqs) < PAGE_SIZE:
158 + break
159 + skip += PAGE_SIZE
160 + return out
added jobka/connectors/agnico_eagle.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/agnico_eagle.py
6 +# Rôle : Connecteur Agnico Eagle Mines — taleo (host='agnicoeagle', section='2')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .taleo import TaleoConnector
11 +
12 +
13 +class AgnicoEagleConnector(TaleoConnector):
14 + source_id = 'agnico_eagle'
15 + EMPLOYER = 'Agnico Eagle Mines'
16 + HOST = 'agnicoeagle'
17 + SECTION = '2'
added jobka/connectors/bayshore.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/bayshore.py
6 +# Rôle : Connecteur Bayshore HealthCare — taleo (host='bayshore', section='bs_ex')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .taleo import TaleoConnector
11 +
12 +
13 +class BayshoreConnector(TaleoConnector):
14 + source_id = 'bayshore'
15 + EMPLOYER = 'Bayshore HealthCare'
16 + HOST = 'bayshore'
17 + SECTION = 'bs_ex'
added jobka/connectors/bell_textron.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/bell_textron.py
6 +# Rôle : Connecteur Bell Textron Canada — taleo (host='textron', section='bell')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .taleo import TaleoConnector
11 +
12 +
13 +class BellTextronConnector(TaleoConnector):
14 + source_id = 'bell_textron'
15 + EMPLOYER = 'Bell Textron Canada'
16 + HOST = 'textron'
17 + SECTION = 'bell'
added jobka/connectors/bombardier.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/bombardier.py
6 +# Rôle : Connecteur Bombardier — successfactors (base='https://jobs.bombardier.com')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .successfactors import SuccessFactorsConnector
11 +
12 +
13 +class BombardierConnector(SuccessFactorsConnector):
14 + source_id = 'bombardier'
15 + EMPLOYER = 'Bombardier'
16 + BASE = 'https://jobs.bombardier.com'
added jobka/connectors/capreit.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/capreit.py
6 +# Rôle : Connecteur CAPREIT — icims (sub='careers-capreit')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .icims import ICIMSConnector
11 +
12 +
13 +class CapreitConnector(ICIMSConnector):
14 + source_id = 'capreit'
15 + EMPLOYER = 'CAPREIT'
16 + SUB = 'careers-capreit'
added jobka/connectors/cascades.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/cascades.py
6 +# Rôle : Connecteur Cascades — successfactors (base='https://jobs.cascades.com')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .successfactors import SuccessFactorsConnector
11 +
12 +
13 +class CascadesConnector(SuccessFactorsConnector):
14 + source_id = 'cascades'
15 + EMPLOYER = 'Cascades'
16 + BASE = 'https://jobs.cascades.com'
added jobka/connectors/ciusss_est_montreal.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/ciusss_est_montreal.py
6 +# Rôle : Connecteur CIUSSS de l'Est-de-l'Île-de-Montréal — taleo (host='ciusssemtl', section='cemtl')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .taleo import TaleoConnector
11 +
12 +
13 +class CiusssEstMontrealConnector(TaleoConnector):
14 + source_id = 'ciusss_est_montreal'
15 + EMPLOYER = "CIUSSS de l'Est-de-l'Île-de-Montréal"
16 + HOST = 'ciusssemtl'
17 + SECTION = 'cemtl'
added jobka/connectors/cssdm.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/cssdm.py
6 +# Rôle : Connecteur Centre de services scolaire de Montréal — workland (list_url='https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .workland import WorklandConnector
11 +
12 +
13 +class CssdmConnector(WorklandConnector):
14 + source_id = 'cssdm'
15 + EMPLOYER = 'Centre de services scolaire de Montréal'
16 + DEFAULT_CITY = "Montréal"
17 + LIST_URL = 'https://www.cssdm.gouv.qc.ca/travailler-cssdm/offres-emploi/'
added jobka/connectors/digitalrecruiters.py +81 −0
@@ -0,0 +1,81 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/digitalrecruiters.py
6 +# Rôle : Classe de plateforme Cegid Digital Recruiters (sites carrières) —
7 +# sitemap + JSON-LD des pages annonce (ex. Santé Québec)
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme Cegid Digital Recruiters.
11 +
12 +Les sites carrières Digital Recruiters (ex. emplois.sante.quebec) sont rendus
13 +côté serveur :
14 +- liste : GET <BASE>/<locale>/sitemap.xml -> URLs /annonce/<id>-<slug>
15 +- détail : chaque page annonce embarque un JSON-LD schema.org/JobPosting
16 + complet (description, lieu, salaire, dates) — cache BD + budget.
17 +"""
18 +from __future__ import annotations
19 +
20 +import os
21 +import re
22 +
23 +from ..schema import JobPosting, is_quebec_location
24 +from . import _jsonld
25 +from .base import BaseConnector
26 +
27 +MAX_DETAILS = int(os.environ.get("JOBKA_DR_DETAIL_LIMIT", "120"))
28 +
29 +_AD_RE = re.compile(r"<loc>([^<]*/annonce/(\d+)[^<]*)</loc>", re.I)
30 +
31 +
32 +class DigitalRecruitersConnector(BaseConnector):
33 + """Base Digital Recruiters — sous-classes : définir source_id, EMPLOYER,
34 + BASE (ex. https://emplois.sante.quebec) et LOCALE."""
35 +
36 + ats = "digitalrecruiters"
37 + request_delay = 0.8
38 +
39 + EMPLOYER = ""
40 + BASE = ""
41 + LOCALE = "fr_CA"
42 + quebec_only = True
43 +
44 + def _fetch_detail(self, url: str) -> dict:
45 + html = self.get(url).text
46 + node = _jsonld.extract_jobposting(html)
47 + return _jsonld.jobposting_fields(node) if node else {}
48 +
49 + def fetch(self) -> list[JobPosting]:
50 + xml = self.get(f"{self.BASE}/{self.LOCALE}/sitemap.xml").text
51 + out: list[JobPosting] = []
52 + details_used = 0
53 + seen: set[str] = set()
54 + for url, eid in _AD_RE.findall(xml):
55 + if eid in seen:
56 + continue
57 + seen.add(eid)
58 + job = JobPosting(source=self.source_id, external_id=eid, url=url,
59 + employer=self.EMPLOYER, title="", ats=self.ats)
60 + fields: dict = {}
61 + if details_used < MAX_DETAILS:
62 + fresh = [False]
63 +
64 + def _fn(u=url, fresh=fresh):
65 + fresh[0] = True
66 + return self._fetch_detail(u)
67 +
68 + fields = self.detail(eid, eid, _fn)
69 + if fresh[0]:
70 + details_used += 1
71 + if not fields:
72 + continue
73 + _jsonld.apply_fields(job, fields)
74 + if self.quebec_only and job.city and not is_quebec_location(
75 + f"{job.city} {fields.get('region_code', '')}"):
76 + if (fields.get("region_code") or "").upper() != "QC":
77 + continue
78 + if not job.title:
79 + continue
80 + out.append(job)
81 + return out
added jobka/connectors/domtar.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/domtar.py
6 +# Rôle : Connecteur Domtar — successfactors (base='https://jobs.domtar.com')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .successfactors import SuccessFactorsConnector
11 +
12 +
13 +class DomtarConnector(SuccessFactorsConnector):
14 + source_id = 'domtar'
15 + EMPLOYER = 'Domtar'
16 + BASE = 'https://jobs.domtar.com'
added jobka/connectors/edf_renouvelables.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/edf_renouvelables.py
6 +# Rôle : Connecteur EDF Renouvelables Canada — icims (sub='cafrench-edf-re')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .icims import ICIMSConnector
11 +
12 +
13 +class EdfRenouvelablesConnector(ICIMSConnector):
14 + source_id = 'edf_renouvelables'
15 + EMPLOYER = 'EDF Renouvelables Canada'
16 + SUB = 'cafrench-edf-re'
added jobka/connectors/icims.py +114 −0
@@ -0,0 +1,114 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/icims.py
6 +# Rôle : Classe de plateforme iCIMS (portail carrières <org>.icims.com)
7 +# — liste iframe paginée + JSON-LD des pages détail
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme iCIMS.
11 +
12 +- liste : GET https://<sub>.icims.com/jobs/search?ss=1&in_iframe=1&pr=<page>
13 + (HTML léger, liens /jobs/<id>/<slug>/job) — pagination pr=0,1,2…
14 +- détail : GET https://<sub>.icims.com/jobs/<id>/<slug>/job?in_iframe=1
15 + -> JSON-LD schema.org/JobPosting complet (description, lieu, dates)
16 +
17 +Le lieu n'apparaissant pas toujours sur la liste, le filtre Québec est
18 +appliqué après lecture du détail (avec cache BD : chaque offre n'est visitée
19 +qu'une fois).
20 +"""
21 +from __future__ import annotations
22 +
23 +import os
24 +import re
25 +
26 +from ..schema import JobPosting, is_quebec_location
27 +from . import _jsonld
28 +from .base import BaseConnector
29 +
30 +MAX_DETAILS = int(os.environ.get("JOBKA_ICIMS_DETAIL_LIMIT", "150"))
31 +
32 +_LINK_RE = re.compile(r'href="https?://[^"]*?/jobs/(\d+)/([^/"]+)/job[^"]*"')
33 +
34 +
35 +class ICIMSConnector(BaseConnector):
36 + """Base iCIMS — sous-classes : définir source_id, EMPLOYER, SUB
37 + (sous-domaine complet, ex. « careers-capreit »)."""
38 +
39 + ats = "icims"
40 + request_delay = 1.0
41 +
42 + EMPLOYER = ""
43 + SUB = "" # <SUB>.icims.com
44 + quebec_only = True
45 + max_pages = 15
46 +
47 + @property
48 + def _base(self) -> str:
49 + return f"https://{self.SUB}.icims.com"
50 +
51 + def _fetch_detail(self, jid: str, slug: str) -> dict:
52 + html = self.get(f"{self._base}/jobs/{jid}/{slug}/job",
53 + params={"in_iframe": "1"}).text
54 + node = _jsonld.extract_jobposting(html)
55 + return _jsonld.jobposting_fields(node) if node else {}
56 +
57 + def fetch(self) -> list[JobPosting]:
58 + seen: dict[str, str] = {}
59 + for page in range(self.max_pages):
60 + html = self.get(f"{self._base}/jobs/search",
61 + params={"ss": "1", "in_iframe": "1",
62 + "pr": str(page)}).text
63 + links = _LINK_RE.findall(html)
64 + new = 0
65 + for jid, slug in links:
66 + if jid not in seen:
67 + seen[jid] = slug
68 + new += 1
69 + if not links or new == 0:
70 + break
71 +
72 + out: list[JobPosting] = []
73 + details_used = 0
74 + for jid, slug in seen.items():
75 + if details_used >= MAX_DETAILS:
76 + # budget épuisé : ne servir que le cache (sans jamais y écrire
77 + # un vide) — l'offre sera reprise à la prochaine synchronisation
78 + from .. import db
79 + if self._detail_con is None:
80 + self._detail_con = db.connect()
81 + cached = db.get_cached_detail(self._detail_con, self.source_id,
82 + jid, jid)
83 + if cached is None:
84 + continue
85 + fields = cached
86 + else:
87 + fresh = [False]
88 +
89 + def _fn(j=jid, s=slug, fresh=fresh):
90 + fresh[0] = True
91 + return self._fetch_detail(j, s)
92 +
93 + fields = self.detail(jid, jid, _fn)
94 + if fresh[0]:
95 + details_used += 1
96 + if not fields:
97 + continue
98 + place = " ".join(str(fields.get(k) or "") for k in
99 + ("city", "region_code", "postal_code"))
100 + if self.quebec_only and not (
101 + (fields.get("region_code") or "").upper() == "QC"
102 + or is_quebec_location(place)):
103 + continue
104 + job = JobPosting(
105 + source=self.source_id, external_id=jid,
106 + url=f"{self._base}/jobs/{jid}/{slug}/job",
107 + employer=self.EMPLOYER,
108 + title="", ats=self.ats,
109 + )
110 + _jsonld.apply_fields(job, fields)
111 + if not job.title:
112 + job.title = slug.replace("-", " ").strip().capitalize()
113 + out.append(job)
114 + return out
added jobka/connectors/lassonde.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/lassonde.py
6 +# Rôle : Connecteur Industries Lassonde — adp (cid='f96ce972-7721-4c7b-99ea-9a970ddc824a', ccid='9200604444876_2')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .adp import ADPWorkforceNowConnector
11 +
12 +
13 +class LassondeConnector(ADPWorkforceNowConnector):
14 + source_id = 'lassonde'
15 + EMPLOYER = 'Industries Lassonde'
16 + CID = 'f96ce972-7721-4c7b-99ea-9a970ddc824a'
17 + CCID = '9200604444876_2'
added jobka/connectors/mccarthy_tetrault.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/mccarthy_tetrault.py
6 +# Rôle : Connecteur McCarthy Tétrault — icims (sub='careers-mccarthyca')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .icims import ICIMSConnector
11 +
12 +
13 +class MccarthyTetraultConnector(ICIMSConnector):
14 + source_id = 'mccarthy_tetrault'
15 + EMPLOYER = 'McCarthy Tétrault'
16 + SUB = 'careers-mccarthyca'
added jobka/connectors/messer_canada.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/messer_canada.py
6 +# Rôle : Connecteur Messer Canada — ultipro (org='MES1005MESR', board='dbd63926-d756-486f-a254-b2a6ede1d26e')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .ultipro import UltiProConnector
11 +
12 +
13 +class MesserCanadaConnector(UltiProConnector):
14 + source_id = 'messer_canada'
15 + EMPLOYER = 'Messer Canada'
16 + ORG = 'MES1005MESR'
17 + BOARD = 'dbd63926-d756-486f-a254-b2a6ede1d26e'
added jobka/connectors/mitsubishi_hc_capital.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/mitsubishi_hc_capital.py
6 +# Rôle : Connecteur Mitsubishi HC Capital Canada — adp (cid='b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7', ccid='9200144510729_2')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .adp import ADPWorkforceNowConnector
11 +
12 +
13 +class MitsubishiHcCapitalConnector(ADPWorkforceNowConnector):
14 + source_id = 'mitsubishi_hc_capital'
15 + EMPLOYER = 'Mitsubishi HC Capital Canada'
16 + CID = 'b3ef4f03-f8ff-4ded-80c8-6dd5c5a224f7'
17 + CCID = '9200144510729_2'
added jobka/connectors/msi_gestion.py +17 −0
@@ -0,0 +1,17 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/msi_gestion.py
6 +# Rôle : Connecteur MSI Gestion immobilière — adp (cid='47d4e792-26e5-471c-92ac-73c60131bab8', ccid='19000101_000001')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .adp import ADPWorkforceNowConnector
11 +
12 +
13 +class MsiGestionConnector(ADPWorkforceNowConnector):
14 + source_id = 'msi_gestion'
15 + EMPLOYER = 'MSI Gestion immobilière'
16 + CID = '47d4e792-26e5-471c-92ac-73c60131bab8'
17 + CCID = '19000101_000001'
added jobka/connectors/njoyn.py +182 −0
@@ -0,0 +1,182 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/njoyn.py
6 +# Rôle : Classe de plateforme Njoyn (CGI) — très répandu au secteur public
7 +# québécois (santé, villes, sociétés d'État). HTML serveur + jeton.
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme Njoyn (CGI).
11 +
12 +Les babillards Njoyn servent un HTML rendu côté serveur :
13 + <BASE>/<CL>/xweb/xweb.asp?clid=<CLID>&page=joblisting&lang=<n>
14 +Le serveur redirige en ajoutant un jeton de session (tbtoken) — il suffit de
15 +suivre la redirection. Chaque offre est un <article> (habillage moderne) ou
16 +une ligne de tableau (habillage classique) pointant vers
17 +``page=jobdetails&jobid=J####-####``.
18 +
19 +⚠ Les domaines *.njoyn.com sont derrière Radware Bot Manager : pour ces
20 +tenants, mettre ``USE_SCRAPFLY = True`` (contournement ASP). Les babillards
21 +servis sur un domaine personnalisé (ex. emplois.santemontreal.qc.ca) se
22 +consultent directement.
23 +"""
24 +from __future__ import annotations
25 +
26 +import hashlib
27 +import html as _html
28 +import os
29 +import re
30 +
31 +from ..schema import JobPosting, clean_html
32 +from .base import BaseConnector
33 +
34 +MAX_DETAILS = int(os.environ.get("JOBKA_NJOYN_DETAIL_LIMIT", "40"))
35 +
36 +_ARTICLE_RE = re.compile(r"<article[^>]*>(.*?)</article>", re.S | re.I)
37 +_H4_RE = re.compile(r"<h4[^>]*>(.*?)</h4>", re.S | re.I)
38 +_TIME_RE = re.compile(r"<time[^>]*>(.*?)</time>", re.S | re.I)
39 +_JOBID_RE = re.compile(r"jobid=(J\d{4}-\d+)", re.I)
40 +_FIELD_RE = re.compile(
41 + r"<li><strong>([^<:]+?)(?:&nbsp;)?\s*:?\s*</strong>\s*(.*?)</li>",
42 + re.S | re.I)
43 +_APPLY_RE = re.compile(
44 + r"<a[^>]*class='btn btn-primary'[^>]*href='([^']+)'", re.I)
45 +_ROW_LINK_RE = re.compile(
46 + r'''<a[^>]*href=["'][^"']*jobid=(J\d{4}-\d+)[^"']*["'][^>]*>(.*?)</a>''',
47 + re.S | re.I)
48 +
49 +
50 +def _txt(s: str) -> str:
51 + return re.sub(r"\s+", " ", _html.unescape(re.sub(r"<[^>]+>", " ", s or ""))).strip()
52 +
53 +
54 +class NjoynConnector(BaseConnector):
55 + """Base Njoyn — sous-classes : définir source_id, EMPLOYER, BASE, CL, CLID."""
56 +
57 + ats = "njoyn"
58 + request_delay = 1.2
59 +
60 + EMPLOYER = ""
61 + BASE = "" # ex. https://emplois.santemontreal.qc.ca
62 + CL = "CL3" # instance (CL2, CL3, CL4…)
63 + CLID = "" # identifiant client Njoyn
64 + LANG = "2" # 2 = français, 1 = anglais
65 + USE_SCRAPFLY = False # requis pour les domaines *.njoyn.com (Radware)
66 +
67 + def _url(self, page: str, jobid: str = "") -> str:
68 + u = (f"{self.BASE}/{self.CL}/xweb/xweb.asp?clid={self.CLID}"
69 + f"&page={page}&lang={self.LANG}")
70 + if jobid:
71 + u += f"&jobid={jobid}"
72 + return u
73 +
74 + def _html(self, url: str) -> str:
75 + if self.USE_SCRAPFLY:
76 + return self.get_scrapfly(url, render_js=False)
77 + text = self.get(url).text
78 + if "perfdrive" in text[:4000]: # mur anti-bot : bascule Scrapfly
79 + return self.get_scrapfly(url, render_js=False)
80 + return text
81 +
82 + def _fetch_detail(self, jobid: str) -> dict:
83 + page = self._html(self._url("jobdetails", jobid))
84 + page = re.sub(r"<(noscript|nav|script|style)[^>]*>.*?</\1>", " ", page,
85 + flags=re.S | re.I)
86 + # habillage moderne : sections <div class="row"><h2>Titre</h2>corps</div>
87 + sections = re.findall(
88 + r'<div class="row">\s*<h2[^>]*>(.*?)</h2>(.*?)</div>', page,
89 + re.S | re.I)
90 + if sections:
91 + body = "\n\n".join(f"{clean_html(h)}\n{clean_html(b)}"
92 + for h, b in sections)
93 + return {"description": body[:20000]}
94 + # habillage classique : contenu principal après le <h1>
95 + m = re.search(r"<h1[^>]*>.*?</h1>(.*?)(?:<form|<footer|</main)", page,
96 + re.S | re.I)
97 + body = m.group(1) if m else ""
98 + return {"description": clean_html(body)[:20000]}
99 +
100 + def _parse_articles(self, page: str) -> list[dict]:
101 + items = []
102 + for art in _ARTICLE_RE.findall(page):
103 + m = _JOBID_RE.search(art)
104 + if not m:
105 + continue
106 + fields = {_txt(k).lower(): _txt(v) for k, v in _FIELD_RE.findall(art)}
107 + apply_m = _APPLY_RE.search(art)
108 + h4 = _H4_RE.search(art)
109 + t = _TIME_RE.search(art)
110 + items.append({
111 + "jobid": m.group(1),
112 + "title": _txt(h4.group(1)) if h4 else "",
113 + "date": _txt(t.group(1)).lstrip("Le ").strip() if t else "",
114 + "fields": fields,
115 + "apply_url": apply_m.group(1) if apply_m else "",
116 + })
117 + return items
118 +
119 + def _parse_rows(self, page: str) -> list[dict]:
120 + """Habillage classique : simples liens de tableau vers jobdetails."""
121 + items, seen = [], set()
122 + for jobid, label in _ROW_LINK_RE.findall(page):
123 + title = _txt(label)
124 + if not title or title.lower().startswith(("en savoir", "postuler",
125 + "apply", "more")):
126 + title = ""
127 + if jobid in seen:
128 + # garder le premier libellé non vide
129 + if title:
130 + for it in items:
131 + if it["jobid"] == jobid and not it["title"]:
132 + it["title"] = title
133 + continue
134 + seen.add(jobid)
135 + items.append({"jobid": jobid, "title": title, "date": "",
136 + "fields": {}, "apply_url": ""})
137 + return [it for it in items if it["title"]]
138 +
139 + def fetch(self) -> list[JobPosting]:
140 + page = self._html(self._url("joblisting"))
141 + items = self._parse_articles(page) or self._parse_rows(page)
142 + out: list[JobPosting] = []
143 + details_used = 0
144 + for it in items:
145 + f = it["fields"]
146 + salary = f.get("salaire", "")
147 + if re.match(r"^0\s*\$\s*-\s*0\s*\$", salary):
148 + salary = ""
149 + job = JobPosting(
150 + source=self.source_id, external_id=it["jobid"],
151 + url=self._url("jobdetails", it["jobid"]),
152 + employer=f.get("établissement") or f.get("etablissement")
153 + or self.EMPLOYER,
154 + title=it["title"],
155 + city=f.get("villes et arrondissements", "").split(",")[0],
156 + location_label=f.get("villes et arrondissements", ""),
157 + salary_label=salary,
158 + date_posted=it["date"] or None,
159 + ats=self.ats,
160 + )
161 + if f.get("statut de l'employé") or f.get("statut de l'employe"):
162 + job.details["employment_label"] = (
163 + f.get("statut de l'employé") or f.get("statut de l'employe"))
164 + if f.get("niveau de scolarité"):
165 + job.requirements["scolarite"] = f["niveau de scolarité"]
166 + if it["apply_url"]:
167 + job.details["apply_url"] = it["apply_url"]
168 + key = hashlib.sha1(
169 + f"{it['title']}|{it['date']}".encode()).hexdigest()[:12]
170 + if details_used < MAX_DETAILS:
171 + fresh = [False]
172 +
173 + def _fn(j=it["jobid"], fresh=fresh):
174 + fresh[0] = True
175 + return self._fetch_detail(j)
176 +
177 + d = self.detail(it["jobid"], key, _fn)
178 + if fresh[0]:
179 + details_used += 1
180 + job.description = d.get("description", "")
181 + out.append(job)
182 + return out
added jobka/connectors/omni_hotels_gestion.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/omni_hotels_gestion.py
6 +# Rôle : Connecteur Omni Hôtel Mont-Royal — icims (sub='externalmanager-omnihotels')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .icims import ICIMSConnector
11 +
12 +
13 +class OmniHotelsGestionConnector(ICIMSConnector):
14 + source_id = 'omni_hotels_gestion'
15 + EMPLOYER = 'Omni Hôtel Mont-Royal'
16 + SUB = 'externalmanager-omnihotels'
added jobka/connectors/omni_hotels_horaire.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/omni_hotels_horaire.py
6 +# Rôle : Connecteur Omni Hôtel Mont-Royal — icims (sub='externalhourly-omnihotels')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .icims import ICIMSConnector
11 +
12 +
13 +class OmniHotelsHoraireConnector(ICIMSConnector):
14 + source_id = 'omni_hotels_horaire'
15 + EMPLOYER = 'Omni Hôtel Mont-Royal'
16 + SUB = 'externalhourly-omnihotels'
added jobka/connectors/rtc_quebec.py +19 −0
@@ -0,0 +1,19 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/rtc_quebec.py
6 +# Rôle : Connecteur Réseau de transport de la Capitale — njoyn (base='https://rtc.njoyn.com', cl='CGI')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .njoyn import NjoynConnector
11 +
12 +
13 +class RtcQuebecConnector(NjoynConnector):
14 + source_id = 'rtc_quebec'
15 + EMPLOYER = 'Réseau de transport de la Capitale'
16 + BASE = 'https://rtc.njoyn.com'
17 + CL = 'CGI'
18 + CLID = '23009'
19 + USE_SCRAPFLY = True
added jobka/connectors/sante_montreal.py +18 −0
@@ -0,0 +1,18 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/sante_montreal.py
6 +# Rôle : Connecteur Santé Québec — Montréal — njoyn (base='https://emplois.santemontreal.qc.ca', cl='CL3')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .njoyn import NjoynConnector
11 +
12 +
13 +class SanteMontrealConnector(NjoynConnector):
14 + source_id = 'sante_montreal'
15 + EMPLOYER = 'Santé Québec — Montréal'
16 + BASE = 'https://emplois.santemontreal.qc.ca'
17 + CL = 'CL3'
18 + CLID = '54327'
added jobka/connectors/sante_quebec.py +16 −0
@@ -0,0 +1,16 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/sante_quebec.py
6 +# Rôle : Connecteur Santé Québec — digitalrecruiters (base='https://emplois.sante.quebec')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .digitalrecruiters import DigitalRecruitersConnector
11 +
12 +
13 +class SanteQuebecConnector(DigitalRecruitersConnector):
14 + source_id = 'sante_quebec'
15 + EMPLOYER = 'Santé Québec'
16 + BASE = 'https://emplois.sante.quebec'
added jobka/connectors/stl_laval.py +19 −0
@@ -0,0 +1,19 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/stl_laval.py
6 +# Rôle : Connecteur Société de transport de Laval — njoyn (base='https://lavaltransit.njoyn.com', cl='CL2')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .njoyn import NjoynConnector
11 +
12 +
13 +class StlLavalConnector(NjoynConnector):
14 + source_id = 'stl_laval'
15 + EMPLOYER = 'Société de transport de Laval'
16 + BASE = 'https://lavaltransit.njoyn.com'
17 + CL = 'CL2'
18 + CLID = '60406'
19 + USE_SCRAPFLY = True
added jobka/connectors/successfactors.py +161 −0
@@ -0,0 +1,161 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/successfactors.py
6 +# Rôle : Classe de plateforme SAP SuccessFactors (sites carrières « Career
7 +# Site Builder » / jobs2web) — sitemap RSS ou XML + microdonnées
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme SAP SuccessFactors (Career Site Builder, ex-jobs2web).
11 +
12 +Deux formats de « sitemap.xml » selon la génération du site :
13 +- RSS 2.0 : <item> avec titre « Poste (VILLE, PROV, PAYS[, CP]) » ET la
14 + description HTML complète — tout en une requête ;
15 +- urlset : URLs /job/<Ville>-<titre>-<Prov>-<CP>/<id>/ + lastmod ; la
16 + description est lue sur la page détail (microdonnées itemprop, avec cache
17 + BD + budget). Le filtre Québec s'appuie sur la province/le code postal
18 + présents dans le slug — aucun détail hors Québec n'est visité.
19 +"""
20 +from __future__ import annotations
21 +
22 +import html as _html
23 +import os
24 +import re
25 +import urllib.parse
26 +
27 +from ..schema import JobPosting, clean_html, is_quebec_location
28 +from .base import BaseConnector
29 +
30 +MAX_DETAILS = int(os.environ.get("JOBKA_SF_DETAIL_LIMIT", "150"))
31 +
32 +_ITEM_RE = re.compile(r"<item>(.*?)</item>", re.S | re.I)
33 +_TAG_RE = {t: re.compile(rf"<{t}>(.*?)</{t}>", re.S | re.I)
34 + for t in ("title", "link", "description", "pubDate")}
35 +_LOC_RE = re.compile(r"<loc>([^<]+)</loc>\s*(?:<lastmod>([^<]+)</lastmod>)?",
36 + re.I)
37 +_RSS_TITLE_RE = re.compile(r"^(.*)\(([^,()]+),\s*([^,()]+?),\s*([A-Z]{2})"
38 + r"(?:,\s*([^()]+))?\)\s*$", re.S)
39 +_QC_PROV = {"QC", "QUÉBEC", "QUEBEC"}
40 +_QC_SLUG_RE = re.compile(r"-(qu[ée]b|qc)-|-[ghj]\d[a-z][- ]?\d[a-z]\d/", re.I)
41 +_ITEMPROP_RE = re.compile(
42 + r'<(?:span|div|p)[^>]*itemprop="description"[^>]*>(.*?)'
43 + r"</(?:span|div|p)>", re.S | re.I)
44 +_META_RE = re.compile(
45 + r'itemprop="(datePosted|addressLocality|postalCode|employmentType)"'
46 + r'[^>]*content="([^"]*)"', re.I)
47 +
48 +
49 +class SuccessFactorsConnector(BaseConnector):
50 + """Base SuccessFactors CSB — sous-classes : définir source_id, EMPLOYER,
51 + BASE (URL du site carrières, ex. https://jobs.bombardier.com)."""
52 +
53 + ats = "successfactors"
54 + request_delay = 1.0
55 +
56 + EMPLOYER = ""
57 + BASE = ""
58 + quebec_only = True
59 +
60 + def _cdata(self, s: str) -> str:
61 + s = (s or "").strip()
62 + if s.startswith("<![CDATA["):
63 + s = s[9:]
64 + if s.endswith("]]>"):
65 + s = s[:-3]
66 + return _html.unescape(s.strip())
67 +
68 + # -- format RSS ------------------------------------------------------------
69 + def _fetch_rss(self, xml: str) -> list[JobPosting]:
70 + out = []
71 + for raw in _ITEM_RE.findall(xml):
72 + def g(tag: str, raw=raw) -> str:
73 + m = _TAG_RE[tag].search(raw)
74 + return self._cdata(m.group(1)) if m else ""
75 + vals = {t: g(t) for t in _TAG_RE}
76 + link = vals["link"]
77 + m_id = re.search(r"/(\d+)/?$", link)
78 + if not link or not m_id:
79 + continue
80 + title, city, prov = vals["title"], "", ""
81 + m = _RSS_TITLE_RE.match(vals["title"])
82 + if m:
83 + title = m.group(1).strip()
84 + city, prov = m.group(2).strip(), m.group(3).strip()
85 + if self.quebec_only and not (
86 + prov.upper() in _QC_PROV
87 + or (city and is_quebec_location(city))):
88 + continue
89 + job = JobPosting(
90 + source=self.source_id, external_id=m_id.group(1),
91 + url=link, employer=self.EMPLOYER, title=title,
92 + description=clean_html(self._cdata(vals["description"])),
93 + city=city.title() if city.isupper() else city,
94 + date_posted=vals["pubDate"] or None,
95 + ats=self.ats,
96 + )
97 + out.append(job)
98 + return out
99 +
100 + # -- format urlset ---------------------------------------------------------
101 + def _fetch_detail(self, url: str) -> dict:
102 + html = self.get(url).text
103 + d: dict = {}
104 + chunks = _ITEMPROP_RE.findall(html)
105 + if chunks:
106 + d["description"] = clean_html("\n".join(chunks))[:20000]
107 + for prop, content in _META_RE.findall(html):
108 + d[prop] = content
109 + return d
110 +
111 + def _fetch_urlset(self, xml: str) -> list[JobPosting]:
112 + out = []
113 + details_used = 0
114 + for loc, lastmod in _LOC_RE.findall(xml):
115 + loc = _html.unescape(loc.strip())
116 + m = re.search(r"/job/([^/]+)/(\d+)/?$", loc)
117 + if not m:
118 + continue
119 + slug_raw, eid = m.group(1), m.group(2)
120 + slug = urllib.parse.unquote(slug_raw)
121 + if self.quebec_only and not _QC_SLUG_RE.search(slug + "/"):
122 + continue
123 + city = slug.split("-")[0]
124 + postal = ""
125 + m_cp = re.search(r"([GHJ]\d[A-Z])[- ]?(\d[A-Z]\d)", slug, re.I)
126 + if m_cp:
127 + postal = f"{m_cp.group(1)}{m_cp.group(2)}".upper()
128 + job = JobPosting(
129 + source=self.source_id, external_id=eid, url=loc,
130 + employer=self.EMPLOYER,
131 + title=" ".join(slug.split("-")[1:-3]) or slug,
132 + city=city, postal_code=postal,
133 + date_posted=lastmod or None,
134 + ats=self.ats,
135 + )
136 + if details_used < MAX_DETAILS:
137 + fresh = [False]
138 +
139 + def _fn(u=loc, fresh=fresh):
140 + fresh[0] = True
141 + return self._fetch_detail(u)
142 +
143 + d = self.detail(eid, lastmod or eid, _fn)
144 + if fresh[0]:
145 + details_used += 1
146 + if d.get("description"):
147 + job.description = d["description"]
148 + if d.get("datePosted"):
149 + job.date_posted = d["datePosted"]
150 + if d.get("addressLocality"):
151 + job.city = d["addressLocality"]
152 + if d.get("employmentType"):
153 + job.details["employment_label"] = d["employmentType"]
154 + out.append(job)
155 + return out
156 +
157 + def fetch(self) -> list[JobPosting]:
158 + xml = self.get(f"{self.BASE}/sitemap.xml").text
159 + if "<rss" in xml[:300].lower():
160 + return self._fetch_rss(xml)
161 + return self._fetch_urlset(xml)
added jobka/connectors/taleo.py +185 −0
@@ -0,0 +1,185 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/taleo.py
6 +# Rôle : Classe de plateforme Oracle Taleo Enterprise (careersection REST)
7 +# — un employeur = une sous-classe (HOST, SECTION, EMPLOYER)
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme Oracle Taleo Enterprise.
11 +
12 +Les sections carrières Taleo exposent une API REST publique (celle que le
13 +frontal jobsearch.ftl consomme lui-même) :
14 +- amorce : GET https://<host>.taleo.net/careersection/<section>/jobsearch.ftl
15 + (cookies de session + numéro de portail « portal=NNN » dans le HTML)
16 +- liste : POST https://<host>.taleo.net/careersection/rest/jobboard/searchjobs
17 + ?lang=fr&portal=<NNN> body JSON {pageNo, sorting…}
18 + -> {requisitionList:[{jobId, contestNo, column:[...],
19 + locationsColumns:[i]}], pagingData:{totalCount, pageSize}}
20 +- détail : la page jobdetail.ftl?job=<contestNo> (description HTML rendue
21 + côté serveur) — visitée avec cache BD et budget.
22 +
23 +Le contenu des colonnes dépend de la configuration du portail : on repère la
24 +colonne des lieux via `locationsColumns` et on suppose la première colonne
25 +comme titre (constaté sur tous les portails testés).
26 +"""
27 +from __future__ import annotations
28 +
29 +import hashlib
30 +import json
31 +import os
32 +import re
33 +
34 +from ..schema import JobPosting, clean_html, is_quebec_location
35 +from .base import BaseConnector
36 +
37 +MAX_DETAILS = int(os.environ.get("JOBKA_TALEO_DETAIL_LIMIT", "120"))
38 +
39 +_DESC_RE = re.compile(
40 + r'requisitionDescriptionInterface[^>]*>', re.I)
41 +
42 +
43 +class TaleoConnector(BaseConnector):
44 + """Base Taleo — sous-classes : définir source_id, EMPLOYER, HOST, SECTION
45 + (et PORTAL si le numéro n'apparaît pas dans le HTML d'amorce)."""
46 +
47 + ats = "taleo"
48 + request_delay = 1.0
49 +
50 + EMPLOYER = ""
51 + HOST = "" # <host>.taleo.net
52 + SECTION = "2" # /careersection/<section>/
53 + PORTAL = "" # numéro de portail (sinon extrait du HTML)
54 + LANG = "fr"
55 + quebec_only = True
56 + max_pages = 40
57 +
58 + @property
59 + def _base(self) -> str:
60 + return f"https://{self.HOST}.taleo.net/careersection"
61 +
62 + def _careers_url(self) -> str:
63 + return f"{self._base}/{self.SECTION}/jobsearch.ftl?lang={self.LANG}"
64 +
65 + def _bootstrap(self) -> str:
66 + """Cookies de session + numéro de portail."""
67 + html = self.get(self._careers_url()).text
68 + if self.PORTAL:
69 + return str(self.PORTAL)
70 + m = re.search(r"portal=(\d+)", html)
71 + if not m:
72 + raise RuntimeError(f"taleo {self.HOST}: numéro de portail introuvable")
73 + return m.group(1)
74 +
75 + @staticmethod
76 + def _parse_locations(raw: str) -> list[str]:
77 + """La colonne des lieux est un JSON sérialisé : '["Québec-Malartic"]'."""
78 + if not raw:
79 + return []
80 + try:
81 + val = json.loads(raw)
82 + if isinstance(val, list):
83 + return [str(v) for v in val]
84 + except ValueError:
85 + pass
86 + return [raw]
87 +
88 + def _keep(self, locations: list[str]) -> bool:
89 + if not self.quebec_only:
90 + return True
91 + return any(is_quebec_location(l.replace("-", " ")) for l in locations)
92 +
93 + def _fetch_detail(self, contest_no: str) -> dict:
94 + html = self.get(f"{self._base}/{self.SECTION}/jobdetail.ftl",
95 + params={"job": contest_no, "lang": self.LANG}).text
96 + # description : contenu des blocs « requisitionDescriptionInterface »
97 + # (rendus côté serveur), sinon la zone principale de la page
98 + chunks = re.findall(
99 + r'<(?:span|div)[^>]*id="requisitionDescriptionInterface[^"]*"[^>]*>'
100 + r"(.*?)</(?:span|div)>", html, re.S)
101 + text = clean_html("\n".join(c for c in chunks if len(c) > 40))
102 + if not text:
103 + m = re.search(r'<div[^>]*class="[^"]*mastercontentpanel[^"]*"[^>]*>'
104 + r"(.*?)<!--", html, re.S)
105 + text = clean_html(m.group(1)) if m else ""
106 + return {"description": text}
107 +
108 + def fetch(self) -> list[JobPosting]:
109 + portal = self._bootstrap()
110 + url = (f"{self._base}/rest/jobboard/searchjobs"
111 + f"?lang={self.LANG}&portal={portal}")
112 + body_base = {
113 + "multilineEnabled": False,
114 + "sortingSelection": {"sortBySelectionParam": "3",
115 + "ascendingSortingOrder": "false"},
116 + "fieldData": {"fields": {"KEYWORD": "", "LOCATION": ""},
117 + "valid": True},
118 + "filterSelectionParam": {"searchFilterSelections": [
119 + {"id": "POSTING_DATE", "selectedValues": []},
120 + {"id": "LOCATION", "selectedValues": []}]},
121 + "advancedSearchFiltersSelectionParam":
122 + {"searchFilterSelections": []},
123 + }
124 + out: list[JobPosting] = []
125 + details_used = 0
126 + page = 1
127 + seen: set[str] = set()
128 + while page <= self.max_pages:
129 + data = self.post(url, json={**body_base, "pageNo": page},
130 + headers={"tz": "GMT-04:00"}).json()
131 + reqs = data.get("requisitionList") or []
132 + if not reqs and page == 1 and self.LANG != "en":
133 + # section carrières anglophone seulement : rebasculer en « en »
134 + self.LANG = "en"
135 + portal = self._bootstrap()
136 + url = (f"{self._base}/rest/jobboard/searchjobs"
137 + f"?lang={self.LANG}&portal={portal}")
138 + continue
139 + if not reqs:
140 + break
141 + for r in reqs:
142 + cols = r.get("column") or []
143 + loc_idx = (r.get("locationsColumns") or [1])[0]
144 + title = str(cols[0]) if cols else ""
145 + locations = self._parse_locations(
146 + str(cols[loc_idx])) if len(cols) > loc_idx else []
147 + date_raw = str(cols[-1]) if len(cols) > 2 else ""
148 + contest = str(r.get("contestNo") or r.get("jobId") or "")
149 + if not contest or contest in seen or not title:
150 + continue
151 + seen.add(contest)
152 + if not self._keep(locations):
153 + continue
154 + job = JobPosting(
155 + source=self.source_id, external_id=contest,
156 + url=(f"{self._base}/{self.SECTION}/jobdetail.ftl"
157 + f"?job={contest}&lang={self.LANG}"),
158 + employer=self.EMPLOYER,
159 + title=title,
160 + location_label=locations[0].replace("-", ", ")
161 + if locations else "",
162 + date_posted=date_raw or None,
163 + ats=self.ats,
164 + )
165 + key = hashlib.sha1(
166 + f"{title}|{date_raw}".encode()).hexdigest()[:12]
167 + if details_used < MAX_DETAILS:
168 + fresh = [False]
169 +
170 + def _fn(c=contest, fresh=fresh):
171 + fresh[0] = True
172 + return self._fetch_detail(c)
173 +
174 + d = self.detail(contest, key, _fn)
175 + if fresh[0]:
176 + details_used += 1
177 + job.description = d.get("description", "")
178 + out.append(job)
179 + paging = data.get("pagingData") or {}
180 + total = int(paging.get("totalCount") or 0)
181 + size = int(paging.get("pageSize") or len(reqs) or 25)
182 + if total and page * size >= total:
183 + break
184 + page += 1
185 + return out
added jobka/connectors/ultipro.py +162 −0
@@ -0,0 +1,162 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/ultipro.py
6 +# Rôle : Classe de plateforme UKG Pro Recruiting (UltiPro) — API JSON
7 +# publique LoadSearchResults, un employeur = une sous-classe
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +"""Plateforme UKG Pro Recruiting (recruiting.ultipro.com).
11 +
12 +API JSON publique (celle du frontal) :
13 +- liste : POST https://recruiting.ultipro.com/<ORG>/JobBoard/<BOARD>/
14 + JobBoardView/LoadSearchResults
15 + body {opportunitySearch:{Top, Skip, OrderBy…}}
16 + -> {opportunities:[{Id, Title, RequisitionNumber, BriefDescription,
17 + FullTime, JobCategoryName, PostedDate, Locations:[{Address}]}],
18 + totalCount}
19 +- détail : POST .../JobBoardView/LoadOpportunity {opportunityId} -> Description
20 + complète (HTML). Visité avec cache BD + budget.
21 +"""
22 +from __future__ import annotations
23 +
24 +import os
25 +import re
26 +
27 +from ..schema import JobPosting, clean_html, is_quebec_location
28 +from .base import BaseConnector
29 +
30 +PAGE_SIZE = 50
31 +MAX_DETAILS = int(os.environ.get("JOBKA_ULTIPRO_DETAIL_LIMIT", "80"))
32 +
33 +
34 +class UltiProConnector(BaseConnector):
35 + """Base UltiPro — sous-classes : définir source_id, EMPLOYER, ORG, BOARD."""
36 +
37 + ats = "ultipro"
38 + request_delay = 0.8
39 +
40 + EMPLOYER = ""
41 + ORG = "" # ex. MES1005MESR
42 + BOARD = "" # GUID du JobBoard
43 + quebec_only = True
44 + max_pages = 20
45 +
46 + @property
47 + def _board_url(self) -> str:
48 + return f"https://recruiting.ultipro.com/{self.ORG}/JobBoard/{self.BOARD}"
49 +
50 + @staticmethod
51 + def _loc_fields(loc: dict) -> tuple[str, str, str, str]:
52 + """(ville, province, code postal, libellé) d'un objet Location."""
53 + addr = loc.get("Address") or {}
54 + city = addr.get("City") or ""
55 + state = ((addr.get("State") or {}).get("Code")
56 + if isinstance(addr.get("State"), dict)
57 + else addr.get("State")) or ""
58 + postal = addr.get("PostalCode") or ""
59 + label = loc.get("LocalizedDescription") or city
60 + return city, str(state), postal, label
61 +
62 + def _keep(self, locations: list[dict]) -> bool:
63 + if not self.quebec_only:
64 + return True
65 + for loc in locations:
66 + city, state, postal, label = self._loc_fields(loc)
67 + if state.upper() == "QC" or re.match(r"^[GHJ]\d[A-Z]", postal.upper()):
68 + return True
69 + if is_quebec_location(f"{label} {city}"):
70 + return True
71 + return False
72 +
73 + _DESC_RE = re.compile(r'"Description"\s*:\s*"((?:[^"\\]|\\.)*)"')
74 +
75 + def _fetch_detail(self, opp_id: str) -> dict:
76 + """La page OpportunityDetail embarque l'objet JSON de l'offre —
77 + on en extrait la description complète (chaîne JSON échappée)."""
78 + html = self.get(f"{self._board_url}/OpportunityDetail",
79 + params={"opportunityId": opp_id}).text
80 + best = ""
81 + for m in self._DESC_RE.finditer(html):
82 + try:
83 + import json as _json
84 + val = _json.loads(f'"{m.group(1)}"')
85 + except ValueError:
86 + continue
87 + if len(val) > len(best):
88 + best = val
89 + return {"description": clean_html(best)}
90 +
91 + def fetch(self) -> list[JobPosting]:
92 + url = f"{self._board_url}/JobBoardView/LoadSearchResults"
93 + out: list[JobPosting] = []
94 + details_used = 0
95 + skip = 0
96 + for _ in range(self.max_pages):
97 + body = {
98 + "opportunitySearch": {
99 + "Top": PAGE_SIZE, "Skip": skip, "QueryString": "",
100 + "OrderBy": [{"Value": "postedDateDesc",
101 + "PropertyName": "PostedDate",
102 + "Ascending": False}],
103 + "Filters": [],
104 + },
105 + "matchCriteria": {"PreferredJobs": [], "Educations": [],
106 + "LicenseAndCertifications": [], "Skills": [],
107 + "hasNoLicenses": False, "SkippedSkills": []},
108 + }
109 + data = self.post(url, json=body,
110 + headers={"Accept": "application/json"}).json()
111 + opps = data.get("opportunities") or []
112 + if not opps:
113 + break
114 + for o in opps:
115 + locations = o.get("Locations") or []
116 + if not self._keep(locations):
117 + continue
118 + oid = str(o.get("Id") or "")
119 + if not oid:
120 + continue
121 + city = label = postal = ""
122 + for loc in locations:
123 + c, s, p, lbl = self._loc_fields(loc)
124 + if s.upper() == "QC" or is_quebec_location(f"{lbl} {c}"):
125 + city, postal, label = c or lbl, p, lbl
126 + break
127 + job = JobPosting(
128 + source=self.source_id, external_id=oid,
129 + url=f"{self._board_url}/OpportunityDetail?opportunityId={oid}",
130 + employer=self.EMPLOYER,
131 + title=o.get("Title") or "",
132 + city=city, postal_code=postal, location_label=label,
133 + date_posted=o.get("PostedDate") or None,
134 + ats=self.ats,
135 + )
136 + job.details["requisition"] = o.get("RequisitionNumber") or ""
137 + if o.get("JobCategoryName"):
138 + job.details["team"] = o["JobCategoryName"]
139 + if o.get("FullTime") is True:
140 + job.details["employment_label"] = "Temps plein"
141 + elif o.get("FullTime") is False:
142 + job.details["employment_label"] = "Temps partiel"
143 + key = (o.get("PostedDate") or "") + (o.get("Title") or "")[:40]
144 + if details_used < MAX_DETAILS:
145 + fresh = [False]
146 +
147 + def _fn(i=oid, fresh=fresh):
148 + fresh[0] = True
149 + return self._fetch_detail(i)
150 +
151 + d = self.detail(oid, key, _fn)
152 + if fresh[0]:
153 + details_used += 1
154 + job.description = d.get("description", "")
155 + if not job.description:
156 + job.description = clean_html(o.get("BriefDescription") or "")
157 + out.append(job)
158 + skip += PAGE_SIZE
159 + total = int(data.get("totalCount") or 0)
160 + if total and skip >= total:
161 + break
162 + return out
added jobka/connectors/ville_gatineau.py +19 −0
@@ -0,0 +1,19 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/ville_gatineau.py
6 +# Rôle : Connecteur Ville de Gatineau — njoyn (base='https://gatineau.njoyn.com', cl='CL2')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .njoyn import NjoynConnector
11 +
12 +
13 +class VilleGatineauConnector(NjoynConnector):
14 + source_id = 'ville_gatineau'
15 + EMPLOYER = 'Ville de Gatineau'
16 + BASE = 'https://gatineau.njoyn.com'
17 + CL = 'CL2'
18 + CLID = '27082'
19 + USE_SCRAPFLY = True
added jobka/connectors/ville_terrebonne.py +19 −0
@@ -0,0 +1,19 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/ville_terrebonne.py
6 +# Rôle : Connecteur Ville de Terrebonne — njoyn (base='https://clients.njoyn.com', cl='cl4')
7 +# [généré par scripts/gen_connectors.py, flux vérifié le 2026-08-18]
8 +# Créé : 2026-08-18 Modifié : 2026-08-18
9 +# =============================================================================
10 +from .njoyn import NjoynConnector
11 +
12 +
13 +class VilleTerrebonneConnector(NjoynConnector):
14 + source_id = 'ville_terrebonne'
15 + EMPLOYER = 'Ville de Terrebonne'
16 + BASE = 'https://clients.njoyn.com'
17 + CL = 'cl4'
18 + CLID = '71764'
19 + USE_SCRAPFLY = True
added jobka/connectors/workland.py +90 −0
@@ -0,0 +1,90 @@
1 +# =============================================================================
2 +# Job·Ka — Groupe KA
3 +# Auteur : Simon-Pierre Boucher
4 +# Contact : contact@spboucher.ai
5 +# Fichier : jobka/connectors/workland.py
6 +# Rôle : Classe de plateforme Workland / Atlas (ATS québécois utilisé par
7 +# les centres de services scolaires, PME…) — liste rendue côté
8 +# serveur sur la page carrières de l'employeur
9 +# Créé : 2026-08-18 Modifié : 2026-08-18
10 +# =============================================================================
11 +"""Plateforme Workland (atlas.workland.com).
12 +
13 +Les pages atlas.workland.com sont une SPA Angular sans données côté serveur
14 +(le bloc JSON-LD reste vide même après rendu) : on s'appuie donc sur la page
15 +carrières de l'employeur (LIST_URL), rendue côté serveur, qui liste les liens
16 +https://atlas.workland.com/work/<id>/<slug> — typiquement dans un tableau
17 +« titre | date limite | lien » (ex. CSSDM). Le titre et la date limite
18 +proviennent de la ligne du tableau ; à défaut, du slug.
19 +"""
20 +from __future__ import annotations
21 +
22 +import html as _html
23 +import re
24 +
25 +from ..schema import JobPosting
26 +from .base import BaseConnector
27 +
28 +_WORK_RE = re.compile(
29 + r"https?://atlas\.workland\.com/work/(\d+)/([a-z0-9-]+)", re.I)
30 +_ROW_RE = re.compile(r"<tr[^>]*>(.*?)</tr>", re.S | re.I)
31 +_TD_RE = re.compile(r"<td[^>]*>(.*?)</td>", re.S | re.I)
32 +_DATE_RE = re.compile(r"\d{1,2}(?:er)?\s+[a-zû]{3,9}\.?\s+\d{4}|\d{4}-\d{2}-\d{2}",
33 + re.I)
34 +
35 +
36 +def _txt(s: str) -> str:
37 + return re.sub(r"\s+", " ",
38 + _html.unescape(re.sub(r"<[^>]+>", " ", s or ""))).strip()
39 +
40 +
41 +class WorklandConnector(BaseConnector):
42 + """Base Workland — sous-classes : définir source_id, EMPLOYER, LIST_URL
43 + (page carrières de l'employeur qui liste les liens atlas)."""
44 +
45 + ats = "workland"
46 + request_delay = 1.0
47 +
48 + EMPLOYER = ""
49 + LIST_URL = ""
50 + DEFAULT_CITY = "" # ville par défaut (ex. « Montréal » pour le CSSDM)
51 +
52 + def fetch(self) -> list[JobPosting]:
53 + page = self.get(self.LIST_URL).text
54 + out: list[JobPosting] = []
55 + seen: set[str] = set()
56 +
57 + # 1) lignes de tableau « titre | date limite | lien atlas »
58 + for row in _ROW_RE.findall(page):
59 + m = _WORK_RE.search(row)
60 + if not m or m.group(1) in seen:
61 + continue
62 + tds = [_txt(td) for td in _TD_RE.findall(row)]
63 + title = next((t for t in tds if t and not _DATE_RE.fullmatch(t)), "")
64 + deadline = next((t for t in tds if t and _DATE_RE.fullmatch(t)), None)
65 + if not title:
66 + title = m.group(2).replace("-", " ").capitalize()
67 + seen.add(m.group(1))
68 + out.append(JobPosting(
69 + source=self.source_id, external_id=m.group(1),
70 + url=f"https://atlas.workland.com/work/{m.group(1)}/{m.group(2)}",
71 + employer=self.EMPLOYER, title=title,
72 + city=self.DEFAULT_CITY,
73 + date_deadline=deadline,
74 + ats=self.ats,
75 + ))
76 +
77 + # 2) liens atlas hors tableau (autres habillages)
78 + for wid, slug in _WORK_RE.findall(page):
79 + if wid in seen:
80 + continue
81 + seen.add(wid)
82 + out.append(JobPosting(
83 + source=self.source_id, external_id=wid,
84 + url=f"https://atlas.workland.com/work/{wid}/{slug}",
85 + employer=self.EMPLOYER,
86 + title=slug.replace("-", " ").capitalize(),
87 + city=self.DEFAULT_CITY,
88 + ats=self.ats,
89 + ))
90 + return out
modified scripts/gen_connectors.py +39 −0
@@ -54,6 +54,20 @@ ORG_ATTR = {"lever": "ORG", "greenhouse": "BOARD",
54 54 "smartrecruiters": "COMPANY", "ashby": "ORG", "workable": "ORG",
55 55 "recruitee": "ORG", "breezy": "ORG", "bamboohr": "ORG"}
56 56
57 +# ATS à attributs multiples (secteur public/parapublic et grands employeurs) :
58 +# feed[clé] -> attribut de classe. Voir data/feeds-public.json.
59 +EXTRA_ATTRS = {
60 + "taleo": [("host", "HOST"), ("section", "SECTION"), ("portal", "PORTAL")],
61 + "njoyn": [("base", "BASE"), ("cl", "CL"), ("clid", "CLID"),
62 + ("use_scrapfly", "USE_SCRAPFLY")],
63 + "ultipro": [("org", "ORG"), ("board", "BOARD")],
64 + "icims": [("sub", "SUB")],
65 + "successfactors": [("base", "BASE")],
66 + "adp": [("cid", "CID"), ("ccid", "CCID")],
67 + "digitalrecruiters": [("base", "BASE")],
68 + "workland": [("list_url", "LIST_URL")],
69 +}
70 +
57 71 CAREERS_URL = {
58 72 "workday": "https://{tenant}.{host}.myworkdayjobs.com/{site}",
59 73 "lever": "https://jobs.lever.co/{org}",
@@ -64,8 +78,28 @@ CAREERS_URL = {
64 78 "recruitee": "https://{org}.recruitee.com",
65 79 "breezy": "https://{org}.breezy.hr",
66 80 "bamboohr": "https://{org}.bamboohr.com/careers",
81 + "taleo": "https://{host}.taleo.net/careersection/{section}/jobsearch.ftl?lang=fr",
82 + "njoyn": "{base}/{cl}/xweb/xweb.asp?clid={clid}&page=joblisting&lang=2",
83 + "ultipro": "https://recruiting.ultipro.com/{org}/JobBoard/{board}",
84 + "icims": "https://{sub}.icims.com/jobs/search?ss=1",
85 + "successfactors": "{base}",
86 + "adp": ("https://workforcenow.adp.com/mascsr/default/mdf/recruitment/"
87 + "recruitment.html?cid={cid}&ccId={ccid}&lang=fr_CA"),
88 + "digitalrecruiters": "{base}",
89 + "workland": "{list_url}",
67 90 }
68 91
92 +PLATFORMS.update({
93 + "taleo": ("taleo", "TaleoConnector"),
94 + "njoyn": ("njoyn", "NjoynConnector"),
95 + "ultipro": ("ultipro", "UltiProConnector"),
96 + "icims": ("icims", "ICIMSConnector"),
97 + "successfactors": ("successfactors", "SuccessFactorsConnector"),
98 + "adp": ("adp", "ADPWorkforceNowConnector"),
99 + "digitalrecruiters": ("digitalrecruiters", "DigitalRecruitersConnector"),
100 + "workland": ("workland", "WorklandConnector"),
101 +})
102 +
69 103
70 104 def slugify(s: str) -> str:
71 105 s = unicodedata.normalize("NFD", s)
@@ -110,6 +144,11 @@ def gen_one(feed: dict) -> tuple[str, str] | None:
110 144 body = (f" TENANT = {feed['tenant']!r}\n"
111 145 f" HOST = {feed['host']!r}\n"
112 146 f" SITE = {feed['site']!r}\n")
147 + elif ats in EXTRA_ATTRS:
148 + pairs = [(k, a) for k, a in EXTRA_ATTRS[ats]
149 + if feed.get(k) not in (None, "")]
150 + detail = ", ".join(f"{a.lower()}={feed[k]!r}" for k, a in pairs[:2])
151 + body = "".join(f" {a} = {feed[k]!r}\n" for k, a in pairs)
113 152 else:
114 153 attr = ORG_ATTR[ats]
115 154 detail = f"org « {feed['org']} »"
116 155