SPB Git forge

spb/lou-ka

Public

Lou·Ka — tous les logements à louer du Québec, un seul endroit.

232commits 1branches 0releases
172.9 MBsize
maindefault branch
2 days agolast push
HTML 98.9% Python 0.6%

Court terme : upgrade majeur — 21 nouveaux connecteurs, 12 enrichis, fiches améliorées

- 21 nouveaux connecteurs (chaleto, hébergement-charlevoix, chaletsalpins,
  memoria, pourvoiries FPQ, campingquebec, hebergia, rezerve, accestremblant,
  chaletsdanslenord, alouerauxiles, rvmt, boraboreal, homminichalets…)
- enrichissement des existants : airbnb (desc/capacité/commodités via PDP),
  booking/vrbo/expedia (Apollo SSR, bug slug vrbo corrigé), sepaq (desc+géo),
  tremblantliving (prix API Streamline), qldc/kijiji/parcscanada/bonjourquebec/
  gitespassant ; sinistar : prix vérifiés inaccessibles (documenté)
- fiches CT : /api/ct/listings/{uid}/context (analyse prix/nuit par segment,
  histogramme, centile) + badge marché + hébergements similaires à proximité
- registre sources_ct.json : 52 sources (7 mortes retirées, 21 ajoutées)

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
Simon-Pierre Boucher committed 1 mo ago (Aug 25, 2026) parent bf51f56

41 changed files +5,841 −109

modified data/sources_ct.json +324 −34
@@ -1,35 +1,325 @@
1 1 {
2 − "sources": [
3 − { "id": "airbnb", "name": "Airbnb", "url": "https://www.airbnb.ca", "category": "internationale" },
4 − { "id": "booking", "name": "Booking.com", "url": "https://www.booking.com", "category": "internationale" },
5 − { "id": "vrbo", "name": "Vrbo", "url": "https://www.vrbo.com", "category": "internationale" },
6 − { "id": "expedia", "name": "Expedia (locations de vacances)", "url": "https://www.expedia.ca", "category": "internationale" },
7 − { "id": "agoda", "name": "Agoda Homes", "url": "https://www.agoda.com", "category": "internationale", "status": "en attente de connecteur" },
8 − { "id": "plumguide", "name": "Plum Guide", "url": "https://www.plumguide.com", "category": "internationale", "status": "en attente de connecteur" },
9 − { "id": "glampinghub", "name": "Glamping Hub", "url": "https://glampinghub.com", "category": "internationale" },
10 − { "id": "hipcamp", "name": "Hipcamp", "url": "https://www.hipcamp.com", "category": "internationale" },
11 − { "id": "marriott_hvmi", "name": "Marriott Homes & Villas", "url": "https://homes-and-villas.marriott.com", "category": "internationale", "status": "en attente de connecteur" },
12 − { "id": "furnishedfinder", "name": "Furnished Finder", "url": "https://www.furnishedfinder.com", "category": "internationale", "status": "en attente de connecteur" },
13 − { "id": "corporatestays", "name": "Corporate Stays", "url": "https://www.corporatestays.com", "category": "internationale" },
14 − { "id": "monsieurchalets", "name": "MonsieurChalets", "url": "https://www.monsieurchalets.com", "category": "quebecoise" },
15 − { "id": "wechalet", "name": "WeChalet", "url": "https://wechalet.com", "category": "quebecoise" },
16 − { "id": "chaletsarabais", "name": "Chalets à Rabais", "url": "https://www.chaletsarabais.com", "category": "quebecoise" },
17 − { "id": "chaletsalouer", "name": "ChaletsÀLouer.com", "url": "https://www.chaletsalouer.com", "category": "quebecoise" },
18 − { "id": "mcal", "name": "Maisons et chalets à louer", "url": "https://maisonsetchaletsalouer.com", "category": "quebecoise" },
19 − { "id": "qldc", "name": "Québec Location de Chalets", "url": "https://www.quebeclocationdechalets.com", "category": "quebecoise" },
20 − { "id": "rsvpchalets", "name": "RSVP Chalets", "url": "https://rsvpchalets.com", "category": "quebecoise" },
21 − { "id": "chaletsauquebec", "name": "Chalets au Québec", "url": "https://www.chaletsauquebec.com", "category": "quebecoise" },
22 − { "id": "tremblant_living", "name": "Tremblant Living", "url": "https://www.tremblantliving.com", "category": "agence" },
23 − { "id": "gites_passant", "name": "Gîtes et Auberges du Passant", "url": "https://www.terroiretsaveurs.com", "category": "gites" },
24 − { "id": "bonjourquebec", "name": "Bonjour Québec (CITQ)", "url": "https://www.bonjourquebec.com", "category": "officielle" },
25 − { "id": "sepaq", "name": "Sépaq", "url": "https://www.sepaq.com", "category": "officielle" },
26 − { "id": "parcscanada", "name": "Parcs Canada", "url": "https://reservation.pc.gc.ca", "category": "officielle" },
27 − { "id": "sinistar", "name": "Sinistar", "url": "https://www.sinistar.ca", "category": "relogement" },
28 − { "id": "louer_ca_ct", "name": "Louer.ca (court terme)", "url": "https://www.louer.ca", "category": "relogement", "status": "en attente de connecteur" },
29 − { "id": "kijiji_ct", "name": "Kijiji (court terme)", "url": "https://www.kijiji.ca", "category": "petites-annonces" },
30 − { "id": "lespac_ct", "name": "LesPAC (court terme)", "url": "https://www.lespac.com", "category": "petites-annonces", "status": "en attente de connecteur" },
31 − { "id": "fb_marketplace_ct", "name": "Facebook Marketplace (court terme)", "url": "https://www.facebook.com/marketplace", "category": "petites-annonces", "status": "en attente de connecteur" },
32 − { "id": "craigslist_ct", "name": "Craigslist (court terme)", "url": "https://montreal.craigslist.org", "category": "petites-annonces", "status": "en attente de connecteur" },
33 − { "id": "bonjourresidences", "name": "Bonjour Résidences", "url": "https://www.bonjourresidences.com", "category": "cas-particuliers", "status": "en attente de connecteur" }
34 − ]
35 −}
2 + "sources": [
3 + {
4 + "id": "airbnb",
5 + "name": "Airbnb",
6 + "url": "https://www.airbnb.ca",
7 + "category": "internationale"
8 + },
9 + {
10 + "id": "booking",
11 + "name": "Booking.com",
12 + "url": "https://www.booking.com",
13 + "category": "internationale"
14 + },
15 + {
16 + "id": "vrbo",
17 + "name": "Vrbo",
18 + "url": "https://www.vrbo.com",
19 + "category": "internationale"
20 + },
21 + {
22 + "id": "expedia",
23 + "name": "Expedia (locations de vacances)",
24 + "url": "https://www.expedia.ca",
25 + "category": "internationale"
26 + },
27 + {
28 + "id": "agoda",
29 + "name": "Agoda Homes",
30 + "url": "https://www.agoda.com",
31 + "category": "internationale",
32 + "status": "en attente de connecteur"
33 + },
34 + {
35 + "id": "plumguide",
36 + "name": "Plum Guide",
37 + "url": "https://www.plumguide.com",
38 + "category": "internationale",
39 + "status": "en attente de connecteur"
40 + },
41 + {
42 + "id": "glampinghub",
43 + "name": "Glamping Hub",
44 + "url": "https://glampinghub.com",
45 + "category": "internationale"
46 + },
47 + {
48 + "id": "hipcamp",
49 + "name": "Hipcamp",
50 + "url": "https://www.hipcamp.com",
51 + "category": "internationale"
52 + },
53 + {
54 + "id": "marriott_hvmi",
55 + "name": "Marriott Homes & Villas",
56 + "url": "https://homes-and-villas.marriott.com",
57 + "category": "internationale",
58 + "status": "en attente de connecteur"
59 + },
60 + {
61 + "id": "furnishedfinder",
62 + "name": "Furnished Finder",
63 + "url": "https://www.furnishedfinder.com",
64 + "category": "internationale",
65 + "status": "en attente de connecteur"
66 + },
67 + {
68 + "id": "corporatestays",
69 + "name": "Corporate Stays",
70 + "url": "https://www.corporatestays.com",
71 + "category": "internationale"
72 + },
73 + {
74 + "id": "monsieurchalets",
75 + "name": "MonsieurChalets",
76 + "url": "https://www.monsieurchalets.com",
77 + "category": "quebecoise"
78 + },
79 + {
80 + "id": "wechalet",
81 + "name": "WeChalet",
82 + "url": "https://wechalet.com",
83 + "category": "quebecoise"
84 + },
85 + {
86 + "id": "chaletsarabais",
87 + "name": "Chalets à Rabais",
88 + "url": "https://www.chaletsarabais.com",
89 + "category": "quebecoise"
90 + },
91 + {
92 + "id": "chaletsalouer",
93 + "name": "ChaletsÀLouer.com",
94 + "url": "https://www.chaletsalouer.com",
95 + "category": "quebecoise"
96 + },
97 + {
98 + "id": "mcal",
99 + "name": "Maisons et chalets à louer",
100 + "url": "https://maisonsetchaletsalouer.com",
101 + "category": "quebecoise"
102 + },
103 + {
104 + "id": "qldc",
105 + "name": "Québec Location de Chalets",
106 + "url": "https://www.quebeclocationdechalets.com",
107 + "category": "quebecoise"
108 + },
109 + {
110 + "id": "rsvpchalets",
111 + "name": "RSVP Chalets",
112 + "url": "https://rsvpchalets.com",
113 + "category": "quebecoise"
114 + },
115 + {
116 + "id": "chaletsauquebec",
117 + "name": "Chalets au Québec",
118 + "url": "https://www.chaletsauquebec.com",
119 + "category": "quebecoise"
120 + },
121 + {
122 + "id": "tremblant_living",
123 + "name": "Tremblant Living",
124 + "url": "https://www.tremblantliving.com",
125 + "category": "agence"
126 + },
127 + {
128 + "id": "gites_passant",
129 + "name": "Gîtes et Auberges du Passant",
130 + "url": "https://www.terroiretsaveurs.com",
131 + "category": "gites"
132 + },
133 + {
134 + "id": "bonjourquebec",
135 + "name": "Bonjour Québec (CITQ)",
136 + "url": "https://www.bonjourquebec.com",
137 + "category": "officielle"
138 + },
139 + {
140 + "id": "sepaq",
141 + "name": "Sépaq",
142 + "url": "https://www.sepaq.com",
143 + "category": "officielle"
144 + },
145 + {
146 + "id": "parcscanada",
147 + "name": "Parcs Canada",
148 + "url": "https://reservation.pc.gc.ca",
149 + "category": "officielle"
150 + },
151 + {
152 + "id": "sinistar",
153 + "name": "Sinistar",
154 + "url": "https://www.sinistar.ca",
155 + "category": "relogement"
156 + },
157 + {
158 + "id": "louer_ca_ct",
159 + "name": "Louer.ca (court terme)",
160 + "url": "https://www.louer.ca",
161 + "category": "relogement",
162 + "status": "en attente de connecteur"
163 + },
164 + {
165 + "id": "kijiji_ct",
166 + "name": "Kijiji (court terme)",
167 + "url": "https://www.kijiji.ca",
168 + "category": "petites-annonces"
169 + },
170 + {
171 + "id": "lespac_ct",
172 + "name": "LesPAC (court terme)",
173 + "url": "https://www.lespac.com",
174 + "category": "petites-annonces",
175 + "status": "en attente de connecteur"
176 + },
177 + {
178 + "id": "fb_marketplace_ct",
179 + "name": "Facebook Marketplace (court terme)",
180 + "url": "https://www.facebook.com/marketplace",
181 + "category": "petites-annonces",
182 + "status": "en attente de connecteur"
183 + },
184 + {
185 + "id": "craigslist_ct",
186 + "name": "Craigslist (court terme)",
187 + "url": "https://montreal.craigslist.org",
188 + "category": "petites-annonces",
189 + "status": "en attente de connecteur"
190 + },
191 + {
192 + "id": "bonjourresidences",
193 + "name": "Bonjour Résidences",
194 + "url": "https://www.bonjourresidences.com",
195 + "category": "cas-particuliers",
196 + "status": "en attente de connecteur"
197 + },
198 + {
199 + "id": "chaleto",
200 + "name": "Chaletô",
201 + "url": "https://chaleto.ca",
202 + "category": "quebecoise"
203 + },
204 + {
205 + "id": "hebergia",
206 + "name": "Hebergia",
207 + "url": "https://hebergia.ca",
208 + "category": "quebecoise"
209 + },
210 + {
211 + "id": "rezerve",
212 + "name": "Rëzerve",
213 + "url": "https://reserver.ca",
214 + "category": "quebecoise"
215 + },
216 + {
217 + "id": "campingquebec",
218 + "name": "Camping Québec",
219 + "url": "https://www.campingquebec.com",
220 + "category": "officielle"
221 + },
222 + {
223 + "id": "pourvoiries",
224 + "name": "Pourvoiries du Québec (FPQ)",
225 + "url": "https://www.pourvoiries.com",
226 + "category": "officielle"
227 + },
228 + {
229 + "id": "gitesauquebec",
230 + "name": "Gîtes au Québec",
231 + "url": "https://www.gitesauquebec.com",
232 + "category": "gites"
233 + },
234 + {
235 + "id": "hebergementcharlevoix",
236 + "name": "Hébergement Charlevoix",
237 + "url": "https://www.hebergement-charlevoix.com",
238 + "category": "agence"
239 + },
240 + {
241 + "id": "chaletsalpins",
242 + "name": "Les Chalets Alpins",
243 + "url": "https://chaletsalpins.ca",
244 + "category": "agence"
245 + },
246 + {
247 + "id": "memoriachalets",
248 + "name": "Memoria Chalets",
249 + "url": "https://memoriachalets.com",
250 + "category": "agence"
251 + },
252 + {
253 + "id": "accestremblant",
254 + "name": "Accès Tremblant",
255 + "url": "https://accestremblant.ca",
256 + "category": "agence"
257 + },
258 + {
259 + "id": "chaletsdanslenord",
260 + "name": "Les Chalets dans le Nord",
261 + "url": "https://leschaletsdanslenord.com",
262 + "category": "agence"
263 + },
264 + {
265 + "id": "alouerauxiles",
266 + "name": "À louer aux Îles",
267 + "url": "https://www.alouerauxiles.com",
268 + "category": "agence"
269 + },
270 + {
271 + "id": "lesversants",
272 + "name": "Les Versants Mont-Tremblant",
273 + "url": "https://www.lesversants.com",
274 + "category": "agence"
275 + },
276 + {
277 + "id": "captremblant",
278 + "name": "Cap Tremblant",
279 + "url": "https://captremblant.com",
280 + "category": "agence"
281 + },
282 + {
283 + "id": "rvmt",
284 + "name": "Rendez-vous Mont-Tremblant",
285 + "url": "https://www.rvmt.com",
286 + "category": "agence"
287 + },
288 + {
289 + "id": "locationdechalets",
290 + "name": "Location de Chalets 4 Saisons",
291 + "url": "https://www.locationdechalets.com",
292 + "category": "agence"
293 + },
294 + {
295 + "id": "chaletsbsl",
296 + "name": "Chalets BSL",
297 + "url": "https://www.chaletsbsl.com",
298 + "category": "agence"
299 + },
300 + {
301 + "id": "panoraloges",
302 + "name": "Panora Loges",
303 + "url": "https://www.panoraloges.ca",
304 + "category": "agence"
305 + },
306 + {
307 + "id": "domesstcome",
308 + "name": "Dômes St-Côme",
309 + "url": "https://www.domesstcome.com",
310 + "category": "agence"
311 + },
312 + {
313 + "id": "boraboreal",
314 + "name": "Bora Boréal",
315 + "url": "https://www.boraboreal.com",
316 + "category": "agence"
317 + },
318 + {
319 + "id": "homminichalets",
320 + "name": "HOM Mini-Chalets",
321 + "url": "https://homminichalets.com",
322 + "category": "agence"
323 + }
324 + ]
325 +}
\ No newline at end of file
added frontend/src/components/CtPriceAnalysis.tsx +96 −0
@@ -0,0 +1,96 @@
1 +// -----------------------------------------------------------------------------
2 +// Lou-Ka — Location court terme
3 +// components/CtPriceAnalysis.tsx : bloc « Analyse du prix / nuit » (fiche CT)
4 +// Position du tarif dans la distribution des hébergements comparables
5 +// (segment : type + capacité + région touristique, calculé par l'API).
6 +// -----------------------------------------------------------------------------
7 +import { CtPriceContext, fmtNight } from "../ctapi";
8 +
9 +export function ctDealBadge(p: CtPriceContext | null) {
10 + if (!p || p.deviation == null || p.verdict == null) return null;
11 + const pct = Math.round(Math.abs(p.deviation) * 100);
12 + if (p.verdict === "sous")
13 + return { cls: "deal-good", txt: `≈ ${pct} % sous le prix médian du segment` };
14 + if (p.verdict === "dessus")
15 + return { cls: "deal-high", txt: `≈ ${pct} % au-dessus du prix médian` };
16 + return { cls: "deal-ok", txt: "Dans les prix du segment" };
17 +}
18 +
19 +function Histo({ p, price }: { p: CtPriceContext; price: number }) {
20 + const bins = p.histogram;
21 + if (!bins || bins.length < 4) return null;
22 + const W = 320, H = 96, top = 18, bottom = 16;
23 + const lo = bins[0].x0, hi = bins[bins.length - 1].x1;
24 + if (hi <= lo) return null;
25 + const max = Math.max(...bins.map((b) => b.n), 1);
26 + const x = (v: number) => ((Math.min(Math.max(v, lo), hi) - lo) / (hi - lo)) * W;
27 + const bw = W / bins.length;
28 + const plotH = H - top - bottom;
29 + const priceX = x(price), medX = x(p.median);
30 + const close = Math.abs(priceX - medX) < 64;
31 + const lblAnchor = (px: number) => (px < 56 ? "start" : px > W - 56 ? "end" : "middle");
32 + return (
33 + <svg className="fv-histo" viewBox={`0 0 ${W} ${H}`} role="img"
34 + aria-label={`Position du prix (${price} $/nuit) parmi ${p.n} hébergements comparables`}>
35 + {bins.map((b, i) => {
36 + const h = Math.max(1.5, (b.n / max) * plotH);
37 + const inBin = price >= b.x0 && price < b.x1;
38 + return (
39 + <rect key={i} x={i * bw + 1} y={H - bottom - h} rx="2"
40 + width={Math.max(1, bw - 2)} height={h}
41 + fill={inBin ? "var(--accent, #ff6a00)" : "rgba(204, 85, 0, 0.28)"}>
42 + <title>{`${b.x0} $ – ${b.x1} $ : ${b.n} hébergement${b.n > 1 ? "s" : ""}`}</title>
43 + </rect>
44 + );
45 + })}
46 + {/* fourchette interquartile (bande) */}
47 + <rect x={x(p.p25)} y={H - bottom} width={Math.max(2, x(p.p75) - x(p.p25))}
48 + height="3.5" rx="1.5" fill="rgba(204, 85, 0, 0.45)" />
49 + {/* marqueur médiane */}
50 + <line x1={medX} x2={medX} y1={top - 2} y2={H - bottom}
51 + stroke="var(--accent-deep, #cc5500)" strokeWidth="1.6" strokeDasharray="3 3" />
52 + {!close && (
53 + <text x={medX} y={top - 7} textAnchor={lblAnchor(medX)}
54 + className="fv-histo-lbl fv-histo-lbl-fv">Médiane</text>
55 + )}
56 + {/* marqueur du prix demandé */}
57 + <line x1={priceX} x2={priceX} y1={top - 2} y2={H - bottom}
58 + stroke="var(--ink, #141814)" strokeWidth="2" />
59 + <text x={priceX} y={close ? top - 7 : H - 4} textAnchor={lblAnchor(priceX)}
60 + className="fv-histo-lbl">Ce prix{close ? " / médiane" : ""}</text>
61 + <text x="1" y={H - 4} className="fv-histo-axis" textAnchor="start">{lo} $</text>
62 + <text x={W - 1} y={H - 4} className="fv-histo-axis" textAnchor="end">{hi} $</text>
63 + </svg>
64 + );
65 +}
66 +
67 +export default function CtPriceAnalysis({ p, price }:
68 + { p: CtPriceContext | null; price: number | null }) {
69 + if (!p || price == null) return null;
70 + const badge = ctDealBadge(p);
71 + return (
72 + <section className="f-bloc f-fairvalue" id="analyse-prix">
73 + <h2>Analyse du prix Lou-Ka</h2>
74 + <div className="fv-head">
75 + <div>
76 + <div className="fv-value">{fmtNight(p.median)} <small>/ nuit</small></div>
77 + <div className="fv-range">
78 + Prix médian du segment · 50 % des prix entre {fmtNight(p.p25)} et {fmtNight(p.p75)}
79 + </div>
80 + </div>
81 + {badge && <div className={`fv-badge ${badge.cls}`}>{badge.txt}</div>}
82 + </div>
83 + <Histo p={p} price={price} />
84 + <div className="fv-meta">
85 + <span>Segment : <b>{p.segment}</b></span>
86 + <span>{p.n.toLocaleString("fr-CA")} hébergements comparables</span>
87 + <span>Moins cher que {100 - p.percentile} % du segment</span>
88 + </div>
89 + <p className="fine">
90 + Comparaison indicative des tarifs affichés (« à partir de ») par les
91 + sources pour des hébergements du même type dans la même région — les
92 + prix réels varient selon les dates et la durée du séjour.
93 + </p>
94 + </section>
95 + );
96 +}
modified frontend/src/ctapi.ts +19 −0
@@ -87,6 +87,23 @@ export interface CtFilters {
87 87 sort?: string; // recent | prix | prix_desc | note
88 88 }
89 89
90 +export interface CtPriceContext {
91 + segment: string;
92 + n: number;
93 + median: number;
94 + p25: number;
95 + p75: number;
96 + deviation: number | null; // (prix − médiane) / médiane
97 + verdict: "sous" | "dans" | "dessus" | null;
98 + percentile: number; // position du prix dans le segment (0-100)
99 + histogram: { x0: number; x1: number; n: number }[];
100 +}
101 +
102 +export interface CtContext {
103 + price: CtPriceContext | null;
104 + similar: (CtListing & { distance_km?: number })[];
105 +}
106 +
90 107 const CT_SOURCE_NAMES: Record<string, string> = {};
91 108 export function registerCtSourceNames(sources: CtSource[]) {
92 109 for (const s of sources) CT_SOURCE_NAMES[s.id] = s.name;
@@ -118,6 +135,8 @@ export function fetchCtListings(f: CtFilters, limit?: number, offset?: number) {
118 135
119 136 export const fetchCtListing = (uid: string) =>
120 137 get<CtListing>(`/api/ct/listings/${encodeURIComponent(uid)}`);
138 +export const fetchCtContext = (uid: string) =>
139 + get<CtContext>(`/api/ct/listings/${encodeURIComponent(uid)}/context`);
121 140 export const fetchCtFacets = () => get<CtFacets>("/api/ct/facets");
122 141 export const fetchCtStats = () => get<CtStats>("/api/ct/stats");
123 142 export const fetchCtSources = () =>
modified frontend/src/pages/CourtTermeFiche.tsx +27 −2
@@ -8,10 +8,12 @@
8 8 import { lazy, Suspense, useEffect, useRef, useState } from "react";
9 9 import { Link, useParams } from "react-router-dom";
10 10 import {
11 − CtListing, ctSourceName, fetchCtListing, fetchCtSources, fmtNight,
12 − registerCtSourceNames,
11 + CtContext, CtListing, ctSourceName, fetchCtContext, fetchCtListing,
12 + fetchCtSources, fmtNight, registerCtSourceNames,
13 13 } from "../ctapi";
14 14 import SmartImg from "../components/SmartImg";
15 +import CtListingCard from "../components/CtListingCard";
16 +import CtPriceAnalysis, { ctDealBadge } from "../components/CtPriceAnalysis";
15 17 import { IcoAlert } from "../components/Icons";
16 18
17 19 const CtFicheMap = lazy(() => import("../components/CtFicheMap"));
@@ -84,12 +86,15 @@ function Galerie({ images, titre, typeLabel }:
84 86 export default function CourtTermeFichePage() {
85 87 const { uid } = useParams<{ uid: string }>();
86 88 const [l, setL] = useState<CtListing | null>(null);
89 + const [ctx, setCtx] = useState<CtContext | null>(null);
87 90 const [error, setError] = useState<string | null>(null);
88 91
89 92 useEffect(() => {
90 93 fetchCtSources().then((r) => registerCtSourceNames(r.sources)).catch(() => {});
91 94 if (!uid) return;
95 + setCtx(null);
92 96 fetchCtListing(uid).then(setL).catch((e) => setError(String(e)));
97 + fetchCtContext(uid).then(setCtx).catch(() => setCtx(null));
93 98 window.scrollTo(0, 0);
94 99 }, [uid]);
95 100
@@ -152,6 +157,12 @@ export default function CourtTermeFichePage() {
152 157 {fmtNight(l.price_night, l.price_label)}
153 158 {l.price_night != null && <small> /{NBSP}nuit</small>}
154 159 </div>
160 + {(() => {
161 + const badge = ctDealBadge(ctx?.price ?? null);
162 + return badge && (
163 + <div className={`deal-badge ${badge.cls}`}>{badge.txt}</div>
164 + );
165 + })()}
155 166 {l.rating != null && (
156 167 <div className="deal-badge deal-ok">
157 168 ★ {l.rating.toLocaleString("fr-CA", { maximumFractionDigits: 1 })} / 5
@@ -218,6 +229,7 @@ export default function CourtTermeFichePage() {
218 229 </div>
219 230
220 231 <div className="f-col">
232 + <CtPriceAnalysis p={ctx?.price ?? null} price={l.price_night} />
221 233 {l.lat != null && l.lng != null && (
222 234 <section className="f-bloc f-carte" id="emplacement">
223 235 <h2>Emplacement</h2>
@@ -232,6 +244,19 @@ export default function CourtTermeFichePage() {
232 244 </div>
233 245 </div>
234 246
247 + {ctx && ctx.similar.length > 0 && (
248 + <section className="f-bloc f-similaires" aria-label="Hébergements similaires">
249 + <h2>
250 + {l.lat != null && ctx.similar[0]?.distance_km != null
251 + ? "Hébergements similaires à proximité"
252 + : `Hébergements similaires — ${l.region || "même région"}`}
253 + </h2>
254 + <div className="grid">
255 + {ctx.similar.map((s) => <CtListingCard key={s.uid} l={s} />)}
256 + </div>
257 + </section>
258 + )}
259 +
235 260 <div className="fine f-foot">
236 261 {synced && <>Dernière synchronisation : {synced}. </>}
237 262 Les prix et disponibilités sont ceux affichés par la source — chaque fiche
modified frontend/src/styles.css +4 −0
@@ -1055,6 +1055,10 @@ html { scroll-padding-top: 76px; } /* header sticky au-dessus des ancres */
1055 1055 }
1056 1056 .f-foot { margin-top: 26px; }
1057 1057
1058 +/* --- Fiche court terme : hébergements similaires (pleine largeur) --- */
1059 +.f-similaires { margin-top: 22px; }
1060 +.f-similaires .grid { margin-top: 4px; }
1061 +
1058 1062 /* --- Passerelle vers l'annonce originale ------------------------------------ */
1059 1063 .passerelle {
1060 1064 min-height: calc(100dvh - 120px); display: flex; align-items: center;
added louka/shortterm/connectors/_expediadetail.py +155 −0
@@ -0,0 +1,155 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/_expediadetail.py : parseur partagé des pages détail de la
4 +# plateforme Expedia (Vrbo + Expedia — même moteur, même SSR).
5 +#
6 +# La page détail (…/location/pXXXXXX[vb] ou …hXXXXXX.Hotel-Information),
7 +# récupérée via Scrapfly ASP SANS rendu JS, embarque :
8 +# - __APOLLO_STATE__ (JSON.parse("…")) → PropertyInfo :
9 +# . propertyContentSectionGroups(…).aboutThisProperty → description
10 +# (blocs PropertyContentItemMarkup, HTML — inclut souvent le n° CITQ)
11 +# . summary.amenities(…) → commodités localisées (infoItems[].text)
12 +# - microdonnées schema.org SSR : lat/lng (itemProp latitude/longitude),
13 +# capacité (occupancy → value), municipalité (addressLocality)
14 +# - galerie : URLs media.vrbo.com / images.trvl-media.com (~6 photos SSR)
15 +# Vérifié live 2026-08-25 sur p3363083vb (Vrbo) et h130341845 (Expedia).
16 +# -----------------------------------------------------------------------------
17 +from __future__ import annotations
18 +
19 +import html as _html
20 +import json
21 +import re
22 +
23 +_APOLLO_RE = re.compile(
24 + r'__APOLLO_STATE__\s*=\s*JSON\.parse\("(.*?)(?<!\\)"\)', re.S)
25 +_LAT_RE = re.compile(r'itemProp="latitude" content="(-?[\d.]+)"')
26 +_LNG_RE = re.compile(r'itemProp="longitude" content="(-?[\d.]+)"')
27 +_OCC_RE = re.compile(
28 + r'itemProp="occupancy".{0,200}?itemProp="value" content="(\d+)"', re.S)
29 +_CITY_RE = re.compile(r'itemProp="addressLocality" content="([^"]+)"')
30 +_IMG_RE = re.compile(
31 + r'https://(?:media\.vrbo\.com|images\.trvl-media\.com)/lodging/'
32 + r'[^"\s\\)&?]+')
33 +
34 +
35 +def _apollo(html: str) -> dict:
36 + """Entités du store Apollo SSR ({} si absent/illisible)."""
37 + m = _APOLLO_RE.search(html or "")
38 + if not m:
39 + return {}
40 + try:
41 + # la chaîne est un littéral JSON : la re-quoter puis parser deux fois
42 + return json.loads(json.loads('"' + m.group(1) + '"'))
43 + except ValueError:
44 + return {}
45 +
46 +
47 +def _markup_texts(node, out: list[str]) -> None:
48 + """Collecte récursive des blocs PropertyContentItemMarkup (description)."""
49 + if isinstance(node, dict):
50 + if node.get("__typename") == "PropertyContentItemMarkup":
51 + txt = ((node.get("content") or {}).get("text") or "").strip()
52 + if txt:
53 + out.append(txt)
54 + return
55 + for v in node.values():
56 + _markup_texts(v, out)
57 + elif isinstance(node, list):
58 + for v in node:
59 + _markup_texts(v, out)
60 +
61 +
62 +def _strip_html(raw: str) -> str:
63 + t = re.sub(r"<br\s*/?>|</p>", "\n", raw)
64 + t = re.sub(r"<[^>]+>", " ", t)
65 + t = _html.unescape(t)
66 + lines = [re.sub(r"\s+", " ", ln).strip() for ln in t.split("\n")]
67 + return "\n".join(ln for ln in lines if ln).strip()
68 +
69 +
70 +def parse_detail(html: str) -> dict:
71 + """Payload détail {description, amenities, capacity, lat, lng, city,
72 + images} d'une page hébergement Vrbo/Expedia ({} si page invalide)."""
73 + store = _apollo(html or "")
74 + pinfo = next((v for k, v in store.items()
75 + if k.startswith("PropertyInfo") and isinstance(v, dict)), {})
76 + if not pinfo and not _LAT_RE.search(html or ""):
77 + return {} # page vide / redirection hors fiche
78 +
79 + out: dict = {}
80 +
81 + # description : sections « À propos de cet hébergement »
82 + paras: list[str] = []
83 + for k, v in pinfo.items():
84 + if k.startswith("propertyContentSectionGroups") and isinstance(v, dict):
85 + _markup_texts(v.get("aboutThisProperty"), paras)
86 + if not paras: # repli : éditorial du quartier
87 + loc = (pinfo.get("summary") or {}).get("location") or {}
88 + ed = ((loc.get("whatsAround") or {}).get("editorial") or {})
89 + paras = [t for t in (ed.get("content") or []) if isinstance(t, str)]
90 + desc = "\n\n".join(_strip_html(p) for p in paras).strip()
91 + if desc:
92 + out["description"] = desc[:6000]
93 +
94 + # commodités localisées (summary.amenities → sections → infoItems)
95 + amenities: list[str] = []
96 + for k, v in (pinfo.get("summary") or {}).items():
97 + if not (k.startswith("amenities") and isinstance(v, dict)):
98 + continue
99 + for sec in v.get("amenities") or []:
100 + for cont in (sec or {}).get("contents") or []:
101 + for it in (cont or {}).get("infoItems") or []:
102 + txt = ((it or {}).get("text") or "").strip()
103 + if txt and txt not in amenities:
104 + amenities.append(txt)
105 + if amenities:
106 + out["amenities"] = amenities[:80]
107 +
108 + # microdonnées SSR : géo, capacité, municipalité
109 + m = _LAT_RE.search(html)
110 + n = _LNG_RE.search(html)
111 + if m and n:
112 + try:
113 + out["lat"], out["lng"] = float(m.group(1)), float(n.group(1))
114 + except ValueError:
115 + pass
116 + m = _OCC_RE.search(html)
117 + if m:
118 + out["capacity"] = float(m.group(1))
119 + m = _CITY_RE.search(html)
120 + if m:
121 + out["city"] = _html.unescape(m.group(1)).strip()
122 +
123 + # galerie SSR : dédupliquée par chemin, servie en 1200 px
124 + images: list[str] = []
125 + for u in _IMG_RE.findall(html):
126 + big = u + "?impolicy=resizecrop&rw=1200&ra=fit"
127 + if big not in images:
128 + images.append(big)
129 + if len(images) >= 15:
130 + break
131 + if images:
132 + out["images"] = images
133 + return out
134 +
135 +
136 +def apply_detail(lst, d: dict) -> None:
137 + """Applique le payload détail sans écraser ce que la carte a fourni
138 + (sauf galerie : on garde la plus grande)."""
139 + if not d:
140 + return
141 + if d.get("description") and len(d["description"]) > len(lst.description or ""):
142 + lst.description = d["description"]
143 + if d.get("amenities"):
144 + seen = {a.lower() for a in lst.amenities}
145 + for a in d["amenities"]:
146 + if a.lower() not in seen:
147 + lst.amenities.append(a)
148 + seen.add(a.lower())
149 + for f in ("capacity", "lat", "lng"):
150 + if d.get(f) is not None and getattr(lst, f, None) is None:
151 + setattr(lst, f, d[f])
152 + if d.get("city") and not lst.city:
153 + lst.city = d["city"]
154 + if d.get("images") and len(d["images"]) > len(lst.images):
155 + lst.images = list(d["images"])
added louka/shortterm/connectors/accestremblant.py +173 −0
@@ -0,0 +1,173 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/accestremblant.py : Accès Tremblant (accestremblant.ca)
4 +#
5 +# Agence de condos à Mont-Tremblant (~17 fiches) — WordPress Avada
6 +# (portfolio) + moteur Guesty (guestybookings.com).
7 +#
8 +# Méthode :
9 +# 1. LISTE : sitemap https://accestremblant.ca/avada_portfolio-sitemap.xml
10 +# → /condos/<slug>/ (lastmod = clé du cache détail).
11 +# 2. DÉTAIL (cache self.detail) : la fiche WP fournit spécs (spans
12 +# icon-condos : « 8 Personnes », « 3 chambres », « 3 salles de bain » +
13 +# extras type « Foyer au gaz »), description (JSON-LD Yoast), photos
14 +# (wp-content/uploads) et l'id Guesty (lien guestybookings.com).
15 +# 3. PRIX : chaque fiche contient un carrousel « autres condos » avec
16 +# « À partir de N $ » pour les AUTRES fiches ; chaque payload détail
17 +# mémorise cette carte slug→prix et fetch() fusionne le tout (le prix
18 +# d'une fiche vient donc des autres pages, même à froid depuis le cache).
19 +# External_id = slug WP (stable) ; l'id Guesty est gardé dans details.
20 +# -----------------------------------------------------------------------------
21 +from __future__ import annotations
22 +
23 +import html as _html
24 +import json
25 +import re
26 +
27 +from ..schema import StListing
28 +from .base import StConnector
29 +
30 +SITE = "https://accestremblant.ca"
31 +SITEMAP = f"{SITE}/avada_portfolio-sitemap.xml"
32 +
33 +_TAG_RE = re.compile(r"<[^>]+>")
34 +
35 +
36 +def _text(fragment: str) -> str:
37 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
38 +
39 +
40 +class AccesTremblant(StConnector):
41 + source_id = "accestremblant"
42 +
43 + # -- liste (sitemap) ----------------------------------------------------
44 + def _sitemap(self) -> list[tuple[str, str]]:
45 + """[(url fiche, lastmod)] — dédupliqué, sans la page index."""
46 + xml = self.get(SITEMAP).text
47 + seen: dict[str, str] = {}
48 + for m in re.finditer(r"(?s)<url>\s*<loc>([^<]+)</loc>"
49 + r"(?:\s*<lastmod>([^<]+)</lastmod>)?", xml):
50 + url, lastmod = m.group(1).strip(), (m.group(2) or "").strip()
51 + if re.fullmatch(rf"{re.escape(SITE)}/condos/[^/]+/", url):
52 + seen.setdefault(url, lastmod)
53 + return sorted(seen.items())
54 +
55 + # -- page détail ------------------------------------------------------
56 + def _detail(self, url: str) -> dict:
57 + h = self.get(url).text
58 + d: dict = {}
59 +
60 + # spécs : <span class="icon-condos">… 8 Personnes / 3 chambres / …
61 + extras: list[str] = []
62 + for raw in re.findall(r'(?s)<span class="icon-condos"[^>]*>(.*?)</span>',
63 + h):
64 + t = _text(raw)
65 + if not t:
66 + continue
67 + m = re.match(r"(\d+)\s*[Pp]ersonnes?", t)
68 + if m:
69 + d["capacity"] = float(m.group(1))
70 + continue
71 + m = re.match(r"(\d+)\s*[Cc]hambres?", t)
72 + if m:
73 + d["bedrooms"] = float(m.group(1))
74 + continue
75 + m = re.match(r"(\d+)\s*[Ss]alles?\s*de\s*bain", t)
76 + if m:
77 + d["bathrooms"] = float(m.group(1))
78 + continue
79 + if len(t) <= 60:
80 + extras.append(t)
81 + if extras:
82 + d["amenities"] = extras
83 +
84 + # description : JSON-LD Yoast (WebPage.description)
85 + m = re.search(r'(?s)<script type="application/ld\+json"[^>]*>'
86 + r"(.*?)</script>", h)
87 + if m:
88 + try:
89 + graph = json.loads(m.group(1)).get("@graph") or []
90 + for node in graph:
91 + if node.get("@type") == "WebPage" and node.get("description"):
92 + d["description"] = _text(node["description"])[:4000]
93 + break
94 + except ValueError:
95 + pass
96 +
97 + # id Guesty (lien « réserver » guestybookings.com)
98 + m = re.search(r"guestybookings\.com/(?:fr/)?properties/([a-f0-9]{24})",
99 + h)
100 + if m:
101 + d["guesty_id"] = m.group(1)
102 +
103 + # photos : uploads WP (originaux, sans logos ni vignettes -NxN)
104 + imgs: list[str] = []
105 + for u in re.findall(r'(https://accestremblant\.ca/wp-content/uploads/'
106 + r'20\d\d/\d\d/[^" ]+\.(?:jpe?g|png|webp))', h):
107 + if re.search(r"-\d{2,4}x\d{2,4}\.", u):
108 + continue
109 + if re.search(r"logo|favicon|icon", u, re.I):
110 + continue
111 + if u not in imgs:
112 + imgs.append(u)
113 + d["images"] = imgs[:20]
114 +
115 + # carte des prix « autres condos » : slug → à partir de N $
116 + prices: dict[str, float] = {}
117 + for slug, val in re.findall(
118 + r'href="https://accestremblant\.ca/condos/([^"/]+)/"[^>]*>'
119 + r"\s*À partir de\s*([\d ,]+)\s*\$", h):
120 + try:
121 + prices[slug] = float(val.replace(" ", "").replace(",", "."))
122 + except ValueError:
123 + continue
124 + d["prices_seen"] = prices
125 + return d
126 +
127 + # -- contrat ----------------------------------------------------------
128 + def fetch(self) -> list[StListing]:
129 + rows = []
130 + price_map: dict[str, float] = {}
131 + for url, lastmod in self._sitemap():
132 + slug = url.rstrip("/").rsplit("/", 1)[-1]
133 + try:
134 + det = self.detail(slug, lastmod, lambda u=url: self._detail(u))
135 + except Exception:
136 + continue
137 + price_map.update(det.get("prices_seen") or {})
138 + rows.append((slug, url, det))
139 +
140 + listings: list[StListing] = []
141 + for slug, url, det in rows:
142 + title = slug.replace("-", " ").title()
143 + desc = det.get("description") or ""
144 + m = re.match(r"([^|]{2,60})\|", desc)
145 + if m: # « Verbier C | Découvrez… »
146 + title = m.group(1).strip()
147 + desc = desc.split("|", 1)[1].strip()
148 + price = price_map.get(slug)
149 +
150 + details = {k: v for k, v in {
151 + "guesty_id": det.get("guesty_id"),
152 + }.items() if v}
153 +
154 + listings.append(StListing(
155 + source=self.source_id,
156 + external_id=slug,
157 + url=url,
158 + title=title,
159 + property_type="Condo",
160 + city="Mont-Tremblant",
161 + region="Laurentides",
162 + price_night=price,
163 + price_label=(f"à partir de {price:.0f} $ / nuit"
164 + if price else ""),
165 + capacity=det.get("capacity"),
166 + bedrooms=det.get("bedrooms"),
167 + bathrooms=det.get("bathrooms"),
168 + description=desc[:4000],
169 + amenities=det.get("amenities") or [],
170 + details=details,
171 + images=det.get("images") or [],
172 + ))
173 + return listings
modified louka/shortterm/connectors/airbnb.py +183 −1
@@ -22,9 +22,21 @@
22 22 # L'ordre des cellules est mélangé à chaque run pour que les cellules restantes
23 23 # après épuisement du budget tournent d'un run à l'autre.
24 24 #
25 +# Enrichissement détail : la fiche /rooms/<id> embarque le même genre de JSON
26 +# (<script data-deferred-state-0> → data.node.pdpPresentation) avec description
27 +# longue (descriptions.longDescriptionHtml), capacité (personCapacity +
28 +# overview.items "6 guests · 2 bedrooms…"), commodités groupées
29 +# (amenities.seeAllAmenitiesGroups, drapeau available), règles (rules.groupItems
30 +# → animaux) et municipalité (localizedLocation). Visite via le cache détail
31 +# self.detail() (louka_ct.db) avec budget par run : le parc (~15 000) se
32 +# remplit au fil des syncs. Les échecs réseau/anti-bot ne sont PAS mis en
33 +# cache (retentés au prochain run) ; 8 échecs consécutifs coupent
34 +# l'enrichissement du run (tempête anti-bot).
35 +#
25 36 # Réglages env : LOUKA_AIRBNB_PAGES (pages par cellule feuille, défaut 15),
26 37 # LOUKA_AIRBNB_BUDGET (budget de requêtes HTML, défaut 1200),
27 −# LOUKA_AIRBNB_DEPTH (profondeur max du quadtree, défaut 7).
38 +# LOUKA_AIRBNB_DEPTH (profondeur max du quadtree, défaut 7),
39 +# LOUKA_AIRBNB_DETAIL_LIMIT (fiches détail par run, défaut 800).
28 40 # -----------------------------------------------------------------------------
29 41 from __future__ import annotations
30 42
@@ -147,6 +159,25 @@ _NIGHTLY_RE = re.compile(r"(\d+)\s*nights?\s*x\s*\$\s*([\d,]+(?:\.\d+)?)", re.I)
147 159 _MONEY_RE = re.compile(r"\$\s*([\d,]+(?:\.\d+)?)")
148 160 _NUM_RE = re.compile(r"(\d+(?:\.\d+)?)")
149 161
162 +# Clé de version du parseur de fiche détail (bump → re-visite du parc)
163 +_PDP_KEY = "pdp-v1"
164 +
165 +
166 +class _DetailSkip(Exception):
167 + """Fiche détail indisponible ce run (budget épuisé, blocage anti-bot) —
168 + on ne met RIEN en cache pour retenter au prochain sync."""
169 +
170 +
171 +def _html_to_text(fragment: str) -> str:
172 + """HTML de description Airbnb (<br />, <b>…) → texte propre."""
173 + import html as _h
174 + txt = re.sub(r"<br\s*/?>", "\n", fragment)
175 + txt = re.sub(r"<[^>]+>", " ", txt)
176 + txt = _h.unescape(txt)
177 + txt = re.sub(r"[ \t]+", " ", txt)
178 + txt = re.sub(r" ?\n ?", "\n", txt)
179 + return re.sub(r"\n{3,}", "\n\n", txt).strip()
180 +
150 181
151 182 def _b64_room_id(demand_id: str) -> str:
152 183 """"RGVtYW5kU3RheUxpc3Rpbmc6MTIz" → "123" (DemandStayListing:<id>)."""
@@ -210,6 +241,156 @@ class Airbnb(StConnector):
210 241 return self.get_scrapfly(url, render_js=True, asp=True,
211 242 rendering_wait=3000)
212 243
244 + # -- fiche détail (/rooms/<id>) --------------------------------------------
245 + def _pdp_html(self, room_id: str) -> str:
246 + """HTML d'une fiche : Bright Data d'abord, Scrapfly ASP en secours
247 + (pas de rendu JS : le JSON est embarqué côté serveur)."""
248 + url = f"https://www.airbnb.ca/rooms/{room_id}?locale=en&currency=CAD"
249 + html = self._brightdata(url)
250 + if "data-deferred-state" in html:
251 + return html
252 + try:
253 + return self.get_scrapfly(url, render_js=False, asp=True)
254 + except Exception: # noqa: BLE001 — 429/403/timeout : simple échec
255 + return ""
256 +
257 + @staticmethod
258 + def _parse_pdp(html: str) -> dict:
259 + """Champs riches depuis data.node.pdpPresentation du JSON embarqué."""
260 + pp = None
261 + for blob in re.findall(
262 + r'<script[^>]+id="data-deferred-state[^"]*"[^>]*>(.*?)</script>',
263 + html, re.S):
264 + try:
265 + data = json.loads(blob)
266 + except ValueError:
267 + continue
268 + for entry in data.get("niobeClientData") or []:
269 + if not (isinstance(entry, list) and len(entry) > 1
270 + and isinstance(entry[1], dict)):
271 + continue
272 + node = ((entry[1].get("data") or {}).get("node") or {})
273 + if isinstance(node.get("pdpPresentation"), dict):
274 + pp = node["pdpPresentation"]
275 + break
276 + if pp:
277 + break
278 + if not pp:
279 + return {}
280 +
281 + out: dict = {}
282 + # description longue : texte ORIGINAL de l'hôte (souvent français au
283 + # Québec), repli sur la version traduite
284 + desc = (pp.get("descriptions") or {}).get("longDescriptionHtml") or {}
285 + txt = (desc.get("localizedString")
286 + or desc.get("localizedStringWithTranslationPreference") or "")
287 + if txt:
288 + out["description"] = _html_to_text(txt)[:6000]
289 +
290 + cap = pp.get("personCapacity")
291 + if isinstance(cap, (int, float)) and 0 < cap <= 200:
292 + out["capacity"] = float(cap)
293 +
294 + # overview.items : "6 guests", "2 bedrooms", "3 beds", "2 baths"
295 + ov = pp.get("overview") or {}
296 + for item in ov.get("items") or []:
297 + low = (item or "").lower()
298 + m = _NUM_RE.search(low)
299 + if not m:
300 + continue
301 + val = float(m.group(1))
302 + if "guest" in low:
303 + out.setdefault("capacity", val)
304 + elif "bedroom" in low:
305 + out["bedrooms"] = val
306 + elif "bed" in low:
307 + out["beds"] = val
308 + elif "bath" in low:
309 + out["bathrooms"] = val
310 + if ov.get("title"):
311 + out["overview_title"] = ov["title"]
312 +
313 + # commodités disponibles (les groupes "Not included" ont available=False)
314 + amen: list[str] = []
315 + for grp in (pp.get("amenities") or {}).get("seeAllAmenitiesGroups") or []:
316 + for a in grp.get("amenities") or []:
317 + t = (a.get("title") or "").strip()
318 + if a.get("available") and t and t not in amen:
319 + amen.append(t)
320 + if amen:
321 + out["amenities"] = amen[:120]
322 +
323 + # règles de la maison → animaux ("No pets", "Pets allowed", "2 pets…")
324 + for grp in (pp.get("rules") or {}).get("groupItems") or []:
325 + for it in grp.get("items") or []:
326 + if it.get("type") != "HOUSE_RULES_PETS":
327 + continue
328 + t = (it.get("title") or "").lower()
329 + if "no pets" in t or "pas d" in t or "aucun animal" in t:
330 + out["pets"] = "non"
331 + elif "pets allowed" in t or "animaux accept" in t:
332 + out["pets"] = "oui"
333 + elif t:
334 + out["pets"] = "conditions"
335 +
336 + loc = (pp.get("localizedLocation") or "").split(",")[0].strip()
337 + if loc:
338 + out["city"] = loc
339 + return out
340 +
341 + @staticmethod
342 + def _apply_pdp(lst: StListing, p: dict) -> None:
343 + """Applique un payload détail sans écraser ce que la carte a fourni."""
344 + if p.get("description") and not lst.description:
345 + lst.description = p["description"]
346 + if p.get("amenities") and not lst.amenities:
347 + lst.amenities = list(p["amenities"])
348 + for attr in ("capacity", "bedrooms", "beds", "bathrooms"):
349 + if getattr(lst, attr) is None and p.get(attr) is not None:
350 + setattr(lst, attr, p[attr])
351 + if p.get("pets") and lst.pets is None:
352 + lst.pets = p["pets"]
353 + if p.get("city") and not lst.city:
354 + lst.city = p["city"]
355 + if not lst.property_type and p.get("overview_title"):
356 + ptype, _ = _card_type_and_city(p["overview_title"])
357 + lst.property_type = ptype
358 + lst.finalize() # drapeaux/citq/pets dérivés du nouveau texte
359 +
360 + def _enrich_details(self, listings: list[StListing]) -> None:
361 + """Visite les fiches détail via le cache self.detail() sous budget :
362 + les hits de cache sont gratuits, seuls les fetchs réseau comptent."""
363 + limit = max(0, int(os.environ.get("LOUKA_AIRBNB_DETAIL_LIMIT", "800")
364 + or 800))
365 + used = enriched = 0
366 + streak = 0 # échecs réseau consécutifs
367 +
368 + for lst in listings:
369 + def fetch_fn(rid=lst.external_id):
370 + nonlocal used, streak
371 + if used >= limit or streak >= 8:
372 + raise _DetailSkip
373 + used += 1
374 + html = self._pdp_html(rid)
375 + if "data-deferred-state" not in html:
376 + streak += 1
377 + raise _DetailSkip # blocage/vide : pas de mise en cache
378 + streak = 0
379 + return self._parse_pdp(html)
380 +
381 + try:
382 + payload = self.detail(lst.external_id, _PDP_KEY, fetch_fn)
383 + except _DetailSkip:
384 + continue
385 + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run
386 + continue
387 + if payload:
388 + self._apply_pdp(lst, payload)
389 + enriched += 1
390 + print(f"[airbnb] détail : {enriched} annonces enrichies"
391 + f" ({used}/{limit} fetchs réseau, série d'échecs {streak})",
392 + file=sys.stderr)
393 +
213 394 # -- parse JSON embarqué ----------------------------------------------------
214 395 @staticmethod
215 396 def _deferred_results(html: str) -> tuple[list[dict], list[str]]:
@@ -454,4 +635,5 @@ class Airbnb(StConnector):
454 635 if stack:
455 636 print(f"[airbnb] budget épuisé ({budget} req),"
456 637 f" {len(stack)} cellules non visitées", file=sys.stderr)
638 + self._enrich_details(out)
457 639 return out
added louka/shortterm/connectors/alouerauxiles.py +202 −0
@@ -0,0 +1,202 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/alouerauxiles.py : À louer aux Îles (alouerauxiles.com)
4 +#
5 +# Annuaire local des Îles-de-la-Madeleine (~100 maisons/chalets, région très
6 +# mal couverte ailleurs). Site statique « Tactical Soft » : cartes par île.
7 +#
8 +# Méthode :
9 +# 1. LISTE : les pages d'île FR (/havre-aubert, /cap-aux-meules, …) sont du
10 +# HTML serveur contenant un bloc `<div class=item id=<id> tag=rent …>`
11 +# par annonce : id stable, nom (alt="…"), capacité/chambres
12 +# (cap="3 chambres (5 pers. max)"), prix (dayrate=186, $/nuit calculé)
13 +# et lien public (href=https://alouerauxiles.com/<slug>). Dédup par id
14 +# (une annonce peut apparaître sur plusieurs pages).
15 +# 2. DÉTAIL (cache self.detail) : /php/page_fr.php?id=<id> → type
16 +# d'hébergement, personnes/chambres/salles de bain (attributs title=),
17 +# animaux, description, permis CITQ, adresse + ville (après le code
18 +# postal), commodités (« Commodités : ») et photos (pages/idlm/rent/…).
19 +# External_id = id du bloc item (numérique ou slug, stable).
20 +# -----------------------------------------------------------------------------
21 +from __future__ import annotations
22 +
23 +import html as _html
24 +import re
25 +
26 +from ..schema import StListing
27 +from .base import StConnector
28 +
29 +SITE = "https://alouerauxiles.com"
30 +WWW = "https://www.alouerauxiles.com"
31 +
32 +# pages d'île francophones (les slugs anglais sont des doublons)
33 +ISLANDS = ["havre-aubert", "cap-aux-meules", "havre-aux-maisons",
34 + "pointe-aux-loups", "grosse-ile", "grande-entree", "ile-d-entree"]
35 +
36 +_TAG_RE = re.compile(r"<[^>]+>")
37 +
38 +# libellé du site → type canonique Lou-Ka
39 +_TYPES = {"chalet": "Chalet", "maison": "Maison", "résidence": "Maison",
40 + "studio": "Studio", "appartement": "Appartement", "loft": "Loft",
41 + "chambre": "Chambre", "gîte": "Gîte", "auberge": "Auberge"}
42 +
43 +
44 +def _text(fragment: str) -> str:
45 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))
46 + ).replace("", "").strip()
47 +
48 +
49 +def _num(raw) -> float | None:
50 + m = re.search(r"\d+(?:[.,]\d+)?", str(raw or ""))
51 + return float(m.group(0).replace(",", ".")) if m else None
52 +
53 +
54 +class ALouerAuxIles(StConnector):
55 + source_id = "alouerauxiles"
56 +
57 + # -- liste (pages d'île) --------------------------------------------------
58 + def _island_items(self) -> dict[str, dict]:
59 + items: dict[str, dict] = {}
60 + for island in ISLANDS:
61 + try:
62 + h = self.get(f"{WWW}/{island}").text
63 + except Exception:
64 + continue
65 + for block in re.findall(r"<div class=item ([^>]+)>", h):
66 + if "tag=rent" not in block:
67 + continue
68 + m = re.search(r"\bid=([\w-]+)", block)
69 + if not m:
70 + continue
71 + lid = m.group(1)
72 + it = items.setdefault(lid, {"island": island})
73 + m = re.search(r'alt="([^"]+)"', block)
74 + if m:
75 + it["name"] = _text(m.group(1))
76 + m = re.search(r'cap="(\d+)\s*chambres?\s*\((\d+)\s*pers',
77 + block)
78 + if m:
79 + it["bedrooms"] = float(m.group(1))
80 + it["capacity"] = float(m.group(2))
81 + m = re.search(r"\bdayrate=(\d+(?:\.\d+)?)", block)
82 + if m:
83 + it["dayrate"] = float(m.group(1))
84 + m = re.search(r"href=(https://alouerauxiles\.com/[\w-]+)\b",
85 + block)
86 + if m:
87 + it["url"] = m.group(1)
88 + return items
89 +
90 + # -- page détail ------------------------------------------------------
91 + def _detail(self, lid: str) -> dict:
92 + h = self.get(f"{SITE}/php/page_fr.php", params={"id": lid}).text
93 + d: dict = {}
94 +
95 + m = re.search(r"<b class=t1>([^<]+)</b>", h)
96 + if m:
97 + d["type_label"] = _text(m.group(1))
98 +
99 + m = re.search(r'title="(\d+)\s*personnes', h)
100 + if m:
101 + d["capacity"] = float(m.group(1))
102 + m = re.search(r'title="(\d+)\s*chambres?"', h)
103 + if m:
104 + d["bedrooms"] = float(m.group(1))
105 + m = re.search(r'title="(\d+)\s*salle\(?s?\)?\s*de\s*bain', h)
106 + if m:
107 + d["bathrooms"] = float(m.group(1))
108 + m = re.search(r'title="\s*animaux([^"]*)"', h)
109 + if m:
110 + d["pets"] = "non" if "non" in m.group(1).lower() else "oui"
111 +
112 + # description : bloc txtdiv (nom + texte de présentation)
113 + m = re.search(r"(?s)<div id=txtdiv>(.*?)</div>", h)
114 + if m:
115 + frag = re.sub(r"(?s)<b class=t2>.*?</b>", " ", m.group(1))
116 + d["description"] = _text(frag)[:4000]
117 + m = re.search(r"(?s)<b class=t2>([^<]+)</b>",
118 + h[h.find("txtdiv"):] if "txtdiv" in h else "")
119 + if m:
120 + d["title"] = _text(m.group(1))
121 +
122 + m = re.search(r"CITQ\D{0,12}(\d{6})", h, re.I)
123 + if m:
124 + d["citq"] = m.group(1)
125 +
126 + # adresse : rue + « G4T 3H6 l'Étang-du-Nord » → ville après le code
127 + m = re.search(r"(?s)Adresse\s*:\s*</b>(.*?)(?:<b class|Contact)", h)
128 + if m:
129 + lines = [_text(x) for x in re.split(r"<br\s*/?>", m.group(1))]
130 + lines = [x for x in lines if x]
131 + if lines:
132 + d["address"] = lines[0]
133 + for x in lines:
134 + pm = re.search(r"[A-Z]\d[A-Z]\s?\d[A-Z]\d\s+(.{3,40})$", x)
135 + if pm:
136 + d["city"] = pm.group(1).strip()
137 + break
138 +
139 + # commodités : lignes « - … » entre « Commodités : » et « À proximité »
140 + m = re.search(r"(?s)Commodités\s*:(.*?)(?:À proximité|Contact\s*:|$)",
141 + h)
142 + if m:
143 + amens = [_text(x).lstrip("- ").strip()
144 + for x in re.split(r"<br\s*/?>", m.group(1))]
145 + d["amenities"] = [a for a in amens if 2 <= len(a) <= 80][:50]
146 +
147 + # photos du dossier de l'annonce (originaux img/, pas les vignettes)
148 + imgs: list[str] = []
149 + for u in re.findall(r"[\"'=](?:\.\./)?(pages/idlm/rent/[\w-]+/img/"
150 + r"[^\"'\s>]+\.(?:jpe?g|png|webp))", h, re.I):
151 + full = f"{SITE}/{u}"
152 + if full not in imgs:
153 + imgs.append(full)
154 + d["images"] = imgs[:20]
155 + return d
156 +
157 + # -- contrat ----------------------------------------------------------
158 + def fetch(self) -> list[StListing]:
159 + listings: list[StListing] = []
160 + for lid, it in self._island_items().items():
161 + key = f"{it.get('name', '')}|{it.get('capacity', '')}|" \
162 + f"{it.get('dayrate', '')}|{it.get('bedrooms', '')}"
163 + try:
164 + det = self.detail(lid, key, lambda i=lid: self._detail(i))
165 + except Exception:
166 + det = {}
167 + title = det.get("title") or it.get("name") or ""
168 + if not title:
169 + continue
170 +
171 + type_label = (det.get("type_label") or "").lower()
172 + ptype = ""
173 + for needle, canon in _TYPES.items():
174 + if needle in type_label:
175 + ptype = canon
176 + break
177 +
178 + price = it.get("dayrate")
179 + city = det.get("city") or it["island"].replace("-", " ").title()
180 + listings.append(StListing(
181 + source=self.source_id,
182 + external_id=lid,
183 + url=it.get("url") or f"{SITE}/php/page_fr.php?id={lid}",
184 + title=title,
185 + property_type=ptype or "Maison",
186 + address=det.get("address") or "",
187 + city=city,
188 + region="Îles-de-la-Madeleine",
189 + price_night=price,
190 + price_label=(f"à partir de {price:.0f} $ / nuit"
191 + if price else ""),
192 + capacity=det.get("capacity") or it.get("capacity"),
193 + bedrooms=det.get("bedrooms") or it.get("bedrooms"),
194 + bathrooms=det.get("bathrooms"),
195 + pets=det.get("pets"),
196 + citq=det.get("citq") or "",
197 + description=det.get("description") or "",
198 + amenities=det.get("amenities") or [],
199 + details={"ile": it["island"]},
200 + images=det.get("images") or [],
201 + ))
202 + return listings
modified louka/shortterm/connectors/bonjourquebec.py +55 −5
@@ -18,10 +18,16 @@
18 18 # quelques milliers max) — gîtes et insolites sont gardés en entier ;
19 19 # 3. fiche /fiche/<id> (cache self.detail — 1 seule visite par fiche) :
20 20 # région touristique, ville, adresse, no d'enregistrement CITQ,
21 −# description, services/équipements, animaux, tarifs max (détails),
22 −# photos. Pas de prix « à partir de » sur le site → seuls les maximums
23 −# affichés sont conservés dans details (jamais utilisés comme
24 −# price_night pour ne pas fausser le « à partir de »).
21 +# description, services/équipements, animaux, tarifs, photos.
22 +# PRIX : le site ne publie QUE des maximums par nuitée (widget Tarifs :
23 +# « Maximum pour l'unité la plus chère », « Prix maximum par nuitée
24 +# prêt-à-camper »). On les expose honnêtement via price_label
25 +# (« maximum X $ / nuit ») — finalize() en déduit price_night ; le
26 +# libellé garde la nuance (ce n'est pas un « à partir de »). Les
27 +# emplacements de camping nu sont ignorés (hors mandat).
28 +# CAPACITÉ : jamais publiée en « personnes » sur les fiches — on récupère
29 +# ce qui existe : chambres des gîtes (« Chambre : N unités ») et
30 +# mentions « N personnes » dans la description (rare).
25 31 # -----------------------------------------------------------------------------
26 32 from __future__ import annotations
27 33
@@ -60,6 +66,38 @@ _TYPE_KEYWORDS = [
60 66 ]
61 67
62 68
69 +_PRICE_VAL_RE = re.compile(r"\d[\d\s ,.]*\$")
70 +_CAP_RE = re.compile(r"(\d{1,2})\s*personnes")
71 +_CHAMBRES_RE = re.compile(r"^Chambre\s*:\s*(\d+)\s*unité", re.I)
72 +
73 +
74 +def _price_label(tarifs: list[str]) -> str:
75 + """Libellé prix/nuit depuis le widget Tarifs (le site n'affiche que des
76 + maximums par nuitée). Priorité : unité la plus chère > prêt-à-camper >
77 + autre « par nuitée » — emplacements de camping nu exclus."""
78 + pairs: list[tuple[str, str]] = []
79 + label = ""
80 + for txt in tarifs or []:
81 + m = _PRICE_VAL_RE.search(txt)
82 + if m and label:
83 + pairs.append((label.lower(), re.sub(r"[\s ]+", " ",
84 + m.group(0)).strip()))
85 + label = ""
86 + elif not m and txt:
87 + label = txt
88 +
89 + def pick(needle: str, exclude: str = "") -> str:
90 + for lab, val in pairs:
91 + if needle in lab and (not exclude or exclude not in lab):
92 + return val
93 + return ""
94 +
95 + val = (pick("unité la plus chère")
96 + or pick("prêt-à-camper")
97 + or pick("nuit", exclude="camping"))
98 + return f"maximum {val} / nuit" if val else ""
99 +
100 +
63 101 def _abs(url: str) -> str:
64 102 url = _html.unescape(url or "").strip()
65 103 if not url:
@@ -233,6 +271,16 @@ class BonjourQuebec(StConnector):
233 271 "unites": d.get("unites"),
234 272 }.items() if v}
235 273
274 + # capacité : mention « N personnes » dans la description (rare)
275 + caps = [int(x) for x in _CAP_RE.findall(desc) if 1 <= int(x) <= 40]
276 + capacity = float(max(caps)) if caps else None
277 + # chambres : les gîtes déclarent « Chambre : N unités »
278 + bedrooms = None
279 + for u in d.get("unites") or []:
280 + m = _CHAMBRES_RE.match(u)
281 + if m and 0 < int(m.group(1)) <= 30:
282 + bedrooms = float(m.group(1))
283 +
236 284 url = d.get("url_final") or f"{BASE}/fiche/{ext}"
237 285 lst = StListing(
238 286 source=self.source_id,
@@ -243,7 +291,9 @@ class BonjourQuebec(StConnector):
243 291 address=d.get("address", ""),
244 292 city=d.get("city", ""),
245 293 region=d.get("region", ""),
246 − capacity=None,
294 + price_label=_price_label(d.get("tarifs") or []),
295 + capacity=capacity,
296 + bedrooms=bedrooms,
247 297 pets=d.get("pets"),
248 298 citq=d.get("citq", ""),
249 299 description=desc,
modified louka/shortterm/connectors/booking.py +124 −1
@@ -17,12 +17,21 @@
17 17 # Recherche AVEC dates génériques (~30 jours, 2 nuits) : sans dates, Booking
18 18 # ne renvoie ni prix ni configuration des unités. Le prix est donc indicatif
19 19 # → price_label « à partir de … » + price_night (le plus bas trouvé).
20 +#
21 +# Enrichissement : la page détail /hotel/ca/<pageName>.fr.html (Scrapfly ASP
22 +# SANS rendu JS) embarque son propre store Apollo SSR : description complète
23 +# (data-testid="property-description"), commodités localisées (entités
24 +# Instance/SimpleFacility) et galerie (AccommodationPhoto). Vérifié live
25 +# 2026-08-25 sur /hotel/ca/renarde.fr.html.
26 +# Réglage env : LOUKA_BOOKING_DETAIL_LIMIT (fetchs détail par sync, défaut
27 +# 150 ; cache permanent dans louka_ct.db, le parc se complète au fil des syncs).
20 28 # -----------------------------------------------------------------------------
21 29 from __future__ import annotations
22 30
23 31 import datetime
24 32 import html as _html
25 33 import json
34 +import os
26 35 import re
27 36 import sys
28 37 from urllib.parse import quote
@@ -30,6 +39,10 @@ from urllib.parse import quote
30 39 from ..schema import StListing
31 40 from .base import StConnector
32 41
42 +
43 +class _DetailSkip(Exception):
44 + """Fiche détail sautée (budget épuisé / page bloquée) — pas de cache."""
45 +
33 46 # (texte de recherche Booking, région touristique QC)
34 47 DESTINATIONS = [
35 48 ("Mont-Tremblant", "Laurentides"),
@@ -69,6 +82,8 @@ TYPE_MAP = {
69 82 _CAPLA_RE = re.compile(
70 83 r'<script[^>]*data-capla-store-data="apollo"[^>]*>(.*?)</script>', re.S)
71 84 _IMG_BASE = "https://cf.bstatic.com"
85 +_DESC_RE = re.compile(
86 + r'data-testid="property-description"[^>]*>(.*?)</(?:p|div)>', re.S)
72 87
73 88
74 89 class Booking(StConnector):
@@ -179,6 +194,112 @@ class Booking(StConnector):
179 194 lng=loc.get("longitude"),
180 195 )
181 196
197 + # -- page détail ------------------------------------------------------------
198 + @staticmethod
199 + def _parse_detail(html: str) -> dict:
200 + """Payload {description, amenities, images} d'une page détail Booking
201 + ({} si page bloquée/invalide)."""
202 + out: dict = {}
203 +
204 + # description complète (SSR) — HTML → texte
205 + m = _DESC_RE.search(html or "")
206 + if m:
207 + txt = re.sub(r"<br\s*/?>|</p>", "\n", m.group(1))
208 + txt = _html.unescape(re.sub(r"<[^>]+>", " ", txt))
209 + lines = [re.sub(r"\s+", " ", ln).strip() for ln in txt.split("\n")]
210 + desc = "\n".join(ln for ln in lines if ln).strip()
211 + if desc:
212 + out["description"] = desc[:6000]
213 +
214 + # store Apollo de la page détail : commodités localisées + galerie
215 + mm = _CAPLA_RE.search(html or "")
216 + if mm:
217 + try:
218 + store = json.loads(mm.group(1))
219 + except ValueError:
220 + try:
221 + store = json.loads(_html.unescape(mm.group(1)))
222 + except ValueError:
223 + store = {}
224 + amenities: list[str] = []
225 + for key, val in store.items():
226 + if not isinstance(val, dict):
227 + continue
228 + name = ""
229 + if key.startswith("Instance:"): # équipements du lieu
230 + name = (val.get("title") or "").strip()
231 + elif key.startswith("SimpleFacility:"): # équipements des unités
232 + name = (val.get("name") or "").strip()
233 + if name and name not in amenities:
234 + amenities.append(name)
235 + if amenities:
236 + out["amenities"] = amenities[:80]
237 +
238 + images: list[str] = []
239 + for key, val in store.items():
240 + if not (key.startswith("AccommodationPhoto:")
241 + and isinstance(val, dict)):
242 + continue
243 + for k2, v2 in val.items():
244 + if k2.startswith("resource(") and isinstance(v2, dict) \
245 + and v2.get("relativeUrl"):
246 + rel = re.sub(r"/(?:square|max)\w+/", "/max1024x768/",
247 + v2["relativeUrl"], count=1)
248 + u = _IMG_BASE + rel
249 + if u not in images:
250 + images.append(u)
251 + break
252 + if len(images) >= 15:
253 + break
254 + if images:
255 + out["images"] = images
256 + return out
257 +
258 + def _enrich_details(self, listings: list[StListing]) -> None:
259 + """Visite les fiches détail via le cache self.detail() sous budget :
260 + les hits de cache sont gratuits, seuls les fetchs réseau comptent."""
261 + limit = max(0, int(os.environ.get("LOUKA_BOOKING_DETAIL_LIMIT", "150")
262 + or 150))
263 + used = enriched = streak = 0
264 + for lst in listings:
265 + def fetch_fn(url=lst.url):
266 + nonlocal used, streak
267 + if used >= limit or streak >= 5: # tempête anti-bot : on coupe
268 + raise _DetailSkip
269 + used += 1
270 + res = self.scrapfly(url, render_js=False, asp=True)
271 + if res.get("status_code") in (404, 410):
272 + return {} # fiche retirée : cacher vide
273 + payload = self._parse_detail(res.get("content") or "")
274 + if not payload:
275 + streak += 1
276 + raise _DetailSkip # blocage/vide : pas de cache
277 + streak = 0
278 + return payload
279 +
280 + try:
281 + d = self.detail(lst.external_id, "v1", fetch_fn)
282 + except _DetailSkip:
283 + continue
284 + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run
285 + continue
286 + if not d:
287 + continue
288 + if d.get("description") and len(d["description"]) > \
289 + len(lst.description or ""):
290 + lst.description = d["description"]
291 + if d.get("amenities"):
292 + seen = {a.lower() for a in lst.amenities}
293 + for a in d["amenities"]:
294 + if a.lower() not in seen:
295 + lst.amenities.append(a)
296 + seen.add(a.lower())
297 + if d.get("images") and len(d["images"]) > len(lst.images):
298 + lst.images = list(d["images"])
299 + enriched += 1
300 + print(f"[booking] détail : {enriched} annonces enrichies"
301 + f" ({used}/{limit} fetchs réseau)", file=sys.stderr)
302 +
182 303 # -- contrat ---------------------------------------------------------------
183 304 def fetch(self) -> list[StListing]:
184 305 today = datetime.date.today()
@@ -207,4 +328,6 @@ class Booking(StConnector):
207 328 continue
208 329 if lst and lst.external_id not in listings:
209 330 listings[lst.external_id] = lst
210 − return list(listings.values())
331 + out = list(listings.values())
332 + self._enrich_details(out)
333 + return out
added louka/shortterm/connectors/boraboreal.py +174 −0
@@ -0,0 +1,174 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/boraboreal.py : Bora Boréal (boraboreal.com) — chalets FLOTTANTS
4 +# (minibora, boravilla) à Bury (Cantons-de-l'Est) et à Québec, plus un
5 +# chalet en bois rond ; ~13 unités réservables sur Lodgify
6 +# (reserver-boraboreal.lodgify.com, protégé Cloudflare → Scrapfly).
7 +#
8 +# Méthode :
9 +# 1. SLUGS : le sitemap Lodgify est vide → la page « louer-maison-flottante »
10 +# (rendue via Scrapfly, la grille est en JS) liste les 12 chalets
11 +# flottants ; les pages vitrines boraboreal.com (récupérées en direct)
12 +# ajoutent les unités hors grille (bora-bois-rond).
13 +# 2. DÉTAIL : chaque page unité Lodgify (Scrapfly sans render_js, cache
14 +# détail « v1 ») embarque un JSON-LD VacationRental complet :
15 +# identifier (external_id), priceRange « from 229 CAD/night »,
16 +# occupancy/chambres, amenityFeature, adresse, geo lat/lng, ~19 photos
17 +# icdbcdn ; le CITQ est extrait de la description (au-delà de la fenêtre
18 +# de 2000 caractères de finalize()). Animaux : frais au séjour
19 +# → pets = « conditions » si l'amenité l'indique.
20 +# -----------------------------------------------------------------------------
21 +from __future__ import annotations
22 +
23 +import html as _html
24 +import json
25 +import re
26 +import sys
27 +
28 +from ..schema import StListing
29 +from .airbnb import _region_from_latlng
30 +from .base import StConnector
31 +
32 +BOOKING = "https://reserver-boraboreal.lodgify.com"
33 +GRID = BOOKING + "/fr/louer-maison-flottante"
34 +
35 +# pages vitrines boraboreal.com (accessibles en direct) qui pointent vers des
36 +# unités Lodgify absentes de la grille
37 +SHOWCASE = [
38 + "https://boraboreal.com/chalet-flottant-quebec",
39 + "https://boraboreal.com/mini-chalet-a-louer-estrie",
40 + "https://boraboreal.com/chalet-6-personnes-estrie",
41 +]
42 +
43 +_TAG_RE = re.compile(r"<[^>]+>")
44 +
45 +_LD_RE = re.compile(r'(?s)<script[^>]*type="application/ld\+json"[^>]*>'
46 + r"(.*?)</script>")
47 +
48 +
49 +def _strip_html(txt: str) -> str:
50 + txt = re.sub(r"</p>|<br\s*/?>|</li>", "\n", txt or "")
51 + txt = _html.unescape(_TAG_RE.sub(" ", txt))
52 + txt = re.sub(r"[ \t]+", " ", txt)
53 + return re.sub(r"\n\s+", "\n", txt).strip()
54 +
55 +
56 +class BoraBoreal(StConnector):
57 + source_id = "boraboreal"
58 +
59 + # -- découverte des slugs --------------------------------------------------
60 + def _slugs(self) -> list[str]:
61 + slugs: list[str] = []
62 +
63 + def add(s: str):
64 + s = s.strip("/")
65 + if s and s != "louer-maison-flottante" and s not in slugs:
66 + slugs.append(s)
67 +
68 + try:
69 + grid = self.get_scrapfly(GRID, render_js=True,
70 + rendering_wait=3000)
71 + for s in re.findall(r'href="(?:%s)?/fr/([\w~-]+)"'
72 + % re.escape(BOOKING), grid):
73 + add(s)
74 + except Exception as exc: # noqa: BLE001
75 + print(f"[boraboreal] grille : {exc}", file=sys.stderr)
76 +
77 + for page in SHOWCASE:
78 + try:
79 + h = self.get(page).text
80 + except Exception: # noqa: BLE001
81 + continue
82 + for s in re.findall(
83 + r"reserver-boraboreal\.lodgify\.com/(?:fr/)?([\w~-]+)", h):
84 + add(s)
85 + return slugs
86 +
87 + # -- page unité Lodgify ----------------------------------------------------
88 + def _detail(self, slug: str) -> dict:
89 + h = self.get_scrapfly(f"{BOOKING}/fr/{slug}", render_js=False)
90 + for block in _LD_RE.findall(h):
91 + try:
92 + ld = json.loads(block)
93 + except ValueError:
94 + continue
95 + if ld.get("@type") == "VacationRental":
96 + return {"ld": ld}
97 + return {}
98 +
99 + # -- contrat ---------------------------------------------------------------
100 + def fetch(self) -> list[StListing]:
101 + listings: list[StListing] = []
102 + vus: set[str] = set()
103 + for slug in self._slugs():
104 + det = self.detail(slug, "v1", lambda s=slug: self._detail(s))
105 + ld = det.get("ld") or {}
106 + if not ld:
107 + continue
108 + ext_id = str(ld.get("identifier") or slug)
109 + if ext_id in vus:
110 + continue
111 + vus.add(ext_id)
112 +
113 + place = ld.get("containsPlace") or {}
114 + occupancy = (place.get("occupancy") or {}).get("value")
115 + addr = ld.get("address") or {}
116 + geo = ld.get("geo") or {}
117 +
118 + # « from 229 CAD/night » → price_night
119 + price = None
120 + m = re.search(r"from\s+([\d.]+)\s*CAD",
121 + str(ld.get("priceRange") or ""))
122 + if m:
123 + v = float(m.group(1))
124 + if 20 <= v <= 20000:
125 + price = v
126 +
127 + description = _strip_html(str(ld.get("description") or ""))
128 + m = re.search(r"CITQ\D{0,25}(\d{6})", description)
129 + citq = m.group(1) if m else ""
130 +
131 + amen = [a.get("name") for a in ld.get("amenityFeature") or []
132 + if isinstance(a, dict) and a.get("name")
133 + and a.get("value") is not False]
134 + pets = "conditions" if any("animaux" in a.lower()
135 + or "pet" in a.lower()
136 + for a in amen) else None
137 +
138 + imgs = ld.get("image") or []
139 + if isinstance(imgs, str):
140 + imgs = [imgs]
141 +
142 + city = (addr.get("addressLocality") or "").strip()
143 + region = (_region_from_latlng(geo.get("latitude"),
144 + geo.get("longitude"))
145 + or ("Québec" if slug.endswith("---quebec")
146 + else "Cantons-de-l'Est"))
147 +
148 + listings.append(StListing(
149 + source=self.source_id,
150 + external_id=ext_id,
151 + url=f"{BOOKING}/fr/{slug}",
152 + title=_html.unescape(str(ld.get("name") or slug)).strip(),
153 + property_type="Chalet",
154 + address=(addr.get("streetAddress") or "").strip(),
155 + city=city,
156 + region=region,
157 + price_night=price,
158 + price_label=(f"à partir de {price:g} $ / nuit"
159 + if price else ""),
160 + capacity=float(occupancy) if occupancy else None,
161 + bedrooms=(float(place["numberOfBedrooms"])
162 + if place.get("numberOfBedrooms") else None),
163 + pets=pets,
164 + citq=citq,
165 + description=description[:5000],
166 + amenities=amen,
167 + details={"floating": "bois-rond" not in slug,
168 + "postal_code": addr.get("postalCode") or ""},
169 + images=[u for u in imgs if isinstance(u, str)
170 + and u.startswith("https://")][:20],
171 + lat=geo.get("latitude"),
172 + lng=geo.get("longitude"),
173 + ))
174 + return listings
added louka/shortterm/connectors/campingquebec.py +274 −0
@@ -0,0 +1,274 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/campingquebec.py : Camping Québec (campingquebec.com) —
4 +# l'association des ~830 terrains de camping du Québec. On ne retient QUE
5 +# les campings offrant du PRÊT-À-CAMPER / hébergement locatif (tentes
6 +# aménagées, chalets, yourtes, roulottes…) : un camping « emplacements
7 +# seulement » n'est pas un hébergement court terme pour Lou-Ka.
8 +#
9 +# Méthode (WordPress, aucun anti-bot) :
10 +# 1. LISTE : l'endpoint AJAX de « Trouver un camping » est ouvert :
11 +# GET /fr/wp-json/search/result?lang=fr&view=list
12 +# &ready_to_camps[]=tous-types-de-pret-a-camper-disponible&paged=N
13 +# → fragments HTML de 24 cartes/page (~600 campings filtrés prêt-à-camper).
14 +# Carte : URL /fr/campings/<région>/<slug> (= external_id), nom, région.
15 +# 2. FICHE (cache self.detail, clé mensuelle pour suivre les tarifs) :
16 +# description, adresse + ville (bloc Informations), coordonnées (lien
17 +# google.ca/maps?q=lat,lng), no d'enregistrement CITQ, tarifs (ligne
18 +# « Nuitée, Prêt-à-camper » min-max → price_night), unités prêt-à-camper
19 +# (« Prêt-à-camper disponibles : Tentes : 2 »), services (amenities),
20 +# nb d'emplacements, dates de saison, photos. Garde-fou : la fiche doit
21 +# confirmer le prêt-à-camper (unités ou tarif), sinon elle est écartée.
22 +#
23 +# Réglage env : LOUKA_CAMPINGQUEBEC_LIMIT (nb max de fiches, 0 = tout).
24 +# -----------------------------------------------------------------------------
25 +from __future__ import annotations
26 +
27 +import os
28 +import re
29 +import sys
30 +import time
31 +
32 +from ..schema import StListing, normalize_region
33 +from .base import StConnector
34 +
35 +SITE = "https://www.campingquebec.com"
36 +API = f"{SITE}/fr/wp-json/search/result"
37 +PREFIX_FICHE = f"{SITE}/fr/campings/"
38 +PAGE_MAX = 60 # garde-fou pagination
39 +
40 +_MAPS_RE = re.compile(r"google\.ca/maps\?q=(-?\d+\.\d+),(-?\d+\.\d+)")
41 +_CITQ_RE = re.compile(r"No d[’']enregistrement\s*(\d{5,7})")
42 +_PAGE_RE = re.compile(r'aria-label="Page (\d+)"')
43 +_MONTANT_RE = re.compile(r"([\d\s ]+(?:[.,]\d{2})?)\s*\$")
44 +
45 +# Libellé d'unité prêt-à-camper → type canonique Lou-Ka (si type unique) ;
46 +# le préfixe « location de » est retiré avant consultation.
47 +_TYPE_UNITE = {
48 + "tente": "Prêt-à-camper", "tentes": "Prêt-à-camper",
49 + "chalet": "Chalet", "chalets": "Chalet",
50 + "yourte": "Yourte", "yourtes": "Yourte",
51 + "dôme": "Dôme", "dômes": "Dôme", "bulle ou dôme": "Dôme",
52 + "refuge": "Refuge", "refuges": "Refuge",
53 + "tipi": "Prêt-à-camper", "tipis": "Prêt-à-camper",
54 + "cabine": "Prêt-à-camper", "cabines": "Prêt-à-camper",
55 + "caravane": "Prêt-à-camper", "caravanes": "Prêt-à-camper",
56 +}
57 +
58 +
59 +def _montant(txt: str) -> float | None:
60 + m = _MONTANT_RE.search(txt or "")
61 + if not m:
62 + return None
63 + try:
64 + return float(re.sub(r"[\s ]", "", m.group(1)).replace(",", "."))
65 + except ValueError:
66 + return None
67 +
68 +
69 +class CampingQuebec(StConnector):
70 + source_id = "campingquebec"
71 + request_delay = 0.8
72 +
73 + # -- liste (fragments HTML paginés) ----------------------------------------
74 + def _liste(self, limit: int = 0) -> list[dict]:
75 + from bs4 import BeautifulSoup
76 + items, vus = [], set()
77 + page, total_pages = 1, 1
78 + while page <= min(total_pages, PAGE_MAX):
79 + if limit and len(items) >= limit:
80 + break
81 + html = self.get(API, params={
82 + "lang": "fr", "view": "list",
83 + "ready_to_camps[]": "tous-types-de-pret-a-camper-disponible",
84 + "paged": page,
85 + }).text
86 + pages = [int(p) for p in _PAGE_RE.findall(html)]
87 + if pages:
88 + total_pages = max(pages)
89 + soup = BeautifulSoup(html, "html.parser")
90 + nouveaux = 0
91 + for a in soup.select(f'a.c-card[href^="{PREFIX_FICHE}"]'):
92 + path = a["href"][len(PREFIX_FICHE):].strip("/")
93 + if path.count("/") != 1 or path in vus:
94 + continue
95 + vus.add(path)
96 + nouveaux += 1
97 + h4 = a.find("h4")
98 + span = a.select_one("span.u-text-transform-none")
99 + items.append({
100 + "id": path, # <région>/<slug>
101 + "nom": h4.get_text(" ", strip=True) if h4 else "",
102 + "region": span.get_text(" ", strip=True) if span else "",
103 + })
104 + if not nouveaux: # page vide → fin
105 + break
106 + page += 1
107 + return items
108 +
109 + # -- fiche camping ----------------------------------------------------------
110 + def _fetch_fiche(self, path: str) -> dict:
111 + from bs4 import BeautifulSoup
112 + html = self.get(PREFIX_FICHE + path).text
113 + soup = BeautifulSoup(html, "html.parser")
114 + d: dict = {}
115 +
116 + m = _MAPS_RE.search(html)
117 + if m:
118 + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2))
119 + m = _CITQ_RE.search(html)
120 + if m:
121 + d["citq"] = m.group(1)
122 +
123 + # description : bloc typographique sous l'en-tête « Description »
124 + for div in soup.find_all("div"):
125 + if div.get_text(strip=True) == "Description":
126 + typo = div.find_next_sibling("div")
127 + if typo is not None:
128 + d["description"] = typo.get_text("\n", strip=True)[:2500]
129 + break
130 +
131 + # adresse + ville : paragraphe précédant « Voir sur la carte »
132 + carte = soup.find("a", string=re.compile("Voir sur la carte"))
133 + if carte is None:
134 + for a in soup.find_all("a"):
135 + if "Voir sur la carte" in a.get_text():
136 + carte = a
137 + break
138 + if carte is not None:
139 + p = carte.find_previous("p")
140 + if p is not None:
141 + lignes = [x.strip() for x in p.get_text("\n").split("\n")
142 + if x.strip()]
143 + if lignes:
144 + d["adresse"] = ", ".join(lignes)
145 + # « Saint-Sulpice J5W 3V5 » → ville sans le code postal
146 + d["ville"] = re.sub(
147 + r"\s*[A-Z]\d[A-Z]\s*\d[A-Z]\d\s*$", "",
148 + lignes[-1]).strip(" ,")
149 +
150 + # sections h4 → listes (unités PAC, emplacements…)
151 + sections: dict[str, list[str]] = {}
152 + for h4 in soup.find_all("h4"):
153 + titre = h4.get_text(" ", strip=True)
154 + parent = h4.find_parent("div")
155 + bloc = parent.find_next_sibling("div") if parent else None
156 + if bloc is not None:
157 + lis = [li.get_text(" ", strip=True)
158 + for li in bloc.find_all("li")]
159 + if lis:
160 + sections[titre] = lis
161 +
162 + pac: dict[str, int] = {}
163 + for titre, lis in sections.items():
164 + if titre.lower().startswith("prêt-à-camper"):
165 + for li in lis:
166 + nom, _, nb = li.partition(":")
167 + try:
168 + pac[nom.strip()] = int(nb.strip())
169 + except ValueError:
170 + pac[nom.strip()] = 0
171 + d["pac"] = pac
172 + for titre, lis in sections.items():
173 + if titre.lower().startswith("types d'emplacements"):
174 + d["emplacements"] = lis[:12]
175 +
176 + # tarifs : lignes de la table « Durée / Min. / Max. »
177 + for tr in soup.select("table.c-table tr"):
178 + tds = [td.get_text(" ", strip=True) for td in tr.find_all("td")]
179 + if len(tds) >= 2 and "prêt-à-camper" in tds[0].lower():
180 + d["tarif_pac_min"] = _montant(tds[1])
181 + d["tarif_pac_max"] = _montant(tds[2]) if len(tds) > 2 else None
182 + elif len(tds) >= 2 and tds[0].lower() == "nuitée":
183 + d["tarif_nuit_min"] = _montant(tds[1])
184 +
185 + # services offerts → amenities (panneau d'accordéon « services »)
186 + panneau = soup.select_one(
187 + 'div.c-accordion__target[data-toggler-target*="services"]')
188 + if panneau is not None:
189 + d["services"] = [li.get_text(" ", strip=True)
190 + for li in panneau.find_all("li")][:40]
191 +
192 + # saison
193 + texte = soup.get_text(" ", strip=True)
194 + m = re.search(r"Date d['’]ouverture\s*:\s*([\d]{1,2} \S+ \d{4})", texte)
195 + if m:
196 + d["ouverture"] = m.group(1)
197 + m = re.search(r"Date de fermeture\s*:\s*([\d]{1,2} \S+ \d{4})", texte)
198 + if m:
199 + d["fermeture"] = m.group(1)
200 +
201 + # photos (galerie WordPress, en excluant logos et gabarits)
202 + imgs: list[str] = []
203 + for img in soup.find_all("img"):
204 + u = img.get("data-lazy-src") or img.get("src") or ""
205 + if (u.startswith(f"{SITE}/wp-content/uploads/20")
206 + and "logo" not in u.lower() and u not in imgs):
207 + imgs.append(u)
208 + d["images"] = imgs[:12]
209 + return d
210 +
211 + # -- contrat ----------------------------------------------------------------
212 + def fetch(self) -> list[StListing]:
213 + limit = int(os.environ.get("LOUKA_CAMPINGQUEBEC_LIMIT", "0") or 0)
214 + month = time.strftime("%Y-%m") # re-visite mensuelle (tarifs)
215 + listings: list[StListing] = []
216 + # marge : certaines fiches de la liste seront écartées au garde-fou
217 + for it in self._liste(limit * 3 if limit else 0):
218 + path = it["id"]
219 + try:
220 + d = self.detail(path, month,
221 + lambda p=path: self._fetch_fiche(p))
222 + except Exception as exc: # noqa: BLE001
223 + print(f"[campingquebec] fiche {path} : {exc}", file=sys.stderr)
224 + continue
225 +
226 + pac = d.get("pac") or {}
227 + prix_pac = d.get("tarif_pac_min")
228 + if not pac and not prix_pac: # aucun hébergement locatif confirmé
229 + continue
230 +
231 + price_label = ""
232 + if prix_pac:
233 + pmax = d.get("tarif_pac_max")
234 + price_label = (f"prêt-à-camper {prix_pac:.2f} $"
235 + + (f" à {pmax:.2f} $" if pmax else "")
236 + + " / nuit")
237 +
238 + # type : celui de l'unique famille d'unités, sinon Prêt-à-camper
239 + ptype = "Prêt-à-camper"
240 + if len(pac) == 1:
241 + libelle = re.sub(r"^location (de |d')", "",
242 + next(iter(pac)).lower()).strip()
243 + ptype = _TYPE_UNITE.get(libelle, ptype)
244 +
245 + details = {k: v for k, v in {
246 + "unites_pret_a_camper": pac or None,
247 + "emplacements": d.get("emplacements"),
248 + "tarif_emplacement_min": d.get("tarif_nuit_min"),
249 + "ouverture": d.get("ouverture", ""),
250 + "fermeture": d.get("fermeture", ""),
251 + }.items() if v}
252 +
253 + listings.append(StListing(
254 + source=self.source_id,
255 + external_id=path,
256 + url=PREFIX_FICHE + path,
257 + title=it.get("nom") or path.rsplit("/", 1)[-1],
258 + property_type=ptype,
259 + address=d.get("adresse", ""),
260 + city=d.get("ville", ""),
261 + region=normalize_region(it.get("region", "")),
262 + price_night=prix_pac,
263 + price_label=price_label,
264 + citq=d.get("citq", ""),
265 + description=d.get("description", ""),
266 + amenities=d.get("services") or [],
267 + details=details,
268 + images=d.get("images") or [],
269 + lat=d.get("lat"),
270 + lng=d.get("lng"),
271 + ))
272 + if limit and len(listings) >= limit:
273 + break
274 + return listings
added louka/shortterm/connectors/captremblant.py +157 −0
@@ -0,0 +1,157 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/captremblant.py : Cap Tremblant Mountain Resort
4 +# (captremblant.com) — complexe de résidences de tourisme à Mont-Tremblant.
5 +# 7 catégories de résidences (1 à 5 chambres), chacune regroupant plusieurs
6 +# unités identiques (fiches « catégorie », pas unité par unité).
7 +#
8 +# Méthode : sitemap.xml → 7 URLs /residences/<slug> (version fr ; les /en/
9 +# sont ignorées). Pas de lastmod → cache détail « v1 ». Chaque page embarque
10 +# un JSON-LD @type ["HotelRoom","Product"] : nom, description, image
11 +# (galaxy.tf) et offers.price = tarif À LA NUIT en CAD (ex. 251,10 $) →
12 +# price_night « à partir de ». La liste des caractéristiques vit dans le
13 +# <ul> du bloc <div class="m-content-object--content"> → amenities ;
14 +# capacité extraite de « jusqu'à six personnes » (nombres en toutes
15 +# lettres), chambres déduites du slug/titre. La galerie est chargée en JS :
16 +# seule l'image JSON-LD est disponible statiquement.
17 +# -----------------------------------------------------------------------------
18 +from __future__ import annotations
19 +
20 +import html as _html
21 +import json
22 +import re
23 +
24 +from ..schema import StListing
25 +from .base import StConnector
26 +
27 +SITE = "https://www.captremblant.com"
28 +SITEMAP = SITE + "/sitemap.xml"
29 +
30 +_WORDS = {"un": 1, "une": 1, "deux": 2, "trois": 3, "quatre": 4, "cinq": 5,
31 + "six": 6, "sept": 7, "huit": 8, "neuf": 9, "dix": 10, "onze": 11,
32 + "douze": 12}
33 +
34 +_TAG_RE = re.compile(r"<[^>]+>")
35 +
36 +
37 +def _text(fragment: str) -> str:
38 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
39 +
40 +
41 +def _count(txt: str) -> int | None:
42 + """« six » ou « 6 » → 6."""
43 + t = txt.strip().lower()
44 + if t.isdigit():
45 + return int(t)
46 + return _WORDS.get(t)
47 +
48 +
49 +class CapTremblant(StConnector):
50 + source_id = "captremblant"
51 + request_delay = 0.8
52 +
53 + # -- page détail ----------------------------------------------------------
54 + def _detail(self, url: str) -> dict:
55 + h = self.get(url).text
56 + d: dict = {}
57 +
58 + for block in re.findall(r'<script type="application/ld\+json"[^>]*>'
59 + r"(.*?)</script>", h, re.S):
60 + try:
61 + ld = json.loads(block)
62 + except ValueError:
63 + continue
64 + types = ld.get("@type")
65 + types = types if isinstance(types, list) else [types]
66 + if "HotelRoom" in types or "Product" in types:
67 + d["ld"] = ld
68 + break
69 +
70 + # caractéristiques : <ul> du bloc m-content-object--content
71 + m = re.search(r'(?s)<div class="m-content-object--content[^"]*"[^>]*>'
72 + r"(.*?)</div>", h)
73 + if m:
74 + amen = []
75 + for li in re.findall(r"(?s)<li[^>]*>(.*?)</li>", m.group(1)):
76 + t = _text(li)
77 + if t and t not in amen:
78 + amen.append(t)
79 + if amen:
80 + d["amenities"] = amen
81 + return d
82 +
83 + # -- contrat --------------------------------------------------------------
84 + def fetch(self) -> list[StListing]:
85 + xml = self.get(SITEMAP).text
86 + urls = [u for u in re.findall(r"<loc>([^<]+)</loc>", xml)
87 + if "/residences/" in u and "/en/" not in u]
88 +
89 + listings: list[StListing] = []
90 + vus: set[str] = set()
91 + for url in urls:
92 + slug = url.rstrip("/").split("/")[-1]
93 + if slug in vus:
94 + continue
95 + vus.add(slug)
96 +
97 + det = self.detail(slug, "v1", lambda u=url: self._detail(u))
98 + ld = det.get("ld") or {}
99 + if not ld:
100 + continue
101 +
102 + title = _text(str(ld.get("name") or slug))
103 + amen = det.get("amenities") or []
104 + description = _text(str(ld.get("description") or ""))
105 + if amen and len(" ".join(amen)) > len(description):
106 + description = description or " ".join(amen[:1])
107 +
108 + # capacité : « jusqu'à six personnes » / « de dix à douze
109 + # personnes » (description ou 1re puce) — on prend le maximum
110 + capacity = None
111 + hay = f"{description} {' '.join(amen[:2])}"
112 + m = (re.search(r"jusqu[’']à\s+(\w+)\s+personnes", hay, re.I)
113 + or re.search(r"de\s+\w+\s+à\s+(\w+)\s+personnes", hay, re.I))
114 + if m:
115 + capacity = _count(m.group(1))
116 +
117 + # chambres : déduites du slug/titre (« deux-chambres », « cinq… »)
118 + bedrooms = None
119 + m = re.search(r"(\w+)[- ]chambres?", f"{slug} {title}".lower())
120 + if m:
121 + bedrooms = _count(m.group(1))
122 +
123 + price = None
124 + try:
125 + price = float((ld.get("offers") or {}).get("price"))
126 + except (TypeError, ValueError):
127 + pass
128 + # garde-fou : la fiche « quatre chambres » publie 29520.00 $
129 + # (bogue du site, sans doute 295,20 $) → tarif rejeté
130 + if price is not None and not 50 <= price <= 5000:
131 + price = None
132 + price_label = (f"à partir de {price:g} $ / nuit"
133 + if price else "")
134 +
135 + img = ld.get("image") or ""
136 + images = [img] if isinstance(img, str) \
137 + and img.startswith("https://") else []
138 +
139 + listings.append(StListing(
140 + source=self.source_id,
141 + external_id=slug,
142 + url=url,
143 + title=title,
144 + property_type="Condo",
145 + address="240 Rue du Mont-Plaisant, Mont-Tremblant",
146 + city="Mont-Tremblant",
147 + region="Laurentides",
148 + price_night=price,
149 + price_label=price_label,
150 + capacity=float(capacity) if capacity else None,
151 + bedrooms=float(bedrooms) if bedrooms else None,
152 + description=description[:3000],
153 + amenities=amen,
154 + details={"multi_unit": True, "resort": "Cap Tremblant"},
155 + images=images,
156 + ))
157 + return listings
added louka/shortterm/connectors/chaleto.py +198 −0
@@ -0,0 +1,198 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/chaleto.py : Chaleto (chaleto.ca) — gestionnaire québécois
4 +# (Charlevoix, Capitale-Nationale, Laurentides…), ~440 fiches FR.
5 +#
6 +# Méthode : WordPress (vitrine) + Guesty (inventaire). Sitemap dédié
7 +# /listings-sitemap.xml → URLs /chalets-et-condos-a-louer/<slug>-<guestyId>/
8 +# (doublées en /en/ : on garde le FR). Tout est dans le HTML serveur de la
9 +# page détail : h1, « Ville, Région », prix « À partir de N$ / nuit »,
10 +# pictos (voyageurs/chambres/lits/salles de bain), commodités, description
11 +# (contenant le no CITQ) et TOUTES les photos Guesty pleine résolution dans
12 +# l'attribut data-images de la lightbox. Le lastmod du sitemap est identique
13 +# partout : clé de cache mensuelle (refetch complet 1×/mois, ~440 requêtes).
14 +# -----------------------------------------------------------------------------
15 +from __future__ import annotations
16 +
17 +import html as _html
18 +import json
19 +import re
20 +import time
21 +
22 +from ...normalize import strip_accents
23 +from ..schema import StListing, parse_price_night
24 +from .base import StConnector
25 +
26 +SITEMAP = "https://chaleto.ca/listings-sitemap.xml"
27 +
28 +# URL FR : /chalets-et-condos-a-louer/<slug>-<id Guesty 24 hex>/
29 +_URL_FR = re.compile(
30 + r"https://(?:www\.)?chaleto\.ca/chalets-et-condos-a-louer/"
31 + r"[\w-]+-([0-9a-f]{24})/?$")
32 +
33 +_TAG_RE = re.compile(r"<[^>]+>")
34 +
35 +# le site affiche la région ADMINISTRATIVE (« Capitale-Nationale »,
36 +# « Gaspésie--Îles-de-la-Madeleine ») → région touristique canonique
37 +_REGION_FIX = {
38 + "capitale-nationale": "Québec",
39 + "estrie": "Cantons-de-l'Est",
40 + "gaspesie-iles-de-la-madeleine": "Gaspésie",
41 + "saguenay-lac-saint-jean": "Saguenay–Lac-Saint-Jean",
42 +}
43 +
44 +# municipalités de la Capitale-Nationale qui relèvent touristiquement
45 +# de Charlevoix
46 +_VILLES_CHARLEVOIX = {
47 + "baie-saint-paul", "petite-riviere-saint-francois", "la-malbaie",
48 + "les-eboulements", "saint-urbain", "saint-irenee", "saint-hilarion",
49 + "isle-aux-coudres", "l'isle-aux-coudres", "notre-dame-des-monts",
50 + "saint-aime-des-lacs", "clermont", "saint-simeon",
51 + "baie-sainte-catherine",
52 +}
53 +
54 +# mot-clé du titre → type canonique Lou-Ka
55 +_TYPE_HINTS = [
56 + ("condo", "Condo"), ("loft", "Loft"), ("studio", "Studio"),
57 + ("appartement", "Appartement"), ("maison", "Maison"),
58 + ("mini-maison", "Mini-maison"), ("chalet", "Chalet"),
59 +]
60 +
61 +
62 +def _text(fragment: str) -> str:
63 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
64 +
65 +
66 +class Chaleto(StConnector):
67 + source_id = "chaleto"
68 +
69 + # -- inventaire (sitemap) ----------------------------------------------
70 + def _sitemap_urls(self) -> dict[str, str]:
71 + """id Guesty -> URL détail FR."""
72 + xml = self.get(SITEMAP).text
73 + urls: dict[str, str] = {}
74 + for loc in re.findall(r"<loc>([^<]+)</loc>", xml):
75 + m = _URL_FR.match(loc.strip())
76 + if m:
77 + urls.setdefault(m.group(1), loc.strip())
78 + return urls
79 +
80 + # -- page détail ---------------------------------------------------------
81 + def _detail(self, url: str) -> dict:
82 + h = self.get(url).text
83 + d: dict = {}
84 +
85 + m = re.search(r"<h1[^>]*>(.*?)</h1>", h, re.S)
86 + if m:
87 + d["title"] = _text(m.group(1))
88 +
89 + # « Baie-Saint-Paul, Capitale-Nationale » sous le titre
90 + m = re.search(r'<p class="text-lg">([^<]+)</p>', h)
91 + if m:
92 + parts = [p.strip() for p in _text(m.group(1)).split(",")]
93 + if parts:
94 + d["city"] = parts[0]
95 + if len(parts) > 1:
96 + cle = re.sub(r"-{2,}", "-",
97 + strip_accents(parts[-1]).lower().replace(" ", "-"))
98 + region = _REGION_FIX.get(cle, parts[-1])
99 + ville = strip_accents(d.get("city", "")).lower().replace(" ", "-")
100 + if region == "Québec" and ville in _VILLES_CHARLEVOIX:
101 + region = "Charlevoix"
102 + d["region"] = region
103 +
104 + # « À partir de <strong>200$</strong> / nuit »
105 + m = re.search(r"À partir de</span>\s*<span[^>]*>\s*"
106 + r"<strong>([^<]+)</strong>\s*/\s*nuit", h)
107 + if m:
108 + d["price_label"] = f"à partir de {_text(m.group(1))} / nuit"
109 +
110 + # pictos : « 5 voyageurs », « 3 chambres », « 3 lits », « 2 salles de bain »
111 + for val, label in re.findall(
112 + r"<p>\s*([\d.,]+)\s+(voyageurs?|chambres?|lits?|"
113 + r"salles? de bain)\s*</p>", h):
114 + n = float(val.replace(",", "."))
115 + lab = label.lower()
116 + if lab.startswith("voyageur"):
117 + d["capacity"] = n
118 + elif lab.startswith("chambre"):
119 + d["bedrooms"] = n
120 + elif lab.startswith("lit"):
121 + d["beds"] = n
122 + else:
123 + d["bathrooms"] = n
124 +
125 + # commodités : spans du bloc « Commodités » (grille + accordéon)
126 + i = h.find("Commodités</h2>")
127 + if i >= 0:
128 + j = h.find("<h2", i + 10)
129 + bloc = h[i:j if j > 0 else i + 20000]
130 + amen = []
131 + for a in re.findall(r'<span class="text-white">([^<]+)</span>', bloc):
132 + a = _text(a)
133 + if a and a not in amen:
134 + amen.append(a)
135 + d["amenities"] = amen
136 +
137 + # description (1er paragraphe long — contient « CITQ : NNNNNN | Exp: … »)
138 + m = re.search(r'<p class="max-w-\[50rem\]">(.*?)</p>', h, re.S)
139 + if m:
140 + txt = _html.unescape(re.sub(r"<br\s*/?>", "\n",
141 + m.group(1)))
142 + txt = _TAG_RE.sub(" ", txt)
143 + txt = re.sub(r"[ \t]+", " ", txt).strip()
144 + d["description"] = txt[:5000]
145 + m2 = re.search(r"CITQ\s*:?\s*(\d{6})", txt)
146 + if m2:
147 + d["citq"] = m2.group(1)
148 +
149 + # photos Guesty pleine résolution (lightbox data-images, JSON échappé)
150 + m = re.search(r'data-images="([^"]+)"', h)
151 + if m:
152 + try:
153 + imgs = json.loads(_html.unescape(m.group(1)))
154 + except ValueError:
155 + imgs = []
156 + d["images"] = [u for u in imgs if isinstance(u, str)][:20]
157 + return d
158 +
159 + # -- contrat --------------------------------------------------------------
160 + def fetch(self) -> list[StListing]:
161 + cle = "detail-" + time.strftime("%Y-%m") # lastmod uniforme → mensuel
162 + listings: list[StListing] = []
163 + for eid, url in self._sitemap_urls().items():
164 + try:
165 + d = self.detail(eid, cle, lambda u=url: self._detail(u))
166 + except Exception: # une fiche cassée ≠ inventaire perdu
167 + d = {}
168 + title = d.get("title") or ""
169 + if not title:
170 + continue
171 +
172 + ptype = ""
173 + hay = strip_accents(title).lower()
174 + for needle, canon in _TYPE_HINTS:
175 + if needle in hay:
176 + ptype = canon
177 + break
178 +
179 + listings.append(StListing(
180 + source=self.source_id,
181 + external_id=eid, # id Guesty, stable
182 + url=url,
183 + title=title,
184 + property_type=ptype or "Chalet",
185 + city=d.get("city", ""),
186 + region=d.get("region", ""),
187 + price_night=parse_price_night(d.get("price_label", "")),
188 + price_label=d.get("price_label", ""),
189 + capacity=d.get("capacity"),
190 + bedrooms=d.get("bedrooms"),
191 + beds=d.get("beds"),
192 + bathrooms=d.get("bathrooms"),
193 + citq=d.get("citq", ""),
194 + description=d.get("description", ""),
195 + amenities=d.get("amenities") or [],
196 + images=d.get("images") or [],
197 + ))
198 + return listings
added louka/shortterm/connectors/chaletsalpins.py +200 −0
@@ -0,0 +1,200 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/chaletsalpins.py : Les Chalets Alpins (chaletsalpins.ca)
4 +# — gestionnaire de Stoneham, ~180 chalets (Stoneham, Lac-Beauport,
5 +# Charlevoix, Laurentides).
6 +#
7 +# Méthode : WordPress. Sitemap /chalets-sitemap.xml (lastmod fiable, doublons
8 +# /en/ écartés) → pages /hebergement/<slug>/ (slug = adresse + no CITQ).
9 +# La page détail porte un JSON-LD schema.org Hotel (nom, description,
10 +# addressLocality, petsAllowed, amenityFeature FR) ; les compteurs vivent
11 +# dans des <span class="capacity|bedrooms|beds|restrooms">…N</span> de la
12 +# barre d'entête, et le prix dans l'encadré latéral
13 +# « <h3>2 nuits à partir de (printemps) <span>1 025.00 $</span></h3> »
14 +# (minimum 2 nuits → prix ramené à la nuit). Photos = uploads du carrousel
15 +# (vignettes -150x150 écartées). Pas de géo dans le HTML.
16 +# -----------------------------------------------------------------------------
17 +from __future__ import annotations
18 +
19 +import html as _html
20 +import json
21 +import re
22 +
23 +from ...normalize import strip_accents
24 +from ..schema import StListing
25 +from .base import StConnector
26 +
27 +SITEMAP = "https://chaletsalpins.ca/chalets-sitemap.xml"
28 +
29 +_URL_DETAIL = re.compile(
30 + r"https://(?:www\.)?chaletsalpins\.ca/hebergement/([\w-]+)/?$")
31 +
32 +# localités desservies → région touristique canonique
33 +_VILLE_REGION = {
34 + "stoneham": "Québec",
35 + "stoneham-et-tewkesbury": "Québec",
36 + "lac-beauport": "Québec",
37 + "quebec": "Québec",
38 + "petite-riviere-saint-francois": "Charlevoix",
39 + "baie-saint-paul": "Charlevoix",
40 + "la-malbaie": "Charlevoix",
41 + "les-eboulements": "Charlevoix",
42 + "saint-sauveur": "Laurentides",
43 + "sainte-adele": "Laurentides",
44 + "mont-tremblant": "Laurentides",
45 +}
46 +
47 +_STAT_RE = re.compile(
48 + r'<span class="(capacity|bedrooms|beds|restrooms)">'
49 + r"(?:(?!</span>).)*?([\d.,]+)\s*</span>", re.S)
50 +
51 +_PRIX_RE = re.compile(
52 + r"<h3>\s*(\d+)\s*nuits?\s*à partir de[^<]*<br\s*/?>\s*"
53 + r"<span>\s*([\d\s,. ]+)\s*\$\s*</span>", re.S)
54 +
55 +
56 +def _montant(raw: str) -> float | None:
57 + try:
58 + return float(re.sub(r"[\s ]", "", raw).replace(",", "."))
59 + except ValueError:
60 + return None
61 +
62 +
63 +class ChaletsAlpins(StConnector):
64 + source_id = "chaletsalpins"
65 +
66 + # -- inventaire (sitemap FR + lastmod) -----------------------------------
67 + def _sitemap_urls(self) -> dict[str, tuple[str, str]]:
68 + """slug -> (url détail FR, lastmod)."""
69 + xml = self.get(SITEMAP).text
70 + urls: dict[str, tuple[str, str]] = {}
71 + for bloc in re.findall(r"<url>(.*?)</url>", xml, re.S):
72 + m = re.search(r"<loc>([^<]+)</loc>", bloc)
73 + if not m:
74 + continue
75 + loc = m.group(1).strip()
76 + mu = _URL_DETAIL.match(loc)
77 + if not mu or mu.group(1) in ("hebergement",):
78 + continue
79 + lastmod = re.search(r"<lastmod>([^<]+)</lastmod>", bloc)
80 + urls.setdefault(mu.group(1),
81 + (loc, lastmod.group(1) if lastmod else ""))
82 + return urls
83 +
84 + # -- page détail ---------------------------------------------------------
85 + def _detail(self, url: str) -> dict:
86 + # ⚠️ le serveur ajoute parfois APRÈS </html> un second rendu avec des
87 + # chalets suggérés (autres compteurs/photos) : on tronque au 1er </html>
88 + h = self.get(url).text.split("</html>", 1)[0]
89 + d: dict = {}
90 +
91 + # JSON-LD Hotel : nom, description, adresse, animaux, commodités
92 + for m in re.finditer(r'<script[^>]*application/ld\+json[^>]*>(.*?)'
93 + r"</script>", h, re.S):
94 + try:
95 + data = json.loads(m.group(1), strict=False)
96 + except ValueError:
97 + continue
98 + if not (isinstance(data, dict) and data.get("@type") == "Hotel"):
99 + continue
100 + d["title"] = (data.get("name") or "").strip()
101 + d["description"] = re.sub(
102 + r"\s+", " ", (data.get("description") or "")).strip()[:5000]
103 + addr = data.get("address") or {}
104 + d["city"] = (addr.get("addressLocality") or "").strip()
105 + if data.get("petsAllowed") is not None:
106 + d["pets"] = "oui" if str(data["petsAllowed"]) in (
107 + "True", "true", "1") else "non"
108 + amen = []
109 + for feat in data.get("amenityFeature") or []:
110 + nom = (feat.get("name") or "").strip() \
111 + if isinstance(feat, dict) else ""
112 + if nom and nom not in amen:
113 + amen.append(nom)
114 + if amen:
115 + d["amenities"] = amen
116 + break
117 +
118 + # compteurs de l'entête (capacité, chambres, lits, salles de bain) —
119 + # 1re occurrence seulement (le chalet courant précède toute suggestion)
120 + for cls, val in _STAT_RE.findall(h):
121 + n = _montant(val)
122 + if n is None:
123 + continue
124 + d.setdefault({"capacity": "capacity", "bedrooms": "bedrooms",
125 + "beds": "beds", "restrooms": "bathrooms"}[cls], n)
126 +
127 + # encadré latéral : « 2 nuits à partir de (printemps) 1 025.00 $ »
128 + # → prix / nuit ; certaines unités affichent « Location mensuelle »
129 + # (long terme : pas de prix à la nuit, mention conservée)
130 + m = _PRIX_RE.search(h)
131 + if m:
132 + nuits = int(m.group(1)) or 1
133 + montant = _montant(m.group(2))
134 + if montant:
135 + d["price_night"] = round(montant / nuits, 2)
136 + d["price_label"] = re.sub(
137 + r"\s+", " ", _html.unescape(
138 + re.sub(r"<[^>]+>", " ", m.group(0)))).strip()
139 + elif re.search(r'sidebar scrollbox">\s*<div class="top">\s*'
140 + r"<h3>\s*Location mensuelle", h):
141 + d["location_mensuelle"] = True
142 +
143 + # no CITQ (dans le nom/slug « …(CITQ282170) » ou la description)
144 + m = re.search(r"CITQ\)?\s*:?\s*#?\s*(\d{6})",
145 + d.get("title", "") + " " + d.get("description", ""))
146 + if m:
147 + d["citq"] = m.group(1)
148 +
149 + # photos du carrousel (vignettes et icônes écartées, dédoublonnage
150 + # sur le nom de base sans suffixe de taille -WxH)
151 + imgs, vus = [], set()
152 + for u in re.findall(r'(https://(?:www\.)?chaletsalpins\.ca/'
153 + r'wp-content/uploads/[^"\'\s>]+'
154 + r"\.(?:jpe?g|png|webp))", h):
155 + base = re.sub(r"-\d+x\d+(?=\.\w+$)", "", u)
156 + if base in vus or "-150x150" in u:
157 + continue
158 + vus.add(base)
159 + imgs.append(u)
160 + d["images"] = imgs[:20]
161 + return d
162 +
163 + # -- contrat --------------------------------------------------------------
164 + def fetch(self) -> list[StListing]:
165 + listings: list[StListing] = []
166 + for slug, (url, lastmod) in self._sitemap_urls().items():
167 + try:
168 + d = self.detail(slug, lastmod or "sans-lastmod",
169 + lambda u=url: self._detail(u))
170 + except Exception: # une fiche cassée ≠ inventaire perdu
171 + d = {}
172 + if not d.get("title"):
173 + continue
174 + ville = d.get("city", "")
175 + region = _VILLE_REGION.get(
176 + strip_accents(ville).lower().replace(" ", "-"), "")
177 + details = ({"location_mensuelle": True}
178 + if d.get("location_mensuelle") else {})
179 + listings.append(StListing(
180 + source=self.source_id,
181 + external_id=slug, # slug WP, stable
182 + url=url,
183 + title=d["title"],
184 + property_type="Chalet",
185 + city=ville,
186 + region=region,
187 + price_night=d.get("price_night"),
188 + price_label=d.get("price_label", ""),
189 + capacity=d.get("capacity"),
190 + bedrooms=d.get("bedrooms"),
191 + beds=d.get("beds"),
192 + bathrooms=d.get("bathrooms"),
193 + pets=d.get("pets"),
194 + citq=d.get("citq", ""),
195 + description=d.get("description", ""),
196 + amenities=d.get("amenities") or [],
197 + details=details,
198 + images=d.get("images") or [],
199 + ))
200 + return listings
added louka/shortterm/connectors/chaletsbsl.py +168 −0
@@ -0,0 +1,168 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/chaletsbsl.py : Chalets BSL (chaletsbsl.com) — 4 chalets avec spa
4 +# sur un domaine privé de Saint-Simon-de-Rimouski (Bas-Saint-Laurent).
5 +#
6 +# Méthode : sitemap.xml (index) → sitemap_sections_*.xml → URLs /chalets/<slug>
7 +# + lastmod (clé du cache détail ; les /en/ sont ignorées). Pages statiques
8 +# (CMS maison Bootstrap) :
9 +# - <h3 class="mt-4 fw-600"> = titre (préfixe « Chalets BSL - » retiré) ;
10 +# - <h5 class="prix"> « À partir de 610$ pour 2 nuits » → price_night ;
11 +# - <p class="capacite"> « 2 pers. 1 chbre. 1 sdb. » ;
12 +# - description = <p> entre la capacité et le bloc country-info ;
13 +# - commodités = <h6> des blocs country-name ;
14 +# - galerie = var chaletImgs = {"ete": […], …} (JSON par saison) ;
15 +# - lat/lng = const myLatLng = { lat: …, lng: … } ; CITQ en pied de fiche.
16 +# -----------------------------------------------------------------------------
17 +from __future__ import annotations
18 +
19 +import html as _html
20 +import json
21 +import re
22 +
23 +from ..schema import StListing
24 +from .base import StConnector
25 +
26 +SITE = "https://chaletsbsl.com"
27 +SITEMAP = SITE + "/sitemap.xml"
28 +
29 +_TAG_RE = re.compile(r"<[^>]+>")
30 +
31 +
32 +def _text(fragment: str) -> str:
33 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
34 +
35 +
36 +class ChaletsBsl(StConnector):
37 + source_id = "chaletsbsl"
38 + request_delay = 1.0
39 +
40 + # -- page détail ----------------------------------------------------------
41 + def _detail(self, url: str) -> dict:
42 + h = self.get(url).text
43 + d: dict = {}
44 +
45 + m = re.search(r'(?s)<h3 class="mt-4 fw-600[^"]*"[^>]*>(.*?)</h3>', h)
46 + if m:
47 + d["title"] = re.sub(r"^Chalets BSL\s*-\s*", "", _text(m.group(1)))
48 +
49 + # « À partir de 610$ pour 2 nuits » → 305 $/nuit
50 + m = re.search(r"À partir de\s*([\d\s]+)\$\s*pour\s*(\d+)\s*nuits", h)
51 + if m:
52 + total = float(m.group(1).replace(" ", ""))
53 + nights = int(m.group(2))
54 + if nights and 20 <= total / nights <= 20000:
55 + d["price_night"] = round(total / nights)
56 + d["price_ref"] = f"{total:g} $ pour {nights} nuits"
57 +
58 + # « 2 pers. 1 chbre. 1 sdb. »
59 + m = re.search(r'(?s)<p class="capacite[^"]*"[^>]*>(.*?)</p>', h)
60 + if m:
61 + frag = _text(m.group(1))
62 + for pat, key in ((r"(\d+)\s*pers", "capacity"),
63 + (r"(\d+)\s*chbre", "bedrooms"),
64 + (r"(\d+)\s*sdb", "bathrooms")):
65 + mm = re.search(pat, frag)
66 + if mm:
67 + d[key] = float(mm.group(1))
68 +
69 + # description : les <p> entre la capacité et le bloc country-info
70 + i = h.find('class="capacite')
71 + j = h.find("country-info")
72 + if 0 < i < j:
73 + paras = [_text(p) for p in
74 + re.findall(r"(?s)<p[^>]*>(.*?)</p>", h[i:j])]
75 + texte = "\n".join(p for p in paras if len(p) > 40)
76 + if texte:
77 + d["description"] = texte[:5000]
78 +
79 + # commodités : les <h6> (tous portés par les blocs country-name)
80 + amen: list[str] = []
81 + for h6 in re.findall(r"(?s)<h6[^>]*>(.*?)</h6>", h):
82 + t = _text(h6)
83 + if t and t not in amen:
84 + amen.append(t)
85 + if amen:
86 + d["amenities"] = amen
87 +
88 + m = re.search(r"const myLatLng = \{\s*lat:\s*(-?\d+\.\d+),"
89 + r"\s*lng:\s*(-?\d+\.\d+)", h)
90 + if m:
91 + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2))
92 + m = re.search(r"CITQ\D{0,25}(\d{6})", h)
93 + if m:
94 + d["citq"] = m.group(1)
95 +
96 + # galerie : var chaletImgs = {"ete": […], "hiver": […]}
97 + m = re.search(r"var chaletImgs = (\{.*?\});", h, re.S)
98 + if m:
99 + try:
100 + seasons = json.loads(m.group(1))
101 + imgs: list[str] = []
102 + for key in ("ete", *sorted(k for k in seasons if k != "ete")):
103 + for u in seasons.get(key) or []:
104 + if isinstance(u, str) and u.startswith("https://") \
105 + and u not in imgs and len(imgs) < 20:
106 + imgs.append(u)
107 + if imgs:
108 + d["images"] = imgs
109 + except ValueError:
110 + pass
111 + return d
112 +
113 + # -- contrat --------------------------------------------------------------
114 + def fetch(self) -> list[StListing]:
115 + index = self.get(SITEMAP).text
116 + entries: list[tuple[str, str]] = []
117 + for sub in re.findall(r"<loc>([^<]+)</loc>", index):
118 + if "sitemap_sections" not in sub:
119 + continue
120 + xml = self.get(sub).text
121 + entries += re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>"
122 + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml)
123 +
124 + listings: list[StListing] = []
125 + vus: set[str] = set()
126 + for url, lastmod in entries:
127 + m = re.match(r"https://chaletsbsl\.com/chalets/([^/]+)/?$", url)
128 + if not m:
129 + continue
130 + slug = m.group(1)
131 + if slug in vus:
132 + continue
133 + vus.add(slug)
134 +
135 + det = self.detail(slug, lastmod or "v1",
136 + lambda u=url: self._detail(u))
137 + title = det.get("title") or ""
138 + if not title:
139 + continue
140 +
141 + price = det.get("price_night")
142 + details = {k: v for k, v in {
143 + "price_ref": det.get("price_ref") or "",
144 + "domain": "Chalets BSL",
145 + }.items() if v}
146 +
147 + listings.append(StListing(
148 + source=self.source_id,
149 + external_id=slug,
150 + url=url,
151 + title=title,
152 + property_type="Chalet",
153 + city="Saint-Simon-de-Rimouski",
154 + region="Bas-Saint-Laurent",
155 + price_night=float(price) if price else None,
156 + price_label=f"à partir de {price:g} $ / nuit" if price else "",
157 + capacity=det.get("capacity"),
158 + bedrooms=det.get("bedrooms"),
159 + bathrooms=det.get("bathrooms"),
160 + citq=det.get("citq") or "",
161 + description=det.get("description") or "",
162 + amenities=det.get("amenities") or [],
163 + details=details,
164 + images=det.get("images") or [],
165 + lat=det.get("lat"),
166 + lng=det.get("lng"),
167 + ))
168 + return listings
added louka/shortterm/connectors/chaletsdanslenord.py +219 −0
@@ -0,0 +1,219 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/chaletsdanslenord.py : Les Chalets dans le Nord
4 +# (leschaletsdanslenord.com — Laurentides)
5 +#
6 +# Petite agence familiale (Sainte-Lucie-des-Laurentides / lac Sarrazin,
7 +# ~6 chalets) — vitrine WordPress + moteur de réservation HOSTAWAY
8 +# (reservation.leschaletsdanslenord.com, compte 96792).
9 +#
10 +# Méthode :
11 +# 1. IDS : la racine du moteur Hostaway (Next.js rendu serveur) référence
12 +# tous les chalets via des liens "/listings/<id>" ; la homepage WP donne
13 +# en plus prix (« dès N $ / nuit ») et lien de la fiche vitrine (l'id
14 +# Hostaway est dans l'URL des photos S3 `96792-<id>-…`).
15 +# 2. DÉTAIL (cache self.detail) : /listings/<id> du moteur embarque le JSON
16 +# complet dans le payload React Flight (`self.__next_f`) : prix de base
17 +# par nuit, lat/lng, ville, capacité, chambres, sdb, lits, type, note
18 +# (sur 10 → /2 par finalize), nb d'avis, ~50 photos, ~70 commodités et
19 +# description (référence Flight « $xx » résolue via les segments T<hex>).
20 +# External_id = id de listing Hostaway (stable, dans l'URL du moteur).
21 +# -----------------------------------------------------------------------------
22 +from __future__ import annotations
23 +
24 +import json
25 +import re
26 +
27 +from ..schema import StListing
28 +from .base import StConnector
29 +
30 +SITE = "https://leschaletsdanslenord.com"
31 +ENGINE = "https://reservation.leschaletsdanslenord.com"
32 +
33 +
34 +def _num(v) -> float | None:
35 + try:
36 + return float(v) if v not in (None, "") else None
37 + except (TypeError, ValueError):
38 + return None
39 +
40 +
41 +def _flight_blob(html: str) -> str:
42 + parts = []
43 + for c in re.findall(r'self\.__next_f\.push\(\[1,"((?:[^"\\]|\\.)*)"\]\)',
44 + html):
45 + try:
46 + parts.append(json.loads(f'"{c}"'))
47 + except ValueError:
48 + continue
49 + return "".join(parts)
50 +
51 +
52 +def _flight_text(blob: str, ref: str) -> str:
53 + """Résout une référence texte Flight « $xx » (segment `xx:T<len hex>,`,
54 + longueur en OCTETS utf-8)."""
55 + rid = ref.lstrip("$")
56 + m = re.search(rf"(?:^|\n){re.escape(rid)}:T([0-9a-f]+),", blob)
57 + if not m:
58 + return ""
59 + n = int(m.group(1), 16)
60 + raw = blob[m.end():].encode("utf-8")[:n]
61 + return raw.decode("utf-8", errors="ignore")
62 +
63 +
64 +class ChaletsDansLeNord(StConnector):
65 + source_id = "chaletsdanslenord"
66 +
67 + # -- ids + carte prix/urls vitrine --------------------------------------
68 + def _engine_ids(self) -> list[str]:
69 + h = self.get(f"{ENGINE}/").text
70 + return sorted(set(re.findall(r'"/listings/(\d+)"', h)))
71 +
72 + def _wp_cards(self) -> dict[str, dict]:
73 + """id Hostaway → {url fiche vitrine, prix « dès N $ »} (homepage WP)."""
74 + try:
75 + h = self.get(f"{SITE}/").text
76 + except Exception:
77 + return {}
78 + cards: dict[str, dict] = {}
79 + for block in re.split(r'<li class="lcdln-hsg__card"', h)[1:]:
80 + m = re.search(r"hostaway-platform[^\"]*/listing/96792-(\d+)-",
81 + block)
82 + if not m:
83 + continue
84 + hid = m.group(1)
85 + card: dict = {}
86 + m = re.search(r'class="lcdln-hsg__card-btn" href="([^"]+)"', block)
87 + if m:
88 + card["url"] = m.group(1)
89 + m = re.search(r"dès\s*([\d ,]+)\s*\$\s*/\s*nuit", block)
90 + if m:
91 + card["price"] = float(m.group(1).replace(" ", "")
92 + .replace(",", "."))
93 + cards[hid] = card
94 + return cards
95 +
96 + def _wp_fiche(self, url: str) -> dict:
97 + """Titre + description EN FRANÇAIS depuis la fiche vitrine WP."""
98 + h = self.get(url).text
99 + d: dict = {}
100 + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h)
101 + if m:
102 + d["title"] = re.sub(r"\s+", " ",
103 + re.sub(r"<[^>]+>", " ", m.group(1))).strip()
104 + m = re.search(r"(?s)<main[^>]*>(.*?)</main>", h)
105 + if m:
106 + import html as _h
107 + paras, seen = [], set()
108 + for p in re.findall(r"(?s)<p[^>]*>(.*?)</p>", m.group(1)):
109 + t = _h.unescape(re.sub(r"\s+", " ",
110 + re.sub(r"<[^>]+>", " ", p))).strip()
111 + if len(t) < 60 or t in seen or "Voir les" in t[:30]:
112 + continue
113 + seen.add(t)
114 + paras.append(t)
115 + if len(paras) >= 10:
116 + break
117 + if paras:
118 + d["description"] = " ".join(paras)[:4000]
119 + return d
120 +
121 + # -- détail (moteur Hostaway + fiche vitrine FR) --------------------------
122 + def _detail(self, hid: str, wp_url: str = "") -> dict:
123 + h = self.get(f"{ENGINE}/listings/{hid}").text
124 + blob = _flight_blob(h)
125 + i = blob.find(f'"listing":{{"id":{hid}')
126 + if i < 0:
127 + return {}
128 + obj, _ = json.JSONDecoder().raw_decode(blob[i + len('"listing":'):])
129 + inner = obj.get("listing") or {}
130 +
131 + desc = str(inner.get("description") or "")
132 + if desc.startswith("$"):
133 + desc = _flight_text(blob, desc)
134 + desc = re.sub(r"\s+", " ", desc).strip()
135 +
136 + images = []
137 + for ph in obj.get("listingImage") or []:
138 + u = (ph or {}).get("url")
139 + if u and u not in images:
140 + images.append(u)
141 + if len(images) >= 20:
142 + break
143 +
144 + # fiche vitrine WP : titre + description en français (prioritaires)
145 + wp: dict = {}
146 + if wp_url:
147 + try:
148 + wp = self._wp_fiche(wp_url)
149 + except Exception:
150 + wp = {}
151 + if wp.get("description"):
152 + desc = wp["description"]
153 +
154 + pt = ((inner.get("propertyType") or {}).get("name") or "").strip()
155 + return {
156 + "title": wp.get("title") or (inner.get("name") or "").strip(),
157 + "price": _num(inner.get("price")),
158 + "lat": _num(inner.get("lat")),
159 + "lng": _num(inner.get("lng")),
160 + "city": (inner.get("city") or "").strip(),
161 + "capacity": _num(inner.get("personCapacity")),
162 + "bedrooms": _num(inner.get("bedroomsNumber")),
163 + "beds": _num(inner.get("bedsNumber")),
164 + "bathrooms": _num(inner.get("bathroomsNumber")),
165 + "property_type": pt,
166 + "rating": _num(obj.get("averageReviewRating")),
167 + "reviews": obj.get("reviewsCount"),
168 + "description": desc[:4000],
169 + "amenities": [n for n in
170 + (((a.get("amenity") or {}).get("name")
171 + or a.get("name") or "").strip()
172 + for a in obj.get("listingAmenity") or []
173 + if isinstance(a, dict)) if n][:80],
174 + "images": images,
175 + }
176 +
177 + # -- contrat ----------------------------------------------------------
178 + def fetch(self) -> list[StListing]:
179 + cards = self._wp_cards()
180 + listings: list[StListing] = []
181 + for hid in self._engine_ids():
182 + card = cards.get(hid) or {}
183 + key = json.dumps([hid, card.get("price"), card.get("url")])
184 + try:
185 + det = self.detail(
186 + hid, key,
187 + lambda i=hid, u=card.get("url") or "": self._detail(i, u))
188 + except Exception:
189 + det = {}
190 + if not det.get("title"):
191 + continue
192 +
193 + price = det.get("price") or card.get("price")
194 + reviews = det.get("reviews")
195 + listings.append(StListing(
196 + source=self.source_id,
197 + external_id=hid,
198 + url=card.get("url") or f"{ENGINE}/listings/{hid}",
199 + title=det["title"],
200 + property_type=det.get("property_type") or "Chalet",
201 + city=det.get("city") or "",
202 + region="Laurentides",
203 + price_night=price,
204 + price_label=(f"à partir de {price:.0f} $ / nuit"
205 + if price else ""),
206 + capacity=det.get("capacity"),
207 + bedrooms=det.get("bedrooms"),
208 + beds=det.get("beds"),
209 + bathrooms=det.get("bathrooms"),
210 + rating=det.get("rating"),
211 + reviews=int(reviews) if reviews else None,
212 + description=det.get("description") or "",
213 + amenities=det.get("amenities") or [],
214 + details={"booking_url": f"{ENGINE}/listings/{hid}"},
215 + images=det.get("images") or [],
216 + lat=det.get("lat"),
217 + lng=det.get("lng"),
218 + ))
219 + return listings
added louka/shortterm/connectors/domesstcome.py +157 −0
@@ -0,0 +1,157 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/domesstcome.py : Dômes St-Côme (domesstcome.com) — 4 dômes
4 +# géodésiques avec spa privé et vue panoramique à Saint-Côme (Lanaudière).
5 +#
6 +# Méthode : sitemap Wix (index) → dynamic-domes_*-sitemap.xml → 4 URLs
7 +# /domes/<slug> + lastmod (clé du cache détail). Pages Wix statiques :
8 +# - <h1> = nom du dôme ; sous-titre « Vue panoramique | Spa privé » ;
9 +# - « À partir de 370$/nuit » → price_night ;
10 +# - sections LITS / AUTRES / CUISINE / À L'EXTÉRIEUR / SALLE DE BAIN
11 +# (texte riche Wix) → amenities ; beds = nb de « Lit … » sous LITS ;
12 +# - CITQ (6 chiffres) dans le pied de page ;
13 +# - images : médias wixstatic ~mv2 servis en grand (w ≥ 900) — la galerie
14 +# Pro Gallery est chargée en JS, seuls les héros sont statiques.
15 +# -----------------------------------------------------------------------------
16 +from __future__ import annotations
17 +
18 +import html as _html
19 +import re
20 +
21 +from ..schema import StListing
22 +from .base import StConnector
23 +
24 +SITE = "https://www.domesstcome.com"
25 +SITEMAP = SITE + "/sitemap.xml"
26 +
27 +_SECTIONS = ("LITS", "AUTRES", "CUISINE", "À L'EXTÉRIEUR", "SALLE DE BAIN")
28 +
29 +_TAG_RE = re.compile(r"<[^>]+>")
30 +
31 +
32 +def _lines(fragment: str) -> list[str]:
33 + """HTML riche Wix → lignes de texte propres (CSS inline filtré)."""
34 + txt = re.sub(r"\|(?:\s*\|)+", "\n",
35 + re.sub(r"\s+", " ", _TAG_RE.sub("|", fragment)))
36 + out = []
37 + for x in txt.split("\n"):
38 + x = _html.unescape(x).strip(" |").strip()
39 + x = re.sub(r"\s*\|\s*", " | ", x)
40 + if x and len(x) > 2 and "{" not in x and "--" not in x:
41 + out.append(x)
42 + return out
43 +
44 +
45 +class DomesStCome(StConnector):
46 + source_id = "domesstcome"
47 + request_delay = 1.0
48 +
49 + # -- page détail ----------------------------------------------------------
50 + def _detail(self, url: str) -> dict:
51 + h = self.get(url).text
52 + d: dict = {}
53 +
54 + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h)
55 + if m:
56 + d["title"] = _html.unescape(
57 + re.sub(r"\s+", " ", _TAG_RE.sub(" ", m.group(1)))).strip()
58 +
59 + m = re.search(r"À partir de\s*(\d+)\s*\$\s*/\s*nuit", h)
60 + if m:
61 + d["price_night"] = float(m.group(1))
62 +
63 + m = re.search(r"CITQ\D{0,25}(\d{6})", h)
64 + if m:
65 + d["citq"] = m.group(1)
66 +
67 + # sous-titre (« Vue panoramique | Spa privé ») → description
68 + i, j = h.find("<h1"), h.find("LITS")
69 + if 0 <= i < j:
70 + head = [x for x in _lines(h[i:j])
71 + if x != d.get("title") and "À partir de" not in x
72 + and "Détails" not in x]
73 + if head:
74 + d["description"] = head[0][:500]
75 +
76 + # sections LITS…SALLE DE BAIN → amenities (entêtes exclues)
77 + i = h.find("LITS")
78 + j = h.find("Réserver", max(i, 0))
79 + beds = 0
80 + amen: list[str] = []
81 + if 0 <= i < j:
82 + unescaped_secs = {s for s in _SECTIONS}
83 + current = ""
84 + for x in _lines(h[i:j]):
85 + if x.upper() in unescaped_secs:
86 + current = x.upper()
87 + continue
88 + if x in amen or len(x) > 80:
89 + continue
90 + amen.append(x)
91 + if current == "LITS" and re.match(r"Lit\b", x, re.I):
92 + beds += 1
93 + if amen:
94 + d["amenities"] = amen
95 + if beds:
96 + d["beds"] = float(beds)
97 +
98 + # images : médias wixstatic servis en grand (héros)
99 + big: dict[str, int] = {}
100 + for mid, w in re.findall(r"static\.wixstatic\.com/media/"
101 + r"([\w~%.]+)/v1/fill/w_(\d+)", h):
102 + w = int(w)
103 + if w >= 900:
104 + big[mid] = max(big.get(mid, 0), w)
105 + imgs = [f"https://static.wixstatic.com/media/{mid}"
106 + for mid in big][:15]
107 + if imgs:
108 + d["images"] = imgs
109 + return d
110 +
111 + # -- contrat --------------------------------------------------------------
112 + def fetch(self) -> list[StListing]:
113 + index = self.get(SITEMAP).text
114 + entries: list[tuple[str, str]] = []
115 + for sub in re.findall(r"<loc>([^<]+)</loc>", index):
116 + if "dynamic-domes" not in sub:
117 + continue
118 + xml = self.get(sub).text
119 + entries += re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>"
120 + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml)
121 +
122 + listings: list[StListing] = []
123 + vus: set[str] = set()
124 + for url, lastmod in entries:
125 + m = re.match(r"https://www\.domesstcome\.com/domes/([^/]+)/?$", url)
126 + if not m:
127 + continue
128 + slug = m.group(1)
129 + if slug in vus:
130 + continue
131 + vus.add(slug)
132 +
133 + det = self.detail(slug, lastmod or "v1",
134 + lambda u=url: self._detail(u))
135 + title = det.get("title") or ""
136 + if not title:
137 + continue
138 +
139 + price = det.get("price_night")
140 + listings.append(StListing(
141 + source=self.source_id,
142 + external_id=slug,
143 + url=url,
144 + title=title,
145 + property_type="Dôme",
146 + city="Saint-Côme",
147 + region="Lanaudière",
148 + price_night=price,
149 + price_label=f"À partir de {price:g} $ / nuit" if price else "",
150 + beds=det.get("beds"),
151 + citq=det.get("citq") or "",
152 + description=det.get("description") or "",
153 + amenities=det.get("amenities") or [],
154 + details={"domain": "Dômes St-Côme"},
155 + images=det.get("images") or [],
156 + ))
157 + return listings
modified louka/shortterm/connectors/expedia.py +51 −3
@@ -21,7 +21,13 @@
21 21 # L'inventaire recoupe en partie Vrbo (même groupe) mais avec ses propres ids
22 22 # et des exclusivités hôtelières-résidentielles (apparts-hôtels, glamping).
23 23 #
24 −# Réglage env : LOUKA_EXPEDIA_LIMIT (nb max d'annonces, pour tester petit).
24 +# Enrichissement : la page détail hXXXXXX.Hotel-Information (Scrapfly ASP
25 +# SANS rendu JS — le SSR suffit) porte description, commodités, lat/lng et
26 +# ~6 photos (parseur partagé avec Vrbo : _expediadetail.py).
27 +#
28 +# Réglages env : LOUKA_EXPEDIA_LIMIT (nb max d'annonces, pour tester petit),
29 +# LOUKA_EXPEDIA_DETAIL_LIMIT (fetchs détail par sync, défaut
30 +# 100 ; cache permanent, le parc se complète au fil des syncs).
25 31 # -----------------------------------------------------------------------------
26 32 from __future__ import annotations
27 33
@@ -33,8 +39,13 @@ from urllib.parse import quote
33 39 from bs4 import BeautifulSoup
34 40
35 41 from ..schema import StListing
42 +from . import _expediadetail as _ed
36 43 from .base import StConnector
37 44
45 +
46 +class _DetailSkip(Exception):
47 + """Fiche détail sautée (budget épuisé / page invalide) — pas de cache."""
48 +
38 49 # (destination Expedia, ville affichée par défaut, région touristique QC)
39 50 DESTINATIONS = [
40 51 ("Mont-Tremblant, Quebec, Canada", "Mont-Tremblant", "Laurentides"),
@@ -203,6 +214,39 @@ class Expedia(StConnector):
203 214 images=images,
204 215 )
205 216
217 + # -- enrichissement par la page détail ---------------------------------------
218 + def _enrich_details(self, listings: list[StListing]) -> None:
219 + """Visite les fiches détail via le cache self.detail() sous budget :
220 + les hits de cache sont gratuits, seuls les fetchs réseau comptent."""
221 + limit = max(0, int(os.environ.get("LOUKA_EXPEDIA_DETAIL_LIMIT", "100")
222 + or 100))
223 + used = enriched = streak = 0
224 + for lst in listings:
225 + def fetch_fn(url=lst.url):
226 + nonlocal used, streak
227 + if used >= limit or streak >= 5: # tempête anti-bot : on coupe
228 + raise _DetailSkip
229 + used += 1
230 + html = self.get_scrapfly(url, render_js=False, asp=True)
231 + payload = _ed.parse_detail(html)
232 + if not payload:
233 + streak += 1
234 + raise _DetailSkip # blocage/vide : pas de cache
235 + streak = 0
236 + return payload
237 +
238 + try:
239 + d = self.detail(lst.external_id, "v1", fetch_fn)
240 + except _DetailSkip:
241 + continue
242 + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run
243 + continue
244 + if d:
245 + _ed.apply_detail(lst, d)
246 + enriched += 1
247 + print(f"[expedia] détail : {enriched} annonces enrichies"
248 + f" ({used}/{limit} fetchs réseau)", file=sys.stderr)
249 +
206 250 # -- contrat -----------------------------------------------------------------
207 251 def fetch(self) -> list[StListing]:
208 252 limit = int(os.environ.get("LOUKA_EXPEDIA_LIMIT", "0") or 0)
@@ -216,7 +260,9 @@ class Expedia(StConnector):
216 260 for dest, city, region in DESTINATIONS:
217 261 for sort in SORTS:
218 262 if limit and len(listings) >= limit:
219 − return list(listings.values())
263 + out = list(listings.values())
264 + self._enrich_details(out)
265 + return out
220 266 url = ("https://www.expedia.ca/Hotel-Search?destination="
221 267 + quote(dest)
222 268 + "&adults=2&categorySearch=vacation_rentals_option")
@@ -250,4 +296,6 @@ class Expedia(StConnector):
250 296 print(f"[expedia] {city} ({sort}) :"
251 297 f" {len(listings) - n_before} nouvelles"
252 298 f" (total {len(listings)})", file=sys.stderr)
253 − return list(listings.values())
299 + out = list(listings.values())
300 + self._enrich_details(out)
301 + return out
added louka/shortterm/connectors/gitesauquebec.py +218 −0
@@ -0,0 +1,218 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/gitesauquebec.py : GitesAuQuebec.com — annuaire indépendant de
4 +# gîtes et auberges (B&B). Annuaire en fin de vie : l'inventaire ACTIF est
5 +# aujourd'hui minuscule (~7 fiches, « Page 1 de 1 » sur la recherche sans
6 +# critère), mais les fiches restantes sont riches et le connecteur suivra
7 +# l'inventaire s'il remonte.
8 +#
9 +# Méthode (ASP.NET WebForms, aucun anti-bot) :
10 +# 1. LISTE : GET /Resultats.aspx sans critère → toutes les annonces actives
11 +# (cartes en tables imbriquées : lien /<id>, tarif « 145$ - 205$ nuit »).
12 +# La pagination (« Page 1 de N ») est un postback __VIEWSTATE : tant que
13 +# N = 1 on n'en a pas besoin ; si N > 1 un avertissement est émis (le
14 +# rejouer n'a rien à répliquer aujourd'hui, l'annuaire tient sur 1 page).
15 +# 2. FICHE /<id> (cache self.detail, clé mensuelle) : contrôles ASP.NET
16 +# stables — titre « PL-<id> : Nom », CPH_litInfoGen (type, capacité,
17 +# chambres, salles de bain, lits, animaux, fumeurs), localisation
18 +# (région / ville), CPH_lblDesc, CPH_lblNoEtablissementValeur (CITQ),
19 +# CPH_hidAdrMap (« lat, lng »), CPH_pnlTarif, équipements (img alt),
20 +# photos /_photos/grand/.
21 +# -----------------------------------------------------------------------------
22 +from __future__ import annotations
23 +
24 +import re
25 +import sys
26 +import time
27 +
28 +from ..schema import StListing, normalize_region
29 +from .base import StConnector
30 +
31 +BASE = "https://www.gitesauquebec.com"
32 +RESULTATS = f"{BASE}/Resultats.aspx"
33 +
34 +_ID_RE = re.compile(r"href='/(\d+)'")
35 +_PAGE_RE = re.compile(r"Page\s+(\d+)\s+de\s+(\d+)")
36 +_TARIF_RE = re.compile(r"(\d[\d\s,.]*\$[^<]{0,40})")
37 +_COORD_RE = re.compile(
38 + r'id="CPH_hidAdrMap" value="(-?\d+\.\d+),\s*(-?\d+\.\d+)"')
39 +_NUM_RE = re.compile(r"(\d+)")
40 +
41 +# « Type hébergement » de l'annuaire → type canonique Lou-Ka
42 +_TYPES = {"gîte": "Gîte", "gite": "Gîte", "auberge": "Auberge",
43 + "b&b": "Gîte", "couette et café": "Gîte"}
44 +
45 +# l'annuaire écrit « Cantons de l'est / Estrie », « Centre du Québec »…
46 +_REGIONS = {"cantons de l'est / estrie": "Cantons-de-l'Est",
47 + "centre du québec": "Centre-du-Québec"}
48 +
49 +
50 +class GitesAuQuebec(StConnector):
51 + source_id = "gitesauquebec"
52 + request_delay = 1.0
53 +
54 + # -- liste ------------------------------------------------------------------
55 + def _liste(self) -> dict[str, str]:
56 + """Annonces actives → {id: libellé de tarif de la carte}."""
57 + html = self.get(RESULTATS).text
58 + m = _PAGE_RE.search(re.sub(r"<[^>]+>", " ", html))
59 + if m and int(m.group(2)) > 1:
60 + print(f"[gitesauquebec] pagination inattendue ({m.group(0)}) : "
61 + "seule la page 1 est lue (postback __VIEWSTATE à rejouer)",
62 + file=sys.stderr)
63 + tarifs: dict[str, str] = {}
64 + # une carte = tout le HTML entre deux liens de fiche successifs
65 + morceaux = _ID_RE.split(html)
66 + for i in range(1, len(morceaux), 2):
67 + gid, bloc = morceaux[i], morceaux[i + 1]
68 + if gid in tarifs and tarifs[gid]:
69 + continue
70 + tarif = ""
71 + j = bloc.find("TARIFICATION")
72 + if j >= 0:
73 + mm = _TARIF_RE.search(re.sub(r"<[^>]+>", " ", bloc[j:j + 800]))
74 + if mm:
75 + tarif = re.sub(r"\s+", " ", mm.group(1)).strip()
76 + tarifs[gid] = tarif
77 + return tarifs
78 +
79 + # -- fiche ------------------------------------------------------------------
80 + def _fetch_fiche(self, gid: str) -> dict:
81 + from bs4 import BeautifulSoup
82 + html = self.get(f"{BASE}/{gid}").text
83 + soup = BeautifulSoup(html, "html.parser")
84 + d: dict = {}
85 +
86 + # en-tête « PL-3778 : Gite du Village » (préfixe variable : PL, DI…)
87 + h1 = soup.find(string=re.compile(rf"[A-Z]{{1,3}}-{gid}\s*:"))
88 + if h1:
89 + d["nom"] = h1.split(":", 1)[1].strip()
90 + m = re.match(rf"([A-Z]{{1,3}}-{gid})", h1.strip())
91 + if m:
92 + d["no_annonce"] = m.group(1)
93 + if not d.get("nom"):
94 + # repli : <title> = « Gîte <ville>, <région>, <nom…>, XX-<id> »
95 + title = soup.title.get_text(strip=True) if soup.title else ""
96 + m = re.match(rf"[^,]+,[^,]+,\s*(.+?),?\s*([A-Z]{{1,3}}-{gid})$",
97 + title)
98 + if m:
99 + d["nom"], d["no_annonce"] = m.group(1).strip(), m.group(2)
100 + if not d.get("nom"):
101 + return {}
102 +
103 + # Informations générales : « Étiquette :&nbsp;Valeur » dans CPH_litInfoGen
104 + infos: dict[str, str] = {}
105 + bloc = soup.find(id="CPH_litInfoGen")
106 + if bloc is not None:
107 + for cell in bloc.get_text("\n").split("\n"):
108 + label, sep, val = cell.partition(":")
109 + if sep and val.strip():
110 + infos[label.strip().lower()] = val.strip()
111 + d["type"] = infos.get("type hébergement", "")
112 + m = _NUM_RE.search(infos.get("capacité d'accueil", ""))
113 + if m:
114 + d["capacity"] = int(m.group(1))
115 + m = _NUM_RE.search(infos.get("chambres", ""))
116 + if m:
117 + d["bedrooms"] = int(m.group(1))
118 + m = _NUM_RE.search(infos.get("salles de bain", ""))
119 + if m:
120 + d["bathrooms"] = int(m.group(1))
121 + d["lits"] = infos.get("lits", "")
122 + animaux = infos.get("animaux permis", "").lower()
123 + if animaux.startswith("oui"):
124 + d["pets"] = "oui"
125 + elif animaux.startswith("non"):
126 + d["pets"] = "non"
127 +
128 + # Localisation : lignes « Région : / Ville : » du tableau
129 + texte = soup.get_text("\n")
130 + for champ, cle in (("Région", "region"), ("Ville", "ville")):
131 + m = re.search(rf"{champ}\s*:\s*\n+\s*([^\n]+)", texte)
132 + if m:
133 + d[cle] = m.group(1).strip()
134 +
135 + desc = soup.find(id="CPH_lblDesc")
136 + if desc is not None:
137 + d["description"] = desc.get_text("\n", strip=True)[:2500]
138 +
139 + citq = soup.find(id="CPH_lblNoEtablissementValeur")
140 + if citq is not None:
141 + d["citq"] = citq.get_text(strip=True)
142 +
143 + m = _COORD_RE.search(html)
144 + if m:
145 + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2))
146 +
147 + tarif = soup.find(id="CPH_pnlTarif")
148 + if tarif is not None:
149 + d["tarif"] = re.sub(
150 + r"\s+", " ",
151 + tarif.get_text(" ", strip=True).removeprefix("Tarification")
152 + ).strip()[:300]
153 +
154 + # équipements & activités : alt des pictogrammes
155 + amen: list[str] = []
156 + for img in soup.find_all("img", src=re.compile("equipement")):
157 + alt = (img.get("alt") or "").strip()
158 + if alt and alt not in amen:
159 + amen.append(alt)
160 + d["amenities"] = amen[:40]
161 +
162 + imgs: list[str] = []
163 + for img in soup.find_all("img", src=re.compile(r"/_photos/")):
164 + u = img.get("src") or ""
165 + u = re.sub(r"/_photos/(thumb/)?", "/_photos/grand/", u)
166 + if not u.startswith("http"):
167 + u = BASE + u
168 + if u not in imgs:
169 + imgs.append(u)
170 + d["images"] = imgs[:12]
171 + return d
172 +
173 + # -- contrat ----------------------------------------------------------------
174 + def fetch(self) -> list[StListing]:
175 + month = time.strftime("%Y-%m") # re-visite mensuelle (tarifs)
176 + listings: list[StListing] = []
177 + for gid, tarif_carte in sorted(self._liste().items(),
178 + key=lambda kv: int(kv[0])):
179 + try:
180 + d = self.detail(gid, month,
181 + lambda g=gid: self._fetch_fiche(g))
182 + except Exception as exc: # noqa: BLE001
183 + print(f"[gitesauquebec] fiche {gid} : {exc}", file=sys.stderr)
184 + continue
185 + if not d.get("nom"):
186 + continue
187 +
188 + price_label = d.get("tarif") or tarif_carte
189 + region = d.get("region", "")
190 + region = _REGIONS.get(region.lower(), normalize_region(region))
191 + details = {k: v for k, v in {
192 + "no_annonce": d.get("no_annonce", f"PL-{gid}"),
193 + "lits": d.get("lits", ""),
194 + }.items() if v}
195 +
196 + listings.append(StListing(
197 + source=self.source_id,
198 + external_id=gid,
199 + url=f"{BASE}/{gid}",
200 + title=d["nom"],
201 + property_type=_TYPES.get(d.get("type", "").lower(), "Gîte"),
202 + city=d.get("ville", ""),
203 + region=region,
204 + price_label=price_label,
205 + capacity=float(d["capacity"]) if d.get("capacity") else None,
206 + bedrooms=float(d["bedrooms"]) if d.get("bedrooms") else None,
207 + bathrooms=(float(d["bathrooms"])
208 + if d.get("bathrooms") else None),
209 + pets=d.get("pets"),
210 + citq=d.get("citq", ""),
211 + description=d.get("description", ""),
212 + amenities=d.get("amenities") or [],
213 + details=details,
214 + images=d.get("images") or [],
215 + lat=d.get("lat"),
216 + lng=d.get("lng"),
217 + ))
218 + return listings
modified louka/shortterm/connectors/gitespassant.py +29 −4
@@ -17,7 +17,10 @@
17 17 # /fr/repertoire/detailorganization/id/<id> — HTML serveur : nom (h1),
18 18 # description, « Types d'hébergement » (Gîte/Auberge du Passant, Maison
19 19 # de Campagne…), numéro CITQ (#attestation), classification (soleils),
20 −# adresse postale (spans), téléphone, site web, photos du carrousel.
20 +# adresse postale (spans), téléphone, site web, photos du carrousel,
21 +# coordonnées lat/lng extraites de l'iframe Google Maps (paramètres
22 +# !2d<lng>!3d<lat> de l'embed — présent sur toutes les fiches),
23 +# chambres/capacité glanées dans la description quand mentionnées.
21 24 # Pas de prix affiché (les tarifs sont chez chaque établissement).
22 25 # -----------------------------------------------------------------------------
23 26 from __future__ import annotations
@@ -69,6 +72,11 @@ _TYPES = [
69 72 ]
70 73
71 74 _TAG_RE = re.compile(r"<[^>]+>")
75 +# iframe Google Maps : …/maps/embed?pb=…!2d<longitude>!3d<latitude>!…
76 +_GMAP_RE = re.compile(r"google\.com/maps/embed\?pb=[^\"']*"
77 + r"!2d(-?\d+\.\d+)!3d(-?\d+\.\d+)")
78 +_CAP_RE = re.compile(r"(\d{1,2})\s*personnes")
79 +_CHAMBRES_RE = re.compile(r"(\d{1,2})\s*chambres")
72 80
73 81
74 82 def _text(fragment: str) -> str:
@@ -149,6 +157,12 @@ class GitesPassant(StConnector):
149 157 if m:
150 158 d["website"] = m.group(1)
151 159
160 + # géolocalisation : centre de l'iframe Google Maps (brut : l'iframe
161 + # est parfois injectée par un <script> retiré de `h`)
162 + m = _GMAP_RE.search(brut)
163 + if m:
164 + d["lng"], d["lat"] = float(m.group(1)), float(m.group(2))
165 +
152 166 # photos : le carrousel de la fiche seulement (ailleurs sur la page,
153 167 # les « Suggestions d'articles » ont aussi des images membres)
154 168 logo = ""
@@ -183,8 +197,9 @@ class GitesPassant(StConnector):
183 197
184 198 listings: list[StListing] = []
185 199 for mid in ids:
186 − # fiche rafraîchie une fois par mois (pas de lastmod côté liste)
187 − det = self.detail(mid, time.strftime("%Y-%m"),
200 + # fiche rafraîchie une fois par mois (pas de lastmod côté liste) ;
201 + # v2 : parseur géo (iframe Google Maps) ajouté 2026-08-25
202 + det = self.detail(mid, time.strftime("%Y-%m") + "-v2",
188 203 lambda i=mid: self._fiche(i))
189 204 if not det.get("title"):
190 205 continue
@@ -203,6 +218,12 @@ class GitesPassant(StConnector):
203 218 "(Terroir et Saveurs du Québec)",
204 219 }.items() if v}
205 220
221 + # capacité / chambres : mentions dans la description (rare)
222 + desc = det.get("description") or ""
223 + caps = [int(x) for x in _CAP_RE.findall(desc) if 1 <= int(x) <= 40]
224 + rooms = [int(x) for x in _CHAMBRES_RE.findall(desc)
225 + if 1 <= int(x) <= 30]
226 +
206 227 street = det.get("street") or ""
207 228 city = det.get("city") or ""
208 229 listings.append(StListing(
@@ -214,10 +235,14 @@ class GitesPassant(StConnector):
214 235 address=", ".join(p for p in (street, city) if p),
215 236 city=city,
216 237 region=regions.get(mid, ""),
238 + capacity=float(max(caps)) if caps else None,
239 + bedrooms=float(max(rooms)) if rooms else None,
217 240 citq=det.get("citq") or "",
218 − description=det.get("description") or "",
241 + description=desc,
219 242 details=details,
220 243 images=det.get("images") or [],
244 + lat=det.get("lat"),
245 + lng=det.get("lng"),
221 246 ))
222 247 if limit and len(listings) >= limit:
223 248 break
modified louka/shortterm/connectors/glampinghub.py +71 −2
@@ -11,20 +11,36 @@
11 11 # catégorie, ville + coordonnées, prix (estimated_rate.daily_rate en devise
12 12 # originale, CAD au Québec), chambres/lits, capacité par unité
13 13 # (units_distribution), note sur 5, commodités (nested_features), photos.
14 −# Une seule requête paginée suffit — pas de page détail nécessaire.
14 +# Une seule requête paginée suffit pour la liste ; seule la DESCRIPTION
15 +# n'y figure pas → elle est prise sur la page détail (aucun anti-bot non
16 +# plus), dans le bloc SSR <noscript id="description-content">, avec repli
17 +# sur la <meta name="description">. Cache permanent dans louka_ct.db.
15 18 #
16 19 # URL publique : https://glampinghub.com<absolute_url_en>. Région touristique
17 20 # déduite des coordonnées (centroïdes partagés avec airbnb.py).
18 −# Réglage env : LOUKA_GLAMPINGHUB_LIMIT (nb max de fiches, 0 = tout).
21 +# Réglages env : LOUKA_GLAMPINGHUB_LIMIT (nb max de fiches, 0 = tout),
22 +# LOUKA_GLAMPINGHUB_DETAIL_LIMIT (fetchs détail/sync, défaut 250).
19 23 # -----------------------------------------------------------------------------
20 24 from __future__ import annotations
21 25
26 +import html as _html
22 27 import os
28 +import re
29 +import sys
23 30
24 31 from ..schema import StListing
25 32 from .base import StConnector
26 33 from .airbnb import _region_from_latlng
27 34
35 +
36 +class _DetailSkip(Exception):
37 + """Fiche détail sautée (budget épuisé) — pas de mise en cache."""
38 +
39 +
40 +_DESC_RE = re.compile(r'<noscript id="description-content">(.*?)</noscript>',
41 + re.S)
42 +_META_RE = re.compile(r'<meta name="description" content="([^"]+)"')
43 +
28 44 SITE = "https://glampinghub.com"
29 45 API = f"{SITE}/search-accommodations/"
30 46 PAGE_SIZE = 24
@@ -87,6 +103,58 @@ class GlampingHub(StConnector):
87 103 page += 1
88 104 return items
89 105
106 + # -- description (page détail, SSR ouvert) --------------------------------
107 + @staticmethod
108 + def _parse_description(html: str) -> dict:
109 + """{description} depuis le bloc <noscript> SSR (repli : meta)."""
110 + desc = ""
111 + m = _DESC_RE.search(html or "")
112 + if m:
113 + txt = re.sub(r"<br\s*/?>|</p>|</h2>", "\n", m.group(1))
114 + txt = _html.unescape(re.sub(r"<[^>]+>", " ", txt))
115 + lines = [re.sub(r"\s+", " ", ln).strip() for ln in txt.split("\n")]
116 + # écarter le bruit : lignes « … », slogan SEO de bas de bloc et
117 + # liste de commodités hors-plateforme (pas une description)
118 + lines = [ln for ln in lines if ln and ln not in ("...", "…")
119 + and not re.match(r"^Book your dream .*!$", ln)
120 + and not ln.startswith("Amenities not shown on")]
121 + desc = "\n".join(lines).strip()
122 + if len(desc) < 40: # fiche laconique (« … ») : repli meta
123 + mm = _META_RE.search(html or "")
124 + meta = _html.unescape(mm.group(1)).strip() if mm else ""
125 + if len(meta) > len(desc):
126 + desc = meta
127 + return {"description": desc[:6000]} if desc else {}
128 +
129 + def _enrich_details(self, listings: list[StListing]) -> None:
130 + """Complète la description via la page détail, sous budget (les hits
131 + de cache sont gratuits, seuls les fetchs réseau comptent)."""
132 + limit = max(0, int(os.environ.get("LOUKA_GLAMPINGHUB_DETAIL_LIMIT",
133 + "250") or 250))
134 + used = enriched = 0
135 + for lst in listings:
136 + if not lst.url.startswith(SITE + "/"):
137 + continue
138 + def fetch_fn(url=lst.url):
139 + nonlocal used
140 + if used >= limit:
141 + raise _DetailSkip
142 + used += 1
143 + return self._parse_description(self.get(url).text)
144 +
145 + try:
146 + d = self.detail(lst.external_id, "v1", fetch_fn)
147 + except _DetailSkip:
148 + continue
149 + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run
150 + continue
151 + if d.get("description") and len(d["description"]) > \
152 + len(lst.description or ""):
153 + lst.description = d["description"]
154 + enriched += 1
155 + print(f"[glampinghub] détail : {enriched} descriptions"
156 + f" ({used}/{limit} fetchs réseau)", file=sys.stderr)
157 +
90 158 # -- contrat --------------------------------------------------------------
91 159 def fetch(self) -> list[StListing]:
92 160 limit = int(os.environ.get("LOUKA_GLAMPINGHUB_LIMIT", "0") or 0)
@@ -174,4 +242,5 @@ class GlampingHub(StConnector):
174 242 ))
175 243 if limit and len(listings) >= limit:
176 244 break
245 + self._enrich_details(listings)
177 246 return listings
added louka/shortterm/connectors/hebergementcharlevoix.py +164 −0
@@ -0,0 +1,164 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/hebergementcharlevoix.py : Hébergement Charlevoix
4 +# (hebergement-charlevoix.com) — agence de Charlevoix, ~330 chalets.
5 +#
6 +# Méthode : site PHP maison, HTML 100 % serveur. Sitemap /sitemap.xml
7 +# (301 vers le domaine sans www) → URLs détail
8 +# /fr/chalet-a-louer/<ville>/voir/<CODE> (code alphanum stable, ex. ADE-440 ;
9 +# les autres URLs du sitemap sont des pages de catégories/villes/activités).
10 +# Chaque page détail porte un JSON-LD schema.org VacationRental complet :
11 +# occupancy, containsPlace (chambres, sdb, lits par type), petsAllowed,
12 +# address (ville, rue, code postal), geo (⚠️ décimales à VIRGULE),
13 +# offers.lowPrice (plus bas prix/nuit), images 1080×1080 et licenseNum CITQ.
14 +# Les commodités lisibles (FR) viennent des pictos <div class="bulle">.
15 +# La note affichée (4,6/1370) est GLOBALE au site : on l'ignore.
16 +# Pas de lastmod : clé de cache mensuelle.
17 +# -----------------------------------------------------------------------------
18 +from __future__ import annotations
19 +
20 +import json
21 +import re
22 +import time
23 +
24 +from ..schema import StListing
25 +from .base import StConnector
26 +
27 +BASE = "https://hebergement-charlevoix.com"
28 +SITEMAP = BASE + "/sitemap.xml"
29 +
30 +_URL_DETAIL = re.compile(
31 + r"https://(?:www\.)?hebergement-charlevoix\.com/fr/chalet-a-louer/"
32 + r"([\w-]+)/voir/([\w-]+)/?$")
33 +
34 +# villes desservies hors de la région touristique de Charlevoix
35 +_HORS_CHARLEVOIX = {
36 + "tadoussac": "Côte-Nord",
37 + "saint-ferreol-les-neiges": "Québec",
38 + "saint-tite-des-caps": "Québec",
39 +}
40 +
41 +
42 +def _virgule(v) -> float | None:
43 + """Nombre JSON-LD du site : « 47,568534 » (virgule décimale)."""
44 + try:
45 + return float(str(v).replace(",", "."))
46 + except (TypeError, ValueError):
47 + return None
48 +
49 +
50 +class HebergementCharlevoix(StConnector):
51 + source_id = "hebergementcharlevoix"
52 +
53 + # -- inventaire (sitemap) ----------------------------------------------
54 + def _sitemap_urls(self) -> dict[str, tuple[str, str]]:
55 + """code -> (url détail, slug ville)."""
56 + xml = self.get(SITEMAP).text
57 + urls: dict[str, tuple[str, str]] = {}
58 + for loc in re.findall(r"<loc>([^<]+)</loc>", xml):
59 + m = _URL_DETAIL.match(loc.strip())
60 + if m:
61 + urls.setdefault(m.group(2), (loc.strip(), m.group(1)))
62 + return urls
63 +
64 + # -- page détail (tout est dans le JSON-LD VacationRental) ---------------
65 + def _detail(self, url: str) -> dict:
66 + h = self.get(url).text
67 + d: dict = {}
68 + data = None
69 + for m in re.finditer(r'<script[^>]*application/ld\+json[^>]*>(.*?)'
70 + r"</script>", h, re.S):
71 + try:
72 + cand = json.loads(m.group(1), strict=False)
73 + except ValueError:
74 + continue
75 + if isinstance(cand, dict) and cand.get("@type") == "VacationRental":
76 + data = cand
77 + break
78 + if data:
79 + d["title"] = (data.get("name") or "").strip()
80 + d["description"] = (data.get("description") or "").strip()
81 + d["property_type"] = (data.get("additionalType") or "").strip()
82 + occ = data.get("occupancy") or {}
83 + if occ.get("value") is not None:
84 + d["capacity"] = _virgule(occ["value"])
85 + place = data.get("containsPlace") or {}
86 + if place.get("numberOfBedrooms") is not None:
87 + d["bedrooms"] = _virgule(place["numberOfBedrooms"])
88 + if place.get("numberOfBathroomsTotal") is not None:
89 + d["bathrooms"] = _virgule(place["numberOfBathroomsTotal"])
90 + beds = sum(_virgule(b.get("numberOfBeds")) or 0
91 + for b in place.get("bed") or [] if isinstance(b, dict))
92 + if beds:
93 + d["beds"] = beds
94 + if place.get("petsAllowed") is not None:
95 + d["pets"] = "oui" if place["petsAllowed"] in (
96 + True, "true", "True", 1) else "non"
97 + addr = data.get("address") or {}
98 + d["city"] = (addr.get("addressLocality") or "").strip()
99 + d["address"] = (addr.get("streetAddress") or "").strip()
100 + geo = data.get("geo") or {}
101 + d["lat"] = _virgule(geo.get("latitude"))
102 + d["lng"] = _virgule(geo.get("longitude"))
103 + offers = data.get("offers") or {}
104 + low = _virgule(offers.get("lowPrice"))
105 + if low:
106 + d["price_night"] = round(low, 2)
107 + d["price_label"] = f"à partir de {low:.2f} $ / nuit"
108 + for feat in data.get("amenityFeature") or []:
109 + if not isinstance(feat, dict):
110 + continue
111 + if feat.get("name") == "licenseNum":
112 + m2 = re.search(r"(\d{6})", str(feat.get("value") or ""))
113 + if m2 and m2.group(1) != "000000": # placeholder du site
114 + d["citq"] = m2.group(1)
115 + imgs = data.get("image") or []
116 + if isinstance(imgs, str):
117 + imgs = [imgs]
118 + d["images"] = [u for u in imgs if isinstance(u, str)][:20]
119 +
120 + # commodités lisibles : pictos <div class="bulle">Foyer</div>
121 + amen = []
122 + for a in re.findall(r'<div class="bulle"[^>]*>([^<]+)</div>', h):
123 + a = re.sub(r"\s+", " ", a).strip()
124 + if a and a not in amen:
125 + amen.append(a)
126 + if amen:
127 + d["amenities"] = amen
128 + return d
129 +
130 + # -- contrat --------------------------------------------------------------
131 + def fetch(self) -> list[StListing]:
132 + cle = "detail-" + time.strftime("%Y-%m") # pas de lastmod → mensuel
133 + listings: list[StListing] = []
134 + for code, (url, ville_slug) in self._sitemap_urls().items():
135 + try:
136 + d = self.detail(code, cle, lambda u=url: self._detail(u))
137 + except Exception: # une fiche cassée ≠ inventaire perdu
138 + d = {}
139 + if not d.get("title"):
140 + continue
141 + listings.append(StListing(
142 + source=self.source_id,
143 + external_id=code, # code interne (ADE-440), stable
144 + url=url,
145 + title=d["title"],
146 + property_type=d.get("property_type") or "Chalet",
147 + address=d.get("address", ""),
148 + city=d.get("city", ""),
149 + region=_HORS_CHARLEVOIX.get(ville_slug, "Charlevoix"),
150 + price_night=d.get("price_night"),
151 + price_label=d.get("price_label", ""),
152 + capacity=d.get("capacity"),
153 + bedrooms=d.get("bedrooms"),
154 + beds=d.get("beds"),
155 + bathrooms=d.get("bathrooms"),
156 + pets=d.get("pets"),
157 + citq=d.get("citq", ""),
158 + description=d.get("description", ""),
159 + amenities=d.get("amenities") or [],
160 + images=d.get("images") or [],
161 + lat=d.get("lat"),
162 + lng=d.get("lng"),
163 + ))
164 + return listings
added louka/shortterm/connectors/hebergia.py +164 −0
@@ -0,0 +1,164 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/hebergia.py : Hébergia (hebergia.ca)
4 +#
5 +# Agence multi-régions (Cantons-de-l'Est, Laurentides, Lanaudière, Charlevoix,
6 +# Centre-du-Québec) — WordPress + moteur Guesty, ~60 chalets.
7 +#
8 +# Méthode :
9 +# 1. LISTE : la page /chalets/ embarque `window.chalets = [...]` (JSON
10 +# complet : gid Guesty, prix de base/nuit, url, titre, RÉGION, capacité,
11 +# chambres, lits, lat/lng, vignette). Une seule requête.
12 +# 2. DÉTAIL (cache self.detail, clé = champs stables de la liste) : la page
13 +# WP fr /chalets/<slug>/ fournit ville (« Austin, Cantons-de-l'Est »),
14 +# salles de bain (bloc inshort), description, commodités (section
15 +# « Commodités incluses »), photos (assets.guesty.com) et numéro CITQ
16 +# (dans le règlement).
17 +# External_id = gid (id de listing Guesty, stable).
18 +# -----------------------------------------------------------------------------
19 +from __future__ import annotations
20 +
21 +import html as _html
22 +import json
23 +import re
24 +
25 +from ..schema import StListing
26 +from .base import StConnector
27 +
28 +SITE = "https://hebergia.ca"
29 +
30 +_TAG_RE = re.compile(r"<[^>]+>")
31 +
32 +
33 +def _text(fragment: str) -> str:
34 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
35 +
36 +
37 +def _num(v) -> float | None:
38 + try:
39 + return float(v) if v not in (None, "") else None
40 + except (TypeError, ValueError):
41 + return None
42 +
43 +
44 +class Hebergia(StConnector):
45 + source_id = "hebergia"
46 +
47 + # -- liste ----------------------------------------------------------------
48 + def _list_items(self) -> list[dict]:
49 + h = self.get(f"{SITE}/chalets/").text
50 + m = re.search(r"window\.chalets\s*=\s*(\[.*?\]);", h, re.S)
51 + if not m:
52 + return []
53 + try:
54 + return json.loads(m.group(1))
55 + except ValueError:
56 + return []
57 +
58 + # -- page détail ------------------------------------------------------
59 + def _detail(self, url: str) -> dict:
60 + h = self.get(url).text
61 + d: dict = {}
62 +
63 + m = re.search(r'class="sous-titre-localisation">([^<]+)<', h)
64 + if m:
65 + loc = _text(m.group(1))
66 + d["city"] = loc.split(",")[0].strip()
67 +
68 + # bloc « inshort » : 8 voyageurs / 3 chambres / lits / 2 salles de bain
69 + m = re.search(r"(\d+)\s*salles?\s*de\s*bain", h)
70 + if m:
71 + d["bathrooms"] = float(m.group(1))
72 +
73 + # description : paragraphes de l'article avant les accordéons
74 + m = re.search(r"(?s)<article class=\"hbg-fiche-chalet-content\">"
75 + r".*?</h1>(.*?)<h2 class=\"toggler\"", h)
76 + if m:
77 + frag = re.sub(r'(?s)<p class="sous-titre-localisation">.*?</p>',
78 + " ", m.group(1))
79 + paras = re.findall(r"(?s)<p[^>]*>(.*?)</p>", frag)
80 + d["description"] = " ".join(_text(p) for p in paras
81 + if _text(p))[:4000]
82 +
83 + # commodités : section « Commodités incluses » (lignes à puces)
84 + m = re.search(r'(?s)Commodités incluses</button></h2>\s*'
85 + r'<div class="toToggle"[^>]*>(.*?)</div>', h)
86 + if m:
87 + amens = []
88 + for line in re.split(r"<br\s*/?>|</p>", m.group(1)):
89 + t = _text(line).lstrip("•· ").strip()
90 + if 2 <= len(t) <= 80 and not t.isupper():
91 + amens.append(t)
92 + d["amenities"] = amens[:60]
93 +
94 + # photos (CDN Guesty, dédupliquées)
95 + imgs: list[str] = []
96 + for u in re.findall(r'data-flickity-lazyload-src="'
97 + r'(https://assets\.guesty\.com/[^"]+)"', h):
98 + if u not in imgs:
99 + imgs.append(u)
100 + d["images"] = imgs[:20]
101 +
102 + m = re.search(r"CITQ\D{0,12}(\d{6})", h, re.I)
103 + if m:
104 + d["citq"] = m.group(1)
105 +
106 + m = re.search(r"animaux non admis|pas d.animaux|no pets", h, re.I)
107 + if m:
108 + d["pets"] = "non"
109 + return d
110 +
111 + # -- contrat ----------------------------------------------------------
112 + def fetch(self) -> list[StListing]:
113 + listings: list[StListing] = []
114 + for it in self._list_items():
115 + gid = str(it.get("gid") or "").strip()
116 + url = it.get("url") or ""
117 + title = _text(str(it.get("title") or ""))
118 + if not gid or not url or not title or "/en/" in url:
119 + continue
120 +
121 + key = json.dumps([gid, title, it.get("price"),
122 + it.get("accommodates"), it.get("bedrooms"),
123 + it.get("beds"), it.get("shortDesc")],
124 + ensure_ascii=False)
125 + try:
126 + det = self.detail(gid, key, lambda u=url: self._detail(u))
127 + except Exception: # une fiche détail cassée ≠ annonce perdue
128 + det = {}
129 +
130 + price = _num(it.get("price"))
131 + starting = _num(it.get("startingPrice"))
132 + if starting and starting > 0:
133 + price = starting
134 + desc = det.get("description") or ""
135 + short = _text(str(it.get("shortDesc") or ""))
136 + if short and short not in desc:
137 + desc = f"{short} {desc}".strip()
138 +
139 + listings.append(StListing(
140 + source=self.source_id,
141 + external_id=gid,
142 + url=url,
143 + title=title,
144 + property_type="Chalet",
145 + city=det.get("city") or "",
146 + region=it.get("region") or "",
147 + price_night=price,
148 + price_label=(f"à partir de {price:.0f} $ / nuit"
149 + if price else ""),
150 + capacity=_num(it.get("accommodates")),
151 + bedrooms=_num(it.get("bedrooms")),
152 + beds=_num(it.get("beds")),
153 + bathrooms=det.get("bathrooms"),
154 + pets=det.get("pets"),
155 + citq=det.get("citq") or "",
156 + description=desc[:4000],
157 + amenities=det.get("amenities") or [],
158 + details={"guesty_id": gid},
159 + images=det.get("images")
160 + or ([it["thumbnail"]] if it.get("thumbnail") else []),
161 + lat=_num(it.get("lat")),
162 + lng=_num(it.get("lng")),
163 + ))
164 + return listings
added louka/shortterm/connectors/homminichalets.py +142 −0
@@ -0,0 +1,142 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/homminichalets.py : HOM Mini Chalets (homminichalets.com) —
4 +# 12 mini-chalets numérotés (01. Le Renard … 12. Le Loup) à Val-des-Monts
5 +# (Outaouais), en deux gammes : « avec spa » (01-08) et « avec circuit
6 +# thermal » (09-12, spa + sauna + hammam + douche froide).
7 +#
8 +# Méthode : le site Shopify renvoie un 429 systématique en direct → Scrapfly.
9 +# 1. UIDS : /pages/chalets (rendu JS, le calendrier Hostfully est monté en
10 +# JS) → 12 liens /pages/chalets?uid=<uuid> ; la page liste aussi les noms
11 +# groupés par gamme (« Mini chalets avec circuit thermal » précède
12 +# 09-12) et le pied de page porte les 2 adresses + le CITQ commun 298559.
13 +# 2. DÉTAIL : l'API publique JSONP du moteur Hostfully répond en DIRECT
14 +# (pas de 429) : platform.hostfully.com/getproperty_api.jsp?propertyUID=…
15 +# &aid=ORB-… → {"price": 300, "maximumGuests": 2, "minStay": 1,
16 +# "name": "07. Le Huard"}. Pas d'endpoint photos public → galerie
17 +# GÉNÉRIQUE du site (01.jpg-25.jpg de l'accueil, via Scrapfly sans
18 +# render_js), signalée par details.images_generic. Commodités/description
19 +# par gamme (bandeaux de l'accueil), adresses par numéro d'unité
20 +# (07-08 = chemin du Saphir, le reste = chemin du Rubis).
21 +# -----------------------------------------------------------------------------
22 +from __future__ import annotations
23 +
24 +import json
25 +import re
26 +import sys
27 +
28 +from ..schema import StListing
29 +from .base import StConnector
30 +
31 +SITE = "https://homminichalets.com"
32 +LIST_URL = SITE + "/pages/chalets"
33 +
34 +# API publique du widget de réservation Hostfully (accessible en direct)
35 +AID = "ORB-49587220416635719"
36 +API = ("https://platform.hostfully.com/getproperty_api.jsp"
37 + "?jsoncallback=cb&propertyUID={uid}&aid=" + AID)
38 +
39 +CITQ = "298559"
40 +
41 +# gammes (bandeaux de l'accueil) : commodités + description
42 +_SPA_AMEN = ["Spa privé", "Chaise hamac", "Lit King",
43 + "Foyer intérieur", "Plancher chauffant"]
44 +_SPA_DESC = ("Mini chalet avec spa. Profitez d'un mini chalet de luxe tout "
45 + "équipé avec spa privé sur la galerie, foyer intérieur et "
46 + "plancher chauffant.")
47 +_THERMAL_AMEN = ["Spa", "Sauna", "Hammam", "Douche froide", "Lit Queen",
48 + "Foyer intérieur", "Plancher chauffant"]
49 +_THERMAL_DESC = ("Mini chalet avec circuit thermal (spa, sauna, hammam, "
50 + "douche froide). Vivez une expérience de détente privée et "
51 + "luxueuse dans un mini chalet tout équipé avec foyer "
52 + "intérieur et plancher chauffant.")
53 +
54 +
55 +class HomMiniChalets(StConnector):
56 + source_id = "homminichalets"
57 + request_delay = 1.0
58 +
59 + # -- découverte des uids (Scrapfly, calendrier monté en JS) ---------------
60 + def _uids(self) -> list[str]:
61 + h = self.get_scrapfly(LIST_URL, render_js=True, rendering_wait=4000)
62 + uids: list[str] = []
63 + for u in re.findall(r'href="[^"]*?/pages/chalets\?uid='
64 + r'([0-9a-f-]{36})"', h):
65 + if u not in uids:
66 + uids.append(u)
67 + return uids
68 +
69 + # -- galerie générique du site (accueil Shopify statique) -----------------
70 + def _site_images(self) -> list[str]:
71 + try:
72 + h = self.get_scrapfly(SITE + "/", render_js=False)
73 + except Exception as exc: # noqa: BLE001
74 + print(f"[homminichalets] accueil : {exc}", file=sys.stderr)
75 + return []
76 + imgs: list[str] = []
77 + for num, v in re.findall(r"//homminichalets\.com/cdn/shop/files/"
78 + r"(\d{2}\.jpg)\?v=(\d+)", h):
79 + u = f"{SITE}/cdn/shop/files/{num}?v={v}&width=1600"
80 + if u not in imgs:
81 + imgs.append(u)
82 + return imgs[:15]
83 +
84 + # -- fiche Hostfully (JSONP, en direct) ------------------------------------
85 + def _detail(self, uid: str) -> dict:
86 + txt = self.get(API.format(uid=uid)).text.strip()
87 + m = re.match(r"(?s)cb\(true,(\{.*\})\)$", txt)
88 + return json.loads(m.group(1)) if m else {}
89 +
90 + # -- contrat ---------------------------------------------------------------
91 + def fetch(self) -> list[StListing]:
92 + uids = self._uids()
93 + images = self._site_images() if uids else []
94 +
95 + listings: list[StListing] = []
96 + for uid in uids:
97 + det = self.detail(uid, "v1", lambda u=uid: self._detail(u))
98 + name = str(det.get("name") or "").strip()
99 + if not name:
100 + continue
101 +
102 + m = re.match(r"(\d{1,2})\.", name)
103 + num = int(m.group(1)) if m else 0
104 + thermal = num >= 9
105 + address = ("32 chemin du Saphir" if num in (7, 8)
106 + else "154 chemin du Rubis")
107 +
108 + price = det.get("price")
109 + price = float(price) if price and 20 <= float(price) <= 20000 \
110 + else None
111 +
112 + details = {
113 + "gamme": ("circuit thermal" if thermal else "spa"),
114 + "domain": "HOM Mini Chalets",
115 + }
116 + if images:
117 + details["images_generic"] = True
118 + if det.get("minStay"):
119 + details["min_stay"] = f"{det['minStay']} nuit(s)"
120 +
121 + listings.append(StListing(
122 + source=self.source_id,
123 + external_id=uid,
124 + url=f"{LIST_URL}?uid={uid}",
125 + title=f"{name} — HOM Mini Chalets",
126 + property_type="Mini-chalet",
127 + address=address,
128 + city="Val-des-Monts",
129 + region="Outaouais",
130 + price_night=price,
131 + price_label=(f"à partir de {price:g} $ / nuit"
132 + if price else ""),
133 + capacity=(float(det["maximumGuests"])
134 + if det.get("maximumGuests") else None),
135 + bedrooms=1.0,
136 + citq=CITQ,
137 + description=_THERMAL_DESC if thermal else _SPA_DESC,
138 + amenities=list(_THERMAL_AMEN if thermal else _SPA_AMEN),
139 + details=details,
140 + images=list(images),
141 + ))
142 + return listings
modified louka/shortterm/connectors/kijiji.py +95 −25
@@ -20,11 +20,18 @@
20 20 #
21 21 # Filtres court terme : on ne garde que les annonces OFFER qui ressemblent à
22 22 # un hébergement (attributs chambres/personnes/type de vacances présents —
23 −# la catégorie contient aussi maillots de bain, machines à espresso…) et on
24 −# écarte les locations au mois (« 31 jours et plus », monthly…). Le prix
25 −# Kijiji est un montant sans période : price_night n'est rempli que si le
26 −# texte précise « /nuit » ou « /semaine », sinon le montant affiché va dans
27 −# details.prix_affiche.
23 +# la catégorie contient aussi maillots de bain, machines à espresso… ; si
24 +# les attributs manquent sur la liste mais que le titre évoque un
25 +# hébergement, la fiche détail tranche) et on écarte les locations au mois
26 +# (« 31 jours et plus », monthly, minnights >= 28…).
27 +#
28 +# Prix : le formulaire de la catégorie demande un prix À LA NUIT — un texte
29 +# « X $/nuit » ou « $X/night » dans l'annonce prime (minimum des saisons),
30 +# sinon le montant affiché est pris comme prix/nuit s'il est plausible
31 +# (<= 2 000 $ et séjour min < 28 nuits), sinon details.prix_affiche.
32 +# Salles de bain : la valeur canonique Kijiji est en dixièmes (« 20 » = 2) —
33 +# normalisée. Commodités : aucune dans les attributs de la catégorie — on
34 +# les dérive des mentions explicites du texte (spa, sauna, foyer, BBQ…).
28 35 #
29 36 # Réglage env : LOUKA_KIJIJI_LIMIT (nb max d'annonces, pour tester petit).
30 37 # -----------------------------------------------------------------------------
@@ -37,6 +44,7 @@ import time
37 44
38 45 import requests
39 46
47 +from ...normalize import strip_accents
40 48 from ..schema import StListing, normalize_region, parse_price_night, REGIONS
41 49 from .airbnb import _region_from_latlng
42 50 from .base import StConnector
@@ -54,10 +62,54 @@ _MENSUEL_RE = re.compile(
54 62 r"au mois|par mois|/\s*mois|mensuel|monthly|per\s+month|/\s*month"
55 63 r"|3[01]\s*jours\s*(?:et plus|minimum|min)|month(?:ly)?\s+rental", re.I)
56 64
57 −_NUIT_RE = re.compile(r"(\d[\d\s,.]{0,9})\s*\$\s*(?:/|par|la|per)?\s*"
58 − r"(?:nuit|night)", re.I)
59 −_SEMAINE_RE = re.compile(r"(\d[\d\s,.]{0,9})\s*\$\s*(?:/|par|la|per)?\s*"
60 − r"(?:sem(?:aine)?|week)", re.I)
65 +# « 265 $ / nuit » (fr) comme « $265/night » (en) — $ avant ou après le montant
66 +_NUIT_RE = re.compile(
67 + r"(?:\$\s*(\d[\d\s,.]{0,8}\d|\d)|(\d[\d\s,.]{0,8}\d|\d)\s*\$)\s*"
68 + r"(?:/|par|la|per)?\s*(?:nuit|night)", re.I)
69 +_SEMAINE_RE = re.compile(
70 + r"(?:\$\s*(\d[\d\s,.]{0,8}\d|\d)|(\d[\d\s,.]{0,8}\d|\d)\s*\$)\s*"
71 + r"(?:/|par|la|per)?\s*(?:sem(?:aine)?|week)", re.I)
72 +
73 +
74 +def _prix_min(texte: str, rx: re.Pattern) -> float | None:
75 + """Le plus bas des montants d'une période (les annonces listent souvent
76 + plusieurs saisons : « $265/night … $298/night »)."""
77 + vals = []
78 + for m in rx.finditer(texte or ""):
79 + raw = (m.group(1) or m.group(2) or "").strip()
80 + v = parse_price_night(f"{raw} $")
81 + if v:
82 + vals.append(v)
83 + return min(vals) if vals else None
84 +
85 +
86 +# mentions explicites du texte → commodité affichable (la catégorie Kijiji
87 +# n'a aucun attribut de commodités) ; clés en minuscules sans accents
88 +_AMEN_HINTS = [
89 + ("spa", "Spa"), ("jacuzzi", "Spa"), ("hot tub", "Spa"),
90 + ("sauna", "Sauna"), ("piscine", "Piscine"), ("pool", "Piscine"),
91 + ("foyer", "Foyer"), ("fireplace", "Foyer"),
92 + ("poele a bois", "Poêle à bois"), ("wood stove", "Poêle à bois"),
93 + ("bbq", "BBQ"), ("barbecue", "BBQ"),
94 + ("wifi", "Wi-Fi"), ("wi-fi", "Wi-Fi"), ("internet", "Wi-Fi"),
95 + ("lave-vaisselle", "Lave-vaisselle"), ("dishwasher", "Lave-vaisselle"),
96 + ("laveuse", "Laveuse/sécheuse"), ("washer", "Laveuse/sécheuse"),
97 + ("climatis", "Air climatisé"), ("air conditioning", "Air climatisé"),
98 + ("kayak", "Kayak"), ("canot", "Canot"), ("canoe", "Canot"),
99 + ("stationnement", "Stationnement"), ("parking", "Stationnement"),
100 + ("bord de l'eau", "Bord de l'eau"), ("bord du lac", "Bord de l'eau"),
101 + ("waterfront", "Bord de l'eau"), ("lakefront", "Bord de l'eau"),
102 + ("plage", "Plage à proximité"), ("beach", "Plage à proximité"),
103 +]
104 +
105 +
106 +def _amenities_texte(texte: str) -> list[str]:
107 + hay = strip_accents(texte or "").lower().replace("’", "'")
108 + out: list[str] = []
109 + for needle, label in _AMEN_HINTS:
110 + if needle in hay and label not in out:
111 + out.append(label)
112 + return out
61 113
62 114 _TYPE_HINTS = [
63 115 ("chalet", "Chalet"), ("cottage", "Chalet"), ("cabin", "Chalet"),
@@ -80,6 +132,16 @@ def _num(texts: list[str]) -> float | None:
80 132 return None
81 133
82 134
135 +def _sdb(attrs: dict) -> float | None:
136 + """Salles de bain : la valeur canonique Kijiji est en dixièmes
137 + (« 20 » = 2, « 25 » = 2,5) ; la valeur humaine (« 2 bathrooms ») est
138 + déjà correcte."""
139 + v = _num(attrs.get("numberbathrooms"))
140 + if v is not None and v >= 10 and v % 5 == 0:
141 + v /= 10
142 + return v
143 +
144 +
83 145 def _attrs(entity: dict) -> dict[str, list[str]]:
84 146 """{canonicalName: values (humaines si présentes, sinon canoniques)}."""
85 147 out: dict[str, list[str]] = {}
@@ -212,7 +274,9 @@ class KijijiCt(StConnector):
212 274 if (e.get("type") or "OFFER") != "OFFER":
213 275 continue
214 276 attrs = _attrs(e)
215 − if not (_ATTRS_HEBERGEMENT & set(attrs)):
277 + hay = title.lower()
278 + if not (_ATTRS_HEBERGEMENT & set(attrs)) \
279 + and not any(n in hay for n, _ in _TYPE_HINTS):
216 280 continue # maillots de bain, cafetières, vans…
217 281 texte = f"{title}\n{e.get('description') or ''}"
218 282 if _MENSUEL_RE.search(texte):
@@ -228,30 +292,35 @@ class KijijiCt(StConnector):
228 292 det = {}
229 293 if det.get("attrs"):
230 294 attrs = det["attrs"]
295 + if not (_ATTRS_HEBERGEMENT & set(attrs)):
296 + continue # le détail confirme : pas un hébergement
231 297 texte = (f"{title}\n"
232 298 f"{det.get('description') or e.get('description') or ''}")
233 299 if _MENSUEL_RE.search(texte):
234 300 continue
301 + nuits_min = _num(attrs.get("minnights")) or 0
302 + if nuits_min >= 28:
303 + continue # séjour min d'un mois : hors mandat
235 304
236 − # prix : montant en cents, période seulement si le texte la donne
305 + # prix : « X $/nuit » du texte (minimum des saisons) prime ;
306 + # sinon le montant affiché (en cents) est un prix à la nuit
307 + # (convention de la catégorie) s'il est plausible
237 308 price_night = None
238 309 price_label = ""
239 310 amount = (e.get("price") or {}).get("amount")
240 311 montant = round(amount / 100, 2) if isinstance(
241 312 amount, (int, float)) and amount else None
242 − m = _NUIT_RE.search(texte)
243 − if m:
244 − price_label = f"{m.group(1).strip()} $ / nuit"
245 − elif _SEMAINE_RE.search(texte):
246 − price_label = f"{_SEMAINE_RE.search(texte).group(1).strip()}" \
247 − " $ / semaine"
248 − elif montant:
249 − nuit = (attrs.get("minnights") or ["1"])[0]
250 − if str(nuit) in ("", "1"): # 1 nuit min : montant ≈ par nuit
251 − price_night = montant
252 − price_label = f"{montant:g} $"
253 − if price_label and price_night is None:
254 − price_night = parse_price_night(price_label)
313 + nuit_val = _prix_min(texte, _NUIT_RE)
314 + sem_val = _prix_min(texte, _SEMAINE_RE)
315 + if nuit_val:
316 + price_night = nuit_val
317 + price_label = f"{nuit_val:g} $ / nuit"
318 + elif sem_val:
319 + price_night = round(sem_val / 7, 2)
320 + price_label = f"{sem_val:g} $ / semaine"
321 + elif montant and montant <= 2000:
322 + price_night = montant
323 + price_label = f"{montant:g} $"
255 324
256 325 hay = title.lower()
257 326 ptype = next((canon for needle, canon in _TYPE_HINTS
@@ -302,9 +371,10 @@ class KijijiCt(StConnector):
302 371 price_label=price_label,
303 372 capacity=_num(attrs.get("maxpeople")),
304 373 bedrooms=_num(attrs.get("numberbedrooms")),
305 − bathrooms=_num(attrs.get("numberbathrooms")),
374 + bathrooms=_sdb(attrs),
306 375 pets=pets,
307 376 description=det.get("description") or "",
377 + amenities=_amenities_texte(texte),
308 378 details=details,
309 379 images=det.get("images")
310 380 or [u for u in (e.get("imageUrls") or [])
added louka/shortterm/connectors/lesversants.py +156 −0
@@ -0,0 +1,156 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/lesversants.py : Les Versants Mont-Tremblant (lesversants.com) —
4 +# agence immobilière de Mont-Tremblant dont la division location propose
5 +# ~16 maisons et condos en location saisonnière (Mont-Tremblant,
6 +# Lac-Supérieur, Mont-Blanc).
7 +#
8 +# Méthode : le sitemap.xml est un vestige statique (2 URLs), mais le thème
9 +# WordPress REAL_HOMES expose tout via l'API REST :
10 +# GET /fr/wp-json/wp/v2/property?property-status=79&per_page=100
11 +# (79 = terme « À louer » de la taxonomie property-status). Chaque fiche
12 +# embarque property_meta REAL_HOMES_* : adresse, lat/lng, chambres, salles
13 +# de bain (formats « 3 + 1 »), superficie, galerie (sizes.large), prix.
14 +# ⚠️ PRIX SAISONNIER : le prix affiché est « Hiver 2026-27 | 18 500 $ »
15 +# (un forfait pour LA SAISON, pas à la nuit) → on le range dans
16 +# details.season_price et on laisse price_night/price_label vides pour ne
17 +# pas empoisonner parse_price_night().
18 +# Taxonomies property-type (439 Condo à louer, 440/441 maisons), city et
19 +# mont_tremblant_sector résolues par une passe chacune.
20 +#
21 +# Réglage env : LOUKA_LESVERSANTS_LIMIT (nb max d'annonces, 0 = tout).
22 +# -----------------------------------------------------------------------------
23 +from __future__ import annotations
24 +
25 +import html as _html
26 +import os
27 +import re
28 +
29 +from ..schema import StListing
30 +from .base import StConnector
31 +
32 +API = "https://lesversants.com/fr/wp-json/wp/v2"
33 +STATUS_A_LOUER = 79 # terme « À louer » (property-status)
34 +
35 +# id property-type → type canonique
36 +_TYPES = {439: "Condo", 440: "Maison", 441: "Maison"}
37 +
38 +_TAG_RE = re.compile(r"<[^>]+>")
39 +
40 +
41 +def _strip_html(txt: str) -> str:
42 + return re.sub(r"\s+", " ", _TAG_RE.sub(" ", _html.unescape(txt or ""))).strip()
43 +
44 +
45 +def _rooms(v: str) -> float | None:
46 + """« 3 », « 3 + 1 », « 3.5 » → nombre (les « + N » sont additionnés)."""
47 + nums = re.findall(r"\d+(?:\.\d+)?", str(v or ""))
48 + return sum(float(n) for n in nums) if nums else None
49 +
50 +
51 +class LesVersants(StConnector):
52 + source_id = "lesversants"
53 + request_delay = 0.5
54 +
55 + def _get_json(self, url: str):
56 + return self.get(url, headers={"Accept": "application/json"}).json()
57 +
58 + def _tax(self, name: str) -> dict[int, str]:
59 + try:
60 + terms = self._get_json(f"{API}/{name}?per_page=100"
61 + "&_fields=id,name")
62 + return {t["id"]: _html.unescape(t.get("name") or "").strip()
63 + for t in terms}
64 + except Exception: # noqa: BLE001 — libellés manquants ≠ blocage
65 + return {}
66 +
67 + # -- contrat --------------------------------------------------------------
68 + def fetch(self) -> list[StListing]:
69 + limit = int(os.environ.get("LOUKA_LESVERSANTS_LIMIT", "0") or 0)
70 + items = self._get_json(f"{API}/property?property-status="
71 + f"{STATUS_A_LOUER}&per_page=100")
72 + if not isinstance(items, list):
73 + return []
74 + if limit:
75 + items = items[:limit]
76 +
77 + types = self._tax("property-type")
78 + cities = self._tax("city")
79 + sectors = self._tax("mont_tremblant_sector")
80 +
81 + listings: list[StListing] = []
82 + seen: set[str] = set()
83 + for it in items:
84 + pid = str(it.get("id") or "").strip()
85 + url = (it.get("link") or "").strip()
86 + title = _strip_html((it.get("title") or {}).get("rendered") or "")
87 + if not pid or pid in seen or not url or not title:
88 + continue
89 + seen.add(pid)
90 +
91 + meta = it.get("property_meta") or {}
92 + loc = meta.get("REAL_HOMES_property_location") or {}
93 + lat = float(loc["latitude"]) if loc.get("latitude") else None
94 + lng = float(loc["longitude"]) if loc.get("longitude") else None
95 +
96 + type_ids = it.get("property-type") or []
97 + ptype = next((_TYPES[t] for t in type_ids if t in _TYPES), "")
98 + if not ptype:
99 + ptype = next((types[t] for t in type_ids if t in types), "")
100 +
101 + city_ids = it.get("city") or []
102 + city = next((cities[c] for c in city_ids if c in cities),
103 + "Mont-Tremblant")
104 + sector_ids = it.get("mont_tremblant_sector") or []
105 + sector = next((sectors[s] for s in sector_ids if s in sectors), "")
106 +
107 + # prix SAISONNIER (« Hiver 2026-27 | 18 500 $ ») → details
108 + prefix = (meta.get("REAL_HOMES_property_price_prefix") or "").strip()
109 + raw_price = re.sub(r"[^\d.]", "",
110 + str(meta.get("REAL_HOMES_property_price") or ""))
111 + season_price = ""
112 + if raw_price:
113 + season = prefix.rstrip("|").strip()
114 + season_price = (f"{season} : {raw_price} $" if season
115 + else f"{raw_price} $")
116 +
117 + size = (meta.get("REAL_HOMES_property_size") or "").strip()
118 + size_post = (meta.get("REAL_HOMES_property_size_postfix")
119 + or "").strip()
120 +
121 + images: list[str] = []
122 + for ph in meta.get("REAL_HOMES_property_images") or []:
123 + sizes = (ph or {}).get("sizes") or {}
124 + u = ((sizes.get("1536x1536") or {}).get("url")
125 + or (sizes.get("large") or {}).get("url")
126 + or (sizes.get("medium_large") or {}).get("url") or "")
127 + if u.startswith("https://") and u not in images:
128 + images.append(u)
129 + if len(images) >= 15:
130 + break
131 +
132 + details = {k: v for k, v in {
133 + "season_price": season_price,
134 + "sector": sector,
135 + "size": f"{size} {size_post}".strip() if size else "",
136 + }.items() if v}
137 +
138 + listings.append(StListing(
139 + source=self.source_id,
140 + external_id=pid,
141 + url=url,
142 + title=title,
143 + property_type=ptype,
144 + address=(meta.get("REAL_HOMES_property_address") or "").strip(),
145 + city=city,
146 + region="Laurentides",
147 + bedrooms=_rooms(meta.get("REAL_HOMES_property_bedrooms")),
148 + bathrooms=_rooms(meta.get("REAL_HOMES_property_bathrooms")),
149 + description=_strip_html((it.get("content") or {})
150 + .get("rendered") or "")[:3000],
151 + details=details,
152 + images=images,
153 + lat=lat,
154 + lng=lng,
155 + ))
156 + return listings
added louka/shortterm/connectors/locationdechalets.py +265 −0
@@ -0,0 +1,265 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/locationdechalets.py : Location de Chalets Lanaudière
4 +# (locationdechalets.com) — petit parc de ~10 chalets avec spa privé à
5 +# Notre-Dame-de-la-Merci et Saint-Donat (Lanaudière).
6 +#
7 +# Méthode : sitemap.xml → /fr/chalets/<slug>/ + lastmod (clé du cache détail).
8 +# Pages statiques (CMS maison) riches mais sans JSON-LD :
9 +# - h1 « Chalet Spa Le Héron … Capacité de 2 personnes » ;
10 +# - bloc CARACTÉRISTIQUES (« N personnes maximum / N chambre(s) /
11 +# N salle(s) de bain ») ;
12 +# - sections INTÉRIEUR / EXTÉRIEUR / INCLUS en <ul><li> → amenities ;
13 +# - « EN SUS … Animaux : 10 $ / jour » → pets = conditions ;
14 +# - bloc Coordonnées (rue, ville, « Lanaudière (Québec) », code postal) ;
15 +# - no CITQ dans le pied de page ; lat/lng dans le lien Google Maps (@…) ;
16 +# - TARIF RÉGULIER « 2 nuits : 498 $ … » → price_night = total/2 nuits.
17 +# ⚠️ anti-scrape : zéros de bourrage blancs sur blanc
18 +# (<span style='color: #ffffff;'>0</span>) à retirer AVANT le parsing ;
19 +# - galerie : background:url(/fichiersUploadOpt/…) du slider.
20 +# -----------------------------------------------------------------------------
21 +from __future__ import annotations
22 +
23 +import html as _html
24 +import re
25 +
26 +from ..schema import StListing
27 +from .base import StConnector
28 +
29 +SITE = "https://www.locationdechalets.com"
30 +SITEMAP = SITE + "/sitemap.xml"
31 +
32 +_TAG_RE = re.compile(r"<[^>]+>")
33 +
34 +# ville → région (repli quand la ligne « … (Québec) » manque)
35 +_CITY_REGION = {
36 + "chute-st-philippe": "Laurentides",
37 + "chute-saint-philippe": "Laurentides",
38 + "la macaza": "Laurentides",
39 + "val-david": "Laurentides",
40 + "notre-dame-de-la-merci": "Lanaudière",
41 + "saint-donat": "Lanaudière",
42 + "st-donat": "Lanaudière",
43 +}
44 +
45 +
46 +def _text(fragment: str) -> str:
47 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
48 +
49 +
50 +class LocationDeChalets(StConnector):
51 + source_id = "locationdechalets"
52 + request_delay = 1.0
53 +
54 + # -- page détail ----------------------------------------------------------
55 + def _detail(self, url: str) -> dict:
56 + h = self.get(url).text
57 + d: dict = {}
58 +
59 + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h)
60 + if m:
61 + d["title"] = _text(re.split(r"<br\s*/?>", m.group(1))[0])
62 + # « Capacité de 2 personnes » / « Capacité de 2 à 4 personnes » (max)
63 + m = re.search(r"Capacité de (?:\d+\s+à\s+)?(\d+) personnes", h)
64 + if m:
65 + d["capacity"] = int(m.group(1))
66 +
67 + # description : après <strong>Description</strong>, jusqu'à la
68 + # section suivante ; certaines pages n'ont pas ce marqueur → repli
69 + # sur le plus long <p> éditorial
70 + m = re.search(r"(?s)<strong>Description</strong>(.*?)"
71 + r"(?:CARACTÉRISTIQUES|Caractéristiques|Disponibilités"
72 + r"|<strong>INTÉRIEUR)", h)
73 + if m:
74 + texte = re.sub(r"<br\s*/?>", "\n", m.group(1))
75 + texte = _html.unescape(_TAG_RE.sub(" ", texte))
76 + texte = re.sub(r"[ \t]+", " ", texte)
77 + texte = re.sub(r"\n\s+", "\n", texte).strip()
78 + d["description"] = texte[:5000]
79 + else:
80 + paras = [_text(p) for p in
81 + re.findall(r"(?s)<p[^>]*>(.*?)</p>", h)]
82 + paras = [p for p in paras if len(p) > 150
83 + and "Coordonnées" not in p
84 + and "Contactez-nous" not in p
85 + and "Découvrez nos chalets" not in p]
86 + if paras:
87 + d["description"] = max(paras, key=len)[:5000]
88 +
89 + # sections (deux styles : MAJUSCULES ou « Caractéristiques : ») :
90 + # <ul> qui suit chaque entête → amenities + compteurs
91 + amen: list[str] = []
92 + for sec in (r"CARACT[EÉ]RISTIQUES", r"INT[EÉ]RIEUR",
93 + r"EXT[EÉ]RIEUR", r"INCLUS"):
94 + for m in re.finditer(sec, h, re.I):
95 + mu = re.search(r"(?s)<ul[^>]*>(.*?)</ul>",
96 + h[m.start():m.start() + 3500])
97 + if not mu:
98 + continue
99 + for li in re.findall(r"(?s)<li[^>]*>(.*?)</li>", mu.group(1)):
100 + t = _text(li)
101 + if t and t not in amen:
102 + amen.append(t)
103 + break
104 + # compteurs extraits des puces (« 2 chambres (1 lit queen…) »,
105 + # « 1 salle de bain », « 2 personnes maximum ») + du texte qui suit
106 + # l'entête CARACTÉRISTIQUES (pages où ce sont des <p>, pas des <li>)
107 + blob = " | ".join(amen)
108 + m = re.search(r"CARACT[EÉ]RISTIQUES", h, re.I)
109 + if m:
110 + blob += " | " + _text(h[m.start():m.start() + 700])
111 + m = re.search(r"(\d+)\s+personnes?\s+maximum", blob)
112 + if m:
113 + d.setdefault("capacity", int(m.group(1)))
114 + m = re.search(r"(\d+(?:[.,]5)?)\s+chambres?", blob)
115 + if m:
116 + d["bedrooms"] = float(m.group(1).replace(",", "."))
117 + m = re.search(r"(\d+(?:[.,]5)?)\s+salles?\s+de\s+bain", blob)
118 + if m:
119 + d["bathrooms"] = float(m.group(1).replace(",", "."))
120 + amen = [a for a in amen
121 + if not re.fullmatch(r"\d+\s+(personnes?\s+maximum"
122 + r"|chambres?|salles?\s+de\s+bain)", a)]
123 + if amen:
124 + d["amenities"] = amen
125 +
126 + # frais : « Animaux : 10 $ / jour » (section EN SUS)
127 + m = re.search(r"Animaux\s*:\s*([\d,]+\s*\$[^<]*)", h)
128 + if m:
129 + d["pets_fee"] = _text(m.group(1))
130 +
131 + # coordonnées : rue / ville / « Lanaudière (Québec) » / code postal
132 + m = re.search(r"(?s)<div class=\"coord[^\"]*\"[^>]*>(.*?)</div>", h)
133 + if m:
134 + lines = [_text(x) for x in re.split(r"<br\s*/?>|</p>", m.group(1))]
135 + lines = [x for x in lines if x and "Coordonnées" not in x
136 + and not re.fullmatch(r"[\d,]+", x)]
137 + reg = next((x for x in lines if "(Québec)" in x), "")
138 + if reg:
139 + j = lines.index(reg)
140 + d["region"] = reg.split("(")[0].strip()
141 + if j >= 1:
142 + d["city"] = lines[j - 1].strip(", ")
143 + if j >= 2:
144 + d["address"] = lines[j - 2].strip(", ")
145 + if j + 1 < len(lines):
146 + d["postal_code"] = lines[j + 1]
147 + else:
148 + # variante sans ligne « … (Québec) » : rue / ville / postal
149 + postal = next((x for x in lines
150 + if re.fullmatch(r"[A-Z]\d[A-Z]\s?\d[A-Z]\d",
151 + x.strip())), "")
152 + rest = [x for x in lines if x != postal]
153 + if len(rest) >= 2:
154 + d["address"] = rest[0].strip(", ")
155 + d["city"] = rest[1].strip(", ")
156 + if postal:
157 + d["postal_code"] = postal.strip()
158 +
159 + # (pas de lat/lng : le lien Google Maps pointe l'agence, pas le
160 + # chalet — le géocodage se fera en aval sur adresse+ville)
161 + m = re.search(r"CITQ\D{0,25}(\d{6})", h)
162 + if m:
163 + d["citq"] = m.group(1)
164 +
165 + # TARIF RÉGULIER : « 2 nuits : 498 $ » (en <p> ou en <table>) →
166 + # 249 $/nuit. ⚠️ anti-scrape : chiffres de bourrage blancs sur blanc,
167 + # PARFOIS IMBRIQUÉS (<span #fff>0<span #000>4</span></span>98) → on
168 + # élimine itérativement les <span> les plus internes : blancs = jetés
169 + # avec leur contenu, autres = dépliés (contenu conservé).
170 + i = h.find("TARIF RÉGULIER")
171 + if i >= 0:
172 + frag = h[i:i + 2500]
173 + for _ in range(20):
174 + frag2 = re.sub(
175 + r"<span([^>]*)>([^<]*)</span>",
176 + lambda m: (m.group(2).replace("0", "")
177 + if re.search(r"color:\s*#f{3,6}\b",
178 + m.group(1), re.I)
179 + else m.group(2)),
180 + frag)
181 + if frag2 == frag:
182 + break
183 + frag = frag2
184 + frag = _text(frag)
185 + m = re.search(r"(\d+)\s*nuits?\s*:\s*(\d[\d\s]*)\s*\$", frag)
186 + if m:
187 + nights = int(m.group(1))
188 + total = float(m.group(2).replace(" ", ""))
189 + if nights and 20 <= total / nights <= 20000:
190 + d["price_night"] = round(total / nights)
191 + d["price_ref"] = f"{nights} nuits : {total:g} $"
192 +
193 + # galerie du slider
194 + imgs: list[str] = []
195 + for u in re.findall(r"background:\s*url\(([^)]+)\)", h):
196 + u = u.strip("'\" ")
197 + if u.startswith("/fichiersUploadOpt/"):
198 + u = SITE + u
199 + if u.startswith("https://") and u not in imgs:
200 + imgs.append(u)
201 + if len(imgs) >= 20:
202 + break
203 + if imgs:
204 + d["images"] = imgs
205 + return d
206 +
207 + # -- contrat --------------------------------------------------------------
208 + def fetch(self) -> list[StListing]:
209 + xml = self.get(SITEMAP).text
210 + entries = re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>"
211 + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml)
212 +
213 + listings: list[StListing] = []
214 + vus: set[str] = set()
215 + for url, lastmod in entries:
216 + m = re.match(r"https://www\.locationdechalets\.com/fr/chalets/"
217 + r"([^/]+)/?$", url)
218 + if not m:
219 + continue
220 + slug = m.group(1)
221 + if slug in vus:
222 + continue
223 + vus.add(slug)
224 +
225 + det = self.detail(slug, lastmod or "v1",
226 + lambda u=url: self._detail(u))
227 + title = det.get("title") or ""
228 + if not title:
229 + continue
230 +
231 + city = det.get("city") or ""
232 + region = det.get("region") \
233 + or _CITY_REGION.get(city.lower(), "Lanaudière")
234 +
235 + price = det.get("price_night")
236 + details = {k: v for k, v in {
237 + "postal_code": det.get("postal_code") or "",
238 + "price_ref": det.get("price_ref") or "",
239 + "pets_fee": det.get("pets_fee") or "",
240 + }.items() if v}
241 +
242 + listings.append(StListing(
243 + source=self.source_id,
244 + external_id=slug,
245 + url=url,
246 + title=title,
247 + property_type="Chalet",
248 + address=det.get("address") or "",
249 + city=city,
250 + region=region,
251 + price_night=float(price) if price else None,
252 + price_label=f"à partir de {price:g} $ / nuit" if price else "",
253 + capacity=float(det["capacity"]) if det.get("capacity") else None,
254 + bedrooms=det.get("bedrooms"),
255 + bathrooms=det.get("bathrooms"),
256 + pets="conditions" if det.get("pets_fee") else None,
257 + citq=det.get("citq") or "",
258 + description=det.get("description") or "",
259 + amenities=det.get("amenities") or [],
260 + details=details,
261 + images=det.get("images") or [],
262 + lat=det.get("lat"),
263 + lng=det.get("lng"),
264 + ))
265 + return listings
added louka/shortterm/connectors/memoriachalets.py +248 −0
@@ -0,0 +1,248 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/memoriachalets.py : Memoria Chalets (memoriachalets.com)
4 +# — gestionnaire de Magog, ~110 fiches FR (Cantons-de-l'Est surtout,
5 +# quelques Laurentides / Centre-du-Québec).
6 +#
7 +# Méthode : WordPress, CPT `cpt_chalets`, HTML serveur. Sitemap
8 +# /cpt_chalets-sitemap.xml (lastmod fiable ; doublons /en/ écartés) →
9 +# pages /locations/<slug>/. La fiche est balisée par classes :
10 +# h2.location-title (« Le Hibou – Magog »), p.location-specs
11 +# (« 4 CHAMBRES - 1.5 SALLES DE BAIN - 8 PERSONNES »), p.location-adress
12 +# (adresse complète), p.starting-price (« À partir de 199,00$ par jour » —
13 +# une partie du parc est au mois/31 jours minimum : prix nuit alors absent,
14 +# la mention est conservée dans details), p.citq (« CITQ #296459 … » ou
15 +# « (Location minimum de 31 jours) ») et ul de .accomosdations-filters
16 +# (commodités, dont « Animaux interdit/admis »). Pas de géo dans le HTML.
17 +# -----------------------------------------------------------------------------
18 +from __future__ import annotations
19 +
20 +import re
21 +
22 +from bs4 import BeautifulSoup
23 +
24 +from ...normalize import strip_accents
25 +from ..schema import StListing, parse_price_night
26 +from .base import StConnector
27 +
28 +SITEMAP = "https://memoriachalets.com/cpt_chalets-sitemap.xml"
29 +
30 +_URL_DETAIL = re.compile(
31 + r"https://(?:www\.)?memoriachalets\.com/locations/([\w%-]+)/?$")
32 +
33 +_SPECS_RE = re.compile(
34 + r"([\d.,]+)\s*CHAMBRES?\s*[-·—]\s*([\d.,]+)\s*SALLES?\s+DE\s+BAIN"
35 + r"\s*[-·—]\s*([\d.,]+)\s*PERSONNES?", re.I)
36 +
37 +_CP_RE = re.compile(r"^[A-Za-z]\d[A-Za-z]\s?\d[A-Za-z]\d$") # code postal
38 +
39 +# mot-clé (titre) → type canonique ; défaut : Chalet
40 +_TYPE_HINTS = [("condo", "Condo"), ("appartement", "Appartement"),
41 + ("apt", "Appartement"), ("loft", "Loft"), ("studio", "Studio"),
42 + ("maison", "Maison")]
43 +
44 +# indices de région dans le texte (leur parc : Estrie surtout)
45 +_REGION_HINTS = [
46 + ("estrie", "Cantons-de-l'Est"), ("cantons-de-l'est", "Cantons-de-l'Est"),
47 + ("laurentides", "Laurentides"), ("lanaudiere", "Lanaudière"),
48 + ("centre-du-quebec", "Centre-du-Québec"), ("mauricie", "Mauricie"),
49 + ("charlevoix", "Charlevoix"), ("monteregie", "Montérégie"),
50 +]
51 +
52 +# villes connues du parc → région touristique
53 +_VILLE_REGION = {
54 + "magog": "Cantons-de-l'Est", "orford": "Cantons-de-l'Est",
55 + "eastman": "Cantons-de-l'Est", "austin": "Cantons-de-l'Est",
56 + "ayer's cliff": "Cantons-de-l'Est", "ayers cliff": "Cantons-de-l'Est",
57 + "north hatley": "Cantons-de-l'Est", "sherbrooke": "Cantons-de-l'Est",
58 + "sainte-catherine-de-hatley": "Cantons-de-l'Est",
59 + "canton d'orford": "Cantons-de-l'Est", "bromont": "Cantons-de-l'Est",
60 + "sutton": "Cantons-de-l'Est", "stukely-sud": "Cantons-de-l'Est",
61 + "bolton-est": "Cantons-de-l'Est", "potton": "Cantons-de-l'Est",
62 + "mansonville": "Cantons-de-l'Est", "coaticook": "Cantons-de-l'Est",
63 + "piopolis": "Cantons-de-l'Est", "lac-megantic": "Cantons-de-l'Est",
64 +}
65 +
66 +_RUE_RE = re.compile(r"^(rue|chemin|ch\.|avenue|av\.|route|rte|boulevard|"
67 + r"boul\.?|montee|impasse|allee|place)\b")
68 +
69 +
70 +def _num(raw: str) -> float | None:
71 + try:
72 + return float(raw.replace(",", "."))
73 + except (TypeError, ValueError):
74 + return None
75 +
76 +
77 +class MemoriaChalets(StConnector):
78 + source_id = "memoriachalets"
79 +
80 + # -- inventaire (sitemap FR + lastmod) -----------------------------------
81 + def _sitemap_urls(self) -> dict[str, tuple[str, str]]:
82 + """slug -> (url détail FR, lastmod)."""
83 + xml = self.get(SITEMAP).text
84 + urls: dict[str, tuple[str, str]] = {}
85 + for bloc in re.findall(r"<url>(.*?)</url>", xml, re.S):
86 + m = re.search(r"<loc>([^<]+)</loc>", bloc)
87 + if not m:
88 + continue
89 + loc = m.group(1).strip()
90 + mu = _URL_DETAIL.match(loc)
91 + if not mu or mu.group(1) == "locations":
92 + continue
93 + lastmod = re.search(r"<lastmod>([^<]+)</lastmod>", bloc)
94 + urls.setdefault(mu.group(1),
95 + (loc, lastmod.group(1) if lastmod else ""))
96 + return urls
97 +
98 + # -- page détail ---------------------------------------------------------
99 + def _detail(self, url: str) -> dict:
100 + html = self.get(url).text
101 + soup = BeautifulSoup(html, "html.parser")
102 + d: dict = {}
103 +
104 + h2 = soup.select_one("h2.location-title") or soup.find("h2")
105 + if h2 is not None:
106 + d["title"] = re.sub(r"\s+", " ", h2.get_text(" ", strip=True))
107 +
108 + specs = soup.select_one("p.location-specs")
109 + if specs is not None:
110 + m = _SPECS_RE.search(specs.get_text(" ", strip=True))
111 + if m:
112 + d["bedrooms"] = _num(m.group(1))
113 + d["bathrooms"] = _num(m.group(2))
114 + d["capacity"] = _num(m.group(3))
115 +
116 + # « 207, Rue des Pruches, Magog, Québec, J1X 0M9, Canada »
117 + adresse = soup.select_one("p.location-adress")
118 + if adresse is not None:
119 + texte = re.sub(r"\s+", " ", adresse.get_text(" ", strip=True))
120 + d["address"] = texte
121 + restants = []
122 + for p in texte.split(","):
123 + # code postal parfois collé à la ville (« Bromont J2L 2C1 »)
124 + p = re.sub(r"\b[A-Za-z]\d[A-Za-z]\s?\d[A-Za-z]\d\b", "",
125 + p).strip()
126 + k = strip_accents(p).lower()
127 + if (not p or _CP_RE.match(p) or k in ("quebec", "qc", "canada")
128 + or re.match(r"^(appartement|app|unite)\b", k)):
129 + continue
130 + restants.append(p)
131 + if restants:
132 + ville = restants[-1]
133 + # adresse sans virgule (« Rue des Noyers Orford ») :
134 + # la ville est le dernier mot
135 + if _RUE_RE.match(strip_accents(ville).lower()):
136 + ville = ville.split()[-1]
137 + d["city"] = ville
138 +
139 + prix = soup.select_one("p.starting-price")
140 + if prix is not None:
141 + label = re.sub(r"\s+", " ", prix.get_text(" ", strip=True))
142 + label = re.sub(r"^(À partir de\s+)+", "À partir de ", label, flags=re.I)
143 + d["price_label"] = label
144 +
145 + citq = soup.select_one("p.citq")
146 + if citq is not None:
147 + texte = citq.get_text(" ", strip=True)
148 + m = re.search(r"CITQ\s*#?\s*(\d{6})", texte)
149 + if m:
150 + d["citq"] = m.group(1)
151 + m = re.search(r"\(([^)]*[Ll]ocation minimum[^)]*)\)", texte)
152 + if m:
153 + d["location_minimum"] = m.group(1).strip()
154 +
155 + # commodités (dont le statut animaux)
156 + amen: list[str] = []
157 + for span in soup.select(".accomosdations-filters li span"):
158 + t = re.sub(r"\s+", " ", span.get_text(" ", strip=True))
159 + if t and t not in amen:
160 + amen.append(t)
161 + if amen:
162 + d["amenities"] = amen
163 + for a in amen:
164 + k = strip_accents(a).lower()
165 + if "animaux" in k or "chien" in k:
166 + d["pets"] = "non" if ("interdit" in k or "refus" in k
167 + or "non admis" in k) else "oui"
168 +
169 + # description : paragraphes de la 1re colonne de la fiche
170 + col = soup.select_one('[itemprop="articleBody"] div')
171 + if col is not None:
172 + morceaux = []
173 + for p in col.find_all("p"):
174 + classes = " ".join(p.get("class") or [])
175 + if any(c in classes for c in
176 + ("location-specs", "location-adress",
177 + "starting-price", "citq")):
178 + continue
179 + t = p.get_text(" ", strip=True)
180 + if t:
181 + morceaux.append(re.sub(r"\s+", " ", t))
182 + if morceaux:
183 + d["description"] = "\n".join(morceaux)[:5000]
184 +
185 + # photos (uploads, sans logo/icônes ; dédoublonnage sur le nom de
186 + # base sans suffixe de taille -WxH)
187 + imgs, vus = [], set()
188 + for u in re.findall(r'(https://(?:www\.)?memoriachalets\.com/'
189 + r'wp-content/uploads/[^"\'\s>]+\.(?:jpe?g|webp))',
190 + html):
191 + base = re.sub(r"-\d+x\d+(?=\.\w+$)", "", u)
192 + nom = strip_accents(base).lower()
193 + if base in vus or "logo" in nom or "icon" in nom:
194 + continue
195 + vus.add(base)
196 + imgs.append(re.sub(r"-\d+x\d+(?=\.\w+$)", "", u))
197 + d["images"] = imgs[:20]
198 + return d
199 +
200 + # -- contrat --------------------------------------------------------------
201 + def fetch(self) -> list[StListing]:
202 + listings: list[StListing] = []
203 + for slug, (url, lastmod) in self._sitemap_urls().items():
204 + try:
205 + d = self.detail(slug, lastmod or "sans-lastmod",
206 + lambda u=url: self._detail(u))
207 + except Exception: # une fiche cassée ≠ inventaire perdu
208 + d = {}
209 + if not d.get("title"):
210 + continue
211 +
212 + hay = strip_accents(d["title"]).lower()
213 + ptype = next((c for n, c in _TYPE_HINTS if n in hay), "Chalet")
214 +
215 + # région : ville connue, sinon indice textuel (« en Estrie »…)
216 + ville = d.get("city", "")
217 + region = _VILLE_REGION.get(strip_accents(ville).lower(), "")
218 + if not region:
219 + texte = strip_accents(" ".join(
220 + (d.get("address", ""), d.get("description", "")))).lower()
221 + region = next((r for n, r in _REGION_HINTS if n in texte), "")
222 +
223 + details = {}
224 + if d.get("location_minimum"):
225 + details["location_minimum"] = d["location_minimum"]
226 +
227 + listings.append(StListing(
228 + source=self.source_id,
229 + external_id=slug, # slug WP, stable
230 + url=url,
231 + title=d["title"],
232 + property_type=ptype,
233 + address=d.get("address", ""),
234 + city=ville,
235 + region=region,
236 + price_night=parse_price_night(d.get("price_label", "")),
237 + price_label=d.get("price_label", ""),
238 + capacity=d.get("capacity"),
239 + bedrooms=d.get("bedrooms"),
240 + bathrooms=d.get("bathrooms"),
241 + pets=d.get("pets"),
242 + citq=d.get("citq", ""),
243 + description=d.get("description", ""),
244 + amenities=d.get("amenities") or [],
245 + details=details,
246 + images=d.get("images") or [],
247 + ))
248 + return listings
added louka/shortterm/connectors/panoraloges.py +135 −0
@@ -0,0 +1,135 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/panoraloges.py : Panora Loges Fluviales (panoraloges.com) —
4 +# 10 loges (mini-chalets) en bord de mer à Sainte-Anne-des-Monts (Gaspésie),
5 +# déclinées en DEUX gammes identiques à l'intérieur d'une même gamme :
6 +# loges simples (nos 8-10, 2 pers.) et loges doubles (nos 1-7, 4 pers.).
7 +#
8 +# Méthode : site Squarespace ; l'API ?format=json de la page /fr/loges est
9 +# vide (items=0) → on parse les DEUX pages de gamme statiques
10 +# /fr/logessimples et /fr/logesdoubles. Chaque page expose en <h2-4> :
11 +# prix (« 175 $ » sous « À partir de »), « N invités », « N chambres »,
12 +# « N lits Queen », « 1 salle de bain », puis la liste des commodités
13 +# jusqu'à « Stationnements ». 2 fiches « gamme » (multi_unit) plutôt que
14 +# 10 fiches par unité artificiellement identiques — le site lui-même
15 +# laisse le libre choix de la loge à l'intérieur d'une gamme.
16 +# -----------------------------------------------------------------------------
17 +from __future__ import annotations
18 +
19 +import html as _html
20 +import re
21 +
22 +from ..schema import StListing
23 +from .base import StConnector
24 +
25 +SITE = "https://www.panoraloges.com"
26 +
27 +_PAGES = [
28 + # (external_id, chemin, titre, unités)
29 + ("loges-doubles", "/fr/logesdoubles", "Loges doubles — Panora Loges "
30 + "Fluviales", "Loges 1 à 7"),
31 + ("loges-simples", "/fr/logessimples", "Loges simples — Panora Loges "
32 + "Fluviales", "Loges 8 à 10"),
33 +]
34 +
35 +_TAG_RE = re.compile(r"<[^>]+>")
36 +
37 +# images de gabarit à exclure de la galerie
38 +_IMG_EXCLUDE = ("favicon", "logo", "icon", "texture", "panora_ecran",
39 + "Fond+de+page", "plan")
40 +
41 +
42 +def _text(fragment: str) -> str:
43 + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip()
44 +
45 +
46 +class PanoraLoges(StConnector):
47 + source_id = "panoraloges"
48 + request_delay = 1.0
49 +
50 + def _parse(self, path: str) -> dict:
51 + h = self.get(SITE + path).text
52 + d: dict = {}
53 +
54 + heads = [t for t in (_text(x) for x in re.findall(
55 + r"(?s)<h[1-4][^>]*>(.*?)</h[1-4]>", h)) if t]
56 +
57 + m = re.search(r"(\d+)\s*\$", " ".join(heads))
58 + if m:
59 + d["price_night"] = float(m.group(1))
60 +
61 + # commodités : entêtes entre « N invités » et « Stationnements »
62 + start = next((i for i, t in enumerate(heads)
63 + if re.fullmatch(r"\d+\s+invités?", t)), None)
64 + end = next((i for i, t in enumerate(heads)
65 + if t.lower().startswith("stationnement")), None)
66 + amen: list[str] = []
67 + if start is not None and end is not None and start < end:
68 + amen = heads[start:end + 1]
69 + blob = " | ".join(amen)
70 + for pat, key in ((r"(\d+)\s+invités?", "capacity"),
71 + (r"(\d+)\s+chambres?", "bedrooms"),
72 + (r"(\d+)\s+lits?\s+Queen", "beds"),
73 + (r"(\d+)\s+salles?\s+de\s+bain", "bathrooms")):
74 + m = re.search(pat, blob, re.I)
75 + if m:
76 + d[key] = float(m.group(1))
77 + d["amenities"] = [a for a in amen
78 + if not re.fullmatch(r"\d+\s+(invités?|chambres?"
79 + r"|salles?\s+de\s+bain)", a)]
80 +
81 + # description : sous-titre de gamme + infos pratiques (les <p> du
82 + # gabarit Squarespace contiennent beaucoup de CSS → filtrés)
83 + paras = [_text(p) for p in re.findall(r"(?s)<p[^>]*>(.*?)</p>", h)]
84 + keep = [p for p in paras
85 + if 60 < len(p) < 600 and "{" not in p
86 + and not p.startswith(("FR:", "0 "))
87 + and "Infolettre" not in p and "418-" not in p]
88 + if keep:
89 + d["description"] = "\n".join(keep[:6])[:3000]
90 +
91 + imgs: list[str] = []
92 + for u in re.findall(r"https://images\.squarespace-cdn\.com/[^\"\s?]+",
93 + h):
94 + if any(x.lower() in u.lower() for x in _IMG_EXCLUDE):
95 + continue
96 + u = u.rstrip("\\,&")
97 + if not re.search(r"\.(?:jpe?g|png|webp)$", u, re.I):
98 + continue
99 + if u not in imgs:
100 + imgs.append(u)
101 + if len(imgs) >= 15:
102 + break
103 + d["images"] = imgs
104 + return d
105 +
106 + # -- contrat --------------------------------------------------------------
107 + def fetch(self) -> list[StListing]:
108 + listings: list[StListing] = []
109 + for ext_id, path, title, units in _PAGES:
110 + d = self._parse(path)
111 + if not d.get("price_night") and not d.get("capacity"):
112 + continue
113 + price = d.get("price_night")
114 + listings.append(StListing(
115 + source=self.source_id,
116 + external_id=ext_id,
117 + url=SITE + path,
118 + title=title,
119 + property_type="Chalet",
120 + address="610 boul. Sainte-Anne E",
121 + city="Sainte-Anne-des-Monts",
122 + region="Gaspésie",
123 + price_night=price,
124 + price_label=f"À partir de {price:g} $ / nuit" if price else "",
125 + capacity=d.get("capacity"),
126 + bedrooms=d.get("bedrooms"),
127 + beds=d.get("beds"),
128 + bathrooms=d.get("bathrooms"),
129 + description=d.get("description") or "",
130 + amenities=d.get("amenities") or [],
131 + details={"multi_unit": True, "units": units,
132 + "domain": "Panora Loges Fluviales"},
133 + images=d.get("images") or [],
134 + ))
135 + return listings
modified louka/shortterm/connectors/parcscanada.py +135 −15
@@ -16,7 +16,14 @@
16 16 # description fr, capacité, photos, catégorie. On ne garde que les
17 17 # catégories « hébergement » (table KEEP ci-dessous) ;
18 18 # 3. prix (cache self.detail, clé mensuelle) : POST /api/resource/feeDetails
19 −# ?resourceId=…&startDate=<J+35> → feeTotal = tarif d'une nuit.
19 +# ?resourceId=…&startDate=<J+n> → feeTotal = tarif d'une nuit. Les unités
20 +# fermées à J+35 (saisonnier) sont réessayées à J+95/185/275 ;
21 +# 4. commodités : chaque ressource porte des definedAttributes
22 +# (attributeDefinitionId + valeur/énums) que GET /api/attribute/filterable
23 +# permet de traduire en libellés français (« Foyer sur l'emplacement »,
24 +# « Chiens permis », « Éclairage fourni : solaire »…) — on ne retient que
25 +# les attributs pertinents (table AMEN_DEFS), les dimensions techniques
26 +# d'emplacement sont ignorées.
20 27 #
21 28 # Les ids sont des entiers négatifs (int32 min + n). Aucune géoloc par unité
22 29 # dans l'API : lat/lng et région touristique viennent de la table statique
@@ -78,6 +85,40 @@ KEEP = {
78 85
79 86 _TAG_RE = re.compile(r"<[^>]+>")
80 87
88 +# Attributs « commodités » pertinents (id de définition → retenu). Les autres
89 +# (dimensions, pentes, obstructions, ampérage…) sont du bruit technique.
90 +AMEN_DEFS = {
91 + -32758, # Foyer sur l'emplacement
92 + -32571, # Type de foyer
93 + -32717, # Feux de camp permis
94 + -32721, # Admissible au permis de feu
95 + -32753, # Wi-fi
96 + -32754, # Couverture cellulaire
97 + -32760, # Barbecue fourni
98 + -32570, # Fourneau fourni
99 + -32762, # Source de chaleur interne disponible
100 + -32763, # Éclairage fourni
101 + -32764, # Électricité disponible à l'intérieur
102 + -32761, # Électricité disponible à l'extérieur
103 + -32582, # Électricité
104 + -32736, # Service d'eau
105 + -32756, # Accessible
106 + -32723, # Secteur riverain
107 + -32724, # Accès au rivage
108 + -32725, # Conditions de baignade
109 + -32757, # Tables de pique-nique
110 + -32748, # Ombrage à l'emplacement
111 + -32759, # Casiers à provisions fournis
112 + -32713, # Stationnement
113 + -32709, # Permis de pêche inclus
114 + -32715, # Vélos permis
115 +}
116 +PETS_DEFS = {-32718, -32572} # Chiens permis / Animaux de compagnie permis
117 +MIN_STAY_DEF = -32697 # Séjour minimum (nuits)
118 +
119 +# valeurs d'énum « négatives » : l'attribut est alors omis des commodités
120 +_VAL_NEGATIVES = {"non", "no", "aucun", "aucune", "none", "n/a"}
121 +
81 122
82 123 def _fr(localized: list[dict], *keys: str) -> dict:
83 124 by_culture = {v.get("cultureName"): v for v in (localized or [])}
@@ -105,17 +146,80 @@ class ParcsCanada(StConnector):
105 146
106 147 # -- prix d'une nuit (cache BD, clé mensuelle) -------------------------------
107 148 def _fetch_fee(self, resource_id: int) -> dict:
108 − start = (datetime.date.today()
109 − + datetime.timedelta(days=35)).isoformat()
110 − resp = self.post(
111 − BASE + "/api/resource/feeDetails",
112 − params={"resourceId": resource_id, "startDate": start,
113 − "boatLength": 0, "bookingCategoryId": 0,
114 − "entryPointResourceId": 0, "exitPointResourceId": 0},
115 − json=[], headers=HEADERS)
116 − fees = (resp.json() or {}).get("resourceFeeDetails") or []
117 − total = sum(f.get("feeTotal") or 0 for f in fees)
118 − return {"fee": round(total, 2)} if total > 0 else {}
149 + # unités saisonnières : pas de tarif hors saison → on sonde plusieurs
150 + # dates réparties sur l'année jusqu'à trouver une nuit tarifée
151 + for jours in (35, 95, 185, 275):
152 + start = (datetime.date.today()
153 + + datetime.timedelta(days=jours)).isoformat()
154 + try:
155 + resp = self.post(
156 + BASE + "/api/resource/feeDetails",
157 + params={"resourceId": resource_id, "startDate": start,
158 + "boatLength": 0, "bookingCategoryId": 0,
159 + "entryPointResourceId": 0,
160 + "exitPointResourceId": 0},
161 + json=[], headers=HEADERS)
162 + except Exception: # noqa: BLE001 — on tente la date suivante
163 + continue
164 + fees = (resp.json() or {}).get("resourceFeeDetails") or []
165 + total = sum(f.get("feeTotal") or 0 for f in fees)
166 + if total > 0:
167 + return {"fee": round(total, 2), "date": start}
168 + return {}
169 +
170 + # -- définitions d'attributs (libellés fr des commodités) --------------------
171 + def _fetch_attr_defs(self) -> dict:
172 + data = self._api("/api/attribute/filterable") or {}
173 + defs: dict = {}
174 + for key, v in data.items():
175 + fr = _fr(v.get("localizedValues") or [])
176 + name = fr.get("displayName") or ""
177 + if not name:
178 + continue
179 + values = {}
180 + for ev in v.get("values") or []:
181 + lab = _fr(ev.get("localizedValues") or []).get("displayName")
182 + if lab is not None and ev.get("enumValue") is not None:
183 + values[str(ev["enumValue"])] = lab
184 + defs[str(key)] = {"name": name, "values": values}
185 + return defs
186 +
187 + @staticmethod
188 + def _amenities(res: dict, defs: dict) -> tuple[list[str], str | None,
189 + float | None]:
190 + """(commodités, animaux oui/non, séjour minimum) d'une ressource."""
191 + amen: list[str] = []
192 + pets: str | None = None
193 + min_stay: float | None = None
194 + for da in res.get("definedAttributes") or []:
195 + did = da.get("attributeDefinitionId")
196 + d = defs.get(str(did))
197 + if did == MIN_STAY_DEF and da.get("value") is not None:
198 + min_stay = float(da["value"])
199 + continue
200 + if d is None:
201 + continue
202 + labels = [d["values"].get(str(v)) for v in (da.get("values") or [])]
203 + labels = [l for l in labels if l]
204 + if did in PETS_DEFS and labels:
205 + pets = "non" if all(
206 + l.lower() in _VAL_NEGATIVES for l in labels) else "oui"
207 + continue
208 + if did not in AMEN_DEFS:
209 + continue
210 + positifs = [l for l in labels
211 + if l.lower() not in _VAL_NEGATIVES]
212 + if not positifs:
213 + continue
214 + if all(l.lower() in ("oui", "yes") for l in positifs):
215 + item = d["name"] # booléen : le nom suffit
216 + else:
217 + item = f"{d['name']} : {', '.join(positifs)}"
218 + if item not in amen:
219 + amen.append(item)
220 + # l'API renvoie les attributs dans un ordre variable : trier pour
221 + # un diff stable dans la base
222 + return sorted(amen), pets, min_stay
119 223
120 224 # -- contrat --------------------------------------------------------------
121 225 def fetch(self) -> list[StListing]:
@@ -128,6 +232,14 @@ class ParcsCanada(StConnector):
128 232 cat_names[c.get("resourceCategoryId")] = name
129 233
130 234 month = datetime.date.today().strftime("%Y-%m") # prix revus au mois
235 +
236 + # libellés fr des attributs (une requête, cache BD mensuel)
237 + try:
238 + attr_defs = self.detail("_attrdefs", month, self._fetch_attr_defs)
239 + except Exception as exc: # noqa: BLE001 — les commodités sont optionnelles
240 + print(f"[parcscanada] définitions d'attributs : {exc}",
241 + file=sys.stderr)
242 + attr_defs = {}
131 243 listings: list[StListing] = []
132 244 for loc_id, (loc_name, region, lat, lng) in QC_LOCATIONS.items():
133 245 try:
@@ -162,13 +274,20 @@ class ParcsCanada(StConnector):
162 274 title = f"{cat_fr} {name} – {loc_name}"
163 275
164 276 try:
165 − fee = self.detail(str(rid), month,
277 + # « v2 » : sondage multi-dates (unités saisonnières)
278 + fee = self.detail(str(rid), month + ":v2",
166 279 lambda r=rid: self._fetch_fee(r))
167 280 except Exception as exc: # noqa: BLE001 — le prix est optionnel
168 281 print(f"[parcscanada] tarif {rid} : {exc}", file=sys.stderr)
169 282 fee = {}
170 283 price = fee.get("fee")
171 284
285 + amenities, pets_attr, min_stay = self._amenities(
286 + res, attr_defs)
287 + details = {"parc": loc_name, "categorie": cat_fr}
288 + if min_stay:
289 + details["sejour_minimum"] = min_stay
290 +
172 291 cap = res.get("maxCapacity")
173 292 listings.append(StListing(
174 293 source=self.source_id,
@@ -182,9 +301,10 @@ class ParcsCanada(StConnector):
182 301 price_night=float(price) if price else None,
183 302 price_label=(f"{price:.2f} $ / nuit" if price else ""),
184 303 capacity=float(cap) if cap else None,
185 − pets=_pets(desc),
304 + pets=pets_attr or _pets(desc),
186 305 description=desc[:4000],
187 − details={"parc": loc_name, "categorie": cat_fr},
306 + amenities=amenities,
307 + details=details,
188 308 images=images,
189 309 lat=lat,
190 310 lng=lng,
added louka/shortterm/connectors/pourvoiries.py +288 −0
@@ -0,0 +1,288 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/pourvoiries.py : Fédération des pourvoiries du Québec
4 +# (pourvoiries.com) — ~340 pourvoiries avec hébergement (chalets, camps,
5 +# pavillons, prêt-à-camper) partout en région.
6 +#
7 +# Méthode (TYPO3, aucun anti-bot) :
8 +# 1. sitemap officiel des établissements (/sitemap/outfitters/sitemap.xml)
9 +# → inventaire complet des fiches /pourvoiries/<slug>-<zone>-<permis> ;
10 +# 2. fiche établissement (cache self.detail, clé mensuelle) : nom, ville +
11 +# région (bandeau), description, coordonnées GPS (« Latitude : 48.805 »),
12 +# no d'établissement CITQ, période d'ouverture, type de restauration,
13 +# unités d'hébergement de l'onglet Hébergements (Pavillon / Chalet /
14 +# Camp / Prêt-à-camper… avec capacité et chambres), photos du carrousel ;
15 +# 3. prix : les fiches n'affichent pas de tarif d'hébergement ; le sitemap
16 +# des forfaits (/sitemap/packages/sitemap.xml, slug préfixé du no de
17 +# permis) donne un prix « par personne / nuit » pour ~80 pourvoiries →
18 +# price_label du forfait le plus bas (à défaut d'un vrai prix/nuit).
19 +#
20 +# Une annonce = un établissement (les unités individuelles ne sont pas
21 +# réservables en ligne — la liste des unités va dans details["unites"]).
22 +# Seules les fiches avec au moins une unité d'hébergement sont retenues.
23 +# Réglage env : LOUKA_POURVOIRIES_LIMIT (nb max de fiches, 0 = tout).
24 +# -----------------------------------------------------------------------------
25 +from __future__ import annotations
26 +
27 +import os
28 +import re
29 +import sys
30 +import time
31 +
32 +from ..schema import StListing, normalize_region
33 +from .base import StConnector
34 +
35 +BASE = "https://www.pourvoiries.com"
36 +SITEMAP_FICHES = f"{BASE}/sitemap/outfitters/sitemap.xml"
37 +SITEMAP_FORFAITS = f"{BASE}/sitemap/packages/sitemap.xml"
38 +PREFIX_FICHE = f"{BASE}/pourvoiries/"
39 +PREFIX_FORFAIT = f"{BASE}/forfaits/"
40 +
41 +_LOC_RE = re.compile(r"<loc>\s*(https?://[^<\s]+)")
42 +_PERMIS_RE = re.compile(r"(\d{2}-\d{3})$") # fin du slug établissement
43 +_FORFAIT_PERMIS_RE = re.compile(r"^(\d{2}-\d{3})-") # début du slug forfait
44 +_LAT_RE = re.compile(r"Latitude\s*:\s*(-?\d+\.\d+)")
45 +_LNG_RE = re.compile(r"Longitude\s*:\s*(-?\d+\.\d+)")
46 +_CAP_RE = re.compile(r"Pour\s+(\d+)\s+personne", re.I)
47 +_CH_RE = re.compile(r"(\d+)\s+chambre", re.I)
48 +_PRIX_FORFAIT_RE = re.compile(
49 + r'<strong class="price">\s*([\d\s,. ]+)\s*\$', re.S)
50 +
51 +# Régions FPQ → région touristique canonique (le reste passe par
52 +# normalize_region : Mauricie, Outaouais, Côte-Nord…)
53 +_REGIONS_FPQ = {
54 + "gaspésie et îles-de-la-madeleine": "Gaspésie",
55 + "gaspesie et iles-de-la-madeleine": "Gaspésie",
56 + "saguenay-lac-saint-jean": "Saguenay–Lac-Saint-Jean",
57 + "nord-du-québec": "Nord-du-Québec",
58 + "baie-james": "Eeyou Istchee Baie-James",
59 +}
60 +
61 +# Titre de section d'hébergement → type canonique Lou-Ka
62 +_TYPE_UNITE = {
63 + "chalet": "Chalet", "pavillon": "Auberge", "auberge": "Auberge",
64 + "camp": "Refuge", "refuge": "Refuge", "yourte": "Yourte",
65 + "dôme": "Dôme", "tente": "Prêt-à-camper",
66 + "prêt-à-camper": "Prêt-à-camper", "camping": "Camping",
67 + "chambre": "Chambre", "maison": "Maison", "condo": "Condo",
68 +}
69 +
70 +
71 +def _property_type(types_unites: list[str]) -> str:
72 + """Type dominant de l'établissement — le chalet prime (offre principale)."""
73 + canon = [_TYPE_UNITE.get(t.strip().lower(), "") for t in types_unites]
74 + for pref in ("Chalet", "Auberge", "Yourte", "Dôme", "Prêt-à-camper",
75 + "Refuge", "Maison", "Condo", "Chambre", "Camping"):
76 + if pref in canon:
77 + return pref
78 + return "Chalet"
79 +
80 +
81 +class Pourvoiries(StConnector):
82 + source_id = "pourvoiries"
83 + request_delay = 0.8
84 +
85 + # -- fiche établissement --------------------------------------------------
86 + def _fetch_fiche(self, slug: str) -> dict:
87 + from bs4 import BeautifulSoup
88 + html = self.get(PREFIX_FICHE + slug).text
89 + soup = BeautifulSoup(html, "html.parser")
90 + d: dict = {}
91 +
92 + h1 = soup.select_one("h1.page-title")
93 + if not h1:
94 + return {}
95 + d["nom"] = h1.get_text(" ", strip=True)
96 +
97 + # bandeau : « Rivière-Bonjour, Gaspésie et Îles-de-la-Madeleine »
98 + banner = soup.select_one(".banner-single .region")
99 + if banner:
100 + loc = banner.get_text(" ", strip=True)
101 + ville, _, region = loc.partition(", ")
102 + d["ville"], d["region"] = ville.strip(), region.strip()
103 +
104 + # description (premier bloc sous le h2 « Description »)
105 + for h2 in soup.find_all("h2"):
106 + if h2.get_text(strip=True).lower() == "description":
107 + paras = [p.get_text(" ", strip=True)
108 + for p in h2.find_all_next("p", limit=4)]
109 + d["description"] = "\n".join(x for x in paras if x)[:2500]
110 + break
111 +
112 + # onglet Informations : paires h3 → p
113 + infos: dict[str, str] = {}
114 + for h3 in soup.find_all("h3"):
115 + p = h3.find_next_sibling("p")
116 + if p is not None:
117 + infos[h3.get_text(" ", strip=True).lower()] = \
118 + p.get_text(" ", strip=True)
119 + for label, key in (("numéro d'établissement", "citq"),
120 + ("période d'ouverture", "ouverture"),
121 + ("type de restauration", "restauration"),
122 + ("type de pourvoirie", "type_pourvoirie"),
123 + ("langue de service", "langues")):
124 + for k, v in infos.items():
125 + if k.startswith(label):
126 + d[key] = v
127 + break
128 +
129 + m = _LAT_RE.search(html)
130 + if m:
131 + d["lat"] = float(m.group(1))
132 + m = _LNG_RE.search(html)
133 + if m:
134 + d["lng"] = float(m.group(1))
135 +
136 + # onglet Hébergements : sections (h2.block-title) → cartes d'unités
137 + unites: list[dict] = []
138 + heb = soup.find(id="hebergements")
139 + if heb is not None:
140 + for bloc in heb.select(".block-slides"):
141 + t = bloc.select_one("h2.block-title")
142 + type_u = t.get_text(" ", strip=True) if t else ""
143 + for card in bloc.select(".card"):
144 + titre = card.select_one(".card-title")
145 + if titre is None:
146 + continue
147 + txt = card.get_text(" ", strip=True)
148 + cap = _CAP_RE.search(txt)
149 + ch = _CH_RE.search(txt)
150 + unites.append({
151 + "type": type_u,
152 + "nom": titre.get_text(" ", strip=True),
153 + "capacite": int(cap.group(1)) if cap else None,
154 + "chambres": int(ch.group(1)) if ch else None,
155 + "etoiles": len(card.select(".card-icons-stars "
156 + ".icon-star")) or None,
157 + })
158 + d["unites"] = unites
159 +
160 + # photos du carrousel principal
161 + imgs: list[str] = []
162 + slider = soup.select_one(".block-slider-img")
163 + if slider is not None:
164 + for img in slider.find_all("img"):
165 + u = img.get("data-src") or img.get("src") or ""
166 + if u.startswith("/"):
167 + u = BASE + u
168 + if u.startswith("https://") and u not in imgs:
169 + imgs.append(u)
170 + d["images"] = imgs[:12]
171 + return d
172 +
173 + # -- page forfait (prix « par personne / nuit ») ---------------------------
174 + def _fetch_forfait(self, slug: str) -> dict:
175 + html = self.get(PREFIX_FORFAIT + slug).text
176 + m = _PRIX_FORFAIT_RE.search(html)
177 + if not m:
178 + return {}
179 + try:
180 + prix = float(re.sub(r"[\s ]", "", m.group(1)).replace(",", "."))
181 + except ValueError:
182 + return {}
183 + unite = ""
184 + mm = re.search(r'class="card-pricing">.*?<p>([^<]+)</p>', html, re.S)
185 + if mm:
186 + unite = mm.group(1).strip()
187 + return {"prix": prix, "unite": unite}
188 +
189 + # -- inventaire ------------------------------------------------------------
190 + def fetch(self) -> list[StListing]:
191 + limit = int(os.environ.get("LOUKA_POURVOIRIES_LIMIT", "0") or 0)
192 + month = time.strftime("%Y-%m") # re-visite mensuelle des fiches
193 +
194 + xml = self.get(SITEMAP_FICHES).text
195 + slugs = sorted({loc[len(PREFIX_FICHE):].strip("/")
196 + for loc in _LOC_RE.findall(xml)
197 + if loc.startswith(PREFIX_FICHE)})
198 +
199 + # forfaits groupés par no de permis (slug « 01-501-… »)
200 + forfaits: dict[str, list[str]] = {}
201 + try:
202 + xmlf = self.get(SITEMAP_FORFAITS).text
203 + for loc in _LOC_RE.findall(xmlf):
204 + if not loc.startswith(PREFIX_FORFAIT):
205 + continue
206 + fslug = loc[len(PREFIX_FORFAIT):].strip("/")
207 + m = _FORFAIT_PERMIS_RE.match(fslug)
208 + if m:
209 + forfaits.setdefault(m.group(1), []).append(fslug)
210 + except Exception as exc: # noqa: BLE001 — les forfaits sont optionnels
211 + print(f"[pourvoiries] sitemap forfaits : {exc}", file=sys.stderr)
212 +
213 + listings: list[StListing] = []
214 + for slug in slugs:
215 + try:
216 + d = self.detail(slug, month, lambda s=slug: self._fetch_fiche(s))
217 + except Exception as exc: # noqa: BLE001
218 + print(f"[pourvoiries] fiche {slug} : {exc}", file=sys.stderr)
219 + continue
220 + unites = d.get("unites") or []
221 + if not d.get("nom") or not unites: # pas d'hébergement → hors sujet
222 + continue
223 +
224 + # prix : forfait le moins cher de la pourvoirie (par pers. / nuit)
225 + price_label = ""
226 + m = _PERMIS_RE.search(slug)
227 + permis = m.group(1) if m else ""
228 + best: dict = {}
229 + for fslug in forfaits.get(permis, []):
230 + try:
231 + f = self.detail(f"forfait:{fslug}", month,
232 + lambda s=fslug: self._fetch_forfait(s))
233 + except Exception as exc: # noqa: BLE001
234 + print(f"[pourvoiries] forfait {fslug} : {exc}",
235 + file=sys.stderr)
236 + continue
237 + # seuls les forfaits tarifés à la nuit (ou au jour) sont
238 + # comparables — un prix « par personne / séjour » fausserait
239 + # le prix/nuit dérivé par finalize(). Attention : « séjour »
240 + # contient « jour », d'où l'exclusion explicite.
241 + unite = f.get("unite", "").lower()
242 + if "jour" not in unite and "nuit" not in unite:
243 + continue
244 + if "séjour" in unite or "sejour" in unite:
245 + continue
246 + if f.get("prix") and (not best or f["prix"] < best["prix"]):
247 + best = f
248 + if best:
249 + unite = best.get("unite") or "par personne / nuit"
250 + price_label = (f"forfait à partir de {best['prix']:.0f} $ "
251 + f"{unite}")
252 +
253 + region = d.get("region", "")
254 + region = _REGIONS_FPQ.get(region.lower(), normalize_region(region))
255 +
256 + caps = [u["capacite"] for u in unites if u.get("capacite")]
257 + chs = [u["chambres"] for u in unites if u.get("chambres")]
258 + details = {k: v for k, v in {
259 + "permis": permis,
260 + "nb_unites": len(unites),
261 + "unites": unites[:40],
262 + "ouverture": d.get("ouverture", ""),
263 + "restauration": d.get("restauration", ""),
264 + "type_pourvoirie": d.get("type_pourvoirie", ""),
265 + "langues": d.get("langues", ""),
266 + }.items() if v}
267 +
268 + listings.append(StListing(
269 + source=self.source_id,
270 + external_id=slug,
271 + url=PREFIX_FICHE + slug,
272 + title=d["nom"],
273 + property_type=_property_type([u["type"] for u in unites]),
274 + city=d.get("ville", ""),
275 + region=region,
276 + price_label=price_label,
277 + capacity=float(max(caps)) if caps else None,
278 + bedrooms=float(max(chs)) if chs else None,
279 + citq=d.get("citq", ""),
280 + description=d.get("description", ""),
281 + details=details,
282 + images=d.get("images") or [],
283 + lat=d.get("lat"),
284 + lng=d.get("lng"),
285 + ))
286 + if limit and len(listings) >= limit:
287 + break
288 + return listings
modified louka/shortterm/connectors/qldc.py +54 −7
@@ -7,10 +7,18 @@
7 7 # Méthode : pagination de la liste globale /chalets-a-louer?page=N (site
8 8 # ASP.NET WebForms, 12 cartes/page, HTML statique — la pagination « infinie »
9 9 # accepte le paramètre ?page). Cartes : id stable (/chalet-a-louer/<id>),
10 −# titre, région + ville, capacité, chambres, photo. La page détail (via
11 −# self.detail, cache BD) en variante ?map=o ajoute lat/lng (champs cachés
12 −# InfoLocalisation_hf_lat/long — absents de la page de base), grille de
13 −# tarifs, description, no CITQ, sdb/lits, commodités et photos.
10 +# titre, région + ville, capacité, chambres, photo, et souvent un prix
11 +# « à partir de » (encadré .ListPrix : « Nuit 395$ » ou « Semaine 1030$ »).
12 +# La page détail (via self.detail, cache BD) en variante ?map=o ajoute
13 +# lat/lng (champs cachés InfoLocalisation_hf_lat/long — absents de la page
14 +# de base), grille de tarifs, description, no CITQ, sdb/lits, commodités
15 +# et photos.
16 +#
17 +# Prix : ~40 % des fiches seulement ont la grille de tarifs ; les autres ont
18 +# soit un tarif en texte libre (ctl16_lblvchTarif_Terme, parfois avec
19 +# montants — attention aux dépôts), soit rien du tout (contact direct).
20 +# Ordre de préférence : grille détail > texte libre détail > encadré de la
21 +# carte liste. Beaucoup de fiches n'affichent réellement aucun prix.
14 22 # -----------------------------------------------------------------------------
15 23 from __future__ import annotations
16 24
@@ -88,10 +96,11 @@ class QuebecLocationDeChalets(StConnector):
88 96 page += 1
89 97
90 98 for lst in listings:
99 + # « v2 » : tarif en texte libre ajouté au parseur détail
91 100 cle = hashlib.sha1(("|".join([
92 101 lst.title, lst.city, lst.region,
93 102 str(lst.capacity), str(lst.bedrooms),
94 − ]) + time.strftime("|%Y-%m")).encode("utf-8")).hexdigest()
103 + ]) + time.strftime("|%Y-%m|v2")).encode("utf-8")).hexdigest()
95 104 try:
96 105 d = self.detail(lst.external_id, cle,
97 106 lambda u=lst.url: self._detail(u))
@@ -99,8 +108,12 @@ class QuebecLocationDeChalets(StConnector):
99 108 d = {}
100 109 if not d:
101 110 continue
102 − lst.price_night = d.get("price_night")
103 − lst.price_label = d.get("price_label") or ""
111 + # le prix de la page détail prime ; sinon on garde celui de la
112 + # carte de liste (« à partir de … »)
113 + if d.get("price_night") is not None:
114 + lst.price_night = d["price_night"]
115 + if d.get("price_label"):
116 + lst.price_label = d["price_label"]
104 117 lst.description = d.get("description") or ""
105 118 lst.citq = d.get("citq") or ""
106 119 lst.amenities = d.get("amenities") or []
@@ -159,6 +172,16 @@ class QuebecLocationDeChalets(StConnector):
159 172 if img is not None:
160 173 images.append(urljoin(BASE, img["src"].split("?")[0]))
161 174
175 + # encadré de prix de la carte (« à partir de / Nuit 395$ » ou
176 + # « Semaine 1030$ ») — repli si la page détail n'affiche aucun tarif.
177 + # NE PAS remonter plus haut que la carte : on attraperait le prix
178 + # d'une carte voisine.
179 + prix_label, prix_nuit = "", None
180 + bloc = carte.select_one(".ListPrix") if carte is not None else None
181 + if bloc is not None:
182 + prix_label = re.sub(r"\s+", " ", bloc.get_text(" ", strip=True))
183 + prix_nuit = _prix_nuit(prix_label, prix_label)
184 +
162 185 return StListing(
163 186 source=self.source_id,
164 187 external_id=eid,
@@ -167,6 +190,8 @@ class QuebecLocationDeChalets(StConnector):
167 190 property_type="Chalet",
168 191 city=ville,
169 192 region=_REGIONS.get(region, region),
193 + price_night=prix_nuit,
194 + price_label=prix_label,
170 195 capacity=capacite,
171 196 bedrooms=chambres,
172 197 images=images,
@@ -238,6 +263,28 @@ class QuebecLocationDeChalets(StConnector):
238 263 d["price_night"] = _prix_nuit(d["price_label"],
239 264 d["price_label"])
240 265
266 + # tarif en texte libre (fiches sans grille) : on ne retient que les
267 + # phrases avec un montant ET une période (nuit/jour/semaine), en
268 + # ignorant dépôts et cautions
269 + if "price_night" not in d:
270 + terme = soup.find(id="ctl16_lblvchTarif_Terme")
271 + if terme is not None:
272 + candidats = []
273 + # split en phrases sans casser les décimales (« 129.00$ »)
274 + for phrase in re.split(r"[\n;•]|\.(?!\d)",
275 + terme.get_text("\n", strip=True)):
276 + if "$" not in phrase \
277 + or re.search(r"(?i)d[ée]p[ôo]t|caution|rabais", phrase) \
278 + or not re.search(r"(?i)nuit|jour|sem", phrase):
279 + continue
280 + pn = _prix_nuit(phrase, phrase)
281 + if pn:
282 + candidats.append((pn, phrase.strip()))
283 + if candidats:
284 + pn, phrase = min(candidats)
285 + d["price_night"] = pn
286 + d.setdefault("price_label", re.sub(r"\s+", " ", phrase)[:120])
287 +
241 288 desc = soup.find(id="InfoDescription_pnlDescription")
242 289 if desc is not None:
243 290 texte = desc.get_text("\n", strip=True)
added louka/shortterm/connectors/rezerve.py +116 −0
@@ -0,0 +1,116 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/rezerve.py : Rëzerve / reserver.ca (Charlevoix, Laurentides,
4 +# Estrie, Lanaudière, Mauricie, Outaouais…)
5 +#
6 +# Gestionnaire québécois (~85 chalets) sur moteur Guesty. Le site reserver.ca
7 +# est un Next.js (App Router) : le catalogue /chalets est RENDU SERVEUR et le
8 +# payload React Flight (`self.__next_f.push([1,"…"])`) contient l'objet
9 +# {"listings":[…]} complet — id Guesty, slug, nom, description fr, photos,
10 +# chambres, sdb, lits, capacité, prix « à partir de » CAD, ville, adresse,
11 +# lat/lng, commodités, numéro CITQ, heures d'arrivée/départ. UNE SEULE requête
12 +# suffit, aucune page détail à visiter.
13 +# External_id = id Guesty (stable). URL publique : /chalets/<slug>.
14 +# -----------------------------------------------------------------------------
15 +from __future__ import annotations
16 +
17 +import json
18 +import re
19 +
20 +from ..schema import StListing
21 +from .base import StConnector
22 +
23 +SITE = "https://reserver.ca"
24 +
25 +
26 +def _num(v) -> float | None:
27 + try:
28 + return float(v) if v not in (None, "") else None
29 + except (TypeError, ValueError):
30 + return None
31 +
32 +
33 +def _flight_blob(html: str) -> str:
34 + """Reconstitue le payload React Flight (chunks __next_f concaténés)."""
35 + parts = []
36 + for c in re.findall(r'self\.__next_f\.push\(\[1,"((?:[^"\\]|\\.)*)"\]\)',
37 + html):
38 + try:
39 + parts.append(json.loads(f'"{c}"'))
40 + except ValueError:
41 + continue
42 + return "".join(parts)
43 +
44 +
45 +class Rezerve(StConnector):
46 + source_id = "rezerve"
47 +
48 + def _catalog(self) -> list[dict]:
49 + html = self.get(f"{SITE}/chalets").text
50 + blob = _flight_blob(html)
51 + i = blob.find('{"listings":[')
52 + if i < 0:
53 + return []
54 + obj, _ = json.JSONDecoder().raw_decode(blob[i:])
55 + return obj.get("listings") or []
56 +
57 + @staticmethod
58 + def _description(it: dict) -> str:
59 + desc = str(it.get("description") or "")
60 + if not desc or desc.startswith("$"): # référence Flight non résolue
61 + desc = str((it.get("descriptions") or {}).get("fr") or "")
62 + if desc.startswith("$"):
63 + desc = ""
64 + return re.sub(r"\s+", " ", desc).strip()[:4000]
65 +
66 + # -- contrat ----------------------------------------------------------
67 + def fetch(self) -> list[StListing]:
68 + listings: list[StListing] = []
69 + for it in self._catalog():
70 + lid = str(it.get("id") or "").strip()
71 + slug = (it.get("slug") or "").strip()
72 + title = (it.get("name") or "").strip()
73 + if not lid or not slug or not title:
74 + continue
75 +
76 + geo = it.get("geo") or {}
77 + citq = str((it.get("citq") or {}).get("number") or "").strip()
78 + street = (it.get("addressStreet") or "").strip()
79 + postal = (it.get("addressPostal") or "").strip()
80 + price = _num(it.get("priceFromCAD"))
81 +
82 + details = {k: v for k, v in {
83 + "guesty_id": lid,
84 + "area_sqft": it.get("areaSquareFeet"),
85 + "check_in": it.get("checkInTime"),
86 + "check_out": it.get("checkOutTime"),
87 + "tags": it.get("tags") or None,
88 + }.items() if v}
89 +
90 + listings.append(StListing(
91 + source=self.source_id,
92 + external_id=lid,
93 + url=f"{SITE}/chalets/{slug}",
94 + title=title,
95 + property_type="Chalet",
96 + address=" ".join(p for p in (street, postal) if p),
97 + city=(it.get("city") or "").strip(),
98 + region="", # ville + lat/lng font foi
99 + price_night=price,
100 + price_label=(f"à partir de {price:.0f} $ / nuit"
101 + if price else ""),
102 + capacity=_num(it.get("maxGuests")),
103 + bedrooms=_num(it.get("bedrooms")),
104 + beds=_num(it.get("beds")),
105 + bathrooms=_num(it.get("bathrooms")),
106 + citq=citq if re.fullmatch(r"\d{6}", citq) else "",
107 + description=self._description(it),
108 + amenities=[a for a in (it.get("amenities") or [])
109 + if isinstance(a, str)][:80],
110 + details=details,
111 + images=[u for u in (it.get("images") or [])
112 + if isinstance(u, str)][:20],
113 + lat=_num(geo.get("lat")),
114 + lng=_num(geo.get("lng")),
115 + ))
116 + return listings
added louka/shortterm/connectors/rvmt.py +215 −0
@@ -0,0 +1,215 @@
1 +# -----------------------------------------------------------------------------
2 +# Lou-Ka — Location court terme
3 +# connectors/rvmt.py : Rendez-vous Mont-Tremblant (rvmt.com) — agence de
4 +# gestion locative de la station Mont-Tremblant, ~108 condos, maisons et
5 +# maisons de ville (Plateau, Algonquin, Étoile du Matin, etc.).
6 +#
7 +# Méthode : le site (Nuxt 3) renvoie systématiquement 429 en direct →
8 +# UN SEUL appel Scrapfly (sans render_js) sur une page de complexe
9 +# (/fr/algonquin — la page liste n'embarque PAS les taxonomies, la page
10 +# complexe si). Tout l'inventaire (les 108 unités) vit dans le
11 +# <script id="__NUXT_DATA__"> au format devalue : un tableau plat où les
12 +# valeurs des dicts/listes sont des INDEX entiers vers d'autres cases
13 +# (wrappers ["ShallowReactive", n] à déréférencer).
14 +# On y trouve :
15 +# - le nœud racine {units-rvmt: [108 unités], features-rvmt, beds-rvmt,
16 +# rooms-rvmt, areas-rvmt, amenities-rvmt} — chaque unité est complète :
17 +# code (external_id stable, ex. PLA214-03), adresse+lat/lng, chambres,
18 +# sdb, occupancy, pets_allowed, rental_license (CITQ), min/max_price
19 +# ($/nuit CAD), description fr, galerie Cloudinary, features/amenities
20 +# (ids → libellés fr via les taxonomies), lits par pièce ;
21 +# - le nœud site-config {pages, nav, slug} dont les pages
22 +# data == {"id": <mongo-id du complexe>} donnent l'URL du complexe →
23 +# URL détail = /fr/<complexe>/<slug(nom)> (2 unités non mappées
24 +# retombent sur la page liste).
25 +#
26 +# Réglage env : LOUKA_RVMT_LIMIT (nb max d'annonces, 0 = tout).
27 +# -----------------------------------------------------------------------------
28 +from __future__ import annotations
29 +
30 +import html as _html
31 +import json
32 +import os
33 +import re
34 +import unicodedata
35 +
36 +from ..schema import StListing
37 +from .base import StConnector
38 +
39 +LIST_URL = "https://www.rvmt.com/fr/condos-mont-tremblant"
40 +PAYLOAD_URL = "https://www.rvmt.com/fr/algonquin" # payload complet (taxos)
41 +
42 +# tag pt-* → type canonique
43 +_TYPES = {"pt-condo": "Condo", "pt-townhome": "Maison",
44 + "pt-private-home": "Maison", "pt-hotel": "Condo"}
45 +
46 +_WRAPPERS = ("ShallowReactive", "Reactive", "Ref", "ShallowRef", "EmptyRef",
47 + "EmptyShallowRef")
48 +
49 +
50 +def _slugify(name: str) -> str:
51 + s = unicodedata.normalize("NFKD", name).encode("ascii", "ignore").decode()
52 + return re.sub(r"-{2,}", "-", re.sub(r"[^a-z0-9]+", "-", s.lower())).strip("-")
53 +
54 +
55 +def _fr(node) -> str:
56 + """Nœud {text: {fr, en}} ou {fr, en} → libellé français (repli anglais)."""
57 + if isinstance(node, dict):
58 + t = node.get("text") if isinstance(node.get("text"), dict) else node
59 + if isinstance(t, dict):
60 + return _html.unescape(str(t.get("fr") or t.get("en") or "")).strip()
61 + return ""
62 +
63 +
64 +class Rvmt(StConnector):
65 + source_id = "rvmt"
66 +
67 + # -- devalue --------------------------------------------------------------
68 + def _resolve(self, arr: list, i, seen: frozenset = frozenset()):
69 + v = arr[i] if isinstance(i, int) and 0 <= i < len(arr) else i
70 + if isinstance(v, dict):
71 + if i in seen:
72 + return None
73 + return {k: self._resolve(arr, x, seen | {i}) for k, x in v.items()}
74 + if isinstance(v, list):
75 + if v and v[0] in _WRAPPERS:
76 + return self._resolve(arr, v[1], seen) if len(v) > 1 else None
77 + if i in seen:
78 + return None
79 + return [self._resolve(arr, x, seen | {i}) for x in v]
80 + return v
81 +
82 + # -- contrat --------------------------------------------------------------
83 + def fetch(self) -> list[StListing]:
84 + limit = int(os.environ.get("LOUKA_RVMT_LIMIT", "0") or 0)
85 + h = self.get_scrapfly(PAYLOAD_URL, render_js=False)
86 + m = re.search(r'(?s)<script[^>]*id="__NUXT_DATA__"[^>]*>(.*?)</script>',
87 + h)
88 + if not m:
89 + return []
90 + arr = json.loads(m.group(1))
91 +
92 + # racine = le dict qui contient unités ET taxonomies (une page liste
93 + # n'aurait que units-rvmt : on garde le nœud le plus complet)
94 + root = None
95 + site_cfg = None
96 + fallback_i = None
97 + for i, v in enumerate(arr):
98 + if not isinstance(v, dict):
99 + continue
100 + if root is None and "units-rvmt" in v:
101 + if "features-rvmt" in v:
102 + root = self._resolve(arr, i)
103 + elif fallback_i is None:
104 + fallback_i = i
105 + elif site_cfg is None and {"pages", "nav", "slug"} <= set(v):
106 + site_cfg = self._resolve(arr, i)
107 + if root is not None and site_cfg is not None:
108 + break
109 + if root is None and fallback_i is not None:
110 + root = self._resolve(arr, fallback_i)
111 + if not root:
112 + return []
113 +
114 + units = root.get("units-rvmt") or []
115 + tax = {name: {t.get("value"): _fr(t)
116 + for t in (root.get(f"{name}-rvmt") or [])
117 + if isinstance(t, dict)}
118 + for name in ("features", "beds", "rooms", "areas", "amenities")}
119 +
120 + # id de complexe → slug d'URL (pages du site-config avec data={id})
121 + complexes: dict[str, str] = {}
122 + for p in (site_cfg or {}).get("pages") or []:
123 + data = p.get("data") if isinstance(p, dict) else None
124 + if isinstance(data, dict) and set(data) == {"id"} and p.get("url"):
125 + complexes[str(data["id"])] = str(p["url"]).strip("/")
126 +
127 + listings: list[StListing] = []
128 + vus: set[str] = set()
129 + for u in units:
130 + if not isinstance(u, dict):
131 + continue
132 + code = str(u.get("code") or "").strip()
133 + name = _html.unescape(str(u.get("name") or "")).strip()
134 + if not code or code in vus or not name:
135 + continue
136 + vus.add(code)
137 +
138 + comp = complexes.get(str(u.get("complex") or ""))
139 + url = (f"https://www.rvmt.com/fr/{comp}/{_slugify(name)}"
140 + if comp else LIST_URL)
141 +
142 + addr = u.get("address") or {}
143 + tags = u.get("tags") or []
144 + ptype = next((_TYPES[t] for t in tags if t in _TYPES), "Condo")
145 + area = tax["areas"].get(u.get("area")) or ""
146 +
147 + # lits : somme des quantités des pièces déclarées
148 + beds = 0
149 + for room in u.get("rooms") or []:
150 + for b in (room or {}).get("beds") or []:
151 + try:
152 + beds += int(b.get("quantity") or 0)
153 + except (TypeError, ValueError):
154 + pass
155 +
156 + amen: list[str] = []
157 + for f in u.get("features") or []:
158 + t = tax["features"].get(f)
159 + if t and t not in amen:
160 + amen.append(t)
161 + for a in u.get("amenities") or []:
162 + t = tax["amenities"].get((a or {}).get("amenity"))
163 + if t and t not in amen:
164 + amen.append(t)
165 +
166 + desc = u.get("short_description") or {}
167 + description = _html.unescape(str(desc.get("fr")
168 + or desc.get("en") or "")).strip()
169 +
170 + price = u.get("min_price")
171 + price = float(price) if isinstance(price, (int, float)) \
172 + and 20 <= price <= 20000 else None
173 + price_label = f"À partir de {price:g} $ / nuit" if price else ""
174 +
175 + gallery = [g for g in (u.get("gallery") or [])
176 + if isinstance(g, str) and g.startswith("https://")]
177 +
178 + details = {k: v for k, v in {
179 + "area": area,
180 + "complex": comp or "",
181 + "max_price_night": u.get("max_price"),
182 + "wheelchair_accessible": bool(u.get("wheelchair_accessible")),
183 + }.items() if v}
184 +
185 + line1 = str(addr.get("line_1") or "").strip()
186 + apt = str(addr.get("apt") or "").strip()
187 +
188 + listings.append(StListing(
189 + source=self.source_id,
190 + external_id=code,
191 + url=url,
192 + title=name,
193 + property_type=ptype,
194 + address=f"{line1}, app. {apt}" if line1 and apt else line1,
195 + city=str(addr.get("city") or "Mont-Tremblant").strip(),
196 + region="Laurentides",
197 + price_night=price,
198 + price_label=price_label,
199 + capacity=float(u["occupancy"]) if u.get("occupancy") else None,
200 + bedrooms=float(u["bedrooms"]) if u.get("bedrooms") is not None
201 + else None,
202 + beds=float(beds) if beds else None,
203 + bathrooms=float(u["bathrooms"]) if u.get("bathrooms") else None,
204 + pets="oui" if u.get("pets_allowed") else "non",
205 + citq=str(u.get("rental_license") or "").strip(),
206 + description=description[:5000],
207 + amenities=amen,
208 + details=details,
209 + images=gallery[:20],
210 + lat=addr.get("latitude"),
211 + lng=addr.get("longitude"),
212 + ))
213 + if limit and len(listings) >= limit:
214 + break
215 + return listings
modified louka/shortterm/connectors/sepaq.py +141 −2
@@ -23,6 +23,14 @@
23 23 # distinctif dans le sitemap (il faudrait crawler chaque boucle de camping).
24 24 # Région touristique déduite de l'établissement (table statique ci-dessous).
25 25 # Le site est derrière Cloudflare : self.get() escalade automatiquement.
26 +#
27 +# Description : les pages unité n'ont AUCUN texte descriptif (que des listes
28 +# d'équipements) — la prose vit sur la page éditoriale de l'établissement
29 +# (sepaq.com/<code>/, ex. pq/mot) : on la récupère une fois par établissement
30 +# (cache self.detail) et on l'applique aux unités. Géolocalisation : l'API
31 +# carte échoue souvent (cookie de session perdu derrière l'anti-bot) — repli
32 +# sur les coordonnées statiques de l'établissement (fait géographique stable,
33 +# précision « parc » signalée dans details.geo_precision).
26 34 # -----------------------------------------------------------------------------
27 35 from __future__ import annotations
28 36
@@ -77,6 +85,85 @@ REGION_ETAB = {
77 85 "auberge-de-montagne-des-chic-chocs": "Gaspésie",
78 86 }
79 87
88 +# Coordonnées approximatives de chaque établissement (accueil/centre du
89 +# territoire — fait géographique stable). Repli quand l'API carte ne donne
90 +# pas les coordonnées précises de l'unité.
91 +ETAB_COORDS = {
92 + "centre-touristique-du-lac-kenogami": (48.336, -71.433),
93 + "centre-touristique-du-lac-simon": (46.008, -75.092),
94 + "parc-national-d-aiguebelle": (48.494, -78.712),
95 + "parc-national-d-oka": (45.472, -74.023),
96 + "parc-national-de-frontenac": (45.940, -71.150),
97 + "parc-national-de-la-gaspesie": (48.982, -66.242),
98 + "parc-national-de-la-jacques-cartier": (47.174, -71.374),
99 + "parc-national-de-la-pointe-taillon": (48.665, -72.080),
100 + "parc-national-de-la-yamaska": (45.462, -72.632),
101 + "parc-national-de-plaisance": (45.590, -75.070),
102 + "parc-national-des-grands-jardins": (47.667, -70.628),
103 + "parc-national-des-hautes-gorges-de-la-riviere-malbaie": (47.899, -70.412),
104 + "parc-national-des-monts-valin": (48.597, -70.828),
105 + "parc-national-du-bic": (48.336, -68.802),
106 + "parc-national-du-fjord-du-saguenay": (48.311, -70.324),
107 + "parc-national-du-mont-megantic": (45.455, -71.152),
108 + "parc-national-du-mont-orford": (45.357, -72.235),
109 + "parc-national-du-mont-tremblant": (46.356, -74.462),
110 + "reserve-faunique-ashuapmushuan": (48.890, -72.870),
111 + "reserve-faunique-de-matane": (48.632, -67.114),
112 + "reserve-faunique-de-papineau-labelle": (46.150, -75.300),
113 + "reserve-faunique-de-port-cartier-sept-iles": (50.318, -67.083),
114 + "reserve-faunique-de-port-daniel": (48.240, -64.970),
115 + "reserve-faunique-de-portneuf": (47.100, -72.250),
116 + "reserve-faunique-de-rimouski": (48.075, -68.375),
117 + "reserve-faunique-des-chic-chocs": (48.732, -66.463),
118 + "reserve-faunique-des-laurentides": (47.567, -71.233),
119 + "reserve-faunique-du-saint-maurice": (46.900, -73.100),
120 + "reserve-faunique-la-verendrye": (47.348, -76.868),
121 + "reserve-faunique-mastigouche": (46.600, -73.400),
122 + "reserve-faunique-rouge-matawin": (46.723, -74.500),
123 + "sepaq-anticosti": (49.500, -63.300),
124 + "station-touristique-duchesnay": (46.878, -71.635),
125 + "auberge-de-montagne-des-chic-chocs": (48.900, -66.489),
126 +}
127 +
128 +# Page éditoriale de chaque établissement (sepaq.com/<code>/) — codes vérifiés
129 +# via les titres des sitemaps etablissement/parc-national/reserve-faunique.
130 +ETAB_EDITO = {
131 + "centre-touristique-du-lac-kenogami": "ct/ken",
132 + "centre-touristique-du-lac-simon": "ct/sim",
133 + "parc-national-d-aiguebelle": "pq/aig",
134 + "parc-national-d-oka": "pq/oka",
135 + "parc-national-de-frontenac": "pq/fro",
136 + "parc-national-de-la-gaspesie": "pq/gas",
137 + "parc-national-de-la-jacques-cartier": "pq/jac",
138 + "parc-national-de-la-pointe-taillon": "pq/pta",
139 + "parc-national-de-la-yamaska": "pq/yam",
140 + "parc-national-de-plaisance": "pq/pla",
141 + "parc-national-des-grands-jardins": "pq/grj",
142 + "parc-national-des-hautes-gorges-de-la-riviere-malbaie": "pq/hgo",
143 + "parc-national-des-monts-valin": "pq/mva",
144 + "parc-national-du-bic": "pq/bic",
145 + "parc-national-du-fjord-du-saguenay": "pq/sag",
146 + "parc-national-du-mont-megantic": "pq/mme",
147 + "parc-national-du-mont-orford": "pq/mor",
148 + "parc-national-du-mont-tremblant": "pq/mot",
149 + "reserve-faunique-ashuapmushuan": "rf/ash",
150 + "reserve-faunique-de-matane": "rf/mat",
151 + "reserve-faunique-de-papineau-labelle": "rf/pal",
152 + "reserve-faunique-de-port-cartier-sept-iles": "rf/spc",
153 + "reserve-faunique-de-port-daniel": "rf/pod",
154 + "reserve-faunique-de-portneuf": "rf/por",
155 + "reserve-faunique-de-rimouski": "rf/rim",
156 + "reserve-faunique-des-chic-chocs": "rf/chc",
157 + "reserve-faunique-des-laurentides": "rf/lau",
158 + "reserve-faunique-du-saint-maurice": "rf/stm",
159 + "reserve-faunique-la-verendrye": "rf/lvy",
160 + "reserve-faunique-mastigouche": "rf/mas",
161 + "reserve-faunique-rouge-matawin": "rf/rom",
162 + "sepaq-anticosti": "sepaq-anticosti",
163 + "station-touristique-duchesnay": "ct/duc",
164 + "auberge-de-montagne-des-chic-chocs": "ct/amc",
165 +}
166 +
80 167 _LOC_RE = re.compile(r"<loc>\s*(https?://[^<\s]+)")
81 168 _COOKIE_RE = re.compile(r"(?:JSESSIONID_TRANSAC|__cf_bm)=[^;,\s]+")
82 169 _JSON_ARRAY_RE = re.compile(r"\[.*\]", re.S)
@@ -171,6 +258,34 @@ class Sepaq(StConnector):
171 258 })
172 259 return {"items": items}
173 260
261 + # -- page éditoriale d'un établissement (description riche) -----------------
262 + _EDITO_BRUIT = ("témoins", "cookies", "javascript", "modalités de réserv",
263 + "navigateur", "abonnez-vous", "infolettre")
264 +
265 + def _fetch_edito(self, code: str) -> dict:
266 + from bs4 import BeautifulSoup
267 + html = self.get(f"{BASE}/{code}/").text
268 + soup = BeautifulSoup(html, "html.parser")
269 + if soup.title and "404" in soup.title.get_text():
270 + return {}
271 + parts: list[str] = []
272 + for p in soup.find_all("p"):
273 + txt = re.sub(r"\s+", " ", p.get_text(" ", strip=True))
274 + low = txt.lower()
275 + if len(txt) < 120 or any(b in low for b in self._EDITO_BRUIT):
276 + continue
277 + if txt not in parts:
278 + parts.append(txt)
279 + if sum(len(x) for x in parts) > 1200:
280 + break
281 + desc = "\n\n".join(parts)
282 + if not desc: # repli : meta description de la page
283 + m = re.search(r'<meta name="description" content="([^"]+)"', html)
284 + if m:
285 + import html as _h
286 + desc = _h.unescape(m.group(1)).strip()
287 + return {"description": desc[:2000]} if desc else {}
288 +
174 289 # -- page détail d'une unité --------------------------------------------------
175 290 def _fetch_unit(self, slug: str) -> dict:
176 291 from bs4 import BeautifulSoup
@@ -287,6 +402,21 @@ class Sepaq(StConnector):
287 402 if path:
288 403 geo[path] = it
289 404
405 + # description éditoriale par établissement (une visite, cache BD)
406 + editos: dict[str, str] = {}
407 + for etab in etabs:
408 + code = ETAB_EDITO.get(etab)
409 + if not code:
410 + continue
411 + try:
412 + payload = self.detail(f"edito:{etab}", "v1",
413 + lambda cd=code: self._fetch_edito(cd))
414 + except Exception as exc: # noqa: BLE001 — la description est optionnelle
415 + print(f"[sepaq] édito {etab} : {exc}", file=sys.stderr)
416 + continue
417 + if payload.get("description"):
418 + editos[etab] = payload["description"]
419 +
290 420 month = time.strftime("%Y-%m") # re-visite mensuelle (prix/saison)
291 421 listings: list[StListing] = []
292 422 for slug in sorted(set(units) | set(geo)):
@@ -322,6 +452,14 @@ class Sepaq(StConnector):
322 452
323 453 amenities = (d.get("sections") or {}).get("Description") or []
324 454
455 + # géo : coordonnées précises de l'unité (API carte) sinon repli
456 + # sur celles de l'établissement (précision « parc »)
457 + lat, lng = g.get("lat"), g.get("lng")
458 + if lat is None or lng is None:
459 + lat, lng = ETAB_COORDS.get(etab, (None, None))
460 + if lat is not None:
461 + details["geo_precision"] = "etablissement"
462 +
325 463 lst = StListing(
326 464 source=self.source_id,
327 465 external_id=slug,
@@ -339,11 +477,12 @@ class Sepaq(StConnector):
339 477 beds=float(d["beds"]) if d.get("beds") else None,
340 478 pets=d.get("pets"),
341 479 citq=d.get("citq", ""),
480 + description=editos.get(etab, ""),
342 481 amenities=amenities,
343 482 details=details,
344 483 images=d.get("images") or [],
345 − lat=g.get("lat"),
346 − lng=g.get("lng"),
484 + lat=lat,
485 + lng=lng,
347 486 )
348 487 listings.append(lst)
349 488 return listings
modified louka/shortterm/connectors/sinistar.py +4 −0
@@ -17,6 +17,10 @@
17 17 # (self.__next_f.push) : description, commodités, capacité.
18 18 # 3. AUCUN PRIX PUBLIC : le tarif est négocié entre l'hôte et l'assureur
19 19 # (« couvert par l'assurance ») → price_night=None, prix absent assumé.
20 +# Vérifié 2026-08-25 : ni l'index Algolia, ni le flux React Flight, ni
21 +# le document Firestore public (projects/sinistar-13fdf …/housings/<id>,
22 +# lisible sans auth) ne contiennent de champ prix ou note — les montants
23 +# circulent uniquement dans les soumissions hôte↔assureur (auth requise).
20 24 # La région touristique est déduite des coordonnées (centroïdes Airbnb).
21 25 #
22 26 # Réglage env : LOUKA_SINISTAR_LIMIT (nb max d'annonces, 0 = tout ; utile
modified louka/shortterm/connectors/tremblantliving.py +63 −2
@@ -10,22 +10,31 @@
10 10 # salles de bain, capacité, note/avis, adresse, lat/lng, photos (galerie
11 11 # streamlinevrs.com). La description longue vient du bloc
12 12 # <div class="description block">, les commodités des <li class="amenity_item">.
13 −# Pas de prix statique : les tarifs passent par l'API Streamline
14 −# (admin-ajax.php), bloquée par Cloudflare en POST — price_night reste vide.
13 +# PRIX : pas de prix statique dans le HTML, mais l'API Streamline passe par
14 +# admin-ajax.php avec action=streamlinecore-api-request et le corps JSON
15 +# {methodName, params} DANS LA QUERY STRING (format du plugin Angular) —
16 +# contrairement au POST classique, ce format n'est pas bloqué par
17 +# Cloudflare. GetPropertyRatesRawData(unit_id) retourne la grille des
18 +# tarifs saisonniers ($/nuit) → price_night = minimum des périodes
19 +# courantes/futures (« à partir de »). Rafraîchi à chaque run (37 appels).
15 20 # Les /monthly-rentals/ (units-sitemap.xml) sont du long terme : ignorés.
16 21 # -----------------------------------------------------------------------------
17 22 from __future__ import annotations
18 23
24 +import datetime as _dt
19 25 import html as _html
20 26 import json
21 27 import os
22 28 import re
29 +import sys
30 +from urllib.parse import urlencode
23 31
24 32 from ..schema import StListing
25 33 from .base import StConnector
26 34
27 35 SITE = "https://www.tremblantliving.ca"
28 36 SITEMAP = SITE + "/property-sitemap.xml"
37 +AJAX = SITE + "/wp-admin/admin-ajax.php"
29 38
30 39 # type déduit du nom de la fiche (agence ~100 % chalets et condos)
31 40 _TYPE_HINTS = [
@@ -53,6 +62,45 @@ def _f(v) -> float | None:
53 62 class TremblantLiving(StConnector):
54 63 source_id = "tremblant_living"
55 64
65 + # -- API Streamline (via admin-ajax, JSON en query string) ----------------
66 + def _api(self, method: str, params: dict) -> dict:
67 + req = json.dumps({"methodName": method, "params": params},
68 + separators=(",", ":"))
69 + q = urlencode({"action": "streamlinecore-api-request", "params": req})
70 + resp = self.post(f"{AJAX}?{q}",
71 + headers={"Content-Type": "application/json"})
72 + return resp.json()
73 +
74 + def _price_from_rates(self, unit_id: str) -> tuple[float | None, str]:
75 + """Prix « à partir de » = minimum $/nuit des périodes tarifaires
76 + courantes et futures (GetPropertyRatesRawData). Jamais mis en cache :
77 + les tarifs saisonniers bougent sans que la page change."""
78 + data = self._api("GetPropertyRatesRawData",
79 + {"unit_id": int(unit_id)}).get("data") or {}
80 + rates = data.get("rates") or []
81 + today = _dt.date.today()
82 + prices: list[float] = []
83 + for r in rates if isinstance(rates, list) else [rates]:
84 + try:
85 + end = _dt.datetime.strptime(
86 + str(r.get("period_end") or ""), "%m/%d/%Y").date()
87 + except ValueError:
88 + end = today # période sans date : on la garde
89 + if end < today:
90 + continue # saison passée
91 + for k in ("daily_first_interval_price",
92 + "daily_second_interval_price"):
93 + m = re.search(r"(\d[\d,]*(?:\.\d+)?)", str(r.get(k) or ""))
94 + if m:
95 + v = float(m.group(1).replace(",", ""))
96 + if 20 <= v <= 20000:
97 + prices.append(v)
98 + if not prices:
99 + return None, ""
100 + mn = min(prices)
101 + mn = int(mn) if mn == int(mn) else mn
102 + return float(mn), f"à partir de {mn} $ / nuit"
103 +
56 104 # -- page détail --------------------------------------------------------
57 105 def _detail(self, url: str) -> dict:
58 106 h = self.get(url).text
@@ -134,6 +182,17 @@ class TremblantLiving(StConnector):
134 182 if isinstance(imgs, str):
135 183 imgs = [imgs]
136 184 reviews = agg.get("reviewCount")
185 +
186 + # tarif « à partir de » via l'API Streamline (hors cache détail)
187 + price_night, price_label = None, ""
188 + unit_id = str(ld.get("identifier") or "")
189 + if unit_id.isdigit():
190 + try:
191 + price_night, price_label = self._price_from_rates(unit_id)
192 + except Exception as exc: # tarif manquant ≠ annonce perdue
193 + print(f"[tremblant_living] tarifs {unit_id} : {exc}",
194 + file=sys.stderr)
195 +
137 196 listings.append(StListing(
138 197 source=self.source_id,
139 198 external_id=str(ld.get("identifier") or slug),
@@ -143,6 +202,8 @@ class TremblantLiving(StConnector):
143 202 address=_text(str(addr.get("streetAddress") or "")),
144 203 city=_text(str(addr.get("addressLocality") or "Mont-Tremblant")),
145 204 region="Laurentides",
205 + price_night=price_night,
206 + price_label=price_label,
146 207 capacity=_f(occupancy),
147 208 bedrooms=_f(place.get("numberOfBedrooms")),
148 209 bathrooms=_f(place.get("numberOfBathroomsTotal")),
modified louka/shortterm/connectors/vrbo.py +55 −6
@@ -10,13 +10,21 @@
10 10 #
11 11 # Limites assumées : ~18 cartes rendues par destination (liste virtualisée,
12 12 # le scroll ne persiste pas plus de cartes dans le snapshot DOM), pas de
13 −# lat/lng ni d'adresse sur les cartes (géocodage aval possible), images
14 −# présentes seulement sur les cartes proches du viewport initial.
13 +# lat/lng ni d'adresse sur les cartes, images présentes seulement sur les
14 +# cartes proches du viewport initial.
15 15 # Recherche SANS dates : Vrbo affiche alors un prix « à partir de » par nuit
16 16 # sur les prochaines dates disponibles → price_label + price_night plancher.
17 +#
18 +# Enrichissement : la page détail (Scrapfly ASP SANS rendu JS — le SSR suffit)
19 +# porte description, commodités, capacité, lat/lng et ~6 photos (voir
20 +# _expediadetail.py). ⚠️ certains slugs ont un suffixe (p1234567vb) : l'URL
21 +# sans suffixe redirige vers une page région — on conserve le slug complet.
22 +# Réglage env : LOUKA_VRBO_DETAIL_LIMIT (fetchs détail par sync, défaut 100 ;
23 +# cache permanent dans louka_ct.db, le parc se complète au fil des syncs).
17 24 # -----------------------------------------------------------------------------
18 25 from __future__ import annotations
19 26
27 +import os
20 28 import re
21 29 import sys
22 30 from urllib.parse import quote
@@ -24,8 +32,13 @@ from urllib.parse import quote
24 32 from bs4 import BeautifulSoup
25 33
26 34 from ..schema import StListing
35 +from . import _expediadetail as _ed
27 36 from .base import StConnector
28 37
38 +
39 +class _DetailSkip(Exception):
40 + """Fiche détail sautée (budget épuisé / page invalide) — pas de cache."""
41 +
29 42 # (destination Vrbo, ville affichée, région touristique QC)
30 43 DESTINATIONS = [
31 44 ("Mont-Tremblant, Québec, Canada", "Mont-Tremblant", "Laurentides"),
@@ -61,7 +74,8 @@ TYPE_MAP = {
61 74 "hébergement": "Autre",
62 75 }
63 76
64 −_ID_RE = re.compile(r"/location/p(\d+)")
77 +# slug complet (p123vb) ET id numérique — le suffixe est requis dans l'URL
78 +_ID_RE = re.compile(r"/location/(p(\d+)[a-z]{0,2})")
65 79 _TYPELINE_RE = re.compile(
66 80 r"^([A-ZÀ-Ý][\w’' -]{2,30})\s*·", re.UNICODE)
67 81 _BEDROOMS_RE = re.compile(r"(\d+)\s*chambres?")
@@ -84,8 +98,8 @@ class Vrbo(StConnector):
84 98 m = _ID_RE.search(href)
85 99 if not m:
86 100 return None
87 − external_id = m.group(1)
88 − url = f"https://www.vrbo.com/fr-ca/location/p{external_id}"
101 + external_id = m.group(2)
102 + url = f"https://www.vrbo.com/fr-ca/location/{m.group(1)}"
89 103
90 104 title = ""
91 105 for h in card.find_all("h3"):
@@ -164,6 +178,39 @@ class Vrbo(StConnector):
164 178 images=images,
165 179 )
166 180
181 + # -- enrichissement par la page détail ---------------------------------------
182 + def _enrich_details(self, listings: list[StListing]) -> None:
183 + """Visite les fiches détail via le cache self.detail() sous budget :
184 + les hits de cache sont gratuits, seuls les fetchs réseau comptent."""
185 + limit = max(0, int(os.environ.get("LOUKA_VRBO_DETAIL_LIMIT", "100")
186 + or 100))
187 + used = enriched = streak = 0
188 + for lst in listings:
189 + def fetch_fn(url=lst.url):
190 + nonlocal used, streak
191 + if used >= limit or streak >= 5: # tempête anti-bot : on coupe
192 + raise _DetailSkip
193 + used += 1
194 + html = self.get_scrapfly(url, render_js=False, asp=True)
195 + payload = _ed.parse_detail(html)
196 + if not payload:
197 + streak += 1
198 + raise _DetailSkip # blocage/vide : pas de cache
199 + streak = 0
200 + return payload
201 +
202 + try:
203 + d = self.detail(lst.external_id, "v1", fetch_fn)
204 + except _DetailSkip:
205 + continue
206 + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run
207 + continue
208 + if d:
209 + _ed.apply_detail(lst, d)
210 + enriched += 1
211 + print(f"[vrbo] détail : {enriched} annonces enrichies"
212 + f" ({used}/{limit} fetchs réseau)", file=sys.stderr)
213 +
167 214 # -- contrat -----------------------------------------------------------------
168 215 def fetch(self) -> list[StListing]:
169 216 listings: dict[str, StListing] = {}
@@ -188,4 +235,6 @@ class Vrbo(StConnector):
188 235 continue
189 236 if lst and lst.external_id not in listings:
190 237 listings[lst.external_id] = lst
191 − return list(listings.values())
238 + out = list(listings.values())
239 + self._enrich_details(out)
240 + return out
modified louka/shortterm/web.py +123 −0
@@ -6,6 +6,7 @@
6 6 from __future__ import annotations
7 7
8 8 import json
9 +import math
9 10 import threading
10 11 from pathlib import Path
11 12
@@ -224,6 +225,128 @@ def ct_sources():
224 225 return {"sources": registry}
225 226
226 227
228 +def _percentile(sorted_vals: list[float], q: float) -> float:
229 + if not sorted_vals:
230 + return 0.0
231 + pos = (len(sorted_vals) - 1) * q
232 + lo, hi = int(pos), min(int(pos) + 1, len(sorted_vals) - 1)
233 + return sorted_vals[lo] + (sorted_vals[hi] - sorted_vals[lo]) * (pos - lo)
234 +
235 +
236 +def _haversine_km(lat1, lng1, lat2, lng2) -> float:
237 + rl1, rl2 = math.radians(lat1), math.radians(lat2)
238 + dlat, dlng = rl2 - rl1, math.radians(lng2 - lng1)
239 + a = (math.sin(dlat / 2) ** 2
240 + + math.cos(rl1) * math.cos(rl2) * math.sin(dlng / 2) ** 2)
241 + return 6371.0 * 2 * math.asin(math.sqrt(a))
242 +
243 +
244 +@router.get("/listings/{uid}/context")
245 +def ct_listing_context(uid: str):
246 + """Contexte d'une fiche : analyse du prix/nuit vs segment comparable
247 + (région → + type → + capacité) et hébergements similaires à proximité."""
248 + con = db.connect()
249 + row = con.execute("SELECT * FROM st_listings WHERE uid=?",
250 + (uid,)).fetchone()
251 + if row is None:
252 + con.close()
253 + raise HTTPException(404, "Hébergement introuvable")
254 + l = dict(row)
255 +
256 + # --- analyse de prix : segment le plus précis avec ≥ 12 comparables ------
257 + price_block = None
258 + if l["price_night"] is not None and l["region"]:
259 + candidates: list[tuple[str, str, list]] = []
260 + base_sql = (" AND region=?")
261 + base_args: list = [l["region"]]
262 + if l["property_type"] and l["capacity"]:
263 + candidates.append((
264 + f"{l['property_type']} · {int(l['capacity'])}±2 pers. · {l['region']}",
265 + base_sql + " AND property_type=? AND capacity BETWEEN ? AND ?",
266 + base_args + [l["property_type"], l["capacity"] - 2,
267 + l["capacity"] + 2]))
268 + if l["property_type"]:
269 + candidates.append((
270 + f"{l['property_type']} · {l['region']}",
271 + base_sql + " AND property_type=?",
272 + base_args + [l["property_type"]]))
273 + candidates.append((l["region"], base_sql, base_args))
274 +
275 + for label, extra, args in candidates:
276 + vals = [r["p"] for r in con.execute(
277 + "SELECT price_night p FROM st_listings WHERE active=1"
278 + " AND price_night IS NOT NULL AND uid<>?" + extra
279 + + " ORDER BY price_night", [uid] + args)]
280 + if len(vals) < 12:
281 + continue
282 + med = _percentile(vals, 0.5)
283 + deviation = (l["price_night"] - med) / med if med else None
284 + verdict = None
285 + if deviation is not None:
286 + verdict = ("sous" if deviation <= -0.15
287 + else "dans" if deviation < 0.12 else "dessus")
288 + # histogramme 12 classes entre p5 et p95 (queues écrasées)
289 + lo, hi = _percentile(vals, 0.05), _percentile(vals, 0.95)
290 + bins = []
291 + if hi > lo:
292 + step = (hi - lo) / 12
293 + edges = [lo + i * step for i in range(13)]
294 + counts = [0] * 12
295 + for v in vals:
296 + i = min(11, max(0, int((v - lo) / step)))
297 + counts[i] += 1
298 + bins = [{"x0": round(edges[i]), "x1": round(edges[i + 1]),
299 + "n": counts[i]} for i in range(12)]
300 + rank = sum(1 for v in vals if v <= l["price_night"])
301 + price_block = {
302 + "segment": label, "n": len(vals),
303 + "median": round(med), "p25": round(_percentile(vals, 0.25)),
304 + "p75": round(_percentile(vals, 0.75)),
305 + "deviation": round(deviation, 3) if deviation is not None else None,
306 + "verdict": verdict,
307 + "percentile": round(100 * rank / len(vals)),
308 + "histogram": bins,
309 + }
310 + break
311 +
312 + # --- hébergements similaires ---------------------------------------------
313 + similar: list[dict] = []
314 + if l["lat"] is not None and l["lng"] is not None:
315 + dlat = 0.45 # ≈ 50 km
316 + dlng = dlat / max(0.2, math.cos(math.radians(l["lat"])))
317 + rows = con.execute(
318 + "SELECT * FROM st_listings WHERE active=1 AND uid<>?"
319 + " AND lat BETWEEN ? AND ? AND lng BETWEEN ? AND ?"
320 + " AND images IS NOT NULL AND images<>'[]' LIMIT 400",
321 + (uid, l["lat"] - dlat, l["lat"] + dlat,
322 + l["lng"] - dlng, l["lng"] + dlng)).fetchall()
323 + scored = []
324 + for r in rows:
325 + km = _haversine_km(l["lat"], l["lng"], r["lat"], r["lng"])
326 + if km > 50:
327 + continue
328 + same_type = (l["property_type"]
329 + and r["property_type"] == l["property_type"])
330 + scored.append((0 if same_type else 1, km, r))
331 + scored.sort(key=lambda t: (t[0], t[1]))
332 + for _, km, r in scored[:8]:
333 + d = _row_to_dict(r)
334 + d["distance_km"] = round(km, 1)
335 + similar.append(d)
336 + if not similar and l["region"]:
337 + sql = ("SELECT * FROM st_listings WHERE active=1 AND uid<>?"
338 + " AND region=? AND images IS NOT NULL AND images<>'[]'")
339 + args = [uid, l["region"]]
340 + if l["property_type"]:
341 + sql += " AND property_type=?"
342 + args.append(l["property_type"])
343 + sql += " ORDER BY rating IS NULL, rating DESC, first_seen DESC LIMIT 8"
344 + similar = [_row_to_dict(r) for r in con.execute(sql, args)]
345 +
346 + con.close()
347 + return {"price": price_block, "similar": similar}
348 +
349 +
227 350 @router.get("/listings/{uid}")
228 351 def ct_listing(uid: str):
229 352 con = db.connect()
230 353