Court terme : upgrade majeur — 21 nouveaux connecteurs, 12 enrichis, fiches améliorées
- 21 nouveaux connecteurs (chaleto, hébergement-charlevoix, chaletsalpins,
memoria, pourvoiries FPQ, campingquebec, hebergia, rezerve, accestremblant,
chaletsdanslenord, alouerauxiles, rvmt, boraboreal, homminichalets…)
- enrichissement des existants : airbnb (desc/capacité/commodités via PDP),
booking/vrbo/expedia (Apollo SSR, bug slug vrbo corrigé), sepaq (desc+géo),
tremblantliving (prix API Streamline), qldc/kijiji/parcscanada/bonjourquebec/
gitespassant ; sinistar : prix vérifiés inaccessibles (documenté)
- fiches CT : /api/ct/listings/{uid}/context (analyse prix/nuit par segment,
histogramme, centile) + badge marché + hébergements similaires à proximité
- registre sources_ct.json : 52 sources (7 mortes retirées, 21 ajoutées)
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
41 changed files +5,841 −109
modified
data/sources_ct.json
+324 −34
@@ -1,35 +1,325 @@ | ||
| 1 | 1 | { |
| 2 | − "sources": [ | |
| 3 | − { "id": "airbnb", "name": "Airbnb", "url": "https://www.airbnb.ca", "category": "internationale" }, | |
| 4 | − { "id": "booking", "name": "Booking.com", "url": "https://www.booking.com", "category": "internationale" }, | |
| 5 | − { "id": "vrbo", "name": "Vrbo", "url": "https://www.vrbo.com", "category": "internationale" }, | |
| 6 | − { "id": "expedia", "name": "Expedia (locations de vacances)", "url": "https://www.expedia.ca", "category": "internationale" }, | |
| 7 | − { "id": "agoda", "name": "Agoda Homes", "url": "https://www.agoda.com", "category": "internationale", "status": "en attente de connecteur" }, | |
| 8 | − { "id": "plumguide", "name": "Plum Guide", "url": "https://www.plumguide.com", "category": "internationale", "status": "en attente de connecteur" }, | |
| 9 | − { "id": "glampinghub", "name": "Glamping Hub", "url": "https://glampinghub.com", "category": "internationale" }, | |
| 10 | − { "id": "hipcamp", "name": "Hipcamp", "url": "https://www.hipcamp.com", "category": "internationale" }, | |
| 11 | − { "id": "marriott_hvmi", "name": "Marriott Homes & Villas", "url": "https://homes-and-villas.marriott.com", "category": "internationale", "status": "en attente de connecteur" }, | |
| 12 | − { "id": "furnishedfinder", "name": "Furnished Finder", "url": "https://www.furnishedfinder.com", "category": "internationale", "status": "en attente de connecteur" }, | |
| 13 | − { "id": "corporatestays", "name": "Corporate Stays", "url": "https://www.corporatestays.com", "category": "internationale" }, | |
| 14 | − { "id": "monsieurchalets", "name": "MonsieurChalets", "url": "https://www.monsieurchalets.com", "category": "quebecoise" }, | |
| 15 | − { "id": "wechalet", "name": "WeChalet", "url": "https://wechalet.com", "category": "quebecoise" }, | |
| 16 | − { "id": "chaletsarabais", "name": "Chalets à Rabais", "url": "https://www.chaletsarabais.com", "category": "quebecoise" }, | |
| 17 | − { "id": "chaletsalouer", "name": "ChaletsÀLouer.com", "url": "https://www.chaletsalouer.com", "category": "quebecoise" }, | |
| 18 | − { "id": "mcal", "name": "Maisons et chalets à louer", "url": "https://maisonsetchaletsalouer.com", "category": "quebecoise" }, | |
| 19 | − { "id": "qldc", "name": "Québec Location de Chalets", "url": "https://www.quebeclocationdechalets.com", "category": "quebecoise" }, | |
| 20 | − { "id": "rsvpchalets", "name": "RSVP Chalets", "url": "https://rsvpchalets.com", "category": "quebecoise" }, | |
| 21 | − { "id": "chaletsauquebec", "name": "Chalets au Québec", "url": "https://www.chaletsauquebec.com", "category": "quebecoise" }, | |
| 22 | − { "id": "tremblant_living", "name": "Tremblant Living", "url": "https://www.tremblantliving.com", "category": "agence" }, | |
| 23 | − { "id": "gites_passant", "name": "Gîtes et Auberges du Passant", "url": "https://www.terroiretsaveurs.com", "category": "gites" }, | |
| 24 | − { "id": "bonjourquebec", "name": "Bonjour Québec (CITQ)", "url": "https://www.bonjourquebec.com", "category": "officielle" }, | |
| 25 | − { "id": "sepaq", "name": "Sépaq", "url": "https://www.sepaq.com", "category": "officielle" }, | |
| 26 | − { "id": "parcscanada", "name": "Parcs Canada", "url": "https://reservation.pc.gc.ca", "category": "officielle" }, | |
| 27 | − { "id": "sinistar", "name": "Sinistar", "url": "https://www.sinistar.ca", "category": "relogement" }, | |
| 28 | − { "id": "louer_ca_ct", "name": "Louer.ca (court terme)", "url": "https://www.louer.ca", "category": "relogement", "status": "en attente de connecteur" }, | |
| 29 | − { "id": "kijiji_ct", "name": "Kijiji (court terme)", "url": "https://www.kijiji.ca", "category": "petites-annonces" }, | |
| 30 | − { "id": "lespac_ct", "name": "LesPAC (court terme)", "url": "https://www.lespac.com", "category": "petites-annonces", "status": "en attente de connecteur" }, | |
| 31 | − { "id": "fb_marketplace_ct", "name": "Facebook Marketplace (court terme)", "url": "https://www.facebook.com/marketplace", "category": "petites-annonces", "status": "en attente de connecteur" }, | |
| 32 | − { "id": "craigslist_ct", "name": "Craigslist (court terme)", "url": "https://montreal.craigslist.org", "category": "petites-annonces", "status": "en attente de connecteur" }, | |
| 33 | − { "id": "bonjourresidences", "name": "Bonjour Résidences", "url": "https://www.bonjourresidences.com", "category": "cas-particuliers", "status": "en attente de connecteur" } | |
| 34 | − ] | |
| 35 | −} | |
| 2 | + "sources": [ | |
| 3 | + { | |
| 4 | + "id": "airbnb", | |
| 5 | + "name": "Airbnb", | |
| 6 | + "url": "https://www.airbnb.ca", | |
| 7 | + "category": "internationale" | |
| 8 | + }, | |
| 9 | + { | |
| 10 | + "id": "booking", | |
| 11 | + "name": "Booking.com", | |
| 12 | + "url": "https://www.booking.com", | |
| 13 | + "category": "internationale" | |
| 14 | + }, | |
| 15 | + { | |
| 16 | + "id": "vrbo", | |
| 17 | + "name": "Vrbo", | |
| 18 | + "url": "https://www.vrbo.com", | |
| 19 | + "category": "internationale" | |
| 20 | + }, | |
| 21 | + { | |
| 22 | + "id": "expedia", | |
| 23 | + "name": "Expedia (locations de vacances)", | |
| 24 | + "url": "https://www.expedia.ca", | |
| 25 | + "category": "internationale" | |
| 26 | + }, | |
| 27 | + { | |
| 28 | + "id": "agoda", | |
| 29 | + "name": "Agoda Homes", | |
| 30 | + "url": "https://www.agoda.com", | |
| 31 | + "category": "internationale", | |
| 32 | + "status": "en attente de connecteur" | |
| 33 | + }, | |
| 34 | + { | |
| 35 | + "id": "plumguide", | |
| 36 | + "name": "Plum Guide", | |
| 37 | + "url": "https://www.plumguide.com", | |
| 38 | + "category": "internationale", | |
| 39 | + "status": "en attente de connecteur" | |
| 40 | + }, | |
| 41 | + { | |
| 42 | + "id": "glampinghub", | |
| 43 | + "name": "Glamping Hub", | |
| 44 | + "url": "https://glampinghub.com", | |
| 45 | + "category": "internationale" | |
| 46 | + }, | |
| 47 | + { | |
| 48 | + "id": "hipcamp", | |
| 49 | + "name": "Hipcamp", | |
| 50 | + "url": "https://www.hipcamp.com", | |
| 51 | + "category": "internationale" | |
| 52 | + }, | |
| 53 | + { | |
| 54 | + "id": "marriott_hvmi", | |
| 55 | + "name": "Marriott Homes & Villas", | |
| 56 | + "url": "https://homes-and-villas.marriott.com", | |
| 57 | + "category": "internationale", | |
| 58 | + "status": "en attente de connecteur" | |
| 59 | + }, | |
| 60 | + { | |
| 61 | + "id": "furnishedfinder", | |
| 62 | + "name": "Furnished Finder", | |
| 63 | + "url": "https://www.furnishedfinder.com", | |
| 64 | + "category": "internationale", | |
| 65 | + "status": "en attente de connecteur" | |
| 66 | + }, | |
| 67 | + { | |
| 68 | + "id": "corporatestays", | |
| 69 | + "name": "Corporate Stays", | |
| 70 | + "url": "https://www.corporatestays.com", | |
| 71 | + "category": "internationale" | |
| 72 | + }, | |
| 73 | + { | |
| 74 | + "id": "monsieurchalets", | |
| 75 | + "name": "MonsieurChalets", | |
| 76 | + "url": "https://www.monsieurchalets.com", | |
| 77 | + "category": "quebecoise" | |
| 78 | + }, | |
| 79 | + { | |
| 80 | + "id": "wechalet", | |
| 81 | + "name": "WeChalet", | |
| 82 | + "url": "https://wechalet.com", | |
| 83 | + "category": "quebecoise" | |
| 84 | + }, | |
| 85 | + { | |
| 86 | + "id": "chaletsarabais", | |
| 87 | + "name": "Chalets à Rabais", | |
| 88 | + "url": "https://www.chaletsarabais.com", | |
| 89 | + "category": "quebecoise" | |
| 90 | + }, | |
| 91 | + { | |
| 92 | + "id": "chaletsalouer", | |
| 93 | + "name": "ChaletsÀLouer.com", | |
| 94 | + "url": "https://www.chaletsalouer.com", | |
| 95 | + "category": "quebecoise" | |
| 96 | + }, | |
| 97 | + { | |
| 98 | + "id": "mcal", | |
| 99 | + "name": "Maisons et chalets à louer", | |
| 100 | + "url": "https://maisonsetchaletsalouer.com", | |
| 101 | + "category": "quebecoise" | |
| 102 | + }, | |
| 103 | + { | |
| 104 | + "id": "qldc", | |
| 105 | + "name": "Québec Location de Chalets", | |
| 106 | + "url": "https://www.quebeclocationdechalets.com", | |
| 107 | + "category": "quebecoise" | |
| 108 | + }, | |
| 109 | + { | |
| 110 | + "id": "rsvpchalets", | |
| 111 | + "name": "RSVP Chalets", | |
| 112 | + "url": "https://rsvpchalets.com", | |
| 113 | + "category": "quebecoise" | |
| 114 | + }, | |
| 115 | + { | |
| 116 | + "id": "chaletsauquebec", | |
| 117 | + "name": "Chalets au Québec", | |
| 118 | + "url": "https://www.chaletsauquebec.com", | |
| 119 | + "category": "quebecoise" | |
| 120 | + }, | |
| 121 | + { | |
| 122 | + "id": "tremblant_living", | |
| 123 | + "name": "Tremblant Living", | |
| 124 | + "url": "https://www.tremblantliving.com", | |
| 125 | + "category": "agence" | |
| 126 | + }, | |
| 127 | + { | |
| 128 | + "id": "gites_passant", | |
| 129 | + "name": "Gîtes et Auberges du Passant", | |
| 130 | + "url": "https://www.terroiretsaveurs.com", | |
| 131 | + "category": "gites" | |
| 132 | + }, | |
| 133 | + { | |
| 134 | + "id": "bonjourquebec", | |
| 135 | + "name": "Bonjour Québec (CITQ)", | |
| 136 | + "url": "https://www.bonjourquebec.com", | |
| 137 | + "category": "officielle" | |
| 138 | + }, | |
| 139 | + { | |
| 140 | + "id": "sepaq", | |
| 141 | + "name": "Sépaq", | |
| 142 | + "url": "https://www.sepaq.com", | |
| 143 | + "category": "officielle" | |
| 144 | + }, | |
| 145 | + { | |
| 146 | + "id": "parcscanada", | |
| 147 | + "name": "Parcs Canada", | |
| 148 | + "url": "https://reservation.pc.gc.ca", | |
| 149 | + "category": "officielle" | |
| 150 | + }, | |
| 151 | + { | |
| 152 | + "id": "sinistar", | |
| 153 | + "name": "Sinistar", | |
| 154 | + "url": "https://www.sinistar.ca", | |
| 155 | + "category": "relogement" | |
| 156 | + }, | |
| 157 | + { | |
| 158 | + "id": "louer_ca_ct", | |
| 159 | + "name": "Louer.ca (court terme)", | |
| 160 | + "url": "https://www.louer.ca", | |
| 161 | + "category": "relogement", | |
| 162 | + "status": "en attente de connecteur" | |
| 163 | + }, | |
| 164 | + { | |
| 165 | + "id": "kijiji_ct", | |
| 166 | + "name": "Kijiji (court terme)", | |
| 167 | + "url": "https://www.kijiji.ca", | |
| 168 | + "category": "petites-annonces" | |
| 169 | + }, | |
| 170 | + { | |
| 171 | + "id": "lespac_ct", | |
| 172 | + "name": "LesPAC (court terme)", | |
| 173 | + "url": "https://www.lespac.com", | |
| 174 | + "category": "petites-annonces", | |
| 175 | + "status": "en attente de connecteur" | |
| 176 | + }, | |
| 177 | + { | |
| 178 | + "id": "fb_marketplace_ct", | |
| 179 | + "name": "Facebook Marketplace (court terme)", | |
| 180 | + "url": "https://www.facebook.com/marketplace", | |
| 181 | + "category": "petites-annonces", | |
| 182 | + "status": "en attente de connecteur" | |
| 183 | + }, | |
| 184 | + { | |
| 185 | + "id": "craigslist_ct", | |
| 186 | + "name": "Craigslist (court terme)", | |
| 187 | + "url": "https://montreal.craigslist.org", | |
| 188 | + "category": "petites-annonces", | |
| 189 | + "status": "en attente de connecteur" | |
| 190 | + }, | |
| 191 | + { | |
| 192 | + "id": "bonjourresidences", | |
| 193 | + "name": "Bonjour Résidences", | |
| 194 | + "url": "https://www.bonjourresidences.com", | |
| 195 | + "category": "cas-particuliers", | |
| 196 | + "status": "en attente de connecteur" | |
| 197 | + }, | |
| 198 | + { | |
| 199 | + "id": "chaleto", | |
| 200 | + "name": "Chaletô", | |
| 201 | + "url": "https://chaleto.ca", | |
| 202 | + "category": "quebecoise" | |
| 203 | + }, | |
| 204 | + { | |
| 205 | + "id": "hebergia", | |
| 206 | + "name": "Hebergia", | |
| 207 | + "url": "https://hebergia.ca", | |
| 208 | + "category": "quebecoise" | |
| 209 | + }, | |
| 210 | + { | |
| 211 | + "id": "rezerve", | |
| 212 | + "name": "Rëzerve", | |
| 213 | + "url": "https://reserver.ca", | |
| 214 | + "category": "quebecoise" | |
| 215 | + }, | |
| 216 | + { | |
| 217 | + "id": "campingquebec", | |
| 218 | + "name": "Camping Québec", | |
| 219 | + "url": "https://www.campingquebec.com", | |
| 220 | + "category": "officielle" | |
| 221 | + }, | |
| 222 | + { | |
| 223 | + "id": "pourvoiries", | |
| 224 | + "name": "Pourvoiries du Québec (FPQ)", | |
| 225 | + "url": "https://www.pourvoiries.com", | |
| 226 | + "category": "officielle" | |
| 227 | + }, | |
| 228 | + { | |
| 229 | + "id": "gitesauquebec", | |
| 230 | + "name": "Gîtes au Québec", | |
| 231 | + "url": "https://www.gitesauquebec.com", | |
| 232 | + "category": "gites" | |
| 233 | + }, | |
| 234 | + { | |
| 235 | + "id": "hebergementcharlevoix", | |
| 236 | + "name": "Hébergement Charlevoix", | |
| 237 | + "url": "https://www.hebergement-charlevoix.com", | |
| 238 | + "category": "agence" | |
| 239 | + }, | |
| 240 | + { | |
| 241 | + "id": "chaletsalpins", | |
| 242 | + "name": "Les Chalets Alpins", | |
| 243 | + "url": "https://chaletsalpins.ca", | |
| 244 | + "category": "agence" | |
| 245 | + }, | |
| 246 | + { | |
| 247 | + "id": "memoriachalets", | |
| 248 | + "name": "Memoria Chalets", | |
| 249 | + "url": "https://memoriachalets.com", | |
| 250 | + "category": "agence" | |
| 251 | + }, | |
| 252 | + { | |
| 253 | + "id": "accestremblant", | |
| 254 | + "name": "Accès Tremblant", | |
| 255 | + "url": "https://accestremblant.ca", | |
| 256 | + "category": "agence" | |
| 257 | + }, | |
| 258 | + { | |
| 259 | + "id": "chaletsdanslenord", | |
| 260 | + "name": "Les Chalets dans le Nord", | |
| 261 | + "url": "https://leschaletsdanslenord.com", | |
| 262 | + "category": "agence" | |
| 263 | + }, | |
| 264 | + { | |
| 265 | + "id": "alouerauxiles", | |
| 266 | + "name": "À louer aux Îles", | |
| 267 | + "url": "https://www.alouerauxiles.com", | |
| 268 | + "category": "agence" | |
| 269 | + }, | |
| 270 | + { | |
| 271 | + "id": "lesversants", | |
| 272 | + "name": "Les Versants Mont-Tremblant", | |
| 273 | + "url": "https://www.lesversants.com", | |
| 274 | + "category": "agence" | |
| 275 | + }, | |
| 276 | + { | |
| 277 | + "id": "captremblant", | |
| 278 | + "name": "Cap Tremblant", | |
| 279 | + "url": "https://captremblant.com", | |
| 280 | + "category": "agence" | |
| 281 | + }, | |
| 282 | + { | |
| 283 | + "id": "rvmt", | |
| 284 | + "name": "Rendez-vous Mont-Tremblant", | |
| 285 | + "url": "https://www.rvmt.com", | |
| 286 | + "category": "agence" | |
| 287 | + }, | |
| 288 | + { | |
| 289 | + "id": "locationdechalets", | |
| 290 | + "name": "Location de Chalets 4 Saisons", | |
| 291 | + "url": "https://www.locationdechalets.com", | |
| 292 | + "category": "agence" | |
| 293 | + }, | |
| 294 | + { | |
| 295 | + "id": "chaletsbsl", | |
| 296 | + "name": "Chalets BSL", | |
| 297 | + "url": "https://www.chaletsbsl.com", | |
| 298 | + "category": "agence" | |
| 299 | + }, | |
| 300 | + { | |
| 301 | + "id": "panoraloges", | |
| 302 | + "name": "Panora Loges", | |
| 303 | + "url": "https://www.panoraloges.ca", | |
| 304 | + "category": "agence" | |
| 305 | + }, | |
| 306 | + { | |
| 307 | + "id": "domesstcome", | |
| 308 | + "name": "Dômes St-Côme", | |
| 309 | + "url": "https://www.domesstcome.com", | |
| 310 | + "category": "agence" | |
| 311 | + }, | |
| 312 | + { | |
| 313 | + "id": "boraboreal", | |
| 314 | + "name": "Bora Boréal", | |
| 315 | + "url": "https://www.boraboreal.com", | |
| 316 | + "category": "agence" | |
| 317 | + }, | |
| 318 | + { | |
| 319 | + "id": "homminichalets", | |
| 320 | + "name": "HOM Mini-Chalets", | |
| 321 | + "url": "https://homminichalets.com", | |
| 322 | + "category": "agence" | |
| 323 | + } | |
| 324 | + ] | |
| 325 | +} | |
| \ No newline at end of file | ||
added
frontend/src/components/CtPriceAnalysis.tsx
+96 −0
@@ -0,0 +1,96 @@ | ||
| 1 | +// ----------------------------------------------------------------------------- | |
| 2 | +// Lou-Ka — Location court terme | |
| 3 | +// components/CtPriceAnalysis.tsx : bloc « Analyse du prix / nuit » (fiche CT) | |
| 4 | +// Position du tarif dans la distribution des hébergements comparables | |
| 5 | +// (segment : type + capacité + région touristique, calculé par l'API). | |
| 6 | +// ----------------------------------------------------------------------------- | |
| 7 | +import { CtPriceContext, fmtNight } from "../ctapi"; | |
| 8 | + | |
| 9 | +export function ctDealBadge(p: CtPriceContext | null) { | |
| 10 | + if (!p || p.deviation == null || p.verdict == null) return null; | |
| 11 | + const pct = Math.round(Math.abs(p.deviation) * 100); | |
| 12 | + if (p.verdict === "sous") | |
| 13 | + return { cls: "deal-good", txt: `≈ ${pct} % sous le prix médian du segment` }; | |
| 14 | + if (p.verdict === "dessus") | |
| 15 | + return { cls: "deal-high", txt: `≈ ${pct} % au-dessus du prix médian` }; | |
| 16 | + return { cls: "deal-ok", txt: "Dans les prix du segment" }; | |
| 17 | +} | |
| 18 | + | |
| 19 | +function Histo({ p, price }: { p: CtPriceContext; price: number }) { | |
| 20 | + const bins = p.histogram; | |
| 21 | + if (!bins || bins.length < 4) return null; | |
| 22 | + const W = 320, H = 96, top = 18, bottom = 16; | |
| 23 | + const lo = bins[0].x0, hi = bins[bins.length - 1].x1; | |
| 24 | + if (hi <= lo) return null; | |
| 25 | + const max = Math.max(...bins.map((b) => b.n), 1); | |
| 26 | + const x = (v: number) => ((Math.min(Math.max(v, lo), hi) - lo) / (hi - lo)) * W; | |
| 27 | + const bw = W / bins.length; | |
| 28 | + const plotH = H - top - bottom; | |
| 29 | + const priceX = x(price), medX = x(p.median); | |
| 30 | + const close = Math.abs(priceX - medX) < 64; | |
| 31 | + const lblAnchor = (px: number) => (px < 56 ? "start" : px > W - 56 ? "end" : "middle"); | |
| 32 | + return ( | |
| 33 | + <svg className="fv-histo" viewBox={`0 0 ${W} ${H}`} role="img" | |
| 34 | + aria-label={`Position du prix (${price} $/nuit) parmi ${p.n} hébergements comparables`}> | |
| 35 | + {bins.map((b, i) => { | |
| 36 | + const h = Math.max(1.5, (b.n / max) * plotH); | |
| 37 | + const inBin = price >= b.x0 && price < b.x1; | |
| 38 | + return ( | |
| 39 | + <rect key={i} x={i * bw + 1} y={H - bottom - h} rx="2" | |
| 40 | + width={Math.max(1, bw - 2)} height={h} | |
| 41 | + fill={inBin ? "var(--accent, #ff6a00)" : "rgba(204, 85, 0, 0.28)"}> | |
| 42 | + <title>{`${b.x0} $ – ${b.x1} $ : ${b.n} hébergement${b.n > 1 ? "s" : ""}`}</title> | |
| 43 | + </rect> | |
| 44 | + ); | |
| 45 | + })} | |
| 46 | + {/* fourchette interquartile (bande) */} | |
| 47 | + <rect x={x(p.p25)} y={H - bottom} width={Math.max(2, x(p.p75) - x(p.p25))} | |
| 48 | + height="3.5" rx="1.5" fill="rgba(204, 85, 0, 0.45)" /> | |
| 49 | + {/* marqueur médiane */} | |
| 50 | + <line x1={medX} x2={medX} y1={top - 2} y2={H - bottom} | |
| 51 | + stroke="var(--accent-deep, #cc5500)" strokeWidth="1.6" strokeDasharray="3 3" /> | |
| 52 | + {!close && ( | |
| 53 | + <text x={medX} y={top - 7} textAnchor={lblAnchor(medX)} | |
| 54 | + className="fv-histo-lbl fv-histo-lbl-fv">Médiane</text> | |
| 55 | + )} | |
| 56 | + {/* marqueur du prix demandé */} | |
| 57 | + <line x1={priceX} x2={priceX} y1={top - 2} y2={H - bottom} | |
| 58 | + stroke="var(--ink, #141814)" strokeWidth="2" /> | |
| 59 | + <text x={priceX} y={close ? top - 7 : H - 4} textAnchor={lblAnchor(priceX)} | |
| 60 | + className="fv-histo-lbl">Ce prix{close ? " / médiane" : ""}</text> | |
| 61 | + <text x="1" y={H - 4} className="fv-histo-axis" textAnchor="start">{lo} $</text> | |
| 62 | + <text x={W - 1} y={H - 4} className="fv-histo-axis" textAnchor="end">{hi} $</text> | |
| 63 | + </svg> | |
| 64 | + ); | |
| 65 | +} | |
| 66 | + | |
| 67 | +export default function CtPriceAnalysis({ p, price }: | |
| 68 | + { p: CtPriceContext | null; price: number | null }) { | |
| 69 | + if (!p || price == null) return null; | |
| 70 | + const badge = ctDealBadge(p); | |
| 71 | + return ( | |
| 72 | + <section className="f-bloc f-fairvalue" id="analyse-prix"> | |
| 73 | + <h2>Analyse du prix Lou-Ka</h2> | |
| 74 | + <div className="fv-head"> | |
| 75 | + <div> | |
| 76 | + <div className="fv-value">{fmtNight(p.median)} <small>/ nuit</small></div> | |
| 77 | + <div className="fv-range"> | |
| 78 | + Prix médian du segment · 50 % des prix entre {fmtNight(p.p25)} et {fmtNight(p.p75)} | |
| 79 | + </div> | |
| 80 | + </div> | |
| 81 | + {badge && <div className={`fv-badge ${badge.cls}`}>{badge.txt}</div>} | |
| 82 | + </div> | |
| 83 | + <Histo p={p} price={price} /> | |
| 84 | + <div className="fv-meta"> | |
| 85 | + <span>Segment : <b>{p.segment}</b></span> | |
| 86 | + <span>{p.n.toLocaleString("fr-CA")} hébergements comparables</span> | |
| 87 | + <span>Moins cher que {100 - p.percentile} % du segment</span> | |
| 88 | + </div> | |
| 89 | + <p className="fine"> | |
| 90 | + Comparaison indicative des tarifs affichés (« à partir de ») par les | |
| 91 | + sources pour des hébergements du même type dans la même région — les | |
| 92 | + prix réels varient selon les dates et la durée du séjour. | |
| 93 | + </p> | |
| 94 | + </section> | |
| 95 | + ); | |
| 96 | +} | |
modified
frontend/src/ctapi.ts
+19 −0
@@ -87,6 +87,23 @@ export interface CtFilters { | ||
| 87 | 87 | sort?: string; // recent | prix | prix_desc | note |
| 88 | 88 | } |
| 89 | 89 | |
| 90 | +export interface CtPriceContext { | |
| 91 | + segment: string; | |
| 92 | + n: number; | |
| 93 | + median: number; | |
| 94 | + p25: number; | |
| 95 | + p75: number; | |
| 96 | + deviation: number | null; // (prix − médiane) / médiane | |
| 97 | + verdict: "sous" | "dans" | "dessus" | null; | |
| 98 | + percentile: number; // position du prix dans le segment (0-100) | |
| 99 | + histogram: { x0: number; x1: number; n: number }[]; | |
| 100 | +} | |
| 101 | + | |
| 102 | +export interface CtContext { | |
| 103 | + price: CtPriceContext | null; | |
| 104 | + similar: (CtListing & { distance_km?: number })[]; | |
| 105 | +} | |
| 106 | + | |
| 90 | 107 | const CT_SOURCE_NAMES: Record<string, string> = {}; |
| 91 | 108 | export function registerCtSourceNames(sources: CtSource[]) { |
| 92 | 109 | for (const s of sources) CT_SOURCE_NAMES[s.id] = s.name; |
@@ -118,6 +135,8 @@ export function fetchCtListings(f: CtFilters, limit?: number, offset?: number) { | ||
| 118 | 135 | |
| 119 | 136 | export const fetchCtListing = (uid: string) => |
| 120 | 137 | get<CtListing>(`/api/ct/listings/${encodeURIComponent(uid)}`); |
| 138 | +export const fetchCtContext = (uid: string) => | |
| 139 | + get<CtContext>(`/api/ct/listings/${encodeURIComponent(uid)}/context`); | |
| 121 | 140 | export const fetchCtFacets = () => get<CtFacets>("/api/ct/facets"); |
| 122 | 141 | export const fetchCtStats = () => get<CtStats>("/api/ct/stats"); |
| 123 | 142 | export const fetchCtSources = () => |
modified
frontend/src/pages/CourtTermeFiche.tsx
+27 −2
@@ -8,10 +8,12 @@ | ||
| 8 | 8 | import { lazy, Suspense, useEffect, useRef, useState } from "react"; |
| 9 | 9 | import { Link, useParams } from "react-router-dom"; |
| 10 | 10 | import { |
| 11 | − CtListing, ctSourceName, fetchCtListing, fetchCtSources, fmtNight, | |
| 12 | − registerCtSourceNames, | |
| 11 | + CtContext, CtListing, ctSourceName, fetchCtContext, fetchCtListing, | |
| 12 | + fetchCtSources, fmtNight, registerCtSourceNames, | |
| 13 | 13 | } from "../ctapi"; |
| 14 | 14 | import SmartImg from "../components/SmartImg"; |
| 15 | +import CtListingCard from "../components/CtListingCard"; | |
| 16 | +import CtPriceAnalysis, { ctDealBadge } from "../components/CtPriceAnalysis"; | |
| 15 | 17 | import { IcoAlert } from "../components/Icons"; |
| 16 | 18 | |
| 17 | 19 | const CtFicheMap = lazy(() => import("../components/CtFicheMap")); |
@@ -84,12 +86,15 @@ function Galerie({ images, titre, typeLabel }: | ||
| 84 | 86 | export default function CourtTermeFichePage() { |
| 85 | 87 | const { uid } = useParams<{ uid: string }>(); |
| 86 | 88 | const [l, setL] = useState<CtListing | null>(null); |
| 89 | + const [ctx, setCtx] = useState<CtContext | null>(null); | |
| 87 | 90 | const [error, setError] = useState<string | null>(null); |
| 88 | 91 | |
| 89 | 92 | useEffect(() => { |
| 90 | 93 | fetchCtSources().then((r) => registerCtSourceNames(r.sources)).catch(() => {}); |
| 91 | 94 | if (!uid) return; |
| 95 | + setCtx(null); | |
| 92 | 96 | fetchCtListing(uid).then(setL).catch((e) => setError(String(e))); |
| 97 | + fetchCtContext(uid).then(setCtx).catch(() => setCtx(null)); | |
| 93 | 98 | window.scrollTo(0, 0); |
| 94 | 99 | }, [uid]); |
| 95 | 100 | |
@@ -152,6 +157,12 @@ export default function CourtTermeFichePage() { | ||
| 152 | 157 | {fmtNight(l.price_night, l.price_label)} |
| 153 | 158 | {l.price_night != null && <small> /{NBSP}nuit</small>} |
| 154 | 159 | </div> |
| 160 | + {(() => { | |
| 161 | + const badge = ctDealBadge(ctx?.price ?? null); | |
| 162 | + return badge && ( | |
| 163 | + <div className={`deal-badge ${badge.cls}`}>{badge.txt}</div> | |
| 164 | + ); | |
| 165 | + })()} | |
| 155 | 166 | {l.rating != null && ( |
| 156 | 167 | <div className="deal-badge deal-ok"> |
| 157 | 168 | ★ {l.rating.toLocaleString("fr-CA", { maximumFractionDigits: 1 })} / 5 |
@@ -218,6 +229,7 @@ export default function CourtTermeFichePage() { | ||
| 218 | 229 | </div> |
| 219 | 230 | |
| 220 | 231 | <div className="f-col"> |
| 232 | + <CtPriceAnalysis p={ctx?.price ?? null} price={l.price_night} /> | |
| 221 | 233 | {l.lat != null && l.lng != null && ( |
| 222 | 234 | <section className="f-bloc f-carte" id="emplacement"> |
| 223 | 235 | <h2>Emplacement</h2> |
@@ -232,6 +244,19 @@ export default function CourtTermeFichePage() { | ||
| 232 | 244 | </div> |
| 233 | 245 | </div> |
| 234 | 246 | |
| 247 | + {ctx && ctx.similar.length > 0 && ( | |
| 248 | + <section className="f-bloc f-similaires" aria-label="Hébergements similaires"> | |
| 249 | + <h2> | |
| 250 | + {l.lat != null && ctx.similar[0]?.distance_km != null | |
| 251 | + ? "Hébergements similaires à proximité" | |
| 252 | + : `Hébergements similaires — ${l.region || "même région"}`} | |
| 253 | + </h2> | |
| 254 | + <div className="grid"> | |
| 255 | + {ctx.similar.map((s) => <CtListingCard key={s.uid} l={s} />)} | |
| 256 | + </div> | |
| 257 | + </section> | |
| 258 | + )} | |
| 259 | + | |
| 235 | 260 | <div className="fine f-foot"> |
| 236 | 261 | {synced && <>Dernière synchronisation : {synced}. </>} |
| 237 | 262 | Les prix et disponibilités sont ceux affichés par la source — chaque fiche |
modified
frontend/src/styles.css
+4 −0
@@ -1055,6 +1055,10 @@ html { scroll-padding-top: 76px; } /* header sticky au-dessus des ancres */ | ||
| 1055 | 1055 | } |
| 1056 | 1056 | .f-foot { margin-top: 26px; } |
| 1057 | 1057 | |
| 1058 | +/* --- Fiche court terme : hébergements similaires (pleine largeur) --- */ | |
| 1059 | +.f-similaires { margin-top: 22px; } | |
| 1060 | +.f-similaires .grid { margin-top: 4px; } | |
| 1061 | + | |
| 1058 | 1062 | /* --- Passerelle vers l'annonce originale ------------------------------------ */ |
| 1059 | 1063 | .passerelle { |
| 1060 | 1064 | min-height: calc(100dvh - 120px); display: flex; align-items: center; |
added
louka/shortterm/connectors/_expediadetail.py
+155 −0
@@ -0,0 +1,155 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/_expediadetail.py : parseur partagé des pages détail de la | |
| 4 | +# plateforme Expedia (Vrbo + Expedia — même moteur, même SSR). | |
| 5 | +# | |
| 6 | +# La page détail (…/location/pXXXXXX[vb] ou …hXXXXXX.Hotel-Information), | |
| 7 | +# récupérée via Scrapfly ASP SANS rendu JS, embarque : | |
| 8 | +# - __APOLLO_STATE__ (JSON.parse("…")) → PropertyInfo : | |
| 9 | +# . propertyContentSectionGroups(…).aboutThisProperty → description | |
| 10 | +# (blocs PropertyContentItemMarkup, HTML — inclut souvent le n° CITQ) | |
| 11 | +# . summary.amenities(…) → commodités localisées (infoItems[].text) | |
| 12 | +# - microdonnées schema.org SSR : lat/lng (itemProp latitude/longitude), | |
| 13 | +# capacité (occupancy → value), municipalité (addressLocality) | |
| 14 | +# - galerie : URLs media.vrbo.com / images.trvl-media.com (~6 photos SSR) | |
| 15 | +# Vérifié live 2026-08-25 sur p3363083vb (Vrbo) et h130341845 (Expedia). | |
| 16 | +# ----------------------------------------------------------------------------- | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import html as _html | |
| 20 | +import json | |
| 21 | +import re | |
| 22 | + | |
| 23 | +_APOLLO_RE = re.compile( | |
| 24 | + r'__APOLLO_STATE__\s*=\s*JSON\.parse\("(.*?)(?<!\\)"\)', re.S) | |
| 25 | +_LAT_RE = re.compile(r'itemProp="latitude" content="(-?[\d.]+)"') | |
| 26 | +_LNG_RE = re.compile(r'itemProp="longitude" content="(-?[\d.]+)"') | |
| 27 | +_OCC_RE = re.compile( | |
| 28 | + r'itemProp="occupancy".{0,200}?itemProp="value" content="(\d+)"', re.S) | |
| 29 | +_CITY_RE = re.compile(r'itemProp="addressLocality" content="([^"]+)"') | |
| 30 | +_IMG_RE = re.compile( | |
| 31 | + r'https://(?:media\.vrbo\.com|images\.trvl-media\.com)/lodging/' | |
| 32 | + r'[^"\s\\)&?]+') | |
| 33 | + | |
| 34 | + | |
| 35 | +def _apollo(html: str) -> dict: | |
| 36 | + """Entités du store Apollo SSR ({} si absent/illisible).""" | |
| 37 | + m = _APOLLO_RE.search(html or "") | |
| 38 | + if not m: | |
| 39 | + return {} | |
| 40 | + try: | |
| 41 | + # la chaîne est un littéral JSON : la re-quoter puis parser deux fois | |
| 42 | + return json.loads(json.loads('"' + m.group(1) + '"')) | |
| 43 | + except ValueError: | |
| 44 | + return {} | |
| 45 | + | |
| 46 | + | |
| 47 | +def _markup_texts(node, out: list[str]) -> None: | |
| 48 | + """Collecte récursive des blocs PropertyContentItemMarkup (description).""" | |
| 49 | + if isinstance(node, dict): | |
| 50 | + if node.get("__typename") == "PropertyContentItemMarkup": | |
| 51 | + txt = ((node.get("content") or {}).get("text") or "").strip() | |
| 52 | + if txt: | |
| 53 | + out.append(txt) | |
| 54 | + return | |
| 55 | + for v in node.values(): | |
| 56 | + _markup_texts(v, out) | |
| 57 | + elif isinstance(node, list): | |
| 58 | + for v in node: | |
| 59 | + _markup_texts(v, out) | |
| 60 | + | |
| 61 | + | |
| 62 | +def _strip_html(raw: str) -> str: | |
| 63 | + t = re.sub(r"<br\s*/?>|</p>", "\n", raw) | |
| 64 | + t = re.sub(r"<[^>]+>", " ", t) | |
| 65 | + t = _html.unescape(t) | |
| 66 | + lines = [re.sub(r"\s+", " ", ln).strip() for ln in t.split("\n")] | |
| 67 | + return "\n".join(ln for ln in lines if ln).strip() | |
| 68 | + | |
| 69 | + | |
| 70 | +def parse_detail(html: str) -> dict: | |
| 71 | + """Payload détail {description, amenities, capacity, lat, lng, city, | |
| 72 | + images} d'une page hébergement Vrbo/Expedia ({} si page invalide).""" | |
| 73 | + store = _apollo(html or "") | |
| 74 | + pinfo = next((v for k, v in store.items() | |
| 75 | + if k.startswith("PropertyInfo") and isinstance(v, dict)), {}) | |
| 76 | + if not pinfo and not _LAT_RE.search(html or ""): | |
| 77 | + return {} # page vide / redirection hors fiche | |
| 78 | + | |
| 79 | + out: dict = {} | |
| 80 | + | |
| 81 | + # description : sections « À propos de cet hébergement » | |
| 82 | + paras: list[str] = [] | |
| 83 | + for k, v in pinfo.items(): | |
| 84 | + if k.startswith("propertyContentSectionGroups") and isinstance(v, dict): | |
| 85 | + _markup_texts(v.get("aboutThisProperty"), paras) | |
| 86 | + if not paras: # repli : éditorial du quartier | |
| 87 | + loc = (pinfo.get("summary") or {}).get("location") or {} | |
| 88 | + ed = ((loc.get("whatsAround") or {}).get("editorial") or {}) | |
| 89 | + paras = [t for t in (ed.get("content") or []) if isinstance(t, str)] | |
| 90 | + desc = "\n\n".join(_strip_html(p) for p in paras).strip() | |
| 91 | + if desc: | |
| 92 | + out["description"] = desc[:6000] | |
| 93 | + | |
| 94 | + # commodités localisées (summary.amenities → sections → infoItems) | |
| 95 | + amenities: list[str] = [] | |
| 96 | + for k, v in (pinfo.get("summary") or {}).items(): | |
| 97 | + if not (k.startswith("amenities") and isinstance(v, dict)): | |
| 98 | + continue | |
| 99 | + for sec in v.get("amenities") or []: | |
| 100 | + for cont in (sec or {}).get("contents") or []: | |
| 101 | + for it in (cont or {}).get("infoItems") or []: | |
| 102 | + txt = ((it or {}).get("text") or "").strip() | |
| 103 | + if txt and txt not in amenities: | |
| 104 | + amenities.append(txt) | |
| 105 | + if amenities: | |
| 106 | + out["amenities"] = amenities[:80] | |
| 107 | + | |
| 108 | + # microdonnées SSR : géo, capacité, municipalité | |
| 109 | + m = _LAT_RE.search(html) | |
| 110 | + n = _LNG_RE.search(html) | |
| 111 | + if m and n: | |
| 112 | + try: | |
| 113 | + out["lat"], out["lng"] = float(m.group(1)), float(n.group(1)) | |
| 114 | + except ValueError: | |
| 115 | + pass | |
| 116 | + m = _OCC_RE.search(html) | |
| 117 | + if m: | |
| 118 | + out["capacity"] = float(m.group(1)) | |
| 119 | + m = _CITY_RE.search(html) | |
| 120 | + if m: | |
| 121 | + out["city"] = _html.unescape(m.group(1)).strip() | |
| 122 | + | |
| 123 | + # galerie SSR : dédupliquée par chemin, servie en 1200 px | |
| 124 | + images: list[str] = [] | |
| 125 | + for u in _IMG_RE.findall(html): | |
| 126 | + big = u + "?impolicy=resizecrop&rw=1200&ra=fit" | |
| 127 | + if big not in images: | |
| 128 | + images.append(big) | |
| 129 | + if len(images) >= 15: | |
| 130 | + break | |
| 131 | + if images: | |
| 132 | + out["images"] = images | |
| 133 | + return out | |
| 134 | + | |
| 135 | + | |
| 136 | +def apply_detail(lst, d: dict) -> None: | |
| 137 | + """Applique le payload détail sans écraser ce que la carte a fourni | |
| 138 | + (sauf galerie : on garde la plus grande).""" | |
| 139 | + if not d: | |
| 140 | + return | |
| 141 | + if d.get("description") and len(d["description"]) > len(lst.description or ""): | |
| 142 | + lst.description = d["description"] | |
| 143 | + if d.get("amenities"): | |
| 144 | + seen = {a.lower() for a in lst.amenities} | |
| 145 | + for a in d["amenities"]: | |
| 146 | + if a.lower() not in seen: | |
| 147 | + lst.amenities.append(a) | |
| 148 | + seen.add(a.lower()) | |
| 149 | + for f in ("capacity", "lat", "lng"): | |
| 150 | + if d.get(f) is not None and getattr(lst, f, None) is None: | |
| 151 | + setattr(lst, f, d[f]) | |
| 152 | + if d.get("city") and not lst.city: | |
| 153 | + lst.city = d["city"] | |
| 154 | + if d.get("images") and len(d["images"]) > len(lst.images): | |
| 155 | + lst.images = list(d["images"]) | |
added
louka/shortterm/connectors/accestremblant.py
+173 −0
@@ -0,0 +1,173 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/accestremblant.py : Accès Tremblant (accestremblant.ca) | |
| 4 | +# | |
| 5 | +# Agence de condos à Mont-Tremblant (~17 fiches) — WordPress Avada | |
| 6 | +# (portfolio) + moteur Guesty (guestybookings.com). | |
| 7 | +# | |
| 8 | +# Méthode : | |
| 9 | +# 1. LISTE : sitemap https://accestremblant.ca/avada_portfolio-sitemap.xml | |
| 10 | +# → /condos/<slug>/ (lastmod = clé du cache détail). | |
| 11 | +# 2. DÉTAIL (cache self.detail) : la fiche WP fournit spécs (spans | |
| 12 | +# icon-condos : « 8 Personnes », « 3 chambres », « 3 salles de bain » + | |
| 13 | +# extras type « Foyer au gaz »), description (JSON-LD Yoast), photos | |
| 14 | +# (wp-content/uploads) et l'id Guesty (lien guestybookings.com). | |
| 15 | +# 3. PRIX : chaque fiche contient un carrousel « autres condos » avec | |
| 16 | +# « À partir de N $ » pour les AUTRES fiches ; chaque payload détail | |
| 17 | +# mémorise cette carte slug→prix et fetch() fusionne le tout (le prix | |
| 18 | +# d'une fiche vient donc des autres pages, même à froid depuis le cache). | |
| 19 | +# External_id = slug WP (stable) ; l'id Guesty est gardé dans details. | |
| 20 | +# ----------------------------------------------------------------------------- | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import html as _html | |
| 24 | +import json | |
| 25 | +import re | |
| 26 | + | |
| 27 | +from ..schema import StListing | |
| 28 | +from .base import StConnector | |
| 29 | + | |
| 30 | +SITE = "https://accestremblant.ca" | |
| 31 | +SITEMAP = f"{SITE}/avada_portfolio-sitemap.xml" | |
| 32 | + | |
| 33 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 34 | + | |
| 35 | + | |
| 36 | +def _text(fragment: str) -> str: | |
| 37 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 38 | + | |
| 39 | + | |
| 40 | +class AccesTremblant(StConnector): | |
| 41 | + source_id = "accestremblant" | |
| 42 | + | |
| 43 | + # -- liste (sitemap) ---------------------------------------------------- | |
| 44 | + def _sitemap(self) -> list[tuple[str, str]]: | |
| 45 | + """[(url fiche, lastmod)] — dédupliqué, sans la page index.""" | |
| 46 | + xml = self.get(SITEMAP).text | |
| 47 | + seen: dict[str, str] = {} | |
| 48 | + for m in re.finditer(r"(?s)<url>\s*<loc>([^<]+)</loc>" | |
| 49 | + r"(?:\s*<lastmod>([^<]+)</lastmod>)?", xml): | |
| 50 | + url, lastmod = m.group(1).strip(), (m.group(2) or "").strip() | |
| 51 | + if re.fullmatch(rf"{re.escape(SITE)}/condos/[^/]+/", url): | |
| 52 | + seen.setdefault(url, lastmod) | |
| 53 | + return sorted(seen.items()) | |
| 54 | + | |
| 55 | + # -- page détail ------------------------------------------------------ | |
| 56 | + def _detail(self, url: str) -> dict: | |
| 57 | + h = self.get(url).text | |
| 58 | + d: dict = {} | |
| 59 | + | |
| 60 | + # spécs : <span class="icon-condos">… 8 Personnes / 3 chambres / … | |
| 61 | + extras: list[str] = [] | |
| 62 | + for raw in re.findall(r'(?s)<span class="icon-condos"[^>]*>(.*?)</span>', | |
| 63 | + h): | |
| 64 | + t = _text(raw) | |
| 65 | + if not t: | |
| 66 | + continue | |
| 67 | + m = re.match(r"(\d+)\s*[Pp]ersonnes?", t) | |
| 68 | + if m: | |
| 69 | + d["capacity"] = float(m.group(1)) | |
| 70 | + continue | |
| 71 | + m = re.match(r"(\d+)\s*[Cc]hambres?", t) | |
| 72 | + if m: | |
| 73 | + d["bedrooms"] = float(m.group(1)) | |
| 74 | + continue | |
| 75 | + m = re.match(r"(\d+)\s*[Ss]alles?\s*de\s*bain", t) | |
| 76 | + if m: | |
| 77 | + d["bathrooms"] = float(m.group(1)) | |
| 78 | + continue | |
| 79 | + if len(t) <= 60: | |
| 80 | + extras.append(t) | |
| 81 | + if extras: | |
| 82 | + d["amenities"] = extras | |
| 83 | + | |
| 84 | + # description : JSON-LD Yoast (WebPage.description) | |
| 85 | + m = re.search(r'(?s)<script type="application/ld\+json"[^>]*>' | |
| 86 | + r"(.*?)</script>", h) | |
| 87 | + if m: | |
| 88 | + try: | |
| 89 | + graph = json.loads(m.group(1)).get("@graph") or [] | |
| 90 | + for node in graph: | |
| 91 | + if node.get("@type") == "WebPage" and node.get("description"): | |
| 92 | + d["description"] = _text(node["description"])[:4000] | |
| 93 | + break | |
| 94 | + except ValueError: | |
| 95 | + pass | |
| 96 | + | |
| 97 | + # id Guesty (lien « réserver » guestybookings.com) | |
| 98 | + m = re.search(r"guestybookings\.com/(?:fr/)?properties/([a-f0-9]{24})", | |
| 99 | + h) | |
| 100 | + if m: | |
| 101 | + d["guesty_id"] = m.group(1) | |
| 102 | + | |
| 103 | + # photos : uploads WP (originaux, sans logos ni vignettes -NxN) | |
| 104 | + imgs: list[str] = [] | |
| 105 | + for u in re.findall(r'(https://accestremblant\.ca/wp-content/uploads/' | |
| 106 | + r'20\d\d/\d\d/[^" ]+\.(?:jpe?g|png|webp))', h): | |
| 107 | + if re.search(r"-\d{2,4}x\d{2,4}\.", u): | |
| 108 | + continue | |
| 109 | + if re.search(r"logo|favicon|icon", u, re.I): | |
| 110 | + continue | |
| 111 | + if u not in imgs: | |
| 112 | + imgs.append(u) | |
| 113 | + d["images"] = imgs[:20] | |
| 114 | + | |
| 115 | + # carte des prix « autres condos » : slug → à partir de N $ | |
| 116 | + prices: dict[str, float] = {} | |
| 117 | + for slug, val in re.findall( | |
| 118 | + r'href="https://accestremblant\.ca/condos/([^"/]+)/"[^>]*>' | |
| 119 | + r"\s*À partir de\s*([\d ,]+)\s*\$", h): | |
| 120 | + try: | |
| 121 | + prices[slug] = float(val.replace(" ", "").replace(",", ".")) | |
| 122 | + except ValueError: | |
| 123 | + continue | |
| 124 | + d["prices_seen"] = prices | |
| 125 | + return d | |
| 126 | + | |
| 127 | + # -- contrat ---------------------------------------------------------- | |
| 128 | + def fetch(self) -> list[StListing]: | |
| 129 | + rows = [] | |
| 130 | + price_map: dict[str, float] = {} | |
| 131 | + for url, lastmod in self._sitemap(): | |
| 132 | + slug = url.rstrip("/").rsplit("/", 1)[-1] | |
| 133 | + try: | |
| 134 | + det = self.detail(slug, lastmod, lambda u=url: self._detail(u)) | |
| 135 | + except Exception: | |
| 136 | + continue | |
| 137 | + price_map.update(det.get("prices_seen") or {}) | |
| 138 | + rows.append((slug, url, det)) | |
| 139 | + | |
| 140 | + listings: list[StListing] = [] | |
| 141 | + for slug, url, det in rows: | |
| 142 | + title = slug.replace("-", " ").title() | |
| 143 | + desc = det.get("description") or "" | |
| 144 | + m = re.match(r"([^|]{2,60})\|", desc) | |
| 145 | + if m: # « Verbier C | Découvrez… » | |
| 146 | + title = m.group(1).strip() | |
| 147 | + desc = desc.split("|", 1)[1].strip() | |
| 148 | + price = price_map.get(slug) | |
| 149 | + | |
| 150 | + details = {k: v for k, v in { | |
| 151 | + "guesty_id": det.get("guesty_id"), | |
| 152 | + }.items() if v} | |
| 153 | + | |
| 154 | + listings.append(StListing( | |
| 155 | + source=self.source_id, | |
| 156 | + external_id=slug, | |
| 157 | + url=url, | |
| 158 | + title=title, | |
| 159 | + property_type="Condo", | |
| 160 | + city="Mont-Tremblant", | |
| 161 | + region="Laurentides", | |
| 162 | + price_night=price, | |
| 163 | + price_label=(f"à partir de {price:.0f} $ / nuit" | |
| 164 | + if price else ""), | |
| 165 | + capacity=det.get("capacity"), | |
| 166 | + bedrooms=det.get("bedrooms"), | |
| 167 | + bathrooms=det.get("bathrooms"), | |
| 168 | + description=desc[:4000], | |
| 169 | + amenities=det.get("amenities") or [], | |
| 170 | + details=details, | |
| 171 | + images=det.get("images") or [], | |
| 172 | + )) | |
| 173 | + return listings | |
modified
louka/shortterm/connectors/airbnb.py
+183 −1
@@ -22,9 +22,21 @@ | ||
| 22 | 22 | # L'ordre des cellules est mélangé à chaque run pour que les cellules restantes |
| 23 | 23 | # après épuisement du budget tournent d'un run à l'autre. |
| 24 | 24 | # |
| 25 | +# Enrichissement détail : la fiche /rooms/<id> embarque le même genre de JSON | |
| 26 | +# (<script data-deferred-state-0> → data.node.pdpPresentation) avec description | |
| 27 | +# longue (descriptions.longDescriptionHtml), capacité (personCapacity + | |
| 28 | +# overview.items "6 guests · 2 bedrooms…"), commodités groupées | |
| 29 | +# (amenities.seeAllAmenitiesGroups, drapeau available), règles (rules.groupItems | |
| 30 | +# → animaux) et municipalité (localizedLocation). Visite via le cache détail | |
| 31 | +# self.detail() (louka_ct.db) avec budget par run : le parc (~15 000) se | |
| 32 | +# remplit au fil des syncs. Les échecs réseau/anti-bot ne sont PAS mis en | |
| 33 | +# cache (retentés au prochain run) ; 8 échecs consécutifs coupent | |
| 34 | +# l'enrichissement du run (tempête anti-bot). | |
| 35 | +# | |
| 25 | 36 | # Réglages env : LOUKA_AIRBNB_PAGES (pages par cellule feuille, défaut 15), |
| 26 | 37 | # LOUKA_AIRBNB_BUDGET (budget de requêtes HTML, défaut 1200), |
| 27 | −# LOUKA_AIRBNB_DEPTH (profondeur max du quadtree, défaut 7). | |
| 38 | +# LOUKA_AIRBNB_DEPTH (profondeur max du quadtree, défaut 7), | |
| 39 | +# LOUKA_AIRBNB_DETAIL_LIMIT (fiches détail par run, défaut 800). | |
| 28 | 40 | # ----------------------------------------------------------------------------- |
| 29 | 41 | from __future__ import annotations |
| 30 | 42 | |
@@ -147,6 +159,25 @@ _NIGHTLY_RE = re.compile(r"(\d+)\s*nights?\s*x\s*\$\s*([\d,]+(?:\.\d+)?)", re.I) | ||
| 147 | 159 | _MONEY_RE = re.compile(r"\$\s*([\d,]+(?:\.\d+)?)") |
| 148 | 160 | _NUM_RE = re.compile(r"(\d+(?:\.\d+)?)") |
| 149 | 161 | |
| 162 | +# Clé de version du parseur de fiche détail (bump → re-visite du parc) | |
| 163 | +_PDP_KEY = "pdp-v1" | |
| 164 | + | |
| 165 | + | |
| 166 | +class _DetailSkip(Exception): | |
| 167 | + """Fiche détail indisponible ce run (budget épuisé, blocage anti-bot) — | |
| 168 | + on ne met RIEN en cache pour retenter au prochain sync.""" | |
| 169 | + | |
| 170 | + | |
| 171 | +def _html_to_text(fragment: str) -> str: | |
| 172 | + """HTML de description Airbnb (<br />, <b>…) → texte propre.""" | |
| 173 | + import html as _h | |
| 174 | + txt = re.sub(r"<br\s*/?>", "\n", fragment) | |
| 175 | + txt = re.sub(r"<[^>]+>", " ", txt) | |
| 176 | + txt = _h.unescape(txt) | |
| 177 | + txt = re.sub(r"[ \t]+", " ", txt) | |
| 178 | + txt = re.sub(r" ?\n ?", "\n", txt) | |
| 179 | + return re.sub(r"\n{3,}", "\n\n", txt).strip() | |
| 180 | + | |
| 150 | 181 | |
| 151 | 182 | def _b64_room_id(demand_id: str) -> str: |
| 152 | 183 | """"RGVtYW5kU3RheUxpc3Rpbmc6MTIz" → "123" (DemandStayListing:<id>).""" |
@@ -210,6 +241,156 @@ class Airbnb(StConnector): | ||
| 210 | 241 | return self.get_scrapfly(url, render_js=True, asp=True, |
| 211 | 242 | rendering_wait=3000) |
| 212 | 243 | |
| 244 | + # -- fiche détail (/rooms/<id>) -------------------------------------------- | |
| 245 | + def _pdp_html(self, room_id: str) -> str: | |
| 246 | + """HTML d'une fiche : Bright Data d'abord, Scrapfly ASP en secours | |
| 247 | + (pas de rendu JS : le JSON est embarqué côté serveur).""" | |
| 248 | + url = f"https://www.airbnb.ca/rooms/{room_id}?locale=en¤cy=CAD" | |
| 249 | + html = self._brightdata(url) | |
| 250 | + if "data-deferred-state" in html: | |
| 251 | + return html | |
| 252 | + try: | |
| 253 | + return self.get_scrapfly(url, render_js=False, asp=True) | |
| 254 | + except Exception: # noqa: BLE001 — 429/403/timeout : simple échec | |
| 255 | + return "" | |
| 256 | + | |
| 257 | + @staticmethod | |
| 258 | + def _parse_pdp(html: str) -> dict: | |
| 259 | + """Champs riches depuis data.node.pdpPresentation du JSON embarqué.""" | |
| 260 | + pp = None | |
| 261 | + for blob in re.findall( | |
| 262 | + r'<script[^>]+id="data-deferred-state[^"]*"[^>]*>(.*?)</script>', | |
| 263 | + html, re.S): | |
| 264 | + try: | |
| 265 | + data = json.loads(blob) | |
| 266 | + except ValueError: | |
| 267 | + continue | |
| 268 | + for entry in data.get("niobeClientData") or []: | |
| 269 | + if not (isinstance(entry, list) and len(entry) > 1 | |
| 270 | + and isinstance(entry[1], dict)): | |
| 271 | + continue | |
| 272 | + node = ((entry[1].get("data") or {}).get("node") or {}) | |
| 273 | + if isinstance(node.get("pdpPresentation"), dict): | |
| 274 | + pp = node["pdpPresentation"] | |
| 275 | + break | |
| 276 | + if pp: | |
| 277 | + break | |
| 278 | + if not pp: | |
| 279 | + return {} | |
| 280 | + | |
| 281 | + out: dict = {} | |
| 282 | + # description longue : texte ORIGINAL de l'hôte (souvent français au | |
| 283 | + # Québec), repli sur la version traduite | |
| 284 | + desc = (pp.get("descriptions") or {}).get("longDescriptionHtml") or {} | |
| 285 | + txt = (desc.get("localizedString") | |
| 286 | + or desc.get("localizedStringWithTranslationPreference") or "") | |
| 287 | + if txt: | |
| 288 | + out["description"] = _html_to_text(txt)[:6000] | |
| 289 | + | |
| 290 | + cap = pp.get("personCapacity") | |
| 291 | + if isinstance(cap, (int, float)) and 0 < cap <= 200: | |
| 292 | + out["capacity"] = float(cap) | |
| 293 | + | |
| 294 | + # overview.items : "6 guests", "2 bedrooms", "3 beds", "2 baths" | |
| 295 | + ov = pp.get("overview") or {} | |
| 296 | + for item in ov.get("items") or []: | |
| 297 | + low = (item or "").lower() | |
| 298 | + m = _NUM_RE.search(low) | |
| 299 | + if not m: | |
| 300 | + continue | |
| 301 | + val = float(m.group(1)) | |
| 302 | + if "guest" in low: | |
| 303 | + out.setdefault("capacity", val) | |
| 304 | + elif "bedroom" in low: | |
| 305 | + out["bedrooms"] = val | |
| 306 | + elif "bed" in low: | |
| 307 | + out["beds"] = val | |
| 308 | + elif "bath" in low: | |
| 309 | + out["bathrooms"] = val | |
| 310 | + if ov.get("title"): | |
| 311 | + out["overview_title"] = ov["title"] | |
| 312 | + | |
| 313 | + # commodités disponibles (les groupes "Not included" ont available=False) | |
| 314 | + amen: list[str] = [] | |
| 315 | + for grp in (pp.get("amenities") or {}).get("seeAllAmenitiesGroups") or []: | |
| 316 | + for a in grp.get("amenities") or []: | |
| 317 | + t = (a.get("title") or "").strip() | |
| 318 | + if a.get("available") and t and t not in amen: | |
| 319 | + amen.append(t) | |
| 320 | + if amen: | |
| 321 | + out["amenities"] = amen[:120] | |
| 322 | + | |
| 323 | + # règles de la maison → animaux ("No pets", "Pets allowed", "2 pets…") | |
| 324 | + for grp in (pp.get("rules") or {}).get("groupItems") or []: | |
| 325 | + for it in grp.get("items") or []: | |
| 326 | + if it.get("type") != "HOUSE_RULES_PETS": | |
| 327 | + continue | |
| 328 | + t = (it.get("title") or "").lower() | |
| 329 | + if "no pets" in t or "pas d" in t or "aucun animal" in t: | |
| 330 | + out["pets"] = "non" | |
| 331 | + elif "pets allowed" in t or "animaux accept" in t: | |
| 332 | + out["pets"] = "oui" | |
| 333 | + elif t: | |
| 334 | + out["pets"] = "conditions" | |
| 335 | + | |
| 336 | + loc = (pp.get("localizedLocation") or "").split(",")[0].strip() | |
| 337 | + if loc: | |
| 338 | + out["city"] = loc | |
| 339 | + return out | |
| 340 | + | |
| 341 | + @staticmethod | |
| 342 | + def _apply_pdp(lst: StListing, p: dict) -> None: | |
| 343 | + """Applique un payload détail sans écraser ce que la carte a fourni.""" | |
| 344 | + if p.get("description") and not lst.description: | |
| 345 | + lst.description = p["description"] | |
| 346 | + if p.get("amenities") and not lst.amenities: | |
| 347 | + lst.amenities = list(p["amenities"]) | |
| 348 | + for attr in ("capacity", "bedrooms", "beds", "bathrooms"): | |
| 349 | + if getattr(lst, attr) is None and p.get(attr) is not None: | |
| 350 | + setattr(lst, attr, p[attr]) | |
| 351 | + if p.get("pets") and lst.pets is None: | |
| 352 | + lst.pets = p["pets"] | |
| 353 | + if p.get("city") and not lst.city: | |
| 354 | + lst.city = p["city"] | |
| 355 | + if not lst.property_type and p.get("overview_title"): | |
| 356 | + ptype, _ = _card_type_and_city(p["overview_title"]) | |
| 357 | + lst.property_type = ptype | |
| 358 | + lst.finalize() # drapeaux/citq/pets dérivés du nouveau texte | |
| 359 | + | |
| 360 | + def _enrich_details(self, listings: list[StListing]) -> None: | |
| 361 | + """Visite les fiches détail via le cache self.detail() sous budget : | |
| 362 | + les hits de cache sont gratuits, seuls les fetchs réseau comptent.""" | |
| 363 | + limit = max(0, int(os.environ.get("LOUKA_AIRBNB_DETAIL_LIMIT", "800") | |
| 364 | + or 800)) | |
| 365 | + used = enriched = 0 | |
| 366 | + streak = 0 # échecs réseau consécutifs | |
| 367 | + | |
| 368 | + for lst in listings: | |
| 369 | + def fetch_fn(rid=lst.external_id): | |
| 370 | + nonlocal used, streak | |
| 371 | + if used >= limit or streak >= 8: | |
| 372 | + raise _DetailSkip | |
| 373 | + used += 1 | |
| 374 | + html = self._pdp_html(rid) | |
| 375 | + if "data-deferred-state" not in html: | |
| 376 | + streak += 1 | |
| 377 | + raise _DetailSkip # blocage/vide : pas de mise en cache | |
| 378 | + streak = 0 | |
| 379 | + return self._parse_pdp(html) | |
| 380 | + | |
| 381 | + try: | |
| 382 | + payload = self.detail(lst.external_id, _PDP_KEY, fetch_fn) | |
| 383 | + except _DetailSkip: | |
| 384 | + continue | |
| 385 | + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run | |
| 386 | + continue | |
| 387 | + if payload: | |
| 388 | + self._apply_pdp(lst, payload) | |
| 389 | + enriched += 1 | |
| 390 | + print(f"[airbnb] détail : {enriched} annonces enrichies" | |
| 391 | + f" ({used}/{limit} fetchs réseau, série d'échecs {streak})", | |
| 392 | + file=sys.stderr) | |
| 393 | + | |
| 213 | 394 | # -- parse JSON embarqué ---------------------------------------------------- |
| 214 | 395 | @staticmethod |
| 215 | 396 | def _deferred_results(html: str) -> tuple[list[dict], list[str]]: |
@@ -454,4 +635,5 @@ class Airbnb(StConnector): | ||
| 454 | 635 | if stack: |
| 455 | 636 | print(f"[airbnb] budget épuisé ({budget} req)," |
| 456 | 637 | f" {len(stack)} cellules non visitées", file=sys.stderr) |
| 638 | + self._enrich_details(out) | |
| 457 | 639 | return out |
added
louka/shortterm/connectors/alouerauxiles.py
+202 −0
@@ -0,0 +1,202 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/alouerauxiles.py : À louer aux Îles (alouerauxiles.com) | |
| 4 | +# | |
| 5 | +# Annuaire local des Îles-de-la-Madeleine (~100 maisons/chalets, région très | |
| 6 | +# mal couverte ailleurs). Site statique « Tactical Soft » : cartes par île. | |
| 7 | +# | |
| 8 | +# Méthode : | |
| 9 | +# 1. LISTE : les pages d'île FR (/havre-aubert, /cap-aux-meules, …) sont du | |
| 10 | +# HTML serveur contenant un bloc `<div class=item id=<id> tag=rent …>` | |
| 11 | +# par annonce : id stable, nom (alt="…"), capacité/chambres | |
| 12 | +# (cap="3 chambres (5 pers. max)"), prix (dayrate=186, $/nuit calculé) | |
| 13 | +# et lien public (href=https://alouerauxiles.com/<slug>). Dédup par id | |
| 14 | +# (une annonce peut apparaître sur plusieurs pages). | |
| 15 | +# 2. DÉTAIL (cache self.detail) : /php/page_fr.php?id=<id> → type | |
| 16 | +# d'hébergement, personnes/chambres/salles de bain (attributs title=), | |
| 17 | +# animaux, description, permis CITQ, adresse + ville (après le code | |
| 18 | +# postal), commodités (« Commodités : ») et photos (pages/idlm/rent/…). | |
| 19 | +# External_id = id du bloc item (numérique ou slug, stable). | |
| 20 | +# ----------------------------------------------------------------------------- | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import html as _html | |
| 24 | +import re | |
| 25 | + | |
| 26 | +from ..schema import StListing | |
| 27 | +from .base import StConnector | |
| 28 | + | |
| 29 | +SITE = "https://alouerauxiles.com" | |
| 30 | +WWW = "https://www.alouerauxiles.com" | |
| 31 | + | |
| 32 | +# pages d'île francophones (les slugs anglais sont des doublons) | |
| 33 | +ISLANDS = ["havre-aubert", "cap-aux-meules", "havre-aux-maisons", | |
| 34 | + "pointe-aux-loups", "grosse-ile", "grande-entree", "ile-d-entree"] | |
| 35 | + | |
| 36 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 37 | + | |
| 38 | +# libellé du site → type canonique Lou-Ka | |
| 39 | +_TYPES = {"chalet": "Chalet", "maison": "Maison", "résidence": "Maison", | |
| 40 | + "studio": "Studio", "appartement": "Appartement", "loft": "Loft", | |
| 41 | + "chambre": "Chambre", "gîte": "Gîte", "auberge": "Auberge"} | |
| 42 | + | |
| 43 | + | |
| 44 | +def _text(fragment: str) -> str: | |
| 45 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment)) | |
| 46 | + ).replace("", "").strip() | |
| 47 | + | |
| 48 | + | |
| 49 | +def _num(raw) -> float | None: | |
| 50 | + m = re.search(r"\d+(?:[.,]\d+)?", str(raw or "")) | |
| 51 | + return float(m.group(0).replace(",", ".")) if m else None | |
| 52 | + | |
| 53 | + | |
| 54 | +class ALouerAuxIles(StConnector): | |
| 55 | + source_id = "alouerauxiles" | |
| 56 | + | |
| 57 | + # -- liste (pages d'île) -------------------------------------------------- | |
| 58 | + def _island_items(self) -> dict[str, dict]: | |
| 59 | + items: dict[str, dict] = {} | |
| 60 | + for island in ISLANDS: | |
| 61 | + try: | |
| 62 | + h = self.get(f"{WWW}/{island}").text | |
| 63 | + except Exception: | |
| 64 | + continue | |
| 65 | + for block in re.findall(r"<div class=item ([^>]+)>", h): | |
| 66 | + if "tag=rent" not in block: | |
| 67 | + continue | |
| 68 | + m = re.search(r"\bid=([\w-]+)", block) | |
| 69 | + if not m: | |
| 70 | + continue | |
| 71 | + lid = m.group(1) | |
| 72 | + it = items.setdefault(lid, {"island": island}) | |
| 73 | + m = re.search(r'alt="([^"]+)"', block) | |
| 74 | + if m: | |
| 75 | + it["name"] = _text(m.group(1)) | |
| 76 | + m = re.search(r'cap="(\d+)\s*chambres?\s*\((\d+)\s*pers', | |
| 77 | + block) | |
| 78 | + if m: | |
| 79 | + it["bedrooms"] = float(m.group(1)) | |
| 80 | + it["capacity"] = float(m.group(2)) | |
| 81 | + m = re.search(r"\bdayrate=(\d+(?:\.\d+)?)", block) | |
| 82 | + if m: | |
| 83 | + it["dayrate"] = float(m.group(1)) | |
| 84 | + m = re.search(r"href=(https://alouerauxiles\.com/[\w-]+)\b", | |
| 85 | + block) | |
| 86 | + if m: | |
| 87 | + it["url"] = m.group(1) | |
| 88 | + return items | |
| 89 | + | |
| 90 | + # -- page détail ------------------------------------------------------ | |
| 91 | + def _detail(self, lid: str) -> dict: | |
| 92 | + h = self.get(f"{SITE}/php/page_fr.php", params={"id": lid}).text | |
| 93 | + d: dict = {} | |
| 94 | + | |
| 95 | + m = re.search(r"<b class=t1>([^<]+)</b>", h) | |
| 96 | + if m: | |
| 97 | + d["type_label"] = _text(m.group(1)) | |
| 98 | + | |
| 99 | + m = re.search(r'title="(\d+)\s*personnes', h) | |
| 100 | + if m: | |
| 101 | + d["capacity"] = float(m.group(1)) | |
| 102 | + m = re.search(r'title="(\d+)\s*chambres?"', h) | |
| 103 | + if m: | |
| 104 | + d["bedrooms"] = float(m.group(1)) | |
| 105 | + m = re.search(r'title="(\d+)\s*salle\(?s?\)?\s*de\s*bain', h) | |
| 106 | + if m: | |
| 107 | + d["bathrooms"] = float(m.group(1)) | |
| 108 | + m = re.search(r'title="\s*animaux([^"]*)"', h) | |
| 109 | + if m: | |
| 110 | + d["pets"] = "non" if "non" in m.group(1).lower() else "oui" | |
| 111 | + | |
| 112 | + # description : bloc txtdiv (nom + texte de présentation) | |
| 113 | + m = re.search(r"(?s)<div id=txtdiv>(.*?)</div>", h) | |
| 114 | + if m: | |
| 115 | + frag = re.sub(r"(?s)<b class=t2>.*?</b>", " ", m.group(1)) | |
| 116 | + d["description"] = _text(frag)[:4000] | |
| 117 | + m = re.search(r"(?s)<b class=t2>([^<]+)</b>", | |
| 118 | + h[h.find("txtdiv"):] if "txtdiv" in h else "") | |
| 119 | + if m: | |
| 120 | + d["title"] = _text(m.group(1)) | |
| 121 | + | |
| 122 | + m = re.search(r"CITQ\D{0,12}(\d{6})", h, re.I) | |
| 123 | + if m: | |
| 124 | + d["citq"] = m.group(1) | |
| 125 | + | |
| 126 | + # adresse : rue + « G4T 3H6 l'Étang-du-Nord » → ville après le code | |
| 127 | + m = re.search(r"(?s)Adresse\s*:\s*</b>(.*?)(?:<b class|Contact)", h) | |
| 128 | + if m: | |
| 129 | + lines = [_text(x) for x in re.split(r"<br\s*/?>", m.group(1))] | |
| 130 | + lines = [x for x in lines if x] | |
| 131 | + if lines: | |
| 132 | + d["address"] = lines[0] | |
| 133 | + for x in lines: | |
| 134 | + pm = re.search(r"[A-Z]\d[A-Z]\s?\d[A-Z]\d\s+(.{3,40})$", x) | |
| 135 | + if pm: | |
| 136 | + d["city"] = pm.group(1).strip() | |
| 137 | + break | |
| 138 | + | |
| 139 | + # commodités : lignes « - … » entre « Commodités : » et « À proximité » | |
| 140 | + m = re.search(r"(?s)Commodités\s*:(.*?)(?:À proximité|Contact\s*:|$)", | |
| 141 | + h) | |
| 142 | + if m: | |
| 143 | + amens = [_text(x).lstrip("- ").strip() | |
| 144 | + for x in re.split(r"<br\s*/?>", m.group(1))] | |
| 145 | + d["amenities"] = [a for a in amens if 2 <= len(a) <= 80][:50] | |
| 146 | + | |
| 147 | + # photos du dossier de l'annonce (originaux img/, pas les vignettes) | |
| 148 | + imgs: list[str] = [] | |
| 149 | + for u in re.findall(r"[\"'=](?:\.\./)?(pages/idlm/rent/[\w-]+/img/" | |
| 150 | + r"[^\"'\s>]+\.(?:jpe?g|png|webp))", h, re.I): | |
| 151 | + full = f"{SITE}/{u}" | |
| 152 | + if full not in imgs: | |
| 153 | + imgs.append(full) | |
| 154 | + d["images"] = imgs[:20] | |
| 155 | + return d | |
| 156 | + | |
| 157 | + # -- contrat ---------------------------------------------------------- | |
| 158 | + def fetch(self) -> list[StListing]: | |
| 159 | + listings: list[StListing] = [] | |
| 160 | + for lid, it in self._island_items().items(): | |
| 161 | + key = f"{it.get('name', '')}|{it.get('capacity', '')}|" \ | |
| 162 | + f"{it.get('dayrate', '')}|{it.get('bedrooms', '')}" | |
| 163 | + try: | |
| 164 | + det = self.detail(lid, key, lambda i=lid: self._detail(i)) | |
| 165 | + except Exception: | |
| 166 | + det = {} | |
| 167 | + title = det.get("title") or it.get("name") or "" | |
| 168 | + if not title: | |
| 169 | + continue | |
| 170 | + | |
| 171 | + type_label = (det.get("type_label") or "").lower() | |
| 172 | + ptype = "" | |
| 173 | + for needle, canon in _TYPES.items(): | |
| 174 | + if needle in type_label: | |
| 175 | + ptype = canon | |
| 176 | + break | |
| 177 | + | |
| 178 | + price = it.get("dayrate") | |
| 179 | + city = det.get("city") or it["island"].replace("-", " ").title() | |
| 180 | + listings.append(StListing( | |
| 181 | + source=self.source_id, | |
| 182 | + external_id=lid, | |
| 183 | + url=it.get("url") or f"{SITE}/php/page_fr.php?id={lid}", | |
| 184 | + title=title, | |
| 185 | + property_type=ptype or "Maison", | |
| 186 | + address=det.get("address") or "", | |
| 187 | + city=city, | |
| 188 | + region="Îles-de-la-Madeleine", | |
| 189 | + price_night=price, | |
| 190 | + price_label=(f"à partir de {price:.0f} $ / nuit" | |
| 191 | + if price else ""), | |
| 192 | + capacity=det.get("capacity") or it.get("capacity"), | |
| 193 | + bedrooms=det.get("bedrooms") or it.get("bedrooms"), | |
| 194 | + bathrooms=det.get("bathrooms"), | |
| 195 | + pets=det.get("pets"), | |
| 196 | + citq=det.get("citq") or "", | |
| 197 | + description=det.get("description") or "", | |
| 198 | + amenities=det.get("amenities") or [], | |
| 199 | + details={"ile": it["island"]}, | |
| 200 | + images=det.get("images") or [], | |
| 201 | + )) | |
| 202 | + return listings | |
modified
louka/shortterm/connectors/bonjourquebec.py
+55 −5
@@ -18,10 +18,16 @@ | ||
| 18 | 18 | # quelques milliers max) — gîtes et insolites sont gardés en entier ; |
| 19 | 19 | # 3. fiche /fiche/<id> (cache self.detail — 1 seule visite par fiche) : |
| 20 | 20 | # région touristique, ville, adresse, no d'enregistrement CITQ, |
| 21 | −# description, services/équipements, animaux, tarifs max (détails), | |
| 22 | −# photos. Pas de prix « à partir de » sur le site → seuls les maximums | |
| 23 | −# affichés sont conservés dans details (jamais utilisés comme | |
| 24 | −# price_night pour ne pas fausser le « à partir de »). | |
| 21 | +# description, services/équipements, animaux, tarifs, photos. | |
| 22 | +# PRIX : le site ne publie QUE des maximums par nuitée (widget Tarifs : | |
| 23 | +# « Maximum pour l'unité la plus chère », « Prix maximum par nuitée | |
| 24 | +# prêt-à-camper »). On les expose honnêtement via price_label | |
| 25 | +# (« maximum X $ / nuit ») — finalize() en déduit price_night ; le | |
| 26 | +# libellé garde la nuance (ce n'est pas un « à partir de »). Les | |
| 27 | +# emplacements de camping nu sont ignorés (hors mandat). | |
| 28 | +# CAPACITÉ : jamais publiée en « personnes » sur les fiches — on récupère | |
| 29 | +# ce qui existe : chambres des gîtes (« Chambre : N unités ») et | |
| 30 | +# mentions « N personnes » dans la description (rare). | |
| 25 | 31 | # ----------------------------------------------------------------------------- |
| 26 | 32 | from __future__ import annotations |
| 27 | 33 | |
@@ -60,6 +66,38 @@ _TYPE_KEYWORDS = [ | ||
| 60 | 66 | ] |
| 61 | 67 | |
| 62 | 68 | |
| 69 | +_PRICE_VAL_RE = re.compile(r"\d[\d\s ,.]*\$") | |
| 70 | +_CAP_RE = re.compile(r"(\d{1,2})\s*personnes") | |
| 71 | +_CHAMBRES_RE = re.compile(r"^Chambre\s*:\s*(\d+)\s*unité", re.I) | |
| 72 | + | |
| 73 | + | |
| 74 | +def _price_label(tarifs: list[str]) -> str: | |
| 75 | + """Libellé prix/nuit depuis le widget Tarifs (le site n'affiche que des | |
| 76 | + maximums par nuitée). Priorité : unité la plus chère > prêt-à-camper > | |
| 77 | + autre « par nuitée » — emplacements de camping nu exclus.""" | |
| 78 | + pairs: list[tuple[str, str]] = [] | |
| 79 | + label = "" | |
| 80 | + for txt in tarifs or []: | |
| 81 | + m = _PRICE_VAL_RE.search(txt) | |
| 82 | + if m and label: | |
| 83 | + pairs.append((label.lower(), re.sub(r"[\s ]+", " ", | |
| 84 | + m.group(0)).strip())) | |
| 85 | + label = "" | |
| 86 | + elif not m and txt: | |
| 87 | + label = txt | |
| 88 | + | |
| 89 | + def pick(needle: str, exclude: str = "") -> str: | |
| 90 | + for lab, val in pairs: | |
| 91 | + if needle in lab and (not exclude or exclude not in lab): | |
| 92 | + return val | |
| 93 | + return "" | |
| 94 | + | |
| 95 | + val = (pick("unité la plus chère") | |
| 96 | + or pick("prêt-à-camper") | |
| 97 | + or pick("nuit", exclude="camping")) | |
| 98 | + return f"maximum {val} / nuit" if val else "" | |
| 99 | + | |
| 100 | + | |
| 63 | 101 | def _abs(url: str) -> str: |
| 64 | 102 | url = _html.unescape(url or "").strip() |
| 65 | 103 | if not url: |
@@ -233,6 +271,16 @@ class BonjourQuebec(StConnector): | ||
| 233 | 271 | "unites": d.get("unites"), |
| 234 | 272 | }.items() if v} |
| 235 | 273 | |
| 274 | + # capacité : mention « N personnes » dans la description (rare) | |
| 275 | + caps = [int(x) for x in _CAP_RE.findall(desc) if 1 <= int(x) <= 40] | |
| 276 | + capacity = float(max(caps)) if caps else None | |
| 277 | + # chambres : les gîtes déclarent « Chambre : N unités » | |
| 278 | + bedrooms = None | |
| 279 | + for u in d.get("unites") or []: | |
| 280 | + m = _CHAMBRES_RE.match(u) | |
| 281 | + if m and 0 < int(m.group(1)) <= 30: | |
| 282 | + bedrooms = float(m.group(1)) | |
| 283 | + | |
| 236 | 284 | url = d.get("url_final") or f"{BASE}/fiche/{ext}" |
| 237 | 285 | lst = StListing( |
| 238 | 286 | source=self.source_id, |
@@ -243,7 +291,9 @@ class BonjourQuebec(StConnector): | ||
| 243 | 291 | address=d.get("address", ""), |
| 244 | 292 | city=d.get("city", ""), |
| 245 | 293 | region=d.get("region", ""), |
| 246 | − capacity=None, | |
| 294 | + price_label=_price_label(d.get("tarifs") or []), | |
| 295 | + capacity=capacity, | |
| 296 | + bedrooms=bedrooms, | |
| 247 | 297 | pets=d.get("pets"), |
| 248 | 298 | citq=d.get("citq", ""), |
| 249 | 299 | description=desc, |
modified
louka/shortterm/connectors/booking.py
+124 −1
@@ -17,12 +17,21 @@ | ||
| 17 | 17 | # Recherche AVEC dates génériques (~30 jours, 2 nuits) : sans dates, Booking |
| 18 | 18 | # ne renvoie ni prix ni configuration des unités. Le prix est donc indicatif |
| 19 | 19 | # → price_label « à partir de … » + price_night (le plus bas trouvé). |
| 20 | +# | |
| 21 | +# Enrichissement : la page détail /hotel/ca/<pageName>.fr.html (Scrapfly ASP | |
| 22 | +# SANS rendu JS) embarque son propre store Apollo SSR : description complète | |
| 23 | +# (data-testid="property-description"), commodités localisées (entités | |
| 24 | +# Instance/SimpleFacility) et galerie (AccommodationPhoto). Vérifié live | |
| 25 | +# 2026-08-25 sur /hotel/ca/renarde.fr.html. | |
| 26 | +# Réglage env : LOUKA_BOOKING_DETAIL_LIMIT (fetchs détail par sync, défaut | |
| 27 | +# 150 ; cache permanent dans louka_ct.db, le parc se complète au fil des syncs). | |
| 20 | 28 | # ----------------------------------------------------------------------------- |
| 21 | 29 | from __future__ import annotations |
| 22 | 30 | |
| 23 | 31 | import datetime |
| 24 | 32 | import html as _html |
| 25 | 33 | import json |
| 34 | +import os | |
| 26 | 35 | import re |
| 27 | 36 | import sys |
| 28 | 37 | from urllib.parse import quote |
@@ -30,6 +39,10 @@ from urllib.parse import quote | ||
| 30 | 39 | from ..schema import StListing |
| 31 | 40 | from .base import StConnector |
| 32 | 41 | |
| 42 | + | |
| 43 | +class _DetailSkip(Exception): | |
| 44 | + """Fiche détail sautée (budget épuisé / page bloquée) — pas de cache.""" | |
| 45 | + | |
| 33 | 46 | # (texte de recherche Booking, région touristique QC) |
| 34 | 47 | DESTINATIONS = [ |
| 35 | 48 | ("Mont-Tremblant", "Laurentides"), |
@@ -69,6 +82,8 @@ TYPE_MAP = { | ||
| 69 | 82 | _CAPLA_RE = re.compile( |
| 70 | 83 | r'<script[^>]*data-capla-store-data="apollo"[^>]*>(.*?)</script>', re.S) |
| 71 | 84 | _IMG_BASE = "https://cf.bstatic.com" |
| 85 | +_DESC_RE = re.compile( | |
| 86 | + r'data-testid="property-description"[^>]*>(.*?)</(?:p|div)>', re.S) | |
| 72 | 87 | |
| 73 | 88 | |
| 74 | 89 | class Booking(StConnector): |
@@ -179,6 +194,112 @@ class Booking(StConnector): | ||
| 179 | 194 | lng=loc.get("longitude"), |
| 180 | 195 | ) |
| 181 | 196 | |
| 197 | + # -- page détail ------------------------------------------------------------ | |
| 198 | + @staticmethod | |
| 199 | + def _parse_detail(html: str) -> dict: | |
| 200 | + """Payload {description, amenities, images} d'une page détail Booking | |
| 201 | + ({} si page bloquée/invalide).""" | |
| 202 | + out: dict = {} | |
| 203 | + | |
| 204 | + # description complète (SSR) — HTML → texte | |
| 205 | + m = _DESC_RE.search(html or "") | |
| 206 | + if m: | |
| 207 | + txt = re.sub(r"<br\s*/?>|</p>", "\n", m.group(1)) | |
| 208 | + txt = _html.unescape(re.sub(r"<[^>]+>", " ", txt)) | |
| 209 | + lines = [re.sub(r"\s+", " ", ln).strip() for ln in txt.split("\n")] | |
| 210 | + desc = "\n".join(ln for ln in lines if ln).strip() | |
| 211 | + if desc: | |
| 212 | + out["description"] = desc[:6000] | |
| 213 | + | |
| 214 | + # store Apollo de la page détail : commodités localisées + galerie | |
| 215 | + mm = _CAPLA_RE.search(html or "") | |
| 216 | + if mm: | |
| 217 | + try: | |
| 218 | + store = json.loads(mm.group(1)) | |
| 219 | + except ValueError: | |
| 220 | + try: | |
| 221 | + store = json.loads(_html.unescape(mm.group(1))) | |
| 222 | + except ValueError: | |
| 223 | + store = {} | |
| 224 | + amenities: list[str] = [] | |
| 225 | + for key, val in store.items(): | |
| 226 | + if not isinstance(val, dict): | |
| 227 | + continue | |
| 228 | + name = "" | |
| 229 | + if key.startswith("Instance:"): # équipements du lieu | |
| 230 | + name = (val.get("title") or "").strip() | |
| 231 | + elif key.startswith("SimpleFacility:"): # équipements des unités | |
| 232 | + name = (val.get("name") or "").strip() | |
| 233 | + if name and name not in amenities: | |
| 234 | + amenities.append(name) | |
| 235 | + if amenities: | |
| 236 | + out["amenities"] = amenities[:80] | |
| 237 | + | |
| 238 | + images: list[str] = [] | |
| 239 | + for key, val in store.items(): | |
| 240 | + if not (key.startswith("AccommodationPhoto:") | |
| 241 | + and isinstance(val, dict)): | |
| 242 | + continue | |
| 243 | + for k2, v2 in val.items(): | |
| 244 | + if k2.startswith("resource(") and isinstance(v2, dict) \ | |
| 245 | + and v2.get("relativeUrl"): | |
| 246 | + rel = re.sub(r"/(?:square|max)\w+/", "/max1024x768/", | |
| 247 | + v2["relativeUrl"], count=1) | |
| 248 | + u = _IMG_BASE + rel | |
| 249 | + if u not in images: | |
| 250 | + images.append(u) | |
| 251 | + break | |
| 252 | + if len(images) >= 15: | |
| 253 | + break | |
| 254 | + if images: | |
| 255 | + out["images"] = images | |
| 256 | + return out | |
| 257 | + | |
| 258 | + def _enrich_details(self, listings: list[StListing]) -> None: | |
| 259 | + """Visite les fiches détail via le cache self.detail() sous budget : | |
| 260 | + les hits de cache sont gratuits, seuls les fetchs réseau comptent.""" | |
| 261 | + limit = max(0, int(os.environ.get("LOUKA_BOOKING_DETAIL_LIMIT", "150") | |
| 262 | + or 150)) | |
| 263 | + used = enriched = streak = 0 | |
| 264 | + for lst in listings: | |
| 265 | + def fetch_fn(url=lst.url): | |
| 266 | + nonlocal used, streak | |
| 267 | + if used >= limit or streak >= 5: # tempête anti-bot : on coupe | |
| 268 | + raise _DetailSkip | |
| 269 | + used += 1 | |
| 270 | + res = self.scrapfly(url, render_js=False, asp=True) | |
| 271 | + if res.get("status_code") in (404, 410): | |
| 272 | + return {} # fiche retirée : cacher vide | |
| 273 | + payload = self._parse_detail(res.get("content") or "") | |
| 274 | + if not payload: | |
| 275 | + streak += 1 | |
| 276 | + raise _DetailSkip # blocage/vide : pas de cache | |
| 277 | + streak = 0 | |
| 278 | + return payload | |
| 279 | + | |
| 280 | + try: | |
| 281 | + d = self.detail(lst.external_id, "v1", fetch_fn) | |
| 282 | + except _DetailSkip: | |
| 283 | + continue | |
| 284 | + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run | |
| 285 | + continue | |
| 286 | + if not d: | |
| 287 | + continue | |
| 288 | + if d.get("description") and len(d["description"]) > \ | |
| 289 | + len(lst.description or ""): | |
| 290 | + lst.description = d["description"] | |
| 291 | + if d.get("amenities"): | |
| 292 | + seen = {a.lower() for a in lst.amenities} | |
| 293 | + for a in d["amenities"]: | |
| 294 | + if a.lower() not in seen: | |
| 295 | + lst.amenities.append(a) | |
| 296 | + seen.add(a.lower()) | |
| 297 | + if d.get("images") and len(d["images"]) > len(lst.images): | |
| 298 | + lst.images = list(d["images"]) | |
| 299 | + enriched += 1 | |
| 300 | + print(f"[booking] détail : {enriched} annonces enrichies" | |
| 301 | + f" ({used}/{limit} fetchs réseau)", file=sys.stderr) | |
| 302 | + | |
| 182 | 303 | # -- contrat --------------------------------------------------------------- |
| 183 | 304 | def fetch(self) -> list[StListing]: |
| 184 | 305 | today = datetime.date.today() |
@@ -207,4 +328,6 @@ class Booking(StConnector): | ||
| 207 | 328 | continue |
| 208 | 329 | if lst and lst.external_id not in listings: |
| 209 | 330 | listings[lst.external_id] = lst |
| 210 | − return list(listings.values()) | |
| 331 | + out = list(listings.values()) | |
| 332 | + self._enrich_details(out) | |
| 333 | + return out | |
added
louka/shortterm/connectors/boraboreal.py
+174 −0
@@ -0,0 +1,174 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/boraboreal.py : Bora Boréal (boraboreal.com) — chalets FLOTTANTS | |
| 4 | +# (minibora, boravilla) à Bury (Cantons-de-l'Est) et à Québec, plus un | |
| 5 | +# chalet en bois rond ; ~13 unités réservables sur Lodgify | |
| 6 | +# (reserver-boraboreal.lodgify.com, protégé Cloudflare → Scrapfly). | |
| 7 | +# | |
| 8 | +# Méthode : | |
| 9 | +# 1. SLUGS : le sitemap Lodgify est vide → la page « louer-maison-flottante » | |
| 10 | +# (rendue via Scrapfly, la grille est en JS) liste les 12 chalets | |
| 11 | +# flottants ; les pages vitrines boraboreal.com (récupérées en direct) | |
| 12 | +# ajoutent les unités hors grille (bora-bois-rond). | |
| 13 | +# 2. DÉTAIL : chaque page unité Lodgify (Scrapfly sans render_js, cache | |
| 14 | +# détail « v1 ») embarque un JSON-LD VacationRental complet : | |
| 15 | +# identifier (external_id), priceRange « from 229 CAD/night », | |
| 16 | +# occupancy/chambres, amenityFeature, adresse, geo lat/lng, ~19 photos | |
| 17 | +# icdbcdn ; le CITQ est extrait de la description (au-delà de la fenêtre | |
| 18 | +# de 2000 caractères de finalize()). Animaux : frais au séjour | |
| 19 | +# → pets = « conditions » si l'amenité l'indique. | |
| 20 | +# ----------------------------------------------------------------------------- | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import html as _html | |
| 24 | +import json | |
| 25 | +import re | |
| 26 | +import sys | |
| 27 | + | |
| 28 | +from ..schema import StListing | |
| 29 | +from .airbnb import _region_from_latlng | |
| 30 | +from .base import StConnector | |
| 31 | + | |
| 32 | +BOOKING = "https://reserver-boraboreal.lodgify.com" | |
| 33 | +GRID = BOOKING + "/fr/louer-maison-flottante" | |
| 34 | + | |
| 35 | +# pages vitrines boraboreal.com (accessibles en direct) qui pointent vers des | |
| 36 | +# unités Lodgify absentes de la grille | |
| 37 | +SHOWCASE = [ | |
| 38 | + "https://boraboreal.com/chalet-flottant-quebec", | |
| 39 | + "https://boraboreal.com/mini-chalet-a-louer-estrie", | |
| 40 | + "https://boraboreal.com/chalet-6-personnes-estrie", | |
| 41 | +] | |
| 42 | + | |
| 43 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 44 | + | |
| 45 | +_LD_RE = re.compile(r'(?s)<script[^>]*type="application/ld\+json"[^>]*>' | |
| 46 | + r"(.*?)</script>") | |
| 47 | + | |
| 48 | + | |
| 49 | +def _strip_html(txt: str) -> str: | |
| 50 | + txt = re.sub(r"</p>|<br\s*/?>|</li>", "\n", txt or "") | |
| 51 | + txt = _html.unescape(_TAG_RE.sub(" ", txt)) | |
| 52 | + txt = re.sub(r"[ \t]+", " ", txt) | |
| 53 | + return re.sub(r"\n\s+", "\n", txt).strip() | |
| 54 | + | |
| 55 | + | |
| 56 | +class BoraBoreal(StConnector): | |
| 57 | + source_id = "boraboreal" | |
| 58 | + | |
| 59 | + # -- découverte des slugs -------------------------------------------------- | |
| 60 | + def _slugs(self) -> list[str]: | |
| 61 | + slugs: list[str] = [] | |
| 62 | + | |
| 63 | + def add(s: str): | |
| 64 | + s = s.strip("/") | |
| 65 | + if s and s != "louer-maison-flottante" and s not in slugs: | |
| 66 | + slugs.append(s) | |
| 67 | + | |
| 68 | + try: | |
| 69 | + grid = self.get_scrapfly(GRID, render_js=True, | |
| 70 | + rendering_wait=3000) | |
| 71 | + for s in re.findall(r'href="(?:%s)?/fr/([\w~-]+)"' | |
| 72 | + % re.escape(BOOKING), grid): | |
| 73 | + add(s) | |
| 74 | + except Exception as exc: # noqa: BLE001 | |
| 75 | + print(f"[boraboreal] grille : {exc}", file=sys.stderr) | |
| 76 | + | |
| 77 | + for page in SHOWCASE: | |
| 78 | + try: | |
| 79 | + h = self.get(page).text | |
| 80 | + except Exception: # noqa: BLE001 | |
| 81 | + continue | |
| 82 | + for s in re.findall( | |
| 83 | + r"reserver-boraboreal\.lodgify\.com/(?:fr/)?([\w~-]+)", h): | |
| 84 | + add(s) | |
| 85 | + return slugs | |
| 86 | + | |
| 87 | + # -- page unité Lodgify ---------------------------------------------------- | |
| 88 | + def _detail(self, slug: str) -> dict: | |
| 89 | + h = self.get_scrapfly(f"{BOOKING}/fr/{slug}", render_js=False) | |
| 90 | + for block in _LD_RE.findall(h): | |
| 91 | + try: | |
| 92 | + ld = json.loads(block) | |
| 93 | + except ValueError: | |
| 94 | + continue | |
| 95 | + if ld.get("@type") == "VacationRental": | |
| 96 | + return {"ld": ld} | |
| 97 | + return {} | |
| 98 | + | |
| 99 | + # -- contrat --------------------------------------------------------------- | |
| 100 | + def fetch(self) -> list[StListing]: | |
| 101 | + listings: list[StListing] = [] | |
| 102 | + vus: set[str] = set() | |
| 103 | + for slug in self._slugs(): | |
| 104 | + det = self.detail(slug, "v1", lambda s=slug: self._detail(s)) | |
| 105 | + ld = det.get("ld") or {} | |
| 106 | + if not ld: | |
| 107 | + continue | |
| 108 | + ext_id = str(ld.get("identifier") or slug) | |
| 109 | + if ext_id in vus: | |
| 110 | + continue | |
| 111 | + vus.add(ext_id) | |
| 112 | + | |
| 113 | + place = ld.get("containsPlace") or {} | |
| 114 | + occupancy = (place.get("occupancy") or {}).get("value") | |
| 115 | + addr = ld.get("address") or {} | |
| 116 | + geo = ld.get("geo") or {} | |
| 117 | + | |
| 118 | + # « from 229 CAD/night » → price_night | |
| 119 | + price = None | |
| 120 | + m = re.search(r"from\s+([\d.]+)\s*CAD", | |
| 121 | + str(ld.get("priceRange") or "")) | |
| 122 | + if m: | |
| 123 | + v = float(m.group(1)) | |
| 124 | + if 20 <= v <= 20000: | |
| 125 | + price = v | |
| 126 | + | |
| 127 | + description = _strip_html(str(ld.get("description") or "")) | |
| 128 | + m = re.search(r"CITQ\D{0,25}(\d{6})", description) | |
| 129 | + citq = m.group(1) if m else "" | |
| 130 | + | |
| 131 | + amen = [a.get("name") for a in ld.get("amenityFeature") or [] | |
| 132 | + if isinstance(a, dict) and a.get("name") | |
| 133 | + and a.get("value") is not False] | |
| 134 | + pets = "conditions" if any("animaux" in a.lower() | |
| 135 | + or "pet" in a.lower() | |
| 136 | + for a in amen) else None | |
| 137 | + | |
| 138 | + imgs = ld.get("image") or [] | |
| 139 | + if isinstance(imgs, str): | |
| 140 | + imgs = [imgs] | |
| 141 | + | |
| 142 | + city = (addr.get("addressLocality") or "").strip() | |
| 143 | + region = (_region_from_latlng(geo.get("latitude"), | |
| 144 | + geo.get("longitude")) | |
| 145 | + or ("Québec" if slug.endswith("---quebec") | |
| 146 | + else "Cantons-de-l'Est")) | |
| 147 | + | |
| 148 | + listings.append(StListing( | |
| 149 | + source=self.source_id, | |
| 150 | + external_id=ext_id, | |
| 151 | + url=f"{BOOKING}/fr/{slug}", | |
| 152 | + title=_html.unescape(str(ld.get("name") or slug)).strip(), | |
| 153 | + property_type="Chalet", | |
| 154 | + address=(addr.get("streetAddress") or "").strip(), | |
| 155 | + city=city, | |
| 156 | + region=region, | |
| 157 | + price_night=price, | |
| 158 | + price_label=(f"à partir de {price:g} $ / nuit" | |
| 159 | + if price else ""), | |
| 160 | + capacity=float(occupancy) if occupancy else None, | |
| 161 | + bedrooms=(float(place["numberOfBedrooms"]) | |
| 162 | + if place.get("numberOfBedrooms") else None), | |
| 163 | + pets=pets, | |
| 164 | + citq=citq, | |
| 165 | + description=description[:5000], | |
| 166 | + amenities=amen, | |
| 167 | + details={"floating": "bois-rond" not in slug, | |
| 168 | + "postal_code": addr.get("postalCode") or ""}, | |
| 169 | + images=[u for u in imgs if isinstance(u, str) | |
| 170 | + and u.startswith("https://")][:20], | |
| 171 | + lat=geo.get("latitude"), | |
| 172 | + lng=geo.get("longitude"), | |
| 173 | + )) | |
| 174 | + return listings | |
added
louka/shortterm/connectors/campingquebec.py
+274 −0
@@ -0,0 +1,274 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/campingquebec.py : Camping Québec (campingquebec.com) — | |
| 4 | +# l'association des ~830 terrains de camping du Québec. On ne retient QUE | |
| 5 | +# les campings offrant du PRÊT-À-CAMPER / hébergement locatif (tentes | |
| 6 | +# aménagées, chalets, yourtes, roulottes…) : un camping « emplacements | |
| 7 | +# seulement » n'est pas un hébergement court terme pour Lou-Ka. | |
| 8 | +# | |
| 9 | +# Méthode (WordPress, aucun anti-bot) : | |
| 10 | +# 1. LISTE : l'endpoint AJAX de « Trouver un camping » est ouvert : | |
| 11 | +# GET /fr/wp-json/search/result?lang=fr&view=list | |
| 12 | +# &ready_to_camps[]=tous-types-de-pret-a-camper-disponible&paged=N | |
| 13 | +# → fragments HTML de 24 cartes/page (~600 campings filtrés prêt-à-camper). | |
| 14 | +# Carte : URL /fr/campings/<région>/<slug> (= external_id), nom, région. | |
| 15 | +# 2. FICHE (cache self.detail, clé mensuelle pour suivre les tarifs) : | |
| 16 | +# description, adresse + ville (bloc Informations), coordonnées (lien | |
| 17 | +# google.ca/maps?q=lat,lng), no d'enregistrement CITQ, tarifs (ligne | |
| 18 | +# « Nuitée, Prêt-à-camper » min-max → price_night), unités prêt-à-camper | |
| 19 | +# (« Prêt-à-camper disponibles : Tentes : 2 »), services (amenities), | |
| 20 | +# nb d'emplacements, dates de saison, photos. Garde-fou : la fiche doit | |
| 21 | +# confirmer le prêt-à-camper (unités ou tarif), sinon elle est écartée. | |
| 22 | +# | |
| 23 | +# Réglage env : LOUKA_CAMPINGQUEBEC_LIMIT (nb max de fiches, 0 = tout). | |
| 24 | +# ----------------------------------------------------------------------------- | |
| 25 | +from __future__ import annotations | |
| 26 | + | |
| 27 | +import os | |
| 28 | +import re | |
| 29 | +import sys | |
| 30 | +import time | |
| 31 | + | |
| 32 | +from ..schema import StListing, normalize_region | |
| 33 | +from .base import StConnector | |
| 34 | + | |
| 35 | +SITE = "https://www.campingquebec.com" | |
| 36 | +API = f"{SITE}/fr/wp-json/search/result" | |
| 37 | +PREFIX_FICHE = f"{SITE}/fr/campings/" | |
| 38 | +PAGE_MAX = 60 # garde-fou pagination | |
| 39 | + | |
| 40 | +_MAPS_RE = re.compile(r"google\.ca/maps\?q=(-?\d+\.\d+),(-?\d+\.\d+)") | |
| 41 | +_CITQ_RE = re.compile(r"No d[’']enregistrement\s*(\d{5,7})") | |
| 42 | +_PAGE_RE = re.compile(r'aria-label="Page (\d+)"') | |
| 43 | +_MONTANT_RE = re.compile(r"([\d\s ]+(?:[.,]\d{2})?)\s*\$") | |
| 44 | + | |
| 45 | +# Libellé d'unité prêt-à-camper → type canonique Lou-Ka (si type unique) ; | |
| 46 | +# le préfixe « location de » est retiré avant consultation. | |
| 47 | +_TYPE_UNITE = { | |
| 48 | + "tente": "Prêt-à-camper", "tentes": "Prêt-à-camper", | |
| 49 | + "chalet": "Chalet", "chalets": "Chalet", | |
| 50 | + "yourte": "Yourte", "yourtes": "Yourte", | |
| 51 | + "dôme": "Dôme", "dômes": "Dôme", "bulle ou dôme": "Dôme", | |
| 52 | + "refuge": "Refuge", "refuges": "Refuge", | |
| 53 | + "tipi": "Prêt-à-camper", "tipis": "Prêt-à-camper", | |
| 54 | + "cabine": "Prêt-à-camper", "cabines": "Prêt-à-camper", | |
| 55 | + "caravane": "Prêt-à-camper", "caravanes": "Prêt-à-camper", | |
| 56 | +} | |
| 57 | + | |
| 58 | + | |
| 59 | +def _montant(txt: str) -> float | None: | |
| 60 | + m = _MONTANT_RE.search(txt or "") | |
| 61 | + if not m: | |
| 62 | + return None | |
| 63 | + try: | |
| 64 | + return float(re.sub(r"[\s ]", "", m.group(1)).replace(",", ".")) | |
| 65 | + except ValueError: | |
| 66 | + return None | |
| 67 | + | |
| 68 | + | |
| 69 | +class CampingQuebec(StConnector): | |
| 70 | + source_id = "campingquebec" | |
| 71 | + request_delay = 0.8 | |
| 72 | + | |
| 73 | + # -- liste (fragments HTML paginés) ---------------------------------------- | |
| 74 | + def _liste(self, limit: int = 0) -> list[dict]: | |
| 75 | + from bs4 import BeautifulSoup | |
| 76 | + items, vus = [], set() | |
| 77 | + page, total_pages = 1, 1 | |
| 78 | + while page <= min(total_pages, PAGE_MAX): | |
| 79 | + if limit and len(items) >= limit: | |
| 80 | + break | |
| 81 | + html = self.get(API, params={ | |
| 82 | + "lang": "fr", "view": "list", | |
| 83 | + "ready_to_camps[]": "tous-types-de-pret-a-camper-disponible", | |
| 84 | + "paged": page, | |
| 85 | + }).text | |
| 86 | + pages = [int(p) for p in _PAGE_RE.findall(html)] | |
| 87 | + if pages: | |
| 88 | + total_pages = max(pages) | |
| 89 | + soup = BeautifulSoup(html, "html.parser") | |
| 90 | + nouveaux = 0 | |
| 91 | + for a in soup.select(f'a.c-card[href^="{PREFIX_FICHE}"]'): | |
| 92 | + path = a["href"][len(PREFIX_FICHE):].strip("/") | |
| 93 | + if path.count("/") != 1 or path in vus: | |
| 94 | + continue | |
| 95 | + vus.add(path) | |
| 96 | + nouveaux += 1 | |
| 97 | + h4 = a.find("h4") | |
| 98 | + span = a.select_one("span.u-text-transform-none") | |
| 99 | + items.append({ | |
| 100 | + "id": path, # <région>/<slug> | |
| 101 | + "nom": h4.get_text(" ", strip=True) if h4 else "", | |
| 102 | + "region": span.get_text(" ", strip=True) if span else "", | |
| 103 | + }) | |
| 104 | + if not nouveaux: # page vide → fin | |
| 105 | + break | |
| 106 | + page += 1 | |
| 107 | + return items | |
| 108 | + | |
| 109 | + # -- fiche camping ---------------------------------------------------------- | |
| 110 | + def _fetch_fiche(self, path: str) -> dict: | |
| 111 | + from bs4 import BeautifulSoup | |
| 112 | + html = self.get(PREFIX_FICHE + path).text | |
| 113 | + soup = BeautifulSoup(html, "html.parser") | |
| 114 | + d: dict = {} | |
| 115 | + | |
| 116 | + m = _MAPS_RE.search(html) | |
| 117 | + if m: | |
| 118 | + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2)) | |
| 119 | + m = _CITQ_RE.search(html) | |
| 120 | + if m: | |
| 121 | + d["citq"] = m.group(1) | |
| 122 | + | |
| 123 | + # description : bloc typographique sous l'en-tête « Description » | |
| 124 | + for div in soup.find_all("div"): | |
| 125 | + if div.get_text(strip=True) == "Description": | |
| 126 | + typo = div.find_next_sibling("div") | |
| 127 | + if typo is not None: | |
| 128 | + d["description"] = typo.get_text("\n", strip=True)[:2500] | |
| 129 | + break | |
| 130 | + | |
| 131 | + # adresse + ville : paragraphe précédant « Voir sur la carte » | |
| 132 | + carte = soup.find("a", string=re.compile("Voir sur la carte")) | |
| 133 | + if carte is None: | |
| 134 | + for a in soup.find_all("a"): | |
| 135 | + if "Voir sur la carte" in a.get_text(): | |
| 136 | + carte = a | |
| 137 | + break | |
| 138 | + if carte is not None: | |
| 139 | + p = carte.find_previous("p") | |
| 140 | + if p is not None: | |
| 141 | + lignes = [x.strip() for x in p.get_text("\n").split("\n") | |
| 142 | + if x.strip()] | |
| 143 | + if lignes: | |
| 144 | + d["adresse"] = ", ".join(lignes) | |
| 145 | + # « Saint-Sulpice J5W 3V5 » → ville sans le code postal | |
| 146 | + d["ville"] = re.sub( | |
| 147 | + r"\s*[A-Z]\d[A-Z]\s*\d[A-Z]\d\s*$", "", | |
| 148 | + lignes[-1]).strip(" ,") | |
| 149 | + | |
| 150 | + # sections h4 → listes (unités PAC, emplacements…) | |
| 151 | + sections: dict[str, list[str]] = {} | |
| 152 | + for h4 in soup.find_all("h4"): | |
| 153 | + titre = h4.get_text(" ", strip=True) | |
| 154 | + parent = h4.find_parent("div") | |
| 155 | + bloc = parent.find_next_sibling("div") if parent else None | |
| 156 | + if bloc is not None: | |
| 157 | + lis = [li.get_text(" ", strip=True) | |
| 158 | + for li in bloc.find_all("li")] | |
| 159 | + if lis: | |
| 160 | + sections[titre] = lis | |
| 161 | + | |
| 162 | + pac: dict[str, int] = {} | |
| 163 | + for titre, lis in sections.items(): | |
| 164 | + if titre.lower().startswith("prêt-à-camper"): | |
| 165 | + for li in lis: | |
| 166 | + nom, _, nb = li.partition(":") | |
| 167 | + try: | |
| 168 | + pac[nom.strip()] = int(nb.strip()) | |
| 169 | + except ValueError: | |
| 170 | + pac[nom.strip()] = 0 | |
| 171 | + d["pac"] = pac | |
| 172 | + for titre, lis in sections.items(): | |
| 173 | + if titre.lower().startswith("types d'emplacements"): | |
| 174 | + d["emplacements"] = lis[:12] | |
| 175 | + | |
| 176 | + # tarifs : lignes de la table « Durée / Min. / Max. » | |
| 177 | + for tr in soup.select("table.c-table tr"): | |
| 178 | + tds = [td.get_text(" ", strip=True) for td in tr.find_all("td")] | |
| 179 | + if len(tds) >= 2 and "prêt-à-camper" in tds[0].lower(): | |
| 180 | + d["tarif_pac_min"] = _montant(tds[1]) | |
| 181 | + d["tarif_pac_max"] = _montant(tds[2]) if len(tds) > 2 else None | |
| 182 | + elif len(tds) >= 2 and tds[0].lower() == "nuitée": | |
| 183 | + d["tarif_nuit_min"] = _montant(tds[1]) | |
| 184 | + | |
| 185 | + # services offerts → amenities (panneau d'accordéon « services ») | |
| 186 | + panneau = soup.select_one( | |
| 187 | + 'div.c-accordion__target[data-toggler-target*="services"]') | |
| 188 | + if panneau is not None: | |
| 189 | + d["services"] = [li.get_text(" ", strip=True) | |
| 190 | + for li in panneau.find_all("li")][:40] | |
| 191 | + | |
| 192 | + # saison | |
| 193 | + texte = soup.get_text(" ", strip=True) | |
| 194 | + m = re.search(r"Date d['’]ouverture\s*:\s*([\d]{1,2} \S+ \d{4})", texte) | |
| 195 | + if m: | |
| 196 | + d["ouverture"] = m.group(1) | |
| 197 | + m = re.search(r"Date de fermeture\s*:\s*([\d]{1,2} \S+ \d{4})", texte) | |
| 198 | + if m: | |
| 199 | + d["fermeture"] = m.group(1) | |
| 200 | + | |
| 201 | + # photos (galerie WordPress, en excluant logos et gabarits) | |
| 202 | + imgs: list[str] = [] | |
| 203 | + for img in soup.find_all("img"): | |
| 204 | + u = img.get("data-lazy-src") or img.get("src") or "" | |
| 205 | + if (u.startswith(f"{SITE}/wp-content/uploads/20") | |
| 206 | + and "logo" not in u.lower() and u not in imgs): | |
| 207 | + imgs.append(u) | |
| 208 | + d["images"] = imgs[:12] | |
| 209 | + return d | |
| 210 | + | |
| 211 | + # -- contrat ---------------------------------------------------------------- | |
| 212 | + def fetch(self) -> list[StListing]: | |
| 213 | + limit = int(os.environ.get("LOUKA_CAMPINGQUEBEC_LIMIT", "0") or 0) | |
| 214 | + month = time.strftime("%Y-%m") # re-visite mensuelle (tarifs) | |
| 215 | + listings: list[StListing] = [] | |
| 216 | + # marge : certaines fiches de la liste seront écartées au garde-fou | |
| 217 | + for it in self._liste(limit * 3 if limit else 0): | |
| 218 | + path = it["id"] | |
| 219 | + try: | |
| 220 | + d = self.detail(path, month, | |
| 221 | + lambda p=path: self._fetch_fiche(p)) | |
| 222 | + except Exception as exc: # noqa: BLE001 | |
| 223 | + print(f"[campingquebec] fiche {path} : {exc}", file=sys.stderr) | |
| 224 | + continue | |
| 225 | + | |
| 226 | + pac = d.get("pac") or {} | |
| 227 | + prix_pac = d.get("tarif_pac_min") | |
| 228 | + if not pac and not prix_pac: # aucun hébergement locatif confirmé | |
| 229 | + continue | |
| 230 | + | |
| 231 | + price_label = "" | |
| 232 | + if prix_pac: | |
| 233 | + pmax = d.get("tarif_pac_max") | |
| 234 | + price_label = (f"prêt-à-camper {prix_pac:.2f} $" | |
| 235 | + + (f" à {pmax:.2f} $" if pmax else "") | |
| 236 | + + " / nuit") | |
| 237 | + | |
| 238 | + # type : celui de l'unique famille d'unités, sinon Prêt-à-camper | |
| 239 | + ptype = "Prêt-à-camper" | |
| 240 | + if len(pac) == 1: | |
| 241 | + libelle = re.sub(r"^location (de |d')", "", | |
| 242 | + next(iter(pac)).lower()).strip() | |
| 243 | + ptype = _TYPE_UNITE.get(libelle, ptype) | |
| 244 | + | |
| 245 | + details = {k: v for k, v in { | |
| 246 | + "unites_pret_a_camper": pac or None, | |
| 247 | + "emplacements": d.get("emplacements"), | |
| 248 | + "tarif_emplacement_min": d.get("tarif_nuit_min"), | |
| 249 | + "ouverture": d.get("ouverture", ""), | |
| 250 | + "fermeture": d.get("fermeture", ""), | |
| 251 | + }.items() if v} | |
| 252 | + | |
| 253 | + listings.append(StListing( | |
| 254 | + source=self.source_id, | |
| 255 | + external_id=path, | |
| 256 | + url=PREFIX_FICHE + path, | |
| 257 | + title=it.get("nom") or path.rsplit("/", 1)[-1], | |
| 258 | + property_type=ptype, | |
| 259 | + address=d.get("adresse", ""), | |
| 260 | + city=d.get("ville", ""), | |
| 261 | + region=normalize_region(it.get("region", "")), | |
| 262 | + price_night=prix_pac, | |
| 263 | + price_label=price_label, | |
| 264 | + citq=d.get("citq", ""), | |
| 265 | + description=d.get("description", ""), | |
| 266 | + amenities=d.get("services") or [], | |
| 267 | + details=details, | |
| 268 | + images=d.get("images") or [], | |
| 269 | + lat=d.get("lat"), | |
| 270 | + lng=d.get("lng"), | |
| 271 | + )) | |
| 272 | + if limit and len(listings) >= limit: | |
| 273 | + break | |
| 274 | + return listings | |
added
louka/shortterm/connectors/captremblant.py
+157 −0
@@ -0,0 +1,157 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/captremblant.py : Cap Tremblant Mountain Resort | |
| 4 | +# (captremblant.com) — complexe de résidences de tourisme à Mont-Tremblant. | |
| 5 | +# 7 catégories de résidences (1 à 5 chambres), chacune regroupant plusieurs | |
| 6 | +# unités identiques (fiches « catégorie », pas unité par unité). | |
| 7 | +# | |
| 8 | +# Méthode : sitemap.xml → 7 URLs /residences/<slug> (version fr ; les /en/ | |
| 9 | +# sont ignorées). Pas de lastmod → cache détail « v1 ». Chaque page embarque | |
| 10 | +# un JSON-LD @type ["HotelRoom","Product"] : nom, description, image | |
| 11 | +# (galaxy.tf) et offers.price = tarif À LA NUIT en CAD (ex. 251,10 $) → | |
| 12 | +# price_night « à partir de ». La liste des caractéristiques vit dans le | |
| 13 | +# <ul> du bloc <div class="m-content-object--content"> → amenities ; | |
| 14 | +# capacité extraite de « jusqu'à six personnes » (nombres en toutes | |
| 15 | +# lettres), chambres déduites du slug/titre. La galerie est chargée en JS : | |
| 16 | +# seule l'image JSON-LD est disponible statiquement. | |
| 17 | +# ----------------------------------------------------------------------------- | |
| 18 | +from __future__ import annotations | |
| 19 | + | |
| 20 | +import html as _html | |
| 21 | +import json | |
| 22 | +import re | |
| 23 | + | |
| 24 | +from ..schema import StListing | |
| 25 | +from .base import StConnector | |
| 26 | + | |
| 27 | +SITE = "https://www.captremblant.com" | |
| 28 | +SITEMAP = SITE + "/sitemap.xml" | |
| 29 | + | |
| 30 | +_WORDS = {"un": 1, "une": 1, "deux": 2, "trois": 3, "quatre": 4, "cinq": 5, | |
| 31 | + "six": 6, "sept": 7, "huit": 8, "neuf": 9, "dix": 10, "onze": 11, | |
| 32 | + "douze": 12} | |
| 33 | + | |
| 34 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 35 | + | |
| 36 | + | |
| 37 | +def _text(fragment: str) -> str: | |
| 38 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 39 | + | |
| 40 | + | |
| 41 | +def _count(txt: str) -> int | None: | |
| 42 | + """« six » ou « 6 » → 6.""" | |
| 43 | + t = txt.strip().lower() | |
| 44 | + if t.isdigit(): | |
| 45 | + return int(t) | |
| 46 | + return _WORDS.get(t) | |
| 47 | + | |
| 48 | + | |
| 49 | +class CapTremblant(StConnector): | |
| 50 | + source_id = "captremblant" | |
| 51 | + request_delay = 0.8 | |
| 52 | + | |
| 53 | + # -- page détail ---------------------------------------------------------- | |
| 54 | + def _detail(self, url: str) -> dict: | |
| 55 | + h = self.get(url).text | |
| 56 | + d: dict = {} | |
| 57 | + | |
| 58 | + for block in re.findall(r'<script type="application/ld\+json"[^>]*>' | |
| 59 | + r"(.*?)</script>", h, re.S): | |
| 60 | + try: | |
| 61 | + ld = json.loads(block) | |
| 62 | + except ValueError: | |
| 63 | + continue | |
| 64 | + types = ld.get("@type") | |
| 65 | + types = types if isinstance(types, list) else [types] | |
| 66 | + if "HotelRoom" in types or "Product" in types: | |
| 67 | + d["ld"] = ld | |
| 68 | + break | |
| 69 | + | |
| 70 | + # caractéristiques : <ul> du bloc m-content-object--content | |
| 71 | + m = re.search(r'(?s)<div class="m-content-object--content[^"]*"[^>]*>' | |
| 72 | + r"(.*?)</div>", h) | |
| 73 | + if m: | |
| 74 | + amen = [] | |
| 75 | + for li in re.findall(r"(?s)<li[^>]*>(.*?)</li>", m.group(1)): | |
| 76 | + t = _text(li) | |
| 77 | + if t and t not in amen: | |
| 78 | + amen.append(t) | |
| 79 | + if amen: | |
| 80 | + d["amenities"] = amen | |
| 81 | + return d | |
| 82 | + | |
| 83 | + # -- contrat -------------------------------------------------------------- | |
| 84 | + def fetch(self) -> list[StListing]: | |
| 85 | + xml = self.get(SITEMAP).text | |
| 86 | + urls = [u for u in re.findall(r"<loc>([^<]+)</loc>", xml) | |
| 87 | + if "/residences/" in u and "/en/" not in u] | |
| 88 | + | |
| 89 | + listings: list[StListing] = [] | |
| 90 | + vus: set[str] = set() | |
| 91 | + for url in urls: | |
| 92 | + slug = url.rstrip("/").split("/")[-1] | |
| 93 | + if slug in vus: | |
| 94 | + continue | |
| 95 | + vus.add(slug) | |
| 96 | + | |
| 97 | + det = self.detail(slug, "v1", lambda u=url: self._detail(u)) | |
| 98 | + ld = det.get("ld") or {} | |
| 99 | + if not ld: | |
| 100 | + continue | |
| 101 | + | |
| 102 | + title = _text(str(ld.get("name") or slug)) | |
| 103 | + amen = det.get("amenities") or [] | |
| 104 | + description = _text(str(ld.get("description") or "")) | |
| 105 | + if amen and len(" ".join(amen)) > len(description): | |
| 106 | + description = description or " ".join(amen[:1]) | |
| 107 | + | |
| 108 | + # capacité : « jusqu'à six personnes » / « de dix à douze | |
| 109 | + # personnes » (description ou 1re puce) — on prend le maximum | |
| 110 | + capacity = None | |
| 111 | + hay = f"{description} {' '.join(amen[:2])}" | |
| 112 | + m = (re.search(r"jusqu[’']à\s+(\w+)\s+personnes", hay, re.I) | |
| 113 | + or re.search(r"de\s+\w+\s+à\s+(\w+)\s+personnes", hay, re.I)) | |
| 114 | + if m: | |
| 115 | + capacity = _count(m.group(1)) | |
| 116 | + | |
| 117 | + # chambres : déduites du slug/titre (« deux-chambres », « cinq… ») | |
| 118 | + bedrooms = None | |
| 119 | + m = re.search(r"(\w+)[- ]chambres?", f"{slug} {title}".lower()) | |
| 120 | + if m: | |
| 121 | + bedrooms = _count(m.group(1)) | |
| 122 | + | |
| 123 | + price = None | |
| 124 | + try: | |
| 125 | + price = float((ld.get("offers") or {}).get("price")) | |
| 126 | + except (TypeError, ValueError): | |
| 127 | + pass | |
| 128 | + # garde-fou : la fiche « quatre chambres » publie 29520.00 $ | |
| 129 | + # (bogue du site, sans doute 295,20 $) → tarif rejeté | |
| 130 | + if price is not None and not 50 <= price <= 5000: | |
| 131 | + price = None | |
| 132 | + price_label = (f"à partir de {price:g} $ / nuit" | |
| 133 | + if price else "") | |
| 134 | + | |
| 135 | + img = ld.get("image") or "" | |
| 136 | + images = [img] if isinstance(img, str) \ | |
| 137 | + and img.startswith("https://") else [] | |
| 138 | + | |
| 139 | + listings.append(StListing( | |
| 140 | + source=self.source_id, | |
| 141 | + external_id=slug, | |
| 142 | + url=url, | |
| 143 | + title=title, | |
| 144 | + property_type="Condo", | |
| 145 | + address="240 Rue du Mont-Plaisant, Mont-Tremblant", | |
| 146 | + city="Mont-Tremblant", | |
| 147 | + region="Laurentides", | |
| 148 | + price_night=price, | |
| 149 | + price_label=price_label, | |
| 150 | + capacity=float(capacity) if capacity else None, | |
| 151 | + bedrooms=float(bedrooms) if bedrooms else None, | |
| 152 | + description=description[:3000], | |
| 153 | + amenities=amen, | |
| 154 | + details={"multi_unit": True, "resort": "Cap Tremblant"}, | |
| 155 | + images=images, | |
| 156 | + )) | |
| 157 | + return listings | |
added
louka/shortterm/connectors/chaleto.py
+198 −0
@@ -0,0 +1,198 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/chaleto.py : Chaleto (chaleto.ca) — gestionnaire québécois | |
| 4 | +# (Charlevoix, Capitale-Nationale, Laurentides…), ~440 fiches FR. | |
| 5 | +# | |
| 6 | +# Méthode : WordPress (vitrine) + Guesty (inventaire). Sitemap dédié | |
| 7 | +# /listings-sitemap.xml → URLs /chalets-et-condos-a-louer/<slug>-<guestyId>/ | |
| 8 | +# (doublées en /en/ : on garde le FR). Tout est dans le HTML serveur de la | |
| 9 | +# page détail : h1, « Ville, Région », prix « À partir de N$ / nuit », | |
| 10 | +# pictos (voyageurs/chambres/lits/salles de bain), commodités, description | |
| 11 | +# (contenant le no CITQ) et TOUTES les photos Guesty pleine résolution dans | |
| 12 | +# l'attribut data-images de la lightbox. Le lastmod du sitemap est identique | |
| 13 | +# partout : clé de cache mensuelle (refetch complet 1×/mois, ~440 requêtes). | |
| 14 | +# ----------------------------------------------------------------------------- | |
| 15 | +from __future__ import annotations | |
| 16 | + | |
| 17 | +import html as _html | |
| 18 | +import json | |
| 19 | +import re | |
| 20 | +import time | |
| 21 | + | |
| 22 | +from ...normalize import strip_accents | |
| 23 | +from ..schema import StListing, parse_price_night | |
| 24 | +from .base import StConnector | |
| 25 | + | |
| 26 | +SITEMAP = "https://chaleto.ca/listings-sitemap.xml" | |
| 27 | + | |
| 28 | +# URL FR : /chalets-et-condos-a-louer/<slug>-<id Guesty 24 hex>/ | |
| 29 | +_URL_FR = re.compile( | |
| 30 | + r"https://(?:www\.)?chaleto\.ca/chalets-et-condos-a-louer/" | |
| 31 | + r"[\w-]+-([0-9a-f]{24})/?$") | |
| 32 | + | |
| 33 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 34 | + | |
| 35 | +# le site affiche la région ADMINISTRATIVE (« Capitale-Nationale », | |
| 36 | +# « Gaspésie--Îles-de-la-Madeleine ») → région touristique canonique | |
| 37 | +_REGION_FIX = { | |
| 38 | + "capitale-nationale": "Québec", | |
| 39 | + "estrie": "Cantons-de-l'Est", | |
| 40 | + "gaspesie-iles-de-la-madeleine": "Gaspésie", | |
| 41 | + "saguenay-lac-saint-jean": "Saguenay–Lac-Saint-Jean", | |
| 42 | +} | |
| 43 | + | |
| 44 | +# municipalités de la Capitale-Nationale qui relèvent touristiquement | |
| 45 | +# de Charlevoix | |
| 46 | +_VILLES_CHARLEVOIX = { | |
| 47 | + "baie-saint-paul", "petite-riviere-saint-francois", "la-malbaie", | |
| 48 | + "les-eboulements", "saint-urbain", "saint-irenee", "saint-hilarion", | |
| 49 | + "isle-aux-coudres", "l'isle-aux-coudres", "notre-dame-des-monts", | |
| 50 | + "saint-aime-des-lacs", "clermont", "saint-simeon", | |
| 51 | + "baie-sainte-catherine", | |
| 52 | +} | |
| 53 | + | |
| 54 | +# mot-clé du titre → type canonique Lou-Ka | |
| 55 | +_TYPE_HINTS = [ | |
| 56 | + ("condo", "Condo"), ("loft", "Loft"), ("studio", "Studio"), | |
| 57 | + ("appartement", "Appartement"), ("maison", "Maison"), | |
| 58 | + ("mini-maison", "Mini-maison"), ("chalet", "Chalet"), | |
| 59 | +] | |
| 60 | + | |
| 61 | + | |
| 62 | +def _text(fragment: str) -> str: | |
| 63 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 64 | + | |
| 65 | + | |
| 66 | +class Chaleto(StConnector): | |
| 67 | + source_id = "chaleto" | |
| 68 | + | |
| 69 | + # -- inventaire (sitemap) ---------------------------------------------- | |
| 70 | + def _sitemap_urls(self) -> dict[str, str]: | |
| 71 | + """id Guesty -> URL détail FR.""" | |
| 72 | + xml = self.get(SITEMAP).text | |
| 73 | + urls: dict[str, str] = {} | |
| 74 | + for loc in re.findall(r"<loc>([^<]+)</loc>", xml): | |
| 75 | + m = _URL_FR.match(loc.strip()) | |
| 76 | + if m: | |
| 77 | + urls.setdefault(m.group(1), loc.strip()) | |
| 78 | + return urls | |
| 79 | + | |
| 80 | + # -- page détail --------------------------------------------------------- | |
| 81 | + def _detail(self, url: str) -> dict: | |
| 82 | + h = self.get(url).text | |
| 83 | + d: dict = {} | |
| 84 | + | |
| 85 | + m = re.search(r"<h1[^>]*>(.*?)</h1>", h, re.S) | |
| 86 | + if m: | |
| 87 | + d["title"] = _text(m.group(1)) | |
| 88 | + | |
| 89 | + # « Baie-Saint-Paul, Capitale-Nationale » sous le titre | |
| 90 | + m = re.search(r'<p class="text-lg">([^<]+)</p>', h) | |
| 91 | + if m: | |
| 92 | + parts = [p.strip() for p in _text(m.group(1)).split(",")] | |
| 93 | + if parts: | |
| 94 | + d["city"] = parts[0] | |
| 95 | + if len(parts) > 1: | |
| 96 | + cle = re.sub(r"-{2,}", "-", | |
| 97 | + strip_accents(parts[-1]).lower().replace(" ", "-")) | |
| 98 | + region = _REGION_FIX.get(cle, parts[-1]) | |
| 99 | + ville = strip_accents(d.get("city", "")).lower().replace(" ", "-") | |
| 100 | + if region == "Québec" and ville in _VILLES_CHARLEVOIX: | |
| 101 | + region = "Charlevoix" | |
| 102 | + d["region"] = region | |
| 103 | + | |
| 104 | + # « À partir de <strong>200$</strong> / nuit » | |
| 105 | + m = re.search(r"À partir de</span>\s*<span[^>]*>\s*" | |
| 106 | + r"<strong>([^<]+)</strong>\s*/\s*nuit", h) | |
| 107 | + if m: | |
| 108 | + d["price_label"] = f"à partir de {_text(m.group(1))} / nuit" | |
| 109 | + | |
| 110 | + # pictos : « 5 voyageurs », « 3 chambres », « 3 lits », « 2 salles de bain » | |
| 111 | + for val, label in re.findall( | |
| 112 | + r"<p>\s*([\d.,]+)\s+(voyageurs?|chambres?|lits?|" | |
| 113 | + r"salles? de bain)\s*</p>", h): | |
| 114 | + n = float(val.replace(",", ".")) | |
| 115 | + lab = label.lower() | |
| 116 | + if lab.startswith("voyageur"): | |
| 117 | + d["capacity"] = n | |
| 118 | + elif lab.startswith("chambre"): | |
| 119 | + d["bedrooms"] = n | |
| 120 | + elif lab.startswith("lit"): | |
| 121 | + d["beds"] = n | |
| 122 | + else: | |
| 123 | + d["bathrooms"] = n | |
| 124 | + | |
| 125 | + # commodités : spans du bloc « Commodités » (grille + accordéon) | |
| 126 | + i = h.find("Commodités</h2>") | |
| 127 | + if i >= 0: | |
| 128 | + j = h.find("<h2", i + 10) | |
| 129 | + bloc = h[i:j if j > 0 else i + 20000] | |
| 130 | + amen = [] | |
| 131 | + for a in re.findall(r'<span class="text-white">([^<]+)</span>', bloc): | |
| 132 | + a = _text(a) | |
| 133 | + if a and a not in amen: | |
| 134 | + amen.append(a) | |
| 135 | + d["amenities"] = amen | |
| 136 | + | |
| 137 | + # description (1er paragraphe long — contient « CITQ : NNNNNN | Exp: … ») | |
| 138 | + m = re.search(r'<p class="max-w-\[50rem\]">(.*?)</p>', h, re.S) | |
| 139 | + if m: | |
| 140 | + txt = _html.unescape(re.sub(r"<br\s*/?>", "\n", | |
| 141 | + m.group(1))) | |
| 142 | + txt = _TAG_RE.sub(" ", txt) | |
| 143 | + txt = re.sub(r"[ \t]+", " ", txt).strip() | |
| 144 | + d["description"] = txt[:5000] | |
| 145 | + m2 = re.search(r"CITQ\s*:?\s*(\d{6})", txt) | |
| 146 | + if m2: | |
| 147 | + d["citq"] = m2.group(1) | |
| 148 | + | |
| 149 | + # photos Guesty pleine résolution (lightbox data-images, JSON échappé) | |
| 150 | + m = re.search(r'data-images="([^"]+)"', h) | |
| 151 | + if m: | |
| 152 | + try: | |
| 153 | + imgs = json.loads(_html.unescape(m.group(1))) | |
| 154 | + except ValueError: | |
| 155 | + imgs = [] | |
| 156 | + d["images"] = [u for u in imgs if isinstance(u, str)][:20] | |
| 157 | + return d | |
| 158 | + | |
| 159 | + # -- contrat -------------------------------------------------------------- | |
| 160 | + def fetch(self) -> list[StListing]: | |
| 161 | + cle = "detail-" + time.strftime("%Y-%m") # lastmod uniforme → mensuel | |
| 162 | + listings: list[StListing] = [] | |
| 163 | + for eid, url in self._sitemap_urls().items(): | |
| 164 | + try: | |
| 165 | + d = self.detail(eid, cle, lambda u=url: self._detail(u)) | |
| 166 | + except Exception: # une fiche cassée ≠ inventaire perdu | |
| 167 | + d = {} | |
| 168 | + title = d.get("title") or "" | |
| 169 | + if not title: | |
| 170 | + continue | |
| 171 | + | |
| 172 | + ptype = "" | |
| 173 | + hay = strip_accents(title).lower() | |
| 174 | + for needle, canon in _TYPE_HINTS: | |
| 175 | + if needle in hay: | |
| 176 | + ptype = canon | |
| 177 | + break | |
| 178 | + | |
| 179 | + listings.append(StListing( | |
| 180 | + source=self.source_id, | |
| 181 | + external_id=eid, # id Guesty, stable | |
| 182 | + url=url, | |
| 183 | + title=title, | |
| 184 | + property_type=ptype or "Chalet", | |
| 185 | + city=d.get("city", ""), | |
| 186 | + region=d.get("region", ""), | |
| 187 | + price_night=parse_price_night(d.get("price_label", "")), | |
| 188 | + price_label=d.get("price_label", ""), | |
| 189 | + capacity=d.get("capacity"), | |
| 190 | + bedrooms=d.get("bedrooms"), | |
| 191 | + beds=d.get("beds"), | |
| 192 | + bathrooms=d.get("bathrooms"), | |
| 193 | + citq=d.get("citq", ""), | |
| 194 | + description=d.get("description", ""), | |
| 195 | + amenities=d.get("amenities") or [], | |
| 196 | + images=d.get("images") or [], | |
| 197 | + )) | |
| 198 | + return listings | |
added
louka/shortterm/connectors/chaletsalpins.py
+200 −0
@@ -0,0 +1,200 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/chaletsalpins.py : Les Chalets Alpins (chaletsalpins.ca) | |
| 4 | +# — gestionnaire de Stoneham, ~180 chalets (Stoneham, Lac-Beauport, | |
| 5 | +# Charlevoix, Laurentides). | |
| 6 | +# | |
| 7 | +# Méthode : WordPress. Sitemap /chalets-sitemap.xml (lastmod fiable, doublons | |
| 8 | +# /en/ écartés) → pages /hebergement/<slug>/ (slug = adresse + no CITQ). | |
| 9 | +# La page détail porte un JSON-LD schema.org Hotel (nom, description, | |
| 10 | +# addressLocality, petsAllowed, amenityFeature FR) ; les compteurs vivent | |
| 11 | +# dans des <span class="capacity|bedrooms|beds|restrooms">…N</span> de la | |
| 12 | +# barre d'entête, et le prix dans l'encadré latéral | |
| 13 | +# « <h3>2 nuits à partir de (printemps) <span>1 025.00 $</span></h3> » | |
| 14 | +# (minimum 2 nuits → prix ramené à la nuit). Photos = uploads du carrousel | |
| 15 | +# (vignettes -150x150 écartées). Pas de géo dans le HTML. | |
| 16 | +# ----------------------------------------------------------------------------- | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import html as _html | |
| 20 | +import json | |
| 21 | +import re | |
| 22 | + | |
| 23 | +from ...normalize import strip_accents | |
| 24 | +from ..schema import StListing | |
| 25 | +from .base import StConnector | |
| 26 | + | |
| 27 | +SITEMAP = "https://chaletsalpins.ca/chalets-sitemap.xml" | |
| 28 | + | |
| 29 | +_URL_DETAIL = re.compile( | |
| 30 | + r"https://(?:www\.)?chaletsalpins\.ca/hebergement/([\w-]+)/?$") | |
| 31 | + | |
| 32 | +# localités desservies → région touristique canonique | |
| 33 | +_VILLE_REGION = { | |
| 34 | + "stoneham": "Québec", | |
| 35 | + "stoneham-et-tewkesbury": "Québec", | |
| 36 | + "lac-beauport": "Québec", | |
| 37 | + "quebec": "Québec", | |
| 38 | + "petite-riviere-saint-francois": "Charlevoix", | |
| 39 | + "baie-saint-paul": "Charlevoix", | |
| 40 | + "la-malbaie": "Charlevoix", | |
| 41 | + "les-eboulements": "Charlevoix", | |
| 42 | + "saint-sauveur": "Laurentides", | |
| 43 | + "sainte-adele": "Laurentides", | |
| 44 | + "mont-tremblant": "Laurentides", | |
| 45 | +} | |
| 46 | + | |
| 47 | +_STAT_RE = re.compile( | |
| 48 | + r'<span class="(capacity|bedrooms|beds|restrooms)">' | |
| 49 | + r"(?:(?!</span>).)*?([\d.,]+)\s*</span>", re.S) | |
| 50 | + | |
| 51 | +_PRIX_RE = re.compile( | |
| 52 | + r"<h3>\s*(\d+)\s*nuits?\s*à partir de[^<]*<br\s*/?>\s*" | |
| 53 | + r"<span>\s*([\d\s,. ]+)\s*\$\s*</span>", re.S) | |
| 54 | + | |
| 55 | + | |
| 56 | +def _montant(raw: str) -> float | None: | |
| 57 | + try: | |
| 58 | + return float(re.sub(r"[\s ]", "", raw).replace(",", ".")) | |
| 59 | + except ValueError: | |
| 60 | + return None | |
| 61 | + | |
| 62 | + | |
| 63 | +class ChaletsAlpins(StConnector): | |
| 64 | + source_id = "chaletsalpins" | |
| 65 | + | |
| 66 | + # -- inventaire (sitemap FR + lastmod) ----------------------------------- | |
| 67 | + def _sitemap_urls(self) -> dict[str, tuple[str, str]]: | |
| 68 | + """slug -> (url détail FR, lastmod).""" | |
| 69 | + xml = self.get(SITEMAP).text | |
| 70 | + urls: dict[str, tuple[str, str]] = {} | |
| 71 | + for bloc in re.findall(r"<url>(.*?)</url>", xml, re.S): | |
| 72 | + m = re.search(r"<loc>([^<]+)</loc>", bloc) | |
| 73 | + if not m: | |
| 74 | + continue | |
| 75 | + loc = m.group(1).strip() | |
| 76 | + mu = _URL_DETAIL.match(loc) | |
| 77 | + if not mu or mu.group(1) in ("hebergement",): | |
| 78 | + continue | |
| 79 | + lastmod = re.search(r"<lastmod>([^<]+)</lastmod>", bloc) | |
| 80 | + urls.setdefault(mu.group(1), | |
| 81 | + (loc, lastmod.group(1) if lastmod else "")) | |
| 82 | + return urls | |
| 83 | + | |
| 84 | + # -- page détail --------------------------------------------------------- | |
| 85 | + def _detail(self, url: str) -> dict: | |
| 86 | + # ⚠️ le serveur ajoute parfois APRÈS </html> un second rendu avec des | |
| 87 | + # chalets suggérés (autres compteurs/photos) : on tronque au 1er </html> | |
| 88 | + h = self.get(url).text.split("</html>", 1)[0] | |
| 89 | + d: dict = {} | |
| 90 | + | |
| 91 | + # JSON-LD Hotel : nom, description, adresse, animaux, commodités | |
| 92 | + for m in re.finditer(r'<script[^>]*application/ld\+json[^>]*>(.*?)' | |
| 93 | + r"</script>", h, re.S): | |
| 94 | + try: | |
| 95 | + data = json.loads(m.group(1), strict=False) | |
| 96 | + except ValueError: | |
| 97 | + continue | |
| 98 | + if not (isinstance(data, dict) and data.get("@type") == "Hotel"): | |
| 99 | + continue | |
| 100 | + d["title"] = (data.get("name") or "").strip() | |
| 101 | + d["description"] = re.sub( | |
| 102 | + r"\s+", " ", (data.get("description") or "")).strip()[:5000] | |
| 103 | + addr = data.get("address") or {} | |
| 104 | + d["city"] = (addr.get("addressLocality") or "").strip() | |
| 105 | + if data.get("petsAllowed") is not None: | |
| 106 | + d["pets"] = "oui" if str(data["petsAllowed"]) in ( | |
| 107 | + "True", "true", "1") else "non" | |
| 108 | + amen = [] | |
| 109 | + for feat in data.get("amenityFeature") or []: | |
| 110 | + nom = (feat.get("name") or "").strip() \ | |
| 111 | + if isinstance(feat, dict) else "" | |
| 112 | + if nom and nom not in amen: | |
| 113 | + amen.append(nom) | |
| 114 | + if amen: | |
| 115 | + d["amenities"] = amen | |
| 116 | + break | |
| 117 | + | |
| 118 | + # compteurs de l'entête (capacité, chambres, lits, salles de bain) — | |
| 119 | + # 1re occurrence seulement (le chalet courant précède toute suggestion) | |
| 120 | + for cls, val in _STAT_RE.findall(h): | |
| 121 | + n = _montant(val) | |
| 122 | + if n is None: | |
| 123 | + continue | |
| 124 | + d.setdefault({"capacity": "capacity", "bedrooms": "bedrooms", | |
| 125 | + "beds": "beds", "restrooms": "bathrooms"}[cls], n) | |
| 126 | + | |
| 127 | + # encadré latéral : « 2 nuits à partir de (printemps) 1 025.00 $ » | |
| 128 | + # → prix / nuit ; certaines unités affichent « Location mensuelle » | |
| 129 | + # (long terme : pas de prix à la nuit, mention conservée) | |
| 130 | + m = _PRIX_RE.search(h) | |
| 131 | + if m: | |
| 132 | + nuits = int(m.group(1)) or 1 | |
| 133 | + montant = _montant(m.group(2)) | |
| 134 | + if montant: | |
| 135 | + d["price_night"] = round(montant / nuits, 2) | |
| 136 | + d["price_label"] = re.sub( | |
| 137 | + r"\s+", " ", _html.unescape( | |
| 138 | + re.sub(r"<[^>]+>", " ", m.group(0)))).strip() | |
| 139 | + elif re.search(r'sidebar scrollbox">\s*<div class="top">\s*' | |
| 140 | + r"<h3>\s*Location mensuelle", h): | |
| 141 | + d["location_mensuelle"] = True | |
| 142 | + | |
| 143 | + # no CITQ (dans le nom/slug « …(CITQ282170) » ou la description) | |
| 144 | + m = re.search(r"CITQ\)?\s*:?\s*#?\s*(\d{6})", | |
| 145 | + d.get("title", "") + " " + d.get("description", "")) | |
| 146 | + if m: | |
| 147 | + d["citq"] = m.group(1) | |
| 148 | + | |
| 149 | + # photos du carrousel (vignettes et icônes écartées, dédoublonnage | |
| 150 | + # sur le nom de base sans suffixe de taille -WxH) | |
| 151 | + imgs, vus = [], set() | |
| 152 | + for u in re.findall(r'(https://(?:www\.)?chaletsalpins\.ca/' | |
| 153 | + r'wp-content/uploads/[^"\'\s>]+' | |
| 154 | + r"\.(?:jpe?g|png|webp))", h): | |
| 155 | + base = re.sub(r"-\d+x\d+(?=\.\w+$)", "", u) | |
| 156 | + if base in vus or "-150x150" in u: | |
| 157 | + continue | |
| 158 | + vus.add(base) | |
| 159 | + imgs.append(u) | |
| 160 | + d["images"] = imgs[:20] | |
| 161 | + return d | |
| 162 | + | |
| 163 | + # -- contrat -------------------------------------------------------------- | |
| 164 | + def fetch(self) -> list[StListing]: | |
| 165 | + listings: list[StListing] = [] | |
| 166 | + for slug, (url, lastmod) in self._sitemap_urls().items(): | |
| 167 | + try: | |
| 168 | + d = self.detail(slug, lastmod or "sans-lastmod", | |
| 169 | + lambda u=url: self._detail(u)) | |
| 170 | + except Exception: # une fiche cassée ≠ inventaire perdu | |
| 171 | + d = {} | |
| 172 | + if not d.get("title"): | |
| 173 | + continue | |
| 174 | + ville = d.get("city", "") | |
| 175 | + region = _VILLE_REGION.get( | |
| 176 | + strip_accents(ville).lower().replace(" ", "-"), "") | |
| 177 | + details = ({"location_mensuelle": True} | |
| 178 | + if d.get("location_mensuelle") else {}) | |
| 179 | + listings.append(StListing( | |
| 180 | + source=self.source_id, | |
| 181 | + external_id=slug, # slug WP, stable | |
| 182 | + url=url, | |
| 183 | + title=d["title"], | |
| 184 | + property_type="Chalet", | |
| 185 | + city=ville, | |
| 186 | + region=region, | |
| 187 | + price_night=d.get("price_night"), | |
| 188 | + price_label=d.get("price_label", ""), | |
| 189 | + capacity=d.get("capacity"), | |
| 190 | + bedrooms=d.get("bedrooms"), | |
| 191 | + beds=d.get("beds"), | |
| 192 | + bathrooms=d.get("bathrooms"), | |
| 193 | + pets=d.get("pets"), | |
| 194 | + citq=d.get("citq", ""), | |
| 195 | + description=d.get("description", ""), | |
| 196 | + amenities=d.get("amenities") or [], | |
| 197 | + details=details, | |
| 198 | + images=d.get("images") or [], | |
| 199 | + )) | |
| 200 | + return listings | |
added
louka/shortterm/connectors/chaletsbsl.py
+168 −0
@@ -0,0 +1,168 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/chaletsbsl.py : Chalets BSL (chaletsbsl.com) — 4 chalets avec spa | |
| 4 | +# sur un domaine privé de Saint-Simon-de-Rimouski (Bas-Saint-Laurent). | |
| 5 | +# | |
| 6 | +# Méthode : sitemap.xml (index) → sitemap_sections_*.xml → URLs /chalets/<slug> | |
| 7 | +# + lastmod (clé du cache détail ; les /en/ sont ignorées). Pages statiques | |
| 8 | +# (CMS maison Bootstrap) : | |
| 9 | +# - <h3 class="mt-4 fw-600"> = titre (préfixe « Chalets BSL - » retiré) ; | |
| 10 | +# - <h5 class="prix"> « À partir de 610$ pour 2 nuits » → price_night ; | |
| 11 | +# - <p class="capacite"> « 2 pers. 1 chbre. 1 sdb. » ; | |
| 12 | +# - description = <p> entre la capacité et le bloc country-info ; | |
| 13 | +# - commodités = <h6> des blocs country-name ; | |
| 14 | +# - galerie = var chaletImgs = {"ete": […], …} (JSON par saison) ; | |
| 15 | +# - lat/lng = const myLatLng = { lat: …, lng: … } ; CITQ en pied de fiche. | |
| 16 | +# ----------------------------------------------------------------------------- | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import html as _html | |
| 20 | +import json | |
| 21 | +import re | |
| 22 | + | |
| 23 | +from ..schema import StListing | |
| 24 | +from .base import StConnector | |
| 25 | + | |
| 26 | +SITE = "https://chaletsbsl.com" | |
| 27 | +SITEMAP = SITE + "/sitemap.xml" | |
| 28 | + | |
| 29 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 30 | + | |
| 31 | + | |
| 32 | +def _text(fragment: str) -> str: | |
| 33 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 34 | + | |
| 35 | + | |
| 36 | +class ChaletsBsl(StConnector): | |
| 37 | + source_id = "chaletsbsl" | |
| 38 | + request_delay = 1.0 | |
| 39 | + | |
| 40 | + # -- page détail ---------------------------------------------------------- | |
| 41 | + def _detail(self, url: str) -> dict: | |
| 42 | + h = self.get(url).text | |
| 43 | + d: dict = {} | |
| 44 | + | |
| 45 | + m = re.search(r'(?s)<h3 class="mt-4 fw-600[^"]*"[^>]*>(.*?)</h3>', h) | |
| 46 | + if m: | |
| 47 | + d["title"] = re.sub(r"^Chalets BSL\s*-\s*", "", _text(m.group(1))) | |
| 48 | + | |
| 49 | + # « À partir de 610$ pour 2 nuits » → 305 $/nuit | |
| 50 | + m = re.search(r"À partir de\s*([\d\s]+)\$\s*pour\s*(\d+)\s*nuits", h) | |
| 51 | + if m: | |
| 52 | + total = float(m.group(1).replace(" ", "")) | |
| 53 | + nights = int(m.group(2)) | |
| 54 | + if nights and 20 <= total / nights <= 20000: | |
| 55 | + d["price_night"] = round(total / nights) | |
| 56 | + d["price_ref"] = f"{total:g} $ pour {nights} nuits" | |
| 57 | + | |
| 58 | + # « 2 pers. 1 chbre. 1 sdb. » | |
| 59 | + m = re.search(r'(?s)<p class="capacite[^"]*"[^>]*>(.*?)</p>', h) | |
| 60 | + if m: | |
| 61 | + frag = _text(m.group(1)) | |
| 62 | + for pat, key in ((r"(\d+)\s*pers", "capacity"), | |
| 63 | + (r"(\d+)\s*chbre", "bedrooms"), | |
| 64 | + (r"(\d+)\s*sdb", "bathrooms")): | |
| 65 | + mm = re.search(pat, frag) | |
| 66 | + if mm: | |
| 67 | + d[key] = float(mm.group(1)) | |
| 68 | + | |
| 69 | + # description : les <p> entre la capacité et le bloc country-info | |
| 70 | + i = h.find('class="capacite') | |
| 71 | + j = h.find("country-info") | |
| 72 | + if 0 < i < j: | |
| 73 | + paras = [_text(p) for p in | |
| 74 | + re.findall(r"(?s)<p[^>]*>(.*?)</p>", h[i:j])] | |
| 75 | + texte = "\n".join(p for p in paras if len(p) > 40) | |
| 76 | + if texte: | |
| 77 | + d["description"] = texte[:5000] | |
| 78 | + | |
| 79 | + # commodités : les <h6> (tous portés par les blocs country-name) | |
| 80 | + amen: list[str] = [] | |
| 81 | + for h6 in re.findall(r"(?s)<h6[^>]*>(.*?)</h6>", h): | |
| 82 | + t = _text(h6) | |
| 83 | + if t and t not in amen: | |
| 84 | + amen.append(t) | |
| 85 | + if amen: | |
| 86 | + d["amenities"] = amen | |
| 87 | + | |
| 88 | + m = re.search(r"const myLatLng = \{\s*lat:\s*(-?\d+\.\d+)," | |
| 89 | + r"\s*lng:\s*(-?\d+\.\d+)", h) | |
| 90 | + if m: | |
| 91 | + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2)) | |
| 92 | + m = re.search(r"CITQ\D{0,25}(\d{6})", h) | |
| 93 | + if m: | |
| 94 | + d["citq"] = m.group(1) | |
| 95 | + | |
| 96 | + # galerie : var chaletImgs = {"ete": […], "hiver": […]} | |
| 97 | + m = re.search(r"var chaletImgs = (\{.*?\});", h, re.S) | |
| 98 | + if m: | |
| 99 | + try: | |
| 100 | + seasons = json.loads(m.group(1)) | |
| 101 | + imgs: list[str] = [] | |
| 102 | + for key in ("ete", *sorted(k for k in seasons if k != "ete")): | |
| 103 | + for u in seasons.get(key) or []: | |
| 104 | + if isinstance(u, str) and u.startswith("https://") \ | |
| 105 | + and u not in imgs and len(imgs) < 20: | |
| 106 | + imgs.append(u) | |
| 107 | + if imgs: | |
| 108 | + d["images"] = imgs | |
| 109 | + except ValueError: | |
| 110 | + pass | |
| 111 | + return d | |
| 112 | + | |
| 113 | + # -- contrat -------------------------------------------------------------- | |
| 114 | + def fetch(self) -> list[StListing]: | |
| 115 | + index = self.get(SITEMAP).text | |
| 116 | + entries: list[tuple[str, str]] = [] | |
| 117 | + for sub in re.findall(r"<loc>([^<]+)</loc>", index): | |
| 118 | + if "sitemap_sections" not in sub: | |
| 119 | + continue | |
| 120 | + xml = self.get(sub).text | |
| 121 | + entries += re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>" | |
| 122 | + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml) | |
| 123 | + | |
| 124 | + listings: list[StListing] = [] | |
| 125 | + vus: set[str] = set() | |
| 126 | + for url, lastmod in entries: | |
| 127 | + m = re.match(r"https://chaletsbsl\.com/chalets/([^/]+)/?$", url) | |
| 128 | + if not m: | |
| 129 | + continue | |
| 130 | + slug = m.group(1) | |
| 131 | + if slug in vus: | |
| 132 | + continue | |
| 133 | + vus.add(slug) | |
| 134 | + | |
| 135 | + det = self.detail(slug, lastmod or "v1", | |
| 136 | + lambda u=url: self._detail(u)) | |
| 137 | + title = det.get("title") or "" | |
| 138 | + if not title: | |
| 139 | + continue | |
| 140 | + | |
| 141 | + price = det.get("price_night") | |
| 142 | + details = {k: v for k, v in { | |
| 143 | + "price_ref": det.get("price_ref") or "", | |
| 144 | + "domain": "Chalets BSL", | |
| 145 | + }.items() if v} | |
| 146 | + | |
| 147 | + listings.append(StListing( | |
| 148 | + source=self.source_id, | |
| 149 | + external_id=slug, | |
| 150 | + url=url, | |
| 151 | + title=title, | |
| 152 | + property_type="Chalet", | |
| 153 | + city="Saint-Simon-de-Rimouski", | |
| 154 | + region="Bas-Saint-Laurent", | |
| 155 | + price_night=float(price) if price else None, | |
| 156 | + price_label=f"à partir de {price:g} $ / nuit" if price else "", | |
| 157 | + capacity=det.get("capacity"), | |
| 158 | + bedrooms=det.get("bedrooms"), | |
| 159 | + bathrooms=det.get("bathrooms"), | |
| 160 | + citq=det.get("citq") or "", | |
| 161 | + description=det.get("description") or "", | |
| 162 | + amenities=det.get("amenities") or [], | |
| 163 | + details=details, | |
| 164 | + images=det.get("images") or [], | |
| 165 | + lat=det.get("lat"), | |
| 166 | + lng=det.get("lng"), | |
| 167 | + )) | |
| 168 | + return listings | |
added
louka/shortterm/connectors/chaletsdanslenord.py
+219 −0
@@ -0,0 +1,219 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/chaletsdanslenord.py : Les Chalets dans le Nord | |
| 4 | +# (leschaletsdanslenord.com — Laurentides) | |
| 5 | +# | |
| 6 | +# Petite agence familiale (Sainte-Lucie-des-Laurentides / lac Sarrazin, | |
| 7 | +# ~6 chalets) — vitrine WordPress + moteur de réservation HOSTAWAY | |
| 8 | +# (reservation.leschaletsdanslenord.com, compte 96792). | |
| 9 | +# | |
| 10 | +# Méthode : | |
| 11 | +# 1. IDS : la racine du moteur Hostaway (Next.js rendu serveur) référence | |
| 12 | +# tous les chalets via des liens "/listings/<id>" ; la homepage WP donne | |
| 13 | +# en plus prix (« dès N $ / nuit ») et lien de la fiche vitrine (l'id | |
| 14 | +# Hostaway est dans l'URL des photos S3 `96792-<id>-…`). | |
| 15 | +# 2. DÉTAIL (cache self.detail) : /listings/<id> du moteur embarque le JSON | |
| 16 | +# complet dans le payload React Flight (`self.__next_f`) : prix de base | |
| 17 | +# par nuit, lat/lng, ville, capacité, chambres, sdb, lits, type, note | |
| 18 | +# (sur 10 → /2 par finalize), nb d'avis, ~50 photos, ~70 commodités et | |
| 19 | +# description (référence Flight « $xx » résolue via les segments T<hex>). | |
| 20 | +# External_id = id de listing Hostaway (stable, dans l'URL du moteur). | |
| 21 | +# ----------------------------------------------------------------------------- | |
| 22 | +from __future__ import annotations | |
| 23 | + | |
| 24 | +import json | |
| 25 | +import re | |
| 26 | + | |
| 27 | +from ..schema import StListing | |
| 28 | +from .base import StConnector | |
| 29 | + | |
| 30 | +SITE = "https://leschaletsdanslenord.com" | |
| 31 | +ENGINE = "https://reservation.leschaletsdanslenord.com" | |
| 32 | + | |
| 33 | + | |
| 34 | +def _num(v) -> float | None: | |
| 35 | + try: | |
| 36 | + return float(v) if v not in (None, "") else None | |
| 37 | + except (TypeError, ValueError): | |
| 38 | + return None | |
| 39 | + | |
| 40 | + | |
| 41 | +def _flight_blob(html: str) -> str: | |
| 42 | + parts = [] | |
| 43 | + for c in re.findall(r'self\.__next_f\.push\(\[1,"((?:[^"\\]|\\.)*)"\]\)', | |
| 44 | + html): | |
| 45 | + try: | |
| 46 | + parts.append(json.loads(f'"{c}"')) | |
| 47 | + except ValueError: | |
| 48 | + continue | |
| 49 | + return "".join(parts) | |
| 50 | + | |
| 51 | + | |
| 52 | +def _flight_text(blob: str, ref: str) -> str: | |
| 53 | + """Résout une référence texte Flight « $xx » (segment `xx:T<len hex>,`, | |
| 54 | + longueur en OCTETS utf-8).""" | |
| 55 | + rid = ref.lstrip("$") | |
| 56 | + m = re.search(rf"(?:^|\n){re.escape(rid)}:T([0-9a-f]+),", blob) | |
| 57 | + if not m: | |
| 58 | + return "" | |
| 59 | + n = int(m.group(1), 16) | |
| 60 | + raw = blob[m.end():].encode("utf-8")[:n] | |
| 61 | + return raw.decode("utf-8", errors="ignore") | |
| 62 | + | |
| 63 | + | |
| 64 | +class ChaletsDansLeNord(StConnector): | |
| 65 | + source_id = "chaletsdanslenord" | |
| 66 | + | |
| 67 | + # -- ids + carte prix/urls vitrine -------------------------------------- | |
| 68 | + def _engine_ids(self) -> list[str]: | |
| 69 | + h = self.get(f"{ENGINE}/").text | |
| 70 | + return sorted(set(re.findall(r'"/listings/(\d+)"', h))) | |
| 71 | + | |
| 72 | + def _wp_cards(self) -> dict[str, dict]: | |
| 73 | + """id Hostaway → {url fiche vitrine, prix « dès N $ »} (homepage WP).""" | |
| 74 | + try: | |
| 75 | + h = self.get(f"{SITE}/").text | |
| 76 | + except Exception: | |
| 77 | + return {} | |
| 78 | + cards: dict[str, dict] = {} | |
| 79 | + for block in re.split(r'<li class="lcdln-hsg__card"', h)[1:]: | |
| 80 | + m = re.search(r"hostaway-platform[^\"]*/listing/96792-(\d+)-", | |
| 81 | + block) | |
| 82 | + if not m: | |
| 83 | + continue | |
| 84 | + hid = m.group(1) | |
| 85 | + card: dict = {} | |
| 86 | + m = re.search(r'class="lcdln-hsg__card-btn" href="([^"]+)"', block) | |
| 87 | + if m: | |
| 88 | + card["url"] = m.group(1) | |
| 89 | + m = re.search(r"dès\s*([\d ,]+)\s*\$\s*/\s*nuit", block) | |
| 90 | + if m: | |
| 91 | + card["price"] = float(m.group(1).replace(" ", "") | |
| 92 | + .replace(",", ".")) | |
| 93 | + cards[hid] = card | |
| 94 | + return cards | |
| 95 | + | |
| 96 | + def _wp_fiche(self, url: str) -> dict: | |
| 97 | + """Titre + description EN FRANÇAIS depuis la fiche vitrine WP.""" | |
| 98 | + h = self.get(url).text | |
| 99 | + d: dict = {} | |
| 100 | + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h) | |
| 101 | + if m: | |
| 102 | + d["title"] = re.sub(r"\s+", " ", | |
| 103 | + re.sub(r"<[^>]+>", " ", m.group(1))).strip() | |
| 104 | + m = re.search(r"(?s)<main[^>]*>(.*?)</main>", h) | |
| 105 | + if m: | |
| 106 | + import html as _h | |
| 107 | + paras, seen = [], set() | |
| 108 | + for p in re.findall(r"(?s)<p[^>]*>(.*?)</p>", m.group(1)): | |
| 109 | + t = _h.unescape(re.sub(r"\s+", " ", | |
| 110 | + re.sub(r"<[^>]+>", " ", p))).strip() | |
| 111 | + if len(t) < 60 or t in seen or "Voir les" in t[:30]: | |
| 112 | + continue | |
| 113 | + seen.add(t) | |
| 114 | + paras.append(t) | |
| 115 | + if len(paras) >= 10: | |
| 116 | + break | |
| 117 | + if paras: | |
| 118 | + d["description"] = " ".join(paras)[:4000] | |
| 119 | + return d | |
| 120 | + | |
| 121 | + # -- détail (moteur Hostaway + fiche vitrine FR) -------------------------- | |
| 122 | + def _detail(self, hid: str, wp_url: str = "") -> dict: | |
| 123 | + h = self.get(f"{ENGINE}/listings/{hid}").text | |
| 124 | + blob = _flight_blob(h) | |
| 125 | + i = blob.find(f'"listing":{{"id":{hid}') | |
| 126 | + if i < 0: | |
| 127 | + return {} | |
| 128 | + obj, _ = json.JSONDecoder().raw_decode(blob[i + len('"listing":'):]) | |
| 129 | + inner = obj.get("listing") or {} | |
| 130 | + | |
| 131 | + desc = str(inner.get("description") or "") | |
| 132 | + if desc.startswith("$"): | |
| 133 | + desc = _flight_text(blob, desc) | |
| 134 | + desc = re.sub(r"\s+", " ", desc).strip() | |
| 135 | + | |
| 136 | + images = [] | |
| 137 | + for ph in obj.get("listingImage") or []: | |
| 138 | + u = (ph or {}).get("url") | |
| 139 | + if u and u not in images: | |
| 140 | + images.append(u) | |
| 141 | + if len(images) >= 20: | |
| 142 | + break | |
| 143 | + | |
| 144 | + # fiche vitrine WP : titre + description en français (prioritaires) | |
| 145 | + wp: dict = {} | |
| 146 | + if wp_url: | |
| 147 | + try: | |
| 148 | + wp = self._wp_fiche(wp_url) | |
| 149 | + except Exception: | |
| 150 | + wp = {} | |
| 151 | + if wp.get("description"): | |
| 152 | + desc = wp["description"] | |
| 153 | + | |
| 154 | + pt = ((inner.get("propertyType") or {}).get("name") or "").strip() | |
| 155 | + return { | |
| 156 | + "title": wp.get("title") or (inner.get("name") or "").strip(), | |
| 157 | + "price": _num(inner.get("price")), | |
| 158 | + "lat": _num(inner.get("lat")), | |
| 159 | + "lng": _num(inner.get("lng")), | |
| 160 | + "city": (inner.get("city") or "").strip(), | |
| 161 | + "capacity": _num(inner.get("personCapacity")), | |
| 162 | + "bedrooms": _num(inner.get("bedroomsNumber")), | |
| 163 | + "beds": _num(inner.get("bedsNumber")), | |
| 164 | + "bathrooms": _num(inner.get("bathroomsNumber")), | |
| 165 | + "property_type": pt, | |
| 166 | + "rating": _num(obj.get("averageReviewRating")), | |
| 167 | + "reviews": obj.get("reviewsCount"), | |
| 168 | + "description": desc[:4000], | |
| 169 | + "amenities": [n for n in | |
| 170 | + (((a.get("amenity") or {}).get("name") | |
| 171 | + or a.get("name") or "").strip() | |
| 172 | + for a in obj.get("listingAmenity") or [] | |
| 173 | + if isinstance(a, dict)) if n][:80], | |
| 174 | + "images": images, | |
| 175 | + } | |
| 176 | + | |
| 177 | + # -- contrat ---------------------------------------------------------- | |
| 178 | + def fetch(self) -> list[StListing]: | |
| 179 | + cards = self._wp_cards() | |
| 180 | + listings: list[StListing] = [] | |
| 181 | + for hid in self._engine_ids(): | |
| 182 | + card = cards.get(hid) or {} | |
| 183 | + key = json.dumps([hid, card.get("price"), card.get("url")]) | |
| 184 | + try: | |
| 185 | + det = self.detail( | |
| 186 | + hid, key, | |
| 187 | + lambda i=hid, u=card.get("url") or "": self._detail(i, u)) | |
| 188 | + except Exception: | |
| 189 | + det = {} | |
| 190 | + if not det.get("title"): | |
| 191 | + continue | |
| 192 | + | |
| 193 | + price = det.get("price") or card.get("price") | |
| 194 | + reviews = det.get("reviews") | |
| 195 | + listings.append(StListing( | |
| 196 | + source=self.source_id, | |
| 197 | + external_id=hid, | |
| 198 | + url=card.get("url") or f"{ENGINE}/listings/{hid}", | |
| 199 | + title=det["title"], | |
| 200 | + property_type=det.get("property_type") or "Chalet", | |
| 201 | + city=det.get("city") or "", | |
| 202 | + region="Laurentides", | |
| 203 | + price_night=price, | |
| 204 | + price_label=(f"à partir de {price:.0f} $ / nuit" | |
| 205 | + if price else ""), | |
| 206 | + capacity=det.get("capacity"), | |
| 207 | + bedrooms=det.get("bedrooms"), | |
| 208 | + beds=det.get("beds"), | |
| 209 | + bathrooms=det.get("bathrooms"), | |
| 210 | + rating=det.get("rating"), | |
| 211 | + reviews=int(reviews) if reviews else None, | |
| 212 | + description=det.get("description") or "", | |
| 213 | + amenities=det.get("amenities") or [], | |
| 214 | + details={"booking_url": f"{ENGINE}/listings/{hid}"}, | |
| 215 | + images=det.get("images") or [], | |
| 216 | + lat=det.get("lat"), | |
| 217 | + lng=det.get("lng"), | |
| 218 | + )) | |
| 219 | + return listings | |
added
louka/shortterm/connectors/domesstcome.py
+157 −0
@@ -0,0 +1,157 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/domesstcome.py : Dômes St-Côme (domesstcome.com) — 4 dômes | |
| 4 | +# géodésiques avec spa privé et vue panoramique à Saint-Côme (Lanaudière). | |
| 5 | +# | |
| 6 | +# Méthode : sitemap Wix (index) → dynamic-domes_*-sitemap.xml → 4 URLs | |
| 7 | +# /domes/<slug> + lastmod (clé du cache détail). Pages Wix statiques : | |
| 8 | +# - <h1> = nom du dôme ; sous-titre « Vue panoramique | Spa privé » ; | |
| 9 | +# - « À partir de 370$/nuit » → price_night ; | |
| 10 | +# - sections LITS / AUTRES / CUISINE / À L'EXTÉRIEUR / SALLE DE BAIN | |
| 11 | +# (texte riche Wix) → amenities ; beds = nb de « Lit … » sous LITS ; | |
| 12 | +# - CITQ (6 chiffres) dans le pied de page ; | |
| 13 | +# - images : médias wixstatic ~mv2 servis en grand (w ≥ 900) — la galerie | |
| 14 | +# Pro Gallery est chargée en JS, seuls les héros sont statiques. | |
| 15 | +# ----------------------------------------------------------------------------- | |
| 16 | +from __future__ import annotations | |
| 17 | + | |
| 18 | +import html as _html | |
| 19 | +import re | |
| 20 | + | |
| 21 | +from ..schema import StListing | |
| 22 | +from .base import StConnector | |
| 23 | + | |
| 24 | +SITE = "https://www.domesstcome.com" | |
| 25 | +SITEMAP = SITE + "/sitemap.xml" | |
| 26 | + | |
| 27 | +_SECTIONS = ("LITS", "AUTRES", "CUISINE", "À L'EXTÉRIEUR", "SALLE DE BAIN") | |
| 28 | + | |
| 29 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 30 | + | |
| 31 | + | |
| 32 | +def _lines(fragment: str) -> list[str]: | |
| 33 | + """HTML riche Wix → lignes de texte propres (CSS inline filtré).""" | |
| 34 | + txt = re.sub(r"\|(?:\s*\|)+", "\n", | |
| 35 | + re.sub(r"\s+", " ", _TAG_RE.sub("|", fragment))) | |
| 36 | + out = [] | |
| 37 | + for x in txt.split("\n"): | |
| 38 | + x = _html.unescape(x).strip(" |").strip() | |
| 39 | + x = re.sub(r"\s*\|\s*", " | ", x) | |
| 40 | + if x and len(x) > 2 and "{" not in x and "--" not in x: | |
| 41 | + out.append(x) | |
| 42 | + return out | |
| 43 | + | |
| 44 | + | |
| 45 | +class DomesStCome(StConnector): | |
| 46 | + source_id = "domesstcome" | |
| 47 | + request_delay = 1.0 | |
| 48 | + | |
| 49 | + # -- page détail ---------------------------------------------------------- | |
| 50 | + def _detail(self, url: str) -> dict: | |
| 51 | + h = self.get(url).text | |
| 52 | + d: dict = {} | |
| 53 | + | |
| 54 | + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h) | |
| 55 | + if m: | |
| 56 | + d["title"] = _html.unescape( | |
| 57 | + re.sub(r"\s+", " ", _TAG_RE.sub(" ", m.group(1)))).strip() | |
| 58 | + | |
| 59 | + m = re.search(r"À partir de\s*(\d+)\s*\$\s*/\s*nuit", h) | |
| 60 | + if m: | |
| 61 | + d["price_night"] = float(m.group(1)) | |
| 62 | + | |
| 63 | + m = re.search(r"CITQ\D{0,25}(\d{6})", h) | |
| 64 | + if m: | |
| 65 | + d["citq"] = m.group(1) | |
| 66 | + | |
| 67 | + # sous-titre (« Vue panoramique | Spa privé ») → description | |
| 68 | + i, j = h.find("<h1"), h.find("LITS") | |
| 69 | + if 0 <= i < j: | |
| 70 | + head = [x for x in _lines(h[i:j]) | |
| 71 | + if x != d.get("title") and "À partir de" not in x | |
| 72 | + and "Détails" not in x] | |
| 73 | + if head: | |
| 74 | + d["description"] = head[0][:500] | |
| 75 | + | |
| 76 | + # sections LITS…SALLE DE BAIN → amenities (entêtes exclues) | |
| 77 | + i = h.find("LITS") | |
| 78 | + j = h.find("Réserver", max(i, 0)) | |
| 79 | + beds = 0 | |
| 80 | + amen: list[str] = [] | |
| 81 | + if 0 <= i < j: | |
| 82 | + unescaped_secs = {s for s in _SECTIONS} | |
| 83 | + current = "" | |
| 84 | + for x in _lines(h[i:j]): | |
| 85 | + if x.upper() in unescaped_secs: | |
| 86 | + current = x.upper() | |
| 87 | + continue | |
| 88 | + if x in amen or len(x) > 80: | |
| 89 | + continue | |
| 90 | + amen.append(x) | |
| 91 | + if current == "LITS" and re.match(r"Lit\b", x, re.I): | |
| 92 | + beds += 1 | |
| 93 | + if amen: | |
| 94 | + d["amenities"] = amen | |
| 95 | + if beds: | |
| 96 | + d["beds"] = float(beds) | |
| 97 | + | |
| 98 | + # images : médias wixstatic servis en grand (héros) | |
| 99 | + big: dict[str, int] = {} | |
| 100 | + for mid, w in re.findall(r"static\.wixstatic\.com/media/" | |
| 101 | + r"([\w~%.]+)/v1/fill/w_(\d+)", h): | |
| 102 | + w = int(w) | |
| 103 | + if w >= 900: | |
| 104 | + big[mid] = max(big.get(mid, 0), w) | |
| 105 | + imgs = [f"https://static.wixstatic.com/media/{mid}" | |
| 106 | + for mid in big][:15] | |
| 107 | + if imgs: | |
| 108 | + d["images"] = imgs | |
| 109 | + return d | |
| 110 | + | |
| 111 | + # -- contrat -------------------------------------------------------------- | |
| 112 | + def fetch(self) -> list[StListing]: | |
| 113 | + index = self.get(SITEMAP).text | |
| 114 | + entries: list[tuple[str, str]] = [] | |
| 115 | + for sub in re.findall(r"<loc>([^<]+)</loc>", index): | |
| 116 | + if "dynamic-domes" not in sub: | |
| 117 | + continue | |
| 118 | + xml = self.get(sub).text | |
| 119 | + entries += re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>" | |
| 120 | + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml) | |
| 121 | + | |
| 122 | + listings: list[StListing] = [] | |
| 123 | + vus: set[str] = set() | |
| 124 | + for url, lastmod in entries: | |
| 125 | + m = re.match(r"https://www\.domesstcome\.com/domes/([^/]+)/?$", url) | |
| 126 | + if not m: | |
| 127 | + continue | |
| 128 | + slug = m.group(1) | |
| 129 | + if slug in vus: | |
| 130 | + continue | |
| 131 | + vus.add(slug) | |
| 132 | + | |
| 133 | + det = self.detail(slug, lastmod or "v1", | |
| 134 | + lambda u=url: self._detail(u)) | |
| 135 | + title = det.get("title") or "" | |
| 136 | + if not title: | |
| 137 | + continue | |
| 138 | + | |
| 139 | + price = det.get("price_night") | |
| 140 | + listings.append(StListing( | |
| 141 | + source=self.source_id, | |
| 142 | + external_id=slug, | |
| 143 | + url=url, | |
| 144 | + title=title, | |
| 145 | + property_type="Dôme", | |
| 146 | + city="Saint-Côme", | |
| 147 | + region="Lanaudière", | |
| 148 | + price_night=price, | |
| 149 | + price_label=f"À partir de {price:g} $ / nuit" if price else "", | |
| 150 | + beds=det.get("beds"), | |
| 151 | + citq=det.get("citq") or "", | |
| 152 | + description=det.get("description") or "", | |
| 153 | + amenities=det.get("amenities") or [], | |
| 154 | + details={"domain": "Dômes St-Côme"}, | |
| 155 | + images=det.get("images") or [], | |
| 156 | + )) | |
| 157 | + return listings | |
modified
louka/shortterm/connectors/expedia.py
+51 −3
@@ -21,7 +21,13 @@ | ||
| 21 | 21 | # L'inventaire recoupe en partie Vrbo (même groupe) mais avec ses propres ids |
| 22 | 22 | # et des exclusivités hôtelières-résidentielles (apparts-hôtels, glamping). |
| 23 | 23 | # |
| 24 | −# Réglage env : LOUKA_EXPEDIA_LIMIT (nb max d'annonces, pour tester petit). | |
| 24 | +# Enrichissement : la page détail hXXXXXX.Hotel-Information (Scrapfly ASP | |
| 25 | +# SANS rendu JS — le SSR suffit) porte description, commodités, lat/lng et | |
| 26 | +# ~6 photos (parseur partagé avec Vrbo : _expediadetail.py). | |
| 27 | +# | |
| 28 | +# Réglages env : LOUKA_EXPEDIA_LIMIT (nb max d'annonces, pour tester petit), | |
| 29 | +# LOUKA_EXPEDIA_DETAIL_LIMIT (fetchs détail par sync, défaut | |
| 30 | +# 100 ; cache permanent, le parc se complète au fil des syncs). | |
| 25 | 31 | # ----------------------------------------------------------------------------- |
| 26 | 32 | from __future__ import annotations |
| 27 | 33 | |
@@ -33,8 +39,13 @@ from urllib.parse import quote | ||
| 33 | 39 | from bs4 import BeautifulSoup |
| 34 | 40 | |
| 35 | 41 | from ..schema import StListing |
| 42 | +from . import _expediadetail as _ed | |
| 36 | 43 | from .base import StConnector |
| 37 | 44 | |
| 45 | + | |
| 46 | +class _DetailSkip(Exception): | |
| 47 | + """Fiche détail sautée (budget épuisé / page invalide) — pas de cache.""" | |
| 48 | + | |
| 38 | 49 | # (destination Expedia, ville affichée par défaut, région touristique QC) |
| 39 | 50 | DESTINATIONS = [ |
| 40 | 51 | ("Mont-Tremblant, Quebec, Canada", "Mont-Tremblant", "Laurentides"), |
@@ -203,6 +214,39 @@ class Expedia(StConnector): | ||
| 203 | 214 | images=images, |
| 204 | 215 | ) |
| 205 | 216 | |
| 217 | + # -- enrichissement par la page détail --------------------------------------- | |
| 218 | + def _enrich_details(self, listings: list[StListing]) -> None: | |
| 219 | + """Visite les fiches détail via le cache self.detail() sous budget : | |
| 220 | + les hits de cache sont gratuits, seuls les fetchs réseau comptent.""" | |
| 221 | + limit = max(0, int(os.environ.get("LOUKA_EXPEDIA_DETAIL_LIMIT", "100") | |
| 222 | + or 100)) | |
| 223 | + used = enriched = streak = 0 | |
| 224 | + for lst in listings: | |
| 225 | + def fetch_fn(url=lst.url): | |
| 226 | + nonlocal used, streak | |
| 227 | + if used >= limit or streak >= 5: # tempête anti-bot : on coupe | |
| 228 | + raise _DetailSkip | |
| 229 | + used += 1 | |
| 230 | + html = self.get_scrapfly(url, render_js=False, asp=True) | |
| 231 | + payload = _ed.parse_detail(html) | |
| 232 | + if not payload: | |
| 233 | + streak += 1 | |
| 234 | + raise _DetailSkip # blocage/vide : pas de cache | |
| 235 | + streak = 0 | |
| 236 | + return payload | |
| 237 | + | |
| 238 | + try: | |
| 239 | + d = self.detail(lst.external_id, "v1", fetch_fn) | |
| 240 | + except _DetailSkip: | |
| 241 | + continue | |
| 242 | + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run | |
| 243 | + continue | |
| 244 | + if d: | |
| 245 | + _ed.apply_detail(lst, d) | |
| 246 | + enriched += 1 | |
| 247 | + print(f"[expedia] détail : {enriched} annonces enrichies" | |
| 248 | + f" ({used}/{limit} fetchs réseau)", file=sys.stderr) | |
| 249 | + | |
| 206 | 250 | # -- contrat ----------------------------------------------------------------- |
| 207 | 251 | def fetch(self) -> list[StListing]: |
| 208 | 252 | limit = int(os.environ.get("LOUKA_EXPEDIA_LIMIT", "0") or 0) |
@@ -216,7 +260,9 @@ class Expedia(StConnector): | ||
| 216 | 260 | for dest, city, region in DESTINATIONS: |
| 217 | 261 | for sort in SORTS: |
| 218 | 262 | if limit and len(listings) >= limit: |
| 219 | − return list(listings.values()) | |
| 263 | + out = list(listings.values()) | |
| 264 | + self._enrich_details(out) | |
| 265 | + return out | |
| 220 | 266 | url = ("https://www.expedia.ca/Hotel-Search?destination=" |
| 221 | 267 | + quote(dest) |
| 222 | 268 | + "&adults=2&categorySearch=vacation_rentals_option") |
@@ -250,4 +296,6 @@ class Expedia(StConnector): | ||
| 250 | 296 | print(f"[expedia] {city} ({sort}) :" |
| 251 | 297 | f" {len(listings) - n_before} nouvelles" |
| 252 | 298 | f" (total {len(listings)})", file=sys.stderr) |
| 253 | − return list(listings.values()) | |
| 299 | + out = list(listings.values()) | |
| 300 | + self._enrich_details(out) | |
| 301 | + return out | |
added
louka/shortterm/connectors/gitesauquebec.py
+218 −0
@@ -0,0 +1,218 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/gitesauquebec.py : GitesAuQuebec.com — annuaire indépendant de | |
| 4 | +# gîtes et auberges (B&B). Annuaire en fin de vie : l'inventaire ACTIF est | |
| 5 | +# aujourd'hui minuscule (~7 fiches, « Page 1 de 1 » sur la recherche sans | |
| 6 | +# critère), mais les fiches restantes sont riches et le connecteur suivra | |
| 7 | +# l'inventaire s'il remonte. | |
| 8 | +# | |
| 9 | +# Méthode (ASP.NET WebForms, aucun anti-bot) : | |
| 10 | +# 1. LISTE : GET /Resultats.aspx sans critère → toutes les annonces actives | |
| 11 | +# (cartes en tables imbriquées : lien /<id>, tarif « 145$ - 205$ nuit »). | |
| 12 | +# La pagination (« Page 1 de N ») est un postback __VIEWSTATE : tant que | |
| 13 | +# N = 1 on n'en a pas besoin ; si N > 1 un avertissement est émis (le | |
| 14 | +# rejouer n'a rien à répliquer aujourd'hui, l'annuaire tient sur 1 page). | |
| 15 | +# 2. FICHE /<id> (cache self.detail, clé mensuelle) : contrôles ASP.NET | |
| 16 | +# stables — titre « PL-<id> : Nom », CPH_litInfoGen (type, capacité, | |
| 17 | +# chambres, salles de bain, lits, animaux, fumeurs), localisation | |
| 18 | +# (région / ville), CPH_lblDesc, CPH_lblNoEtablissementValeur (CITQ), | |
| 19 | +# CPH_hidAdrMap (« lat, lng »), CPH_pnlTarif, équipements (img alt), | |
| 20 | +# photos /_photos/grand/. | |
| 21 | +# ----------------------------------------------------------------------------- | |
| 22 | +from __future__ import annotations | |
| 23 | + | |
| 24 | +import re | |
| 25 | +import sys | |
| 26 | +import time | |
| 27 | + | |
| 28 | +from ..schema import StListing, normalize_region | |
| 29 | +from .base import StConnector | |
| 30 | + | |
| 31 | +BASE = "https://www.gitesauquebec.com" | |
| 32 | +RESULTATS = f"{BASE}/Resultats.aspx" | |
| 33 | + | |
| 34 | +_ID_RE = re.compile(r"href='/(\d+)'") | |
| 35 | +_PAGE_RE = re.compile(r"Page\s+(\d+)\s+de\s+(\d+)") | |
| 36 | +_TARIF_RE = re.compile(r"(\d[\d\s,.]*\$[^<]{0,40})") | |
| 37 | +_COORD_RE = re.compile( | |
| 38 | + r'id="CPH_hidAdrMap" value="(-?\d+\.\d+),\s*(-?\d+\.\d+)"') | |
| 39 | +_NUM_RE = re.compile(r"(\d+)") | |
| 40 | + | |
| 41 | +# « Type hébergement » de l'annuaire → type canonique Lou-Ka | |
| 42 | +_TYPES = {"gîte": "Gîte", "gite": "Gîte", "auberge": "Auberge", | |
| 43 | + "b&b": "Gîte", "couette et café": "Gîte"} | |
| 44 | + | |
| 45 | +# l'annuaire écrit « Cantons de l'est / Estrie », « Centre du Québec »… | |
| 46 | +_REGIONS = {"cantons de l'est / estrie": "Cantons-de-l'Est", | |
| 47 | + "centre du québec": "Centre-du-Québec"} | |
| 48 | + | |
| 49 | + | |
| 50 | +class GitesAuQuebec(StConnector): | |
| 51 | + source_id = "gitesauquebec" | |
| 52 | + request_delay = 1.0 | |
| 53 | + | |
| 54 | + # -- liste ------------------------------------------------------------------ | |
| 55 | + def _liste(self) -> dict[str, str]: | |
| 56 | + """Annonces actives → {id: libellé de tarif de la carte}.""" | |
| 57 | + html = self.get(RESULTATS).text | |
| 58 | + m = _PAGE_RE.search(re.sub(r"<[^>]+>", " ", html)) | |
| 59 | + if m and int(m.group(2)) > 1: | |
| 60 | + print(f"[gitesauquebec] pagination inattendue ({m.group(0)}) : " | |
| 61 | + "seule la page 1 est lue (postback __VIEWSTATE à rejouer)", | |
| 62 | + file=sys.stderr) | |
| 63 | + tarifs: dict[str, str] = {} | |
| 64 | + # une carte = tout le HTML entre deux liens de fiche successifs | |
| 65 | + morceaux = _ID_RE.split(html) | |
| 66 | + for i in range(1, len(morceaux), 2): | |
| 67 | + gid, bloc = morceaux[i], morceaux[i + 1] | |
| 68 | + if gid in tarifs and tarifs[gid]: | |
| 69 | + continue | |
| 70 | + tarif = "" | |
| 71 | + j = bloc.find("TARIFICATION") | |
| 72 | + if j >= 0: | |
| 73 | + mm = _TARIF_RE.search(re.sub(r"<[^>]+>", " ", bloc[j:j + 800])) | |
| 74 | + if mm: | |
| 75 | + tarif = re.sub(r"\s+", " ", mm.group(1)).strip() | |
| 76 | + tarifs[gid] = tarif | |
| 77 | + return tarifs | |
| 78 | + | |
| 79 | + # -- fiche ------------------------------------------------------------------ | |
| 80 | + def _fetch_fiche(self, gid: str) -> dict: | |
| 81 | + from bs4 import BeautifulSoup | |
| 82 | + html = self.get(f"{BASE}/{gid}").text | |
| 83 | + soup = BeautifulSoup(html, "html.parser") | |
| 84 | + d: dict = {} | |
| 85 | + | |
| 86 | + # en-tête « PL-3778 : Gite du Village » (préfixe variable : PL, DI…) | |
| 87 | + h1 = soup.find(string=re.compile(rf"[A-Z]{{1,3}}-{gid}\s*:")) | |
| 88 | + if h1: | |
| 89 | + d["nom"] = h1.split(":", 1)[1].strip() | |
| 90 | + m = re.match(rf"([A-Z]{{1,3}}-{gid})", h1.strip()) | |
| 91 | + if m: | |
| 92 | + d["no_annonce"] = m.group(1) | |
| 93 | + if not d.get("nom"): | |
| 94 | + # repli : <title> = « Gîte <ville>, <région>, <nom…>, XX-<id> » | |
| 95 | + title = soup.title.get_text(strip=True) if soup.title else "" | |
| 96 | + m = re.match(rf"[^,]+,[^,]+,\s*(.+?),?\s*([A-Z]{{1,3}}-{gid})$", | |
| 97 | + title) | |
| 98 | + if m: | |
| 99 | + d["nom"], d["no_annonce"] = m.group(1).strip(), m.group(2) | |
| 100 | + if not d.get("nom"): | |
| 101 | + return {} | |
| 102 | + | |
| 103 | + # Informations générales : « Étiquette : Valeur » dans CPH_litInfoGen | |
| 104 | + infos: dict[str, str] = {} | |
| 105 | + bloc = soup.find(id="CPH_litInfoGen") | |
| 106 | + if bloc is not None: | |
| 107 | + for cell in bloc.get_text("\n").split("\n"): | |
| 108 | + label, sep, val = cell.partition(":") | |
| 109 | + if sep and val.strip(): | |
| 110 | + infos[label.strip().lower()] = val.strip() | |
| 111 | + d["type"] = infos.get("type hébergement", "") | |
| 112 | + m = _NUM_RE.search(infos.get("capacité d'accueil", "")) | |
| 113 | + if m: | |
| 114 | + d["capacity"] = int(m.group(1)) | |
| 115 | + m = _NUM_RE.search(infos.get("chambres", "")) | |
| 116 | + if m: | |
| 117 | + d["bedrooms"] = int(m.group(1)) | |
| 118 | + m = _NUM_RE.search(infos.get("salles de bain", "")) | |
| 119 | + if m: | |
| 120 | + d["bathrooms"] = int(m.group(1)) | |
| 121 | + d["lits"] = infos.get("lits", "") | |
| 122 | + animaux = infos.get("animaux permis", "").lower() | |
| 123 | + if animaux.startswith("oui"): | |
| 124 | + d["pets"] = "oui" | |
| 125 | + elif animaux.startswith("non"): | |
| 126 | + d["pets"] = "non" | |
| 127 | + | |
| 128 | + # Localisation : lignes « Région : / Ville : » du tableau | |
| 129 | + texte = soup.get_text("\n") | |
| 130 | + for champ, cle in (("Région", "region"), ("Ville", "ville")): | |
| 131 | + m = re.search(rf"{champ}\s*:\s*\n+\s*([^\n]+)", texte) | |
| 132 | + if m: | |
| 133 | + d[cle] = m.group(1).strip() | |
| 134 | + | |
| 135 | + desc = soup.find(id="CPH_lblDesc") | |
| 136 | + if desc is not None: | |
| 137 | + d["description"] = desc.get_text("\n", strip=True)[:2500] | |
| 138 | + | |
| 139 | + citq = soup.find(id="CPH_lblNoEtablissementValeur") | |
| 140 | + if citq is not None: | |
| 141 | + d["citq"] = citq.get_text(strip=True) | |
| 142 | + | |
| 143 | + m = _COORD_RE.search(html) | |
| 144 | + if m: | |
| 145 | + d["lat"], d["lng"] = float(m.group(1)), float(m.group(2)) | |
| 146 | + | |
| 147 | + tarif = soup.find(id="CPH_pnlTarif") | |
| 148 | + if tarif is not None: | |
| 149 | + d["tarif"] = re.sub( | |
| 150 | + r"\s+", " ", | |
| 151 | + tarif.get_text(" ", strip=True).removeprefix("Tarification") | |
| 152 | + ).strip()[:300] | |
| 153 | + | |
| 154 | + # équipements & activités : alt des pictogrammes | |
| 155 | + amen: list[str] = [] | |
| 156 | + for img in soup.find_all("img", src=re.compile("equipement")): | |
| 157 | + alt = (img.get("alt") or "").strip() | |
| 158 | + if alt and alt not in amen: | |
| 159 | + amen.append(alt) | |
| 160 | + d["amenities"] = amen[:40] | |
| 161 | + | |
| 162 | + imgs: list[str] = [] | |
| 163 | + for img in soup.find_all("img", src=re.compile(r"/_photos/")): | |
| 164 | + u = img.get("src") or "" | |
| 165 | + u = re.sub(r"/_photos/(thumb/)?", "/_photos/grand/", u) | |
| 166 | + if not u.startswith("http"): | |
| 167 | + u = BASE + u | |
| 168 | + if u not in imgs: | |
| 169 | + imgs.append(u) | |
| 170 | + d["images"] = imgs[:12] | |
| 171 | + return d | |
| 172 | + | |
| 173 | + # -- contrat ---------------------------------------------------------------- | |
| 174 | + def fetch(self) -> list[StListing]: | |
| 175 | + month = time.strftime("%Y-%m") # re-visite mensuelle (tarifs) | |
| 176 | + listings: list[StListing] = [] | |
| 177 | + for gid, tarif_carte in sorted(self._liste().items(), | |
| 178 | + key=lambda kv: int(kv[0])): | |
| 179 | + try: | |
| 180 | + d = self.detail(gid, month, | |
| 181 | + lambda g=gid: self._fetch_fiche(g)) | |
| 182 | + except Exception as exc: # noqa: BLE001 | |
| 183 | + print(f"[gitesauquebec] fiche {gid} : {exc}", file=sys.stderr) | |
| 184 | + continue | |
| 185 | + if not d.get("nom"): | |
| 186 | + continue | |
| 187 | + | |
| 188 | + price_label = d.get("tarif") or tarif_carte | |
| 189 | + region = d.get("region", "") | |
| 190 | + region = _REGIONS.get(region.lower(), normalize_region(region)) | |
| 191 | + details = {k: v for k, v in { | |
| 192 | + "no_annonce": d.get("no_annonce", f"PL-{gid}"), | |
| 193 | + "lits": d.get("lits", ""), | |
| 194 | + }.items() if v} | |
| 195 | + | |
| 196 | + listings.append(StListing( | |
| 197 | + source=self.source_id, | |
| 198 | + external_id=gid, | |
| 199 | + url=f"{BASE}/{gid}", | |
| 200 | + title=d["nom"], | |
| 201 | + property_type=_TYPES.get(d.get("type", "").lower(), "Gîte"), | |
| 202 | + city=d.get("ville", ""), | |
| 203 | + region=region, | |
| 204 | + price_label=price_label, | |
| 205 | + capacity=float(d["capacity"]) if d.get("capacity") else None, | |
| 206 | + bedrooms=float(d["bedrooms"]) if d.get("bedrooms") else None, | |
| 207 | + bathrooms=(float(d["bathrooms"]) | |
| 208 | + if d.get("bathrooms") else None), | |
| 209 | + pets=d.get("pets"), | |
| 210 | + citq=d.get("citq", ""), | |
| 211 | + description=d.get("description", ""), | |
| 212 | + amenities=d.get("amenities") or [], | |
| 213 | + details=details, | |
| 214 | + images=d.get("images") or [], | |
| 215 | + lat=d.get("lat"), | |
| 216 | + lng=d.get("lng"), | |
| 217 | + )) | |
| 218 | + return listings | |
modified
louka/shortterm/connectors/gitespassant.py
+29 −4
@@ -17,7 +17,10 @@ | ||
| 17 | 17 | # /fr/repertoire/detailorganization/id/<id> — HTML serveur : nom (h1), |
| 18 | 18 | # description, « Types d'hébergement » (Gîte/Auberge du Passant, Maison |
| 19 | 19 | # de Campagne…), numéro CITQ (#attestation), classification (soleils), |
| 20 | −# adresse postale (spans), téléphone, site web, photos du carrousel. | |
| 20 | +# adresse postale (spans), téléphone, site web, photos du carrousel, | |
| 21 | +# coordonnées lat/lng extraites de l'iframe Google Maps (paramètres | |
| 22 | +# !2d<lng>!3d<lat> de l'embed — présent sur toutes les fiches), | |
| 23 | +# chambres/capacité glanées dans la description quand mentionnées. | |
| 21 | 24 | # Pas de prix affiché (les tarifs sont chez chaque établissement). |
| 22 | 25 | # ----------------------------------------------------------------------------- |
| 23 | 26 | from __future__ import annotations |
@@ -69,6 +72,11 @@ _TYPES = [ | ||
| 69 | 72 | ] |
| 70 | 73 | |
| 71 | 74 | _TAG_RE = re.compile(r"<[^>]+>") |
| 75 | +# iframe Google Maps : …/maps/embed?pb=…!2d<longitude>!3d<latitude>!… | |
| 76 | +_GMAP_RE = re.compile(r"google\.com/maps/embed\?pb=[^\"']*" | |
| 77 | + r"!2d(-?\d+\.\d+)!3d(-?\d+\.\d+)") | |
| 78 | +_CAP_RE = re.compile(r"(\d{1,2})\s*personnes") | |
| 79 | +_CHAMBRES_RE = re.compile(r"(\d{1,2})\s*chambres") | |
| 72 | 80 | |
| 73 | 81 | |
| 74 | 82 | def _text(fragment: str) -> str: |
@@ -149,6 +157,12 @@ class GitesPassant(StConnector): | ||
| 149 | 157 | if m: |
| 150 | 158 | d["website"] = m.group(1) |
| 151 | 159 | |
| 160 | + # géolocalisation : centre de l'iframe Google Maps (brut : l'iframe | |
| 161 | + # est parfois injectée par un <script> retiré de `h`) | |
| 162 | + m = _GMAP_RE.search(brut) | |
| 163 | + if m: | |
| 164 | + d["lng"], d["lat"] = float(m.group(1)), float(m.group(2)) | |
| 165 | + | |
| 152 | 166 | # photos : le carrousel de la fiche seulement (ailleurs sur la page, |
| 153 | 167 | # les « Suggestions d'articles » ont aussi des images membres) |
| 154 | 168 | logo = "" |
@@ -183,8 +197,9 @@ class GitesPassant(StConnector): | ||
| 183 | 197 | |
| 184 | 198 | listings: list[StListing] = [] |
| 185 | 199 | for mid in ids: |
| 186 | − # fiche rafraîchie une fois par mois (pas de lastmod côté liste) | |
| 187 | − det = self.detail(mid, time.strftime("%Y-%m"), | |
| 200 | + # fiche rafraîchie une fois par mois (pas de lastmod côté liste) ; | |
| 201 | + # v2 : parseur géo (iframe Google Maps) ajouté 2026-08-25 | |
| 202 | + det = self.detail(mid, time.strftime("%Y-%m") + "-v2", | |
| 188 | 203 | lambda i=mid: self._fiche(i)) |
| 189 | 204 | if not det.get("title"): |
| 190 | 205 | continue |
@@ -203,6 +218,12 @@ class GitesPassant(StConnector): | ||
| 203 | 218 | "(Terroir et Saveurs du Québec)", |
| 204 | 219 | }.items() if v} |
| 205 | 220 | |
| 221 | + # capacité / chambres : mentions dans la description (rare) | |
| 222 | + desc = det.get("description") or "" | |
| 223 | + caps = [int(x) for x in _CAP_RE.findall(desc) if 1 <= int(x) <= 40] | |
| 224 | + rooms = [int(x) for x in _CHAMBRES_RE.findall(desc) | |
| 225 | + if 1 <= int(x) <= 30] | |
| 226 | + | |
| 206 | 227 | street = det.get("street") or "" |
| 207 | 228 | city = det.get("city") or "" |
| 208 | 229 | listings.append(StListing( |
@@ -214,10 +235,14 @@ class GitesPassant(StConnector): | ||
| 214 | 235 | address=", ".join(p for p in (street, city) if p), |
| 215 | 236 | city=city, |
| 216 | 237 | region=regions.get(mid, ""), |
| 238 | + capacity=float(max(caps)) if caps else None, | |
| 239 | + bedrooms=float(max(rooms)) if rooms else None, | |
| 217 | 240 | citq=det.get("citq") or "", |
| 218 | − description=det.get("description") or "", | |
| 241 | + description=desc, | |
| 219 | 242 | details=details, |
| 220 | 243 | images=det.get("images") or [], |
| 244 | + lat=det.get("lat"), | |
| 245 | + lng=det.get("lng"), | |
| 221 | 246 | )) |
| 222 | 247 | if limit and len(listings) >= limit: |
| 223 | 248 | break |
modified
louka/shortterm/connectors/glampinghub.py
+71 −2
@@ -11,20 +11,36 @@ | ||
| 11 | 11 | # catégorie, ville + coordonnées, prix (estimated_rate.daily_rate en devise |
| 12 | 12 | # originale, CAD au Québec), chambres/lits, capacité par unité |
| 13 | 13 | # (units_distribution), note sur 5, commodités (nested_features), photos. |
| 14 | −# Une seule requête paginée suffit — pas de page détail nécessaire. | |
| 14 | +# Une seule requête paginée suffit pour la liste ; seule la DESCRIPTION | |
| 15 | +# n'y figure pas → elle est prise sur la page détail (aucun anti-bot non | |
| 16 | +# plus), dans le bloc SSR <noscript id="description-content">, avec repli | |
| 17 | +# sur la <meta name="description">. Cache permanent dans louka_ct.db. | |
| 15 | 18 | # |
| 16 | 19 | # URL publique : https://glampinghub.com<absolute_url_en>. Région touristique |
| 17 | 20 | # déduite des coordonnées (centroïdes partagés avec airbnb.py). |
| 18 | −# Réglage env : LOUKA_GLAMPINGHUB_LIMIT (nb max de fiches, 0 = tout). | |
| 21 | +# Réglages env : LOUKA_GLAMPINGHUB_LIMIT (nb max de fiches, 0 = tout), | |
| 22 | +# LOUKA_GLAMPINGHUB_DETAIL_LIMIT (fetchs détail/sync, défaut 250). | |
| 19 | 23 | # ----------------------------------------------------------------------------- |
| 20 | 24 | from __future__ import annotations |
| 21 | 25 | |
| 26 | +import html as _html | |
| 22 | 27 | import os |
| 28 | +import re | |
| 29 | +import sys | |
| 23 | 30 | |
| 24 | 31 | from ..schema import StListing |
| 25 | 32 | from .base import StConnector |
| 26 | 33 | from .airbnb import _region_from_latlng |
| 27 | 34 | |
| 35 | + | |
| 36 | +class _DetailSkip(Exception): | |
| 37 | + """Fiche détail sautée (budget épuisé) — pas de mise en cache.""" | |
| 38 | + | |
| 39 | + | |
| 40 | +_DESC_RE = re.compile(r'<noscript id="description-content">(.*?)</noscript>', | |
| 41 | + re.S) | |
| 42 | +_META_RE = re.compile(r'<meta name="description" content="([^"]+)"') | |
| 43 | + | |
| 28 | 44 | SITE = "https://glampinghub.com" |
| 29 | 45 | API = f"{SITE}/search-accommodations/" |
| 30 | 46 | PAGE_SIZE = 24 |
@@ -87,6 +103,58 @@ class GlampingHub(StConnector): | ||
| 87 | 103 | page += 1 |
| 88 | 104 | return items |
| 89 | 105 | |
| 106 | + # -- description (page détail, SSR ouvert) -------------------------------- | |
| 107 | + @staticmethod | |
| 108 | + def _parse_description(html: str) -> dict: | |
| 109 | + """{description} depuis le bloc <noscript> SSR (repli : meta).""" | |
| 110 | + desc = "" | |
| 111 | + m = _DESC_RE.search(html or "") | |
| 112 | + if m: | |
| 113 | + txt = re.sub(r"<br\s*/?>|</p>|</h2>", "\n", m.group(1)) | |
| 114 | + txt = _html.unescape(re.sub(r"<[^>]+>", " ", txt)) | |
| 115 | + lines = [re.sub(r"\s+", " ", ln).strip() for ln in txt.split("\n")] | |
| 116 | + # écarter le bruit : lignes « … », slogan SEO de bas de bloc et | |
| 117 | + # liste de commodités hors-plateforme (pas une description) | |
| 118 | + lines = [ln for ln in lines if ln and ln not in ("...", "…") | |
| 119 | + and not re.match(r"^Book your dream .*!$", ln) | |
| 120 | + and not ln.startswith("Amenities not shown on")] | |
| 121 | + desc = "\n".join(lines).strip() | |
| 122 | + if len(desc) < 40: # fiche laconique (« … ») : repli meta | |
| 123 | + mm = _META_RE.search(html or "") | |
| 124 | + meta = _html.unescape(mm.group(1)).strip() if mm else "" | |
| 125 | + if len(meta) > len(desc): | |
| 126 | + desc = meta | |
| 127 | + return {"description": desc[:6000]} if desc else {} | |
| 128 | + | |
| 129 | + def _enrich_details(self, listings: list[StListing]) -> None: | |
| 130 | + """Complète la description via la page détail, sous budget (les hits | |
| 131 | + de cache sont gratuits, seuls les fetchs réseau comptent).""" | |
| 132 | + limit = max(0, int(os.environ.get("LOUKA_GLAMPINGHUB_DETAIL_LIMIT", | |
| 133 | + "250") or 250)) | |
| 134 | + used = enriched = 0 | |
| 135 | + for lst in listings: | |
| 136 | + if not lst.url.startswith(SITE + "/"): | |
| 137 | + continue | |
| 138 | + def fetch_fn(url=lst.url): | |
| 139 | + nonlocal used | |
| 140 | + if used >= limit: | |
| 141 | + raise _DetailSkip | |
| 142 | + used += 1 | |
| 143 | + return self._parse_description(self.get(url).text) | |
| 144 | + | |
| 145 | + try: | |
| 146 | + d = self.detail(lst.external_id, "v1", fetch_fn) | |
| 147 | + except _DetailSkip: | |
| 148 | + continue | |
| 149 | + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run | |
| 150 | + continue | |
| 151 | + if d.get("description") and len(d["description"]) > \ | |
| 152 | + len(lst.description or ""): | |
| 153 | + lst.description = d["description"] | |
| 154 | + enriched += 1 | |
| 155 | + print(f"[glampinghub] détail : {enriched} descriptions" | |
| 156 | + f" ({used}/{limit} fetchs réseau)", file=sys.stderr) | |
| 157 | + | |
| 90 | 158 | # -- contrat -------------------------------------------------------------- |
| 91 | 159 | def fetch(self) -> list[StListing]: |
| 92 | 160 | limit = int(os.environ.get("LOUKA_GLAMPINGHUB_LIMIT", "0") or 0) |
@@ -174,4 +242,5 @@ class GlampingHub(StConnector): | ||
| 174 | 242 | )) |
| 175 | 243 | if limit and len(listings) >= limit: |
| 176 | 244 | break |
| 245 | + self._enrich_details(listings) | |
| 177 | 246 | return listings |
added
louka/shortterm/connectors/hebergementcharlevoix.py
+164 −0
@@ -0,0 +1,164 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/hebergementcharlevoix.py : Hébergement Charlevoix | |
| 4 | +# (hebergement-charlevoix.com) — agence de Charlevoix, ~330 chalets. | |
| 5 | +# | |
| 6 | +# Méthode : site PHP maison, HTML 100 % serveur. Sitemap /sitemap.xml | |
| 7 | +# (301 vers le domaine sans www) → URLs détail | |
| 8 | +# /fr/chalet-a-louer/<ville>/voir/<CODE> (code alphanum stable, ex. ADE-440 ; | |
| 9 | +# les autres URLs du sitemap sont des pages de catégories/villes/activités). | |
| 10 | +# Chaque page détail porte un JSON-LD schema.org VacationRental complet : | |
| 11 | +# occupancy, containsPlace (chambres, sdb, lits par type), petsAllowed, | |
| 12 | +# address (ville, rue, code postal), geo (⚠️ décimales à VIRGULE), | |
| 13 | +# offers.lowPrice (plus bas prix/nuit), images 1080×1080 et licenseNum CITQ. | |
| 14 | +# Les commodités lisibles (FR) viennent des pictos <div class="bulle">. | |
| 15 | +# La note affichée (4,6/1370) est GLOBALE au site : on l'ignore. | |
| 16 | +# Pas de lastmod : clé de cache mensuelle. | |
| 17 | +# ----------------------------------------------------------------------------- | |
| 18 | +from __future__ import annotations | |
| 19 | + | |
| 20 | +import json | |
| 21 | +import re | |
| 22 | +import time | |
| 23 | + | |
| 24 | +from ..schema import StListing | |
| 25 | +from .base import StConnector | |
| 26 | + | |
| 27 | +BASE = "https://hebergement-charlevoix.com" | |
| 28 | +SITEMAP = BASE + "/sitemap.xml" | |
| 29 | + | |
| 30 | +_URL_DETAIL = re.compile( | |
| 31 | + r"https://(?:www\.)?hebergement-charlevoix\.com/fr/chalet-a-louer/" | |
| 32 | + r"([\w-]+)/voir/([\w-]+)/?$") | |
| 33 | + | |
| 34 | +# villes desservies hors de la région touristique de Charlevoix | |
| 35 | +_HORS_CHARLEVOIX = { | |
| 36 | + "tadoussac": "Côte-Nord", | |
| 37 | + "saint-ferreol-les-neiges": "Québec", | |
| 38 | + "saint-tite-des-caps": "Québec", | |
| 39 | +} | |
| 40 | + | |
| 41 | + | |
| 42 | +def _virgule(v) -> float | None: | |
| 43 | + """Nombre JSON-LD du site : « 47,568534 » (virgule décimale).""" | |
| 44 | + try: | |
| 45 | + return float(str(v).replace(",", ".")) | |
| 46 | + except (TypeError, ValueError): | |
| 47 | + return None | |
| 48 | + | |
| 49 | + | |
| 50 | +class HebergementCharlevoix(StConnector): | |
| 51 | + source_id = "hebergementcharlevoix" | |
| 52 | + | |
| 53 | + # -- inventaire (sitemap) ---------------------------------------------- | |
| 54 | + def _sitemap_urls(self) -> dict[str, tuple[str, str]]: | |
| 55 | + """code -> (url détail, slug ville).""" | |
| 56 | + xml = self.get(SITEMAP).text | |
| 57 | + urls: dict[str, tuple[str, str]] = {} | |
| 58 | + for loc in re.findall(r"<loc>([^<]+)</loc>", xml): | |
| 59 | + m = _URL_DETAIL.match(loc.strip()) | |
| 60 | + if m: | |
| 61 | + urls.setdefault(m.group(2), (loc.strip(), m.group(1))) | |
| 62 | + return urls | |
| 63 | + | |
| 64 | + # -- page détail (tout est dans le JSON-LD VacationRental) --------------- | |
| 65 | + def _detail(self, url: str) -> dict: | |
| 66 | + h = self.get(url).text | |
| 67 | + d: dict = {} | |
| 68 | + data = None | |
| 69 | + for m in re.finditer(r'<script[^>]*application/ld\+json[^>]*>(.*?)' | |
| 70 | + r"</script>", h, re.S): | |
| 71 | + try: | |
| 72 | + cand = json.loads(m.group(1), strict=False) | |
| 73 | + except ValueError: | |
| 74 | + continue | |
| 75 | + if isinstance(cand, dict) and cand.get("@type") == "VacationRental": | |
| 76 | + data = cand | |
| 77 | + break | |
| 78 | + if data: | |
| 79 | + d["title"] = (data.get("name") or "").strip() | |
| 80 | + d["description"] = (data.get("description") or "").strip() | |
| 81 | + d["property_type"] = (data.get("additionalType") or "").strip() | |
| 82 | + occ = data.get("occupancy") or {} | |
| 83 | + if occ.get("value") is not None: | |
| 84 | + d["capacity"] = _virgule(occ["value"]) | |
| 85 | + place = data.get("containsPlace") or {} | |
| 86 | + if place.get("numberOfBedrooms") is not None: | |
| 87 | + d["bedrooms"] = _virgule(place["numberOfBedrooms"]) | |
| 88 | + if place.get("numberOfBathroomsTotal") is not None: | |
| 89 | + d["bathrooms"] = _virgule(place["numberOfBathroomsTotal"]) | |
| 90 | + beds = sum(_virgule(b.get("numberOfBeds")) or 0 | |
| 91 | + for b in place.get("bed") or [] if isinstance(b, dict)) | |
| 92 | + if beds: | |
| 93 | + d["beds"] = beds | |
| 94 | + if place.get("petsAllowed") is not None: | |
| 95 | + d["pets"] = "oui" if place["petsAllowed"] in ( | |
| 96 | + True, "true", "True", 1) else "non" | |
| 97 | + addr = data.get("address") or {} | |
| 98 | + d["city"] = (addr.get("addressLocality") or "").strip() | |
| 99 | + d["address"] = (addr.get("streetAddress") or "").strip() | |
| 100 | + geo = data.get("geo") or {} | |
| 101 | + d["lat"] = _virgule(geo.get("latitude")) | |
| 102 | + d["lng"] = _virgule(geo.get("longitude")) | |
| 103 | + offers = data.get("offers") or {} | |
| 104 | + low = _virgule(offers.get("lowPrice")) | |
| 105 | + if low: | |
| 106 | + d["price_night"] = round(low, 2) | |
| 107 | + d["price_label"] = f"à partir de {low:.2f} $ / nuit" | |
| 108 | + for feat in data.get("amenityFeature") or []: | |
| 109 | + if not isinstance(feat, dict): | |
| 110 | + continue | |
| 111 | + if feat.get("name") == "licenseNum": | |
| 112 | + m2 = re.search(r"(\d{6})", str(feat.get("value") or "")) | |
| 113 | + if m2 and m2.group(1) != "000000": # placeholder du site | |
| 114 | + d["citq"] = m2.group(1) | |
| 115 | + imgs = data.get("image") or [] | |
| 116 | + if isinstance(imgs, str): | |
| 117 | + imgs = [imgs] | |
| 118 | + d["images"] = [u for u in imgs if isinstance(u, str)][:20] | |
| 119 | + | |
| 120 | + # commodités lisibles : pictos <div class="bulle">Foyer</div> | |
| 121 | + amen = [] | |
| 122 | + for a in re.findall(r'<div class="bulle"[^>]*>([^<]+)</div>', h): | |
| 123 | + a = re.sub(r"\s+", " ", a).strip() | |
| 124 | + if a and a not in amen: | |
| 125 | + amen.append(a) | |
| 126 | + if amen: | |
| 127 | + d["amenities"] = amen | |
| 128 | + return d | |
| 129 | + | |
| 130 | + # -- contrat -------------------------------------------------------------- | |
| 131 | + def fetch(self) -> list[StListing]: | |
| 132 | + cle = "detail-" + time.strftime("%Y-%m") # pas de lastmod → mensuel | |
| 133 | + listings: list[StListing] = [] | |
| 134 | + for code, (url, ville_slug) in self._sitemap_urls().items(): | |
| 135 | + try: | |
| 136 | + d = self.detail(code, cle, lambda u=url: self._detail(u)) | |
| 137 | + except Exception: # une fiche cassée ≠ inventaire perdu | |
| 138 | + d = {} | |
| 139 | + if not d.get("title"): | |
| 140 | + continue | |
| 141 | + listings.append(StListing( | |
| 142 | + source=self.source_id, | |
| 143 | + external_id=code, # code interne (ADE-440), stable | |
| 144 | + url=url, | |
| 145 | + title=d["title"], | |
| 146 | + property_type=d.get("property_type") or "Chalet", | |
| 147 | + address=d.get("address", ""), | |
| 148 | + city=d.get("city", ""), | |
| 149 | + region=_HORS_CHARLEVOIX.get(ville_slug, "Charlevoix"), | |
| 150 | + price_night=d.get("price_night"), | |
| 151 | + price_label=d.get("price_label", ""), | |
| 152 | + capacity=d.get("capacity"), | |
| 153 | + bedrooms=d.get("bedrooms"), | |
| 154 | + beds=d.get("beds"), | |
| 155 | + bathrooms=d.get("bathrooms"), | |
| 156 | + pets=d.get("pets"), | |
| 157 | + citq=d.get("citq", ""), | |
| 158 | + description=d.get("description", ""), | |
| 159 | + amenities=d.get("amenities") or [], | |
| 160 | + images=d.get("images") or [], | |
| 161 | + lat=d.get("lat"), | |
| 162 | + lng=d.get("lng"), | |
| 163 | + )) | |
| 164 | + return listings | |
added
louka/shortterm/connectors/hebergia.py
+164 −0
@@ -0,0 +1,164 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/hebergia.py : Hébergia (hebergia.ca) | |
| 4 | +# | |
| 5 | +# Agence multi-régions (Cantons-de-l'Est, Laurentides, Lanaudière, Charlevoix, | |
| 6 | +# Centre-du-Québec) — WordPress + moteur Guesty, ~60 chalets. | |
| 7 | +# | |
| 8 | +# Méthode : | |
| 9 | +# 1. LISTE : la page /chalets/ embarque `window.chalets = [...]` (JSON | |
| 10 | +# complet : gid Guesty, prix de base/nuit, url, titre, RÉGION, capacité, | |
| 11 | +# chambres, lits, lat/lng, vignette). Une seule requête. | |
| 12 | +# 2. DÉTAIL (cache self.detail, clé = champs stables de la liste) : la page | |
| 13 | +# WP fr /chalets/<slug>/ fournit ville (« Austin, Cantons-de-l'Est »), | |
| 14 | +# salles de bain (bloc inshort), description, commodités (section | |
| 15 | +# « Commodités incluses »), photos (assets.guesty.com) et numéro CITQ | |
| 16 | +# (dans le règlement). | |
| 17 | +# External_id = gid (id de listing Guesty, stable). | |
| 18 | +# ----------------------------------------------------------------------------- | |
| 19 | +from __future__ import annotations | |
| 20 | + | |
| 21 | +import html as _html | |
| 22 | +import json | |
| 23 | +import re | |
| 24 | + | |
| 25 | +from ..schema import StListing | |
| 26 | +from .base import StConnector | |
| 27 | + | |
| 28 | +SITE = "https://hebergia.ca" | |
| 29 | + | |
| 30 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 31 | + | |
| 32 | + | |
| 33 | +def _text(fragment: str) -> str: | |
| 34 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 35 | + | |
| 36 | + | |
| 37 | +def _num(v) -> float | None: | |
| 38 | + try: | |
| 39 | + return float(v) if v not in (None, "") else None | |
| 40 | + except (TypeError, ValueError): | |
| 41 | + return None | |
| 42 | + | |
| 43 | + | |
| 44 | +class Hebergia(StConnector): | |
| 45 | + source_id = "hebergia" | |
| 46 | + | |
| 47 | + # -- liste ---------------------------------------------------------------- | |
| 48 | + def _list_items(self) -> list[dict]: | |
| 49 | + h = self.get(f"{SITE}/chalets/").text | |
| 50 | + m = re.search(r"window\.chalets\s*=\s*(\[.*?\]);", h, re.S) | |
| 51 | + if not m: | |
| 52 | + return [] | |
| 53 | + try: | |
| 54 | + return json.loads(m.group(1)) | |
| 55 | + except ValueError: | |
| 56 | + return [] | |
| 57 | + | |
| 58 | + # -- page détail ------------------------------------------------------ | |
| 59 | + def _detail(self, url: str) -> dict: | |
| 60 | + h = self.get(url).text | |
| 61 | + d: dict = {} | |
| 62 | + | |
| 63 | + m = re.search(r'class="sous-titre-localisation">([^<]+)<', h) | |
| 64 | + if m: | |
| 65 | + loc = _text(m.group(1)) | |
| 66 | + d["city"] = loc.split(",")[0].strip() | |
| 67 | + | |
| 68 | + # bloc « inshort » : 8 voyageurs / 3 chambres / lits / 2 salles de bain | |
| 69 | + m = re.search(r"(\d+)\s*salles?\s*de\s*bain", h) | |
| 70 | + if m: | |
| 71 | + d["bathrooms"] = float(m.group(1)) | |
| 72 | + | |
| 73 | + # description : paragraphes de l'article avant les accordéons | |
| 74 | + m = re.search(r"(?s)<article class=\"hbg-fiche-chalet-content\">" | |
| 75 | + r".*?</h1>(.*?)<h2 class=\"toggler\"", h) | |
| 76 | + if m: | |
| 77 | + frag = re.sub(r'(?s)<p class="sous-titre-localisation">.*?</p>', | |
| 78 | + " ", m.group(1)) | |
| 79 | + paras = re.findall(r"(?s)<p[^>]*>(.*?)</p>", frag) | |
| 80 | + d["description"] = " ".join(_text(p) for p in paras | |
| 81 | + if _text(p))[:4000] | |
| 82 | + | |
| 83 | + # commodités : section « Commodités incluses » (lignes à puces) | |
| 84 | + m = re.search(r'(?s)Commodités incluses</button></h2>\s*' | |
| 85 | + r'<div class="toToggle"[^>]*>(.*?)</div>', h) | |
| 86 | + if m: | |
| 87 | + amens = [] | |
| 88 | + for line in re.split(r"<br\s*/?>|</p>", m.group(1)): | |
| 89 | + t = _text(line).lstrip("•· ").strip() | |
| 90 | + if 2 <= len(t) <= 80 and not t.isupper(): | |
| 91 | + amens.append(t) | |
| 92 | + d["amenities"] = amens[:60] | |
| 93 | + | |
| 94 | + # photos (CDN Guesty, dédupliquées) | |
| 95 | + imgs: list[str] = [] | |
| 96 | + for u in re.findall(r'data-flickity-lazyload-src="' | |
| 97 | + r'(https://assets\.guesty\.com/[^"]+)"', h): | |
| 98 | + if u not in imgs: | |
| 99 | + imgs.append(u) | |
| 100 | + d["images"] = imgs[:20] | |
| 101 | + | |
| 102 | + m = re.search(r"CITQ\D{0,12}(\d{6})", h, re.I) | |
| 103 | + if m: | |
| 104 | + d["citq"] = m.group(1) | |
| 105 | + | |
| 106 | + m = re.search(r"animaux non admis|pas d.animaux|no pets", h, re.I) | |
| 107 | + if m: | |
| 108 | + d["pets"] = "non" | |
| 109 | + return d | |
| 110 | + | |
| 111 | + # -- contrat ---------------------------------------------------------- | |
| 112 | + def fetch(self) -> list[StListing]: | |
| 113 | + listings: list[StListing] = [] | |
| 114 | + for it in self._list_items(): | |
| 115 | + gid = str(it.get("gid") or "").strip() | |
| 116 | + url = it.get("url") or "" | |
| 117 | + title = _text(str(it.get("title") or "")) | |
| 118 | + if not gid or not url or not title or "/en/" in url: | |
| 119 | + continue | |
| 120 | + | |
| 121 | + key = json.dumps([gid, title, it.get("price"), | |
| 122 | + it.get("accommodates"), it.get("bedrooms"), | |
| 123 | + it.get("beds"), it.get("shortDesc")], | |
| 124 | + ensure_ascii=False) | |
| 125 | + try: | |
| 126 | + det = self.detail(gid, key, lambda u=url: self._detail(u)) | |
| 127 | + except Exception: # une fiche détail cassée ≠ annonce perdue | |
| 128 | + det = {} | |
| 129 | + | |
| 130 | + price = _num(it.get("price")) | |
| 131 | + starting = _num(it.get("startingPrice")) | |
| 132 | + if starting and starting > 0: | |
| 133 | + price = starting | |
| 134 | + desc = det.get("description") or "" | |
| 135 | + short = _text(str(it.get("shortDesc") or "")) | |
| 136 | + if short and short not in desc: | |
| 137 | + desc = f"{short} {desc}".strip() | |
| 138 | + | |
| 139 | + listings.append(StListing( | |
| 140 | + source=self.source_id, | |
| 141 | + external_id=gid, | |
| 142 | + url=url, | |
| 143 | + title=title, | |
| 144 | + property_type="Chalet", | |
| 145 | + city=det.get("city") or "", | |
| 146 | + region=it.get("region") or "", | |
| 147 | + price_night=price, | |
| 148 | + price_label=(f"à partir de {price:.0f} $ / nuit" | |
| 149 | + if price else ""), | |
| 150 | + capacity=_num(it.get("accommodates")), | |
| 151 | + bedrooms=_num(it.get("bedrooms")), | |
| 152 | + beds=_num(it.get("beds")), | |
| 153 | + bathrooms=det.get("bathrooms"), | |
| 154 | + pets=det.get("pets"), | |
| 155 | + citq=det.get("citq") or "", | |
| 156 | + description=desc[:4000], | |
| 157 | + amenities=det.get("amenities") or [], | |
| 158 | + details={"guesty_id": gid}, | |
| 159 | + images=det.get("images") | |
| 160 | + or ([it["thumbnail"]] if it.get("thumbnail") else []), | |
| 161 | + lat=_num(it.get("lat")), | |
| 162 | + lng=_num(it.get("lng")), | |
| 163 | + )) | |
| 164 | + return listings | |
added
louka/shortterm/connectors/homminichalets.py
+142 −0
@@ -0,0 +1,142 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/homminichalets.py : HOM Mini Chalets (homminichalets.com) — | |
| 4 | +# 12 mini-chalets numérotés (01. Le Renard … 12. Le Loup) à Val-des-Monts | |
| 5 | +# (Outaouais), en deux gammes : « avec spa » (01-08) et « avec circuit | |
| 6 | +# thermal » (09-12, spa + sauna + hammam + douche froide). | |
| 7 | +# | |
| 8 | +# Méthode : le site Shopify renvoie un 429 systématique en direct → Scrapfly. | |
| 9 | +# 1. UIDS : /pages/chalets (rendu JS, le calendrier Hostfully est monté en | |
| 10 | +# JS) → 12 liens /pages/chalets?uid=<uuid> ; la page liste aussi les noms | |
| 11 | +# groupés par gamme (« Mini chalets avec circuit thermal » précède | |
| 12 | +# 09-12) et le pied de page porte les 2 adresses + le CITQ commun 298559. | |
| 13 | +# 2. DÉTAIL : l'API publique JSONP du moteur Hostfully répond en DIRECT | |
| 14 | +# (pas de 429) : platform.hostfully.com/getproperty_api.jsp?propertyUID=… | |
| 15 | +# &aid=ORB-… → {"price": 300, "maximumGuests": 2, "minStay": 1, | |
| 16 | +# "name": "07. Le Huard"}. Pas d'endpoint photos public → galerie | |
| 17 | +# GÉNÉRIQUE du site (01.jpg-25.jpg de l'accueil, via Scrapfly sans | |
| 18 | +# render_js), signalée par details.images_generic. Commodités/description | |
| 19 | +# par gamme (bandeaux de l'accueil), adresses par numéro d'unité | |
| 20 | +# (07-08 = chemin du Saphir, le reste = chemin du Rubis). | |
| 21 | +# ----------------------------------------------------------------------------- | |
| 22 | +from __future__ import annotations | |
| 23 | + | |
| 24 | +import json | |
| 25 | +import re | |
| 26 | +import sys | |
| 27 | + | |
| 28 | +from ..schema import StListing | |
| 29 | +from .base import StConnector | |
| 30 | + | |
| 31 | +SITE = "https://homminichalets.com" | |
| 32 | +LIST_URL = SITE + "/pages/chalets" | |
| 33 | + | |
| 34 | +# API publique du widget de réservation Hostfully (accessible en direct) | |
| 35 | +AID = "ORB-49587220416635719" | |
| 36 | +API = ("https://platform.hostfully.com/getproperty_api.jsp" | |
| 37 | + "?jsoncallback=cb&propertyUID={uid}&aid=" + AID) | |
| 38 | + | |
| 39 | +CITQ = "298559" | |
| 40 | + | |
| 41 | +# gammes (bandeaux de l'accueil) : commodités + description | |
| 42 | +_SPA_AMEN = ["Spa privé", "Chaise hamac", "Lit King", | |
| 43 | + "Foyer intérieur", "Plancher chauffant"] | |
| 44 | +_SPA_DESC = ("Mini chalet avec spa. Profitez d'un mini chalet de luxe tout " | |
| 45 | + "équipé avec spa privé sur la galerie, foyer intérieur et " | |
| 46 | + "plancher chauffant.") | |
| 47 | +_THERMAL_AMEN = ["Spa", "Sauna", "Hammam", "Douche froide", "Lit Queen", | |
| 48 | + "Foyer intérieur", "Plancher chauffant"] | |
| 49 | +_THERMAL_DESC = ("Mini chalet avec circuit thermal (spa, sauna, hammam, " | |
| 50 | + "douche froide). Vivez une expérience de détente privée et " | |
| 51 | + "luxueuse dans un mini chalet tout équipé avec foyer " | |
| 52 | + "intérieur et plancher chauffant.") | |
| 53 | + | |
| 54 | + | |
| 55 | +class HomMiniChalets(StConnector): | |
| 56 | + source_id = "homminichalets" | |
| 57 | + request_delay = 1.0 | |
| 58 | + | |
| 59 | + # -- découverte des uids (Scrapfly, calendrier monté en JS) --------------- | |
| 60 | + def _uids(self) -> list[str]: | |
| 61 | + h = self.get_scrapfly(LIST_URL, render_js=True, rendering_wait=4000) | |
| 62 | + uids: list[str] = [] | |
| 63 | + for u in re.findall(r'href="[^"]*?/pages/chalets\?uid=' | |
| 64 | + r'([0-9a-f-]{36})"', h): | |
| 65 | + if u not in uids: | |
| 66 | + uids.append(u) | |
| 67 | + return uids | |
| 68 | + | |
| 69 | + # -- galerie générique du site (accueil Shopify statique) ----------------- | |
| 70 | + def _site_images(self) -> list[str]: | |
| 71 | + try: | |
| 72 | + h = self.get_scrapfly(SITE + "/", render_js=False) | |
| 73 | + except Exception as exc: # noqa: BLE001 | |
| 74 | + print(f"[homminichalets] accueil : {exc}", file=sys.stderr) | |
| 75 | + return [] | |
| 76 | + imgs: list[str] = [] | |
| 77 | + for num, v in re.findall(r"//homminichalets\.com/cdn/shop/files/" | |
| 78 | + r"(\d{2}\.jpg)\?v=(\d+)", h): | |
| 79 | + u = f"{SITE}/cdn/shop/files/{num}?v={v}&width=1600" | |
| 80 | + if u not in imgs: | |
| 81 | + imgs.append(u) | |
| 82 | + return imgs[:15] | |
| 83 | + | |
| 84 | + # -- fiche Hostfully (JSONP, en direct) ------------------------------------ | |
| 85 | + def _detail(self, uid: str) -> dict: | |
| 86 | + txt = self.get(API.format(uid=uid)).text.strip() | |
| 87 | + m = re.match(r"(?s)cb\(true,(\{.*\})\)$", txt) | |
| 88 | + return json.loads(m.group(1)) if m else {} | |
| 89 | + | |
| 90 | + # -- contrat --------------------------------------------------------------- | |
| 91 | + def fetch(self) -> list[StListing]: | |
| 92 | + uids = self._uids() | |
| 93 | + images = self._site_images() if uids else [] | |
| 94 | + | |
| 95 | + listings: list[StListing] = [] | |
| 96 | + for uid in uids: | |
| 97 | + det = self.detail(uid, "v1", lambda u=uid: self._detail(u)) | |
| 98 | + name = str(det.get("name") or "").strip() | |
| 99 | + if not name: | |
| 100 | + continue | |
| 101 | + | |
| 102 | + m = re.match(r"(\d{1,2})\.", name) | |
| 103 | + num = int(m.group(1)) if m else 0 | |
| 104 | + thermal = num >= 9 | |
| 105 | + address = ("32 chemin du Saphir" if num in (7, 8) | |
| 106 | + else "154 chemin du Rubis") | |
| 107 | + | |
| 108 | + price = det.get("price") | |
| 109 | + price = float(price) if price and 20 <= float(price) <= 20000 \ | |
| 110 | + else None | |
| 111 | + | |
| 112 | + details = { | |
| 113 | + "gamme": ("circuit thermal" if thermal else "spa"), | |
| 114 | + "domain": "HOM Mini Chalets", | |
| 115 | + } | |
| 116 | + if images: | |
| 117 | + details["images_generic"] = True | |
| 118 | + if det.get("minStay"): | |
| 119 | + details["min_stay"] = f"{det['minStay']} nuit(s)" | |
| 120 | + | |
| 121 | + listings.append(StListing( | |
| 122 | + source=self.source_id, | |
| 123 | + external_id=uid, | |
| 124 | + url=f"{LIST_URL}?uid={uid}", | |
| 125 | + title=f"{name} — HOM Mini Chalets", | |
| 126 | + property_type="Mini-chalet", | |
| 127 | + address=address, | |
| 128 | + city="Val-des-Monts", | |
| 129 | + region="Outaouais", | |
| 130 | + price_night=price, | |
| 131 | + price_label=(f"à partir de {price:g} $ / nuit" | |
| 132 | + if price else ""), | |
| 133 | + capacity=(float(det["maximumGuests"]) | |
| 134 | + if det.get("maximumGuests") else None), | |
| 135 | + bedrooms=1.0, | |
| 136 | + citq=CITQ, | |
| 137 | + description=_THERMAL_DESC if thermal else _SPA_DESC, | |
| 138 | + amenities=list(_THERMAL_AMEN if thermal else _SPA_AMEN), | |
| 139 | + details=details, | |
| 140 | + images=list(images), | |
| 141 | + )) | |
| 142 | + return listings | |
modified
louka/shortterm/connectors/kijiji.py
+95 −25
@@ -20,11 +20,18 @@ | ||
| 20 | 20 | # |
| 21 | 21 | # Filtres court terme : on ne garde que les annonces OFFER qui ressemblent à |
| 22 | 22 | # un hébergement (attributs chambres/personnes/type de vacances présents — |
| 23 | −# la catégorie contient aussi maillots de bain, machines à espresso…) et on | |
| 24 | −# écarte les locations au mois (« 31 jours et plus », monthly…). Le prix | |
| 25 | −# Kijiji est un montant sans période : price_night n'est rempli que si le | |
| 26 | −# texte précise « /nuit » ou « /semaine », sinon le montant affiché va dans | |
| 27 | −# details.prix_affiche. | |
| 23 | +# la catégorie contient aussi maillots de bain, machines à espresso… ; si | |
| 24 | +# les attributs manquent sur la liste mais que le titre évoque un | |
| 25 | +# hébergement, la fiche détail tranche) et on écarte les locations au mois | |
| 26 | +# (« 31 jours et plus », monthly, minnights >= 28…). | |
| 27 | +# | |
| 28 | +# Prix : le formulaire de la catégorie demande un prix À LA NUIT — un texte | |
| 29 | +# « X $/nuit » ou « $X/night » dans l'annonce prime (minimum des saisons), | |
| 30 | +# sinon le montant affiché est pris comme prix/nuit s'il est plausible | |
| 31 | +# (<= 2 000 $ et séjour min < 28 nuits), sinon details.prix_affiche. | |
| 32 | +# Salles de bain : la valeur canonique Kijiji est en dixièmes (« 20 » = 2) — | |
| 33 | +# normalisée. Commodités : aucune dans les attributs de la catégorie — on | |
| 34 | +# les dérive des mentions explicites du texte (spa, sauna, foyer, BBQ…). | |
| 28 | 35 | # |
| 29 | 36 | # Réglage env : LOUKA_KIJIJI_LIMIT (nb max d'annonces, pour tester petit). |
| 30 | 37 | # ----------------------------------------------------------------------------- |
@@ -37,6 +44,7 @@ import time | ||
| 37 | 44 | |
| 38 | 45 | import requests |
| 39 | 46 | |
| 47 | +from ...normalize import strip_accents | |
| 40 | 48 | from ..schema import StListing, normalize_region, parse_price_night, REGIONS |
| 41 | 49 | from .airbnb import _region_from_latlng |
| 42 | 50 | from .base import StConnector |
@@ -54,10 +62,54 @@ _MENSUEL_RE = re.compile( | ||
| 54 | 62 | r"au mois|par mois|/\s*mois|mensuel|monthly|per\s+month|/\s*month" |
| 55 | 63 | r"|3[01]\s*jours\s*(?:et plus|minimum|min)|month(?:ly)?\s+rental", re.I) |
| 56 | 64 | |
| 57 | −_NUIT_RE = re.compile(r"(\d[\d\s,.]{0,9})\s*\$\s*(?:/|par|la|per)?\s*" | |
| 58 | − r"(?:nuit|night)", re.I) | |
| 59 | −_SEMAINE_RE = re.compile(r"(\d[\d\s,.]{0,9})\s*\$\s*(?:/|par|la|per)?\s*" | |
| 60 | − r"(?:sem(?:aine)?|week)", re.I) | |
| 65 | +# « 265 $ / nuit » (fr) comme « $265/night » (en) — $ avant ou après le montant | |
| 66 | +_NUIT_RE = re.compile( | |
| 67 | + r"(?:\$\s*(\d[\d\s,.]{0,8}\d|\d)|(\d[\d\s,.]{0,8}\d|\d)\s*\$)\s*" | |
| 68 | + r"(?:/|par|la|per)?\s*(?:nuit|night)", re.I) | |
| 69 | +_SEMAINE_RE = re.compile( | |
| 70 | + r"(?:\$\s*(\d[\d\s,.]{0,8}\d|\d)|(\d[\d\s,.]{0,8}\d|\d)\s*\$)\s*" | |
| 71 | + r"(?:/|par|la|per)?\s*(?:sem(?:aine)?|week)", re.I) | |
| 72 | + | |
| 73 | + | |
| 74 | +def _prix_min(texte: str, rx: re.Pattern) -> float | None: | |
| 75 | + """Le plus bas des montants d'une période (les annonces listent souvent | |
| 76 | + plusieurs saisons : « $265/night … $298/night »).""" | |
| 77 | + vals = [] | |
| 78 | + for m in rx.finditer(texte or ""): | |
| 79 | + raw = (m.group(1) or m.group(2) or "").strip() | |
| 80 | + v = parse_price_night(f"{raw} $") | |
| 81 | + if v: | |
| 82 | + vals.append(v) | |
| 83 | + return min(vals) if vals else None | |
| 84 | + | |
| 85 | + | |
| 86 | +# mentions explicites du texte → commodité affichable (la catégorie Kijiji | |
| 87 | +# n'a aucun attribut de commodités) ; clés en minuscules sans accents | |
| 88 | +_AMEN_HINTS = [ | |
| 89 | + ("spa", "Spa"), ("jacuzzi", "Spa"), ("hot tub", "Spa"), | |
| 90 | + ("sauna", "Sauna"), ("piscine", "Piscine"), ("pool", "Piscine"), | |
| 91 | + ("foyer", "Foyer"), ("fireplace", "Foyer"), | |
| 92 | + ("poele a bois", "Poêle à bois"), ("wood stove", "Poêle à bois"), | |
| 93 | + ("bbq", "BBQ"), ("barbecue", "BBQ"), | |
| 94 | + ("wifi", "Wi-Fi"), ("wi-fi", "Wi-Fi"), ("internet", "Wi-Fi"), | |
| 95 | + ("lave-vaisselle", "Lave-vaisselle"), ("dishwasher", "Lave-vaisselle"), | |
| 96 | + ("laveuse", "Laveuse/sécheuse"), ("washer", "Laveuse/sécheuse"), | |
| 97 | + ("climatis", "Air climatisé"), ("air conditioning", "Air climatisé"), | |
| 98 | + ("kayak", "Kayak"), ("canot", "Canot"), ("canoe", "Canot"), | |
| 99 | + ("stationnement", "Stationnement"), ("parking", "Stationnement"), | |
| 100 | + ("bord de l'eau", "Bord de l'eau"), ("bord du lac", "Bord de l'eau"), | |
| 101 | + ("waterfront", "Bord de l'eau"), ("lakefront", "Bord de l'eau"), | |
| 102 | + ("plage", "Plage à proximité"), ("beach", "Plage à proximité"), | |
| 103 | +] | |
| 104 | + | |
| 105 | + | |
| 106 | +def _amenities_texte(texte: str) -> list[str]: | |
| 107 | + hay = strip_accents(texte or "").lower().replace("’", "'") | |
| 108 | + out: list[str] = [] | |
| 109 | + for needle, label in _AMEN_HINTS: | |
| 110 | + if needle in hay and label not in out: | |
| 111 | + out.append(label) | |
| 112 | + return out | |
| 61 | 113 | |
| 62 | 114 | _TYPE_HINTS = [ |
| 63 | 115 | ("chalet", "Chalet"), ("cottage", "Chalet"), ("cabin", "Chalet"), |
@@ -80,6 +132,16 @@ def _num(texts: list[str]) -> float | None: | ||
| 80 | 132 | return None |
| 81 | 133 | |
| 82 | 134 | |
| 135 | +def _sdb(attrs: dict) -> float | None: | |
| 136 | + """Salles de bain : la valeur canonique Kijiji est en dixièmes | |
| 137 | + (« 20 » = 2, « 25 » = 2,5) ; la valeur humaine (« 2 bathrooms ») est | |
| 138 | + déjà correcte.""" | |
| 139 | + v = _num(attrs.get("numberbathrooms")) | |
| 140 | + if v is not None and v >= 10 and v % 5 == 0: | |
| 141 | + v /= 10 | |
| 142 | + return v | |
| 143 | + | |
| 144 | + | |
| 83 | 145 | def _attrs(entity: dict) -> dict[str, list[str]]: |
| 84 | 146 | """{canonicalName: values (humaines si présentes, sinon canoniques)}.""" |
| 85 | 147 | out: dict[str, list[str]] = {} |
@@ -212,7 +274,9 @@ class KijijiCt(StConnector): | ||
| 212 | 274 | if (e.get("type") or "OFFER") != "OFFER": |
| 213 | 275 | continue |
| 214 | 276 | attrs = _attrs(e) |
| 215 | − if not (_ATTRS_HEBERGEMENT & set(attrs)): | |
| 277 | + hay = title.lower() | |
| 278 | + if not (_ATTRS_HEBERGEMENT & set(attrs)) \ | |
| 279 | + and not any(n in hay for n, _ in _TYPE_HINTS): | |
| 216 | 280 | continue # maillots de bain, cafetières, vans… |
| 217 | 281 | texte = f"{title}\n{e.get('description') or ''}" |
| 218 | 282 | if _MENSUEL_RE.search(texte): |
@@ -228,30 +292,35 @@ class KijijiCt(StConnector): | ||
| 228 | 292 | det = {} |
| 229 | 293 | if det.get("attrs"): |
| 230 | 294 | attrs = det["attrs"] |
| 295 | + if not (_ATTRS_HEBERGEMENT & set(attrs)): | |
| 296 | + continue # le détail confirme : pas un hébergement | |
| 231 | 297 | texte = (f"{title}\n" |
| 232 | 298 | f"{det.get('description') or e.get('description') or ''}") |
| 233 | 299 | if _MENSUEL_RE.search(texte): |
| 234 | 300 | continue |
| 301 | + nuits_min = _num(attrs.get("minnights")) or 0 | |
| 302 | + if nuits_min >= 28: | |
| 303 | + continue # séjour min d'un mois : hors mandat | |
| 235 | 304 | |
| 236 | − # prix : montant en cents, période seulement si le texte la donne | |
| 305 | + # prix : « X $/nuit » du texte (minimum des saisons) prime ; | |
| 306 | + # sinon le montant affiché (en cents) est un prix à la nuit | |
| 307 | + # (convention de la catégorie) s'il est plausible | |
| 237 | 308 | price_night = None |
| 238 | 309 | price_label = "" |
| 239 | 310 | amount = (e.get("price") or {}).get("amount") |
| 240 | 311 | montant = round(amount / 100, 2) if isinstance( |
| 241 | 312 | amount, (int, float)) and amount else None |
| 242 | − m = _NUIT_RE.search(texte) | |
| 243 | − if m: | |
| 244 | − price_label = f"{m.group(1).strip()} $ / nuit" | |
| 245 | − elif _SEMAINE_RE.search(texte): | |
| 246 | − price_label = f"{_SEMAINE_RE.search(texte).group(1).strip()}" \ | |
| 247 | − " $ / semaine" | |
| 248 | − elif montant: | |
| 249 | − nuit = (attrs.get("minnights") or ["1"])[0] | |
| 250 | − if str(nuit) in ("", "1"): # 1 nuit min : montant ≈ par nuit | |
| 251 | − price_night = montant | |
| 252 | − price_label = f"{montant:g} $" | |
| 253 | − if price_label and price_night is None: | |
| 254 | − price_night = parse_price_night(price_label) | |
| 313 | + nuit_val = _prix_min(texte, _NUIT_RE) | |
| 314 | + sem_val = _prix_min(texte, _SEMAINE_RE) | |
| 315 | + if nuit_val: | |
| 316 | + price_night = nuit_val | |
| 317 | + price_label = f"{nuit_val:g} $ / nuit" | |
| 318 | + elif sem_val: | |
| 319 | + price_night = round(sem_val / 7, 2) | |
| 320 | + price_label = f"{sem_val:g} $ / semaine" | |
| 321 | + elif montant and montant <= 2000: | |
| 322 | + price_night = montant | |
| 323 | + price_label = f"{montant:g} $" | |
| 255 | 324 | |
| 256 | 325 | hay = title.lower() |
| 257 | 326 | ptype = next((canon for needle, canon in _TYPE_HINTS |
@@ -302,9 +371,10 @@ class KijijiCt(StConnector): | ||
| 302 | 371 | price_label=price_label, |
| 303 | 372 | capacity=_num(attrs.get("maxpeople")), |
| 304 | 373 | bedrooms=_num(attrs.get("numberbedrooms")), |
| 305 | − bathrooms=_num(attrs.get("numberbathrooms")), | |
| 374 | + bathrooms=_sdb(attrs), | |
| 306 | 375 | pets=pets, |
| 307 | 376 | description=det.get("description") or "", |
| 377 | + amenities=_amenities_texte(texte), | |
| 308 | 378 | details=details, |
| 309 | 379 | images=det.get("images") |
| 310 | 380 | or [u for u in (e.get("imageUrls") or []) |
added
louka/shortterm/connectors/lesversants.py
+156 −0
@@ -0,0 +1,156 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/lesversants.py : Les Versants Mont-Tremblant (lesversants.com) — | |
| 4 | +# agence immobilière de Mont-Tremblant dont la division location propose | |
| 5 | +# ~16 maisons et condos en location saisonnière (Mont-Tremblant, | |
| 6 | +# Lac-Supérieur, Mont-Blanc). | |
| 7 | +# | |
| 8 | +# Méthode : le sitemap.xml est un vestige statique (2 URLs), mais le thème | |
| 9 | +# WordPress REAL_HOMES expose tout via l'API REST : | |
| 10 | +# GET /fr/wp-json/wp/v2/property?property-status=79&per_page=100 | |
| 11 | +# (79 = terme « À louer » de la taxonomie property-status). Chaque fiche | |
| 12 | +# embarque property_meta REAL_HOMES_* : adresse, lat/lng, chambres, salles | |
| 13 | +# de bain (formats « 3 + 1 »), superficie, galerie (sizes.large), prix. | |
| 14 | +# ⚠️ PRIX SAISONNIER : le prix affiché est « Hiver 2026-27 | 18 500 $ » | |
| 15 | +# (un forfait pour LA SAISON, pas à la nuit) → on le range dans | |
| 16 | +# details.season_price et on laisse price_night/price_label vides pour ne | |
| 17 | +# pas empoisonner parse_price_night(). | |
| 18 | +# Taxonomies property-type (439 Condo à louer, 440/441 maisons), city et | |
| 19 | +# mont_tremblant_sector résolues par une passe chacune. | |
| 20 | +# | |
| 21 | +# Réglage env : LOUKA_LESVERSANTS_LIMIT (nb max d'annonces, 0 = tout). | |
| 22 | +# ----------------------------------------------------------------------------- | |
| 23 | +from __future__ import annotations | |
| 24 | + | |
| 25 | +import html as _html | |
| 26 | +import os | |
| 27 | +import re | |
| 28 | + | |
| 29 | +from ..schema import StListing | |
| 30 | +from .base import StConnector | |
| 31 | + | |
| 32 | +API = "https://lesversants.com/fr/wp-json/wp/v2" | |
| 33 | +STATUS_A_LOUER = 79 # terme « À louer » (property-status) | |
| 34 | + | |
| 35 | +# id property-type → type canonique | |
| 36 | +_TYPES = {439: "Condo", 440: "Maison", 441: "Maison"} | |
| 37 | + | |
| 38 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 39 | + | |
| 40 | + | |
| 41 | +def _strip_html(txt: str) -> str: | |
| 42 | + return re.sub(r"\s+", " ", _TAG_RE.sub(" ", _html.unescape(txt or ""))).strip() | |
| 43 | + | |
| 44 | + | |
| 45 | +def _rooms(v: str) -> float | None: | |
| 46 | + """« 3 », « 3 + 1 », « 3.5 » → nombre (les « + N » sont additionnés).""" | |
| 47 | + nums = re.findall(r"\d+(?:\.\d+)?", str(v or "")) | |
| 48 | + return sum(float(n) for n in nums) if nums else None | |
| 49 | + | |
| 50 | + | |
| 51 | +class LesVersants(StConnector): | |
| 52 | + source_id = "lesversants" | |
| 53 | + request_delay = 0.5 | |
| 54 | + | |
| 55 | + def _get_json(self, url: str): | |
| 56 | + return self.get(url, headers={"Accept": "application/json"}).json() | |
| 57 | + | |
| 58 | + def _tax(self, name: str) -> dict[int, str]: | |
| 59 | + try: | |
| 60 | + terms = self._get_json(f"{API}/{name}?per_page=100" | |
| 61 | + "&_fields=id,name") | |
| 62 | + return {t["id"]: _html.unescape(t.get("name") or "").strip() | |
| 63 | + for t in terms} | |
| 64 | + except Exception: # noqa: BLE001 — libellés manquants ≠ blocage | |
| 65 | + return {} | |
| 66 | + | |
| 67 | + # -- contrat -------------------------------------------------------------- | |
| 68 | + def fetch(self) -> list[StListing]: | |
| 69 | + limit = int(os.environ.get("LOUKA_LESVERSANTS_LIMIT", "0") or 0) | |
| 70 | + items = self._get_json(f"{API}/property?property-status=" | |
| 71 | + f"{STATUS_A_LOUER}&per_page=100") | |
| 72 | + if not isinstance(items, list): | |
| 73 | + return [] | |
| 74 | + if limit: | |
| 75 | + items = items[:limit] | |
| 76 | + | |
| 77 | + types = self._tax("property-type") | |
| 78 | + cities = self._tax("city") | |
| 79 | + sectors = self._tax("mont_tremblant_sector") | |
| 80 | + | |
| 81 | + listings: list[StListing] = [] | |
| 82 | + seen: set[str] = set() | |
| 83 | + for it in items: | |
| 84 | + pid = str(it.get("id") or "").strip() | |
| 85 | + url = (it.get("link") or "").strip() | |
| 86 | + title = _strip_html((it.get("title") or {}).get("rendered") or "") | |
| 87 | + if not pid or pid in seen or not url or not title: | |
| 88 | + continue | |
| 89 | + seen.add(pid) | |
| 90 | + | |
| 91 | + meta = it.get("property_meta") or {} | |
| 92 | + loc = meta.get("REAL_HOMES_property_location") or {} | |
| 93 | + lat = float(loc["latitude"]) if loc.get("latitude") else None | |
| 94 | + lng = float(loc["longitude"]) if loc.get("longitude") else None | |
| 95 | + | |
| 96 | + type_ids = it.get("property-type") or [] | |
| 97 | + ptype = next((_TYPES[t] for t in type_ids if t in _TYPES), "") | |
| 98 | + if not ptype: | |
| 99 | + ptype = next((types[t] for t in type_ids if t in types), "") | |
| 100 | + | |
| 101 | + city_ids = it.get("city") or [] | |
| 102 | + city = next((cities[c] for c in city_ids if c in cities), | |
| 103 | + "Mont-Tremblant") | |
| 104 | + sector_ids = it.get("mont_tremblant_sector") or [] | |
| 105 | + sector = next((sectors[s] for s in sector_ids if s in sectors), "") | |
| 106 | + | |
| 107 | + # prix SAISONNIER (« Hiver 2026-27 | 18 500 $ ») → details | |
| 108 | + prefix = (meta.get("REAL_HOMES_property_price_prefix") or "").strip() | |
| 109 | + raw_price = re.sub(r"[^\d.]", "", | |
| 110 | + str(meta.get("REAL_HOMES_property_price") or "")) | |
| 111 | + season_price = "" | |
| 112 | + if raw_price: | |
| 113 | + season = prefix.rstrip("|").strip() | |
| 114 | + season_price = (f"{season} : {raw_price} $" if season | |
| 115 | + else f"{raw_price} $") | |
| 116 | + | |
| 117 | + size = (meta.get("REAL_HOMES_property_size") or "").strip() | |
| 118 | + size_post = (meta.get("REAL_HOMES_property_size_postfix") | |
| 119 | + or "").strip() | |
| 120 | + | |
| 121 | + images: list[str] = [] | |
| 122 | + for ph in meta.get("REAL_HOMES_property_images") or []: | |
| 123 | + sizes = (ph or {}).get("sizes") or {} | |
| 124 | + u = ((sizes.get("1536x1536") or {}).get("url") | |
| 125 | + or (sizes.get("large") or {}).get("url") | |
| 126 | + or (sizes.get("medium_large") or {}).get("url") or "") | |
| 127 | + if u.startswith("https://") and u not in images: | |
| 128 | + images.append(u) | |
| 129 | + if len(images) >= 15: | |
| 130 | + break | |
| 131 | + | |
| 132 | + details = {k: v for k, v in { | |
| 133 | + "season_price": season_price, | |
| 134 | + "sector": sector, | |
| 135 | + "size": f"{size} {size_post}".strip() if size else "", | |
| 136 | + }.items() if v} | |
| 137 | + | |
| 138 | + listings.append(StListing( | |
| 139 | + source=self.source_id, | |
| 140 | + external_id=pid, | |
| 141 | + url=url, | |
| 142 | + title=title, | |
| 143 | + property_type=ptype, | |
| 144 | + address=(meta.get("REAL_HOMES_property_address") or "").strip(), | |
| 145 | + city=city, | |
| 146 | + region="Laurentides", | |
| 147 | + bedrooms=_rooms(meta.get("REAL_HOMES_property_bedrooms")), | |
| 148 | + bathrooms=_rooms(meta.get("REAL_HOMES_property_bathrooms")), | |
| 149 | + description=_strip_html((it.get("content") or {}) | |
| 150 | + .get("rendered") or "")[:3000], | |
| 151 | + details=details, | |
| 152 | + images=images, | |
| 153 | + lat=lat, | |
| 154 | + lng=lng, | |
| 155 | + )) | |
| 156 | + return listings | |
added
louka/shortterm/connectors/locationdechalets.py
+265 −0
@@ -0,0 +1,265 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/locationdechalets.py : Location de Chalets Lanaudière | |
| 4 | +# (locationdechalets.com) — petit parc de ~10 chalets avec spa privé à | |
| 5 | +# Notre-Dame-de-la-Merci et Saint-Donat (Lanaudière). | |
| 6 | +# | |
| 7 | +# Méthode : sitemap.xml → /fr/chalets/<slug>/ + lastmod (clé du cache détail). | |
| 8 | +# Pages statiques (CMS maison) riches mais sans JSON-LD : | |
| 9 | +# - h1 « Chalet Spa Le Héron … Capacité de 2 personnes » ; | |
| 10 | +# - bloc CARACTÉRISTIQUES (« N personnes maximum / N chambre(s) / | |
| 11 | +# N salle(s) de bain ») ; | |
| 12 | +# - sections INTÉRIEUR / EXTÉRIEUR / INCLUS en <ul><li> → amenities ; | |
| 13 | +# - « EN SUS … Animaux : 10 $ / jour » → pets = conditions ; | |
| 14 | +# - bloc Coordonnées (rue, ville, « Lanaudière (Québec) », code postal) ; | |
| 15 | +# - no CITQ dans le pied de page ; lat/lng dans le lien Google Maps (@…) ; | |
| 16 | +# - TARIF RÉGULIER « 2 nuits : 498 $ … » → price_night = total/2 nuits. | |
| 17 | +# ⚠️ anti-scrape : zéros de bourrage blancs sur blanc | |
| 18 | +# (<span style='color: #ffffff;'>0</span>) à retirer AVANT le parsing ; | |
| 19 | +# - galerie : background:url(/fichiersUploadOpt/…) du slider. | |
| 20 | +# ----------------------------------------------------------------------------- | |
| 21 | +from __future__ import annotations | |
| 22 | + | |
| 23 | +import html as _html | |
| 24 | +import re | |
| 25 | + | |
| 26 | +from ..schema import StListing | |
| 27 | +from .base import StConnector | |
| 28 | + | |
| 29 | +SITE = "https://www.locationdechalets.com" | |
| 30 | +SITEMAP = SITE + "/sitemap.xml" | |
| 31 | + | |
| 32 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 33 | + | |
| 34 | +# ville → région (repli quand la ligne « … (Québec) » manque) | |
| 35 | +_CITY_REGION = { | |
| 36 | + "chute-st-philippe": "Laurentides", | |
| 37 | + "chute-saint-philippe": "Laurentides", | |
| 38 | + "la macaza": "Laurentides", | |
| 39 | + "val-david": "Laurentides", | |
| 40 | + "notre-dame-de-la-merci": "Lanaudière", | |
| 41 | + "saint-donat": "Lanaudière", | |
| 42 | + "st-donat": "Lanaudière", | |
| 43 | +} | |
| 44 | + | |
| 45 | + | |
| 46 | +def _text(fragment: str) -> str: | |
| 47 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 48 | + | |
| 49 | + | |
| 50 | +class LocationDeChalets(StConnector): | |
| 51 | + source_id = "locationdechalets" | |
| 52 | + request_delay = 1.0 | |
| 53 | + | |
| 54 | + # -- page détail ---------------------------------------------------------- | |
| 55 | + def _detail(self, url: str) -> dict: | |
| 56 | + h = self.get(url).text | |
| 57 | + d: dict = {} | |
| 58 | + | |
| 59 | + m = re.search(r"(?s)<h1[^>]*>(.*?)</h1>", h) | |
| 60 | + if m: | |
| 61 | + d["title"] = _text(re.split(r"<br\s*/?>", m.group(1))[0]) | |
| 62 | + # « Capacité de 2 personnes » / « Capacité de 2 à 4 personnes » (max) | |
| 63 | + m = re.search(r"Capacité de (?:\d+\s+à\s+)?(\d+) personnes", h) | |
| 64 | + if m: | |
| 65 | + d["capacity"] = int(m.group(1)) | |
| 66 | + | |
| 67 | + # description : après <strong>Description</strong>, jusqu'à la | |
| 68 | + # section suivante ; certaines pages n'ont pas ce marqueur → repli | |
| 69 | + # sur le plus long <p> éditorial | |
| 70 | + m = re.search(r"(?s)<strong>Description</strong>(.*?)" | |
| 71 | + r"(?:CARACTÉRISTIQUES|Caractéristiques|Disponibilités" | |
| 72 | + r"|<strong>INTÉRIEUR)", h) | |
| 73 | + if m: | |
| 74 | + texte = re.sub(r"<br\s*/?>", "\n", m.group(1)) | |
| 75 | + texte = _html.unescape(_TAG_RE.sub(" ", texte)) | |
| 76 | + texte = re.sub(r"[ \t]+", " ", texte) | |
| 77 | + texte = re.sub(r"\n\s+", "\n", texte).strip() | |
| 78 | + d["description"] = texte[:5000] | |
| 79 | + else: | |
| 80 | + paras = [_text(p) for p in | |
| 81 | + re.findall(r"(?s)<p[^>]*>(.*?)</p>", h)] | |
| 82 | + paras = [p for p in paras if len(p) > 150 | |
| 83 | + and "Coordonnées" not in p | |
| 84 | + and "Contactez-nous" not in p | |
| 85 | + and "Découvrez nos chalets" not in p] | |
| 86 | + if paras: | |
| 87 | + d["description"] = max(paras, key=len)[:5000] | |
| 88 | + | |
| 89 | + # sections (deux styles : MAJUSCULES ou « Caractéristiques : ») : | |
| 90 | + # <ul> qui suit chaque entête → amenities + compteurs | |
| 91 | + amen: list[str] = [] | |
| 92 | + for sec in (r"CARACT[EÉ]RISTIQUES", r"INT[EÉ]RIEUR", | |
| 93 | + r"EXT[EÉ]RIEUR", r"INCLUS"): | |
| 94 | + for m in re.finditer(sec, h, re.I): | |
| 95 | + mu = re.search(r"(?s)<ul[^>]*>(.*?)</ul>", | |
| 96 | + h[m.start():m.start() + 3500]) | |
| 97 | + if not mu: | |
| 98 | + continue | |
| 99 | + for li in re.findall(r"(?s)<li[^>]*>(.*?)</li>", mu.group(1)): | |
| 100 | + t = _text(li) | |
| 101 | + if t and t not in amen: | |
| 102 | + amen.append(t) | |
| 103 | + break | |
| 104 | + # compteurs extraits des puces (« 2 chambres (1 lit queen…) », | |
| 105 | + # « 1 salle de bain », « 2 personnes maximum ») + du texte qui suit | |
| 106 | + # l'entête CARACTÉRISTIQUES (pages où ce sont des <p>, pas des <li>) | |
| 107 | + blob = " | ".join(amen) | |
| 108 | + m = re.search(r"CARACT[EÉ]RISTIQUES", h, re.I) | |
| 109 | + if m: | |
| 110 | + blob += " | " + _text(h[m.start():m.start() + 700]) | |
| 111 | + m = re.search(r"(\d+)\s+personnes?\s+maximum", blob) | |
| 112 | + if m: | |
| 113 | + d.setdefault("capacity", int(m.group(1))) | |
| 114 | + m = re.search(r"(\d+(?:[.,]5)?)\s+chambres?", blob) | |
| 115 | + if m: | |
| 116 | + d["bedrooms"] = float(m.group(1).replace(",", ".")) | |
| 117 | + m = re.search(r"(\d+(?:[.,]5)?)\s+salles?\s+de\s+bain", blob) | |
| 118 | + if m: | |
| 119 | + d["bathrooms"] = float(m.group(1).replace(",", ".")) | |
| 120 | + amen = [a for a in amen | |
| 121 | + if not re.fullmatch(r"\d+\s+(personnes?\s+maximum" | |
| 122 | + r"|chambres?|salles?\s+de\s+bain)", a)] | |
| 123 | + if amen: | |
| 124 | + d["amenities"] = amen | |
| 125 | + | |
| 126 | + # frais : « Animaux : 10 $ / jour » (section EN SUS) | |
| 127 | + m = re.search(r"Animaux\s*:\s*([\d,]+\s*\$[^<]*)", h) | |
| 128 | + if m: | |
| 129 | + d["pets_fee"] = _text(m.group(1)) | |
| 130 | + | |
| 131 | + # coordonnées : rue / ville / « Lanaudière (Québec) » / code postal | |
| 132 | + m = re.search(r"(?s)<div class=\"coord[^\"]*\"[^>]*>(.*?)</div>", h) | |
| 133 | + if m: | |
| 134 | + lines = [_text(x) for x in re.split(r"<br\s*/?>|</p>", m.group(1))] | |
| 135 | + lines = [x for x in lines if x and "Coordonnées" not in x | |
| 136 | + and not re.fullmatch(r"[\d,]+", x)] | |
| 137 | + reg = next((x for x in lines if "(Québec)" in x), "") | |
| 138 | + if reg: | |
| 139 | + j = lines.index(reg) | |
| 140 | + d["region"] = reg.split("(")[0].strip() | |
| 141 | + if j >= 1: | |
| 142 | + d["city"] = lines[j - 1].strip(", ") | |
| 143 | + if j >= 2: | |
| 144 | + d["address"] = lines[j - 2].strip(", ") | |
| 145 | + if j + 1 < len(lines): | |
| 146 | + d["postal_code"] = lines[j + 1] | |
| 147 | + else: | |
| 148 | + # variante sans ligne « … (Québec) » : rue / ville / postal | |
| 149 | + postal = next((x for x in lines | |
| 150 | + if re.fullmatch(r"[A-Z]\d[A-Z]\s?\d[A-Z]\d", | |
| 151 | + x.strip())), "") | |
| 152 | + rest = [x for x in lines if x != postal] | |
| 153 | + if len(rest) >= 2: | |
| 154 | + d["address"] = rest[0].strip(", ") | |
| 155 | + d["city"] = rest[1].strip(", ") | |
| 156 | + if postal: | |
| 157 | + d["postal_code"] = postal.strip() | |
| 158 | + | |
| 159 | + # (pas de lat/lng : le lien Google Maps pointe l'agence, pas le | |
| 160 | + # chalet — le géocodage se fera en aval sur adresse+ville) | |
| 161 | + m = re.search(r"CITQ\D{0,25}(\d{6})", h) | |
| 162 | + if m: | |
| 163 | + d["citq"] = m.group(1) | |
| 164 | + | |
| 165 | + # TARIF RÉGULIER : « 2 nuits : 498 $ » (en <p> ou en <table>) → | |
| 166 | + # 249 $/nuit. ⚠️ anti-scrape : chiffres de bourrage blancs sur blanc, | |
| 167 | + # PARFOIS IMBRIQUÉS (<span #fff>0<span #000>4</span></span>98) → on | |
| 168 | + # élimine itérativement les <span> les plus internes : blancs = jetés | |
| 169 | + # avec leur contenu, autres = dépliés (contenu conservé). | |
| 170 | + i = h.find("TARIF RÉGULIER") | |
| 171 | + if i >= 0: | |
| 172 | + frag = h[i:i + 2500] | |
| 173 | + for _ in range(20): | |
| 174 | + frag2 = re.sub( | |
| 175 | + r"<span([^>]*)>([^<]*)</span>", | |
| 176 | + lambda m: (m.group(2).replace("0", "") | |
| 177 | + if re.search(r"color:\s*#f{3,6}\b", | |
| 178 | + m.group(1), re.I) | |
| 179 | + else m.group(2)), | |
| 180 | + frag) | |
| 181 | + if frag2 == frag: | |
| 182 | + break | |
| 183 | + frag = frag2 | |
| 184 | + frag = _text(frag) | |
| 185 | + m = re.search(r"(\d+)\s*nuits?\s*:\s*(\d[\d\s]*)\s*\$", frag) | |
| 186 | + if m: | |
| 187 | + nights = int(m.group(1)) | |
| 188 | + total = float(m.group(2).replace(" ", "")) | |
| 189 | + if nights and 20 <= total / nights <= 20000: | |
| 190 | + d["price_night"] = round(total / nights) | |
| 191 | + d["price_ref"] = f"{nights} nuits : {total:g} $" | |
| 192 | + | |
| 193 | + # galerie du slider | |
| 194 | + imgs: list[str] = [] | |
| 195 | + for u in re.findall(r"background:\s*url\(([^)]+)\)", h): | |
| 196 | + u = u.strip("'\" ") | |
| 197 | + if u.startswith("/fichiersUploadOpt/"): | |
| 198 | + u = SITE + u | |
| 199 | + if u.startswith("https://") and u not in imgs: | |
| 200 | + imgs.append(u) | |
| 201 | + if len(imgs) >= 20: | |
| 202 | + break | |
| 203 | + if imgs: | |
| 204 | + d["images"] = imgs | |
| 205 | + return d | |
| 206 | + | |
| 207 | + # -- contrat -------------------------------------------------------------- | |
| 208 | + def fetch(self) -> list[StListing]: | |
| 209 | + xml = self.get(SITEMAP).text | |
| 210 | + entries = re.findall(r"(?s)<url>\s*<loc>([^<]+)</loc>" | |
| 211 | + r"(?:\s*<lastmod>([^<]*)</lastmod>)?", xml) | |
| 212 | + | |
| 213 | + listings: list[StListing] = [] | |
| 214 | + vus: set[str] = set() | |
| 215 | + for url, lastmod in entries: | |
| 216 | + m = re.match(r"https://www\.locationdechalets\.com/fr/chalets/" | |
| 217 | + r"([^/]+)/?$", url) | |
| 218 | + if not m: | |
| 219 | + continue | |
| 220 | + slug = m.group(1) | |
| 221 | + if slug in vus: | |
| 222 | + continue | |
| 223 | + vus.add(slug) | |
| 224 | + | |
| 225 | + det = self.detail(slug, lastmod or "v1", | |
| 226 | + lambda u=url: self._detail(u)) | |
| 227 | + title = det.get("title") or "" | |
| 228 | + if not title: | |
| 229 | + continue | |
| 230 | + | |
| 231 | + city = det.get("city") or "" | |
| 232 | + region = det.get("region") \ | |
| 233 | + or _CITY_REGION.get(city.lower(), "Lanaudière") | |
| 234 | + | |
| 235 | + price = det.get("price_night") | |
| 236 | + details = {k: v for k, v in { | |
| 237 | + "postal_code": det.get("postal_code") or "", | |
| 238 | + "price_ref": det.get("price_ref") or "", | |
| 239 | + "pets_fee": det.get("pets_fee") or "", | |
| 240 | + }.items() if v} | |
| 241 | + | |
| 242 | + listings.append(StListing( | |
| 243 | + source=self.source_id, | |
| 244 | + external_id=slug, | |
| 245 | + url=url, | |
| 246 | + title=title, | |
| 247 | + property_type="Chalet", | |
| 248 | + address=det.get("address") or "", | |
| 249 | + city=city, | |
| 250 | + region=region, | |
| 251 | + price_night=float(price) if price else None, | |
| 252 | + price_label=f"à partir de {price:g} $ / nuit" if price else "", | |
| 253 | + capacity=float(det["capacity"]) if det.get("capacity") else None, | |
| 254 | + bedrooms=det.get("bedrooms"), | |
| 255 | + bathrooms=det.get("bathrooms"), | |
| 256 | + pets="conditions" if det.get("pets_fee") else None, | |
| 257 | + citq=det.get("citq") or "", | |
| 258 | + description=det.get("description") or "", | |
| 259 | + amenities=det.get("amenities") or [], | |
| 260 | + details=details, | |
| 261 | + images=det.get("images") or [], | |
| 262 | + lat=det.get("lat"), | |
| 263 | + lng=det.get("lng"), | |
| 264 | + )) | |
| 265 | + return listings | |
added
louka/shortterm/connectors/memoriachalets.py
+248 −0
@@ -0,0 +1,248 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/memoriachalets.py : Memoria Chalets (memoriachalets.com) | |
| 4 | +# — gestionnaire de Magog, ~110 fiches FR (Cantons-de-l'Est surtout, | |
| 5 | +# quelques Laurentides / Centre-du-Québec). | |
| 6 | +# | |
| 7 | +# Méthode : WordPress, CPT `cpt_chalets`, HTML serveur. Sitemap | |
| 8 | +# /cpt_chalets-sitemap.xml (lastmod fiable ; doublons /en/ écartés) → | |
| 9 | +# pages /locations/<slug>/. La fiche est balisée par classes : | |
| 10 | +# h2.location-title (« Le Hibou – Magog »), p.location-specs | |
| 11 | +# (« 4 CHAMBRES - 1.5 SALLES DE BAIN - 8 PERSONNES »), p.location-adress | |
| 12 | +# (adresse complète), p.starting-price (« À partir de 199,00$ par jour » — | |
| 13 | +# une partie du parc est au mois/31 jours minimum : prix nuit alors absent, | |
| 14 | +# la mention est conservée dans details), p.citq (« CITQ #296459 … » ou | |
| 15 | +# « (Location minimum de 31 jours) ») et ul de .accomosdations-filters | |
| 16 | +# (commodités, dont « Animaux interdit/admis »). Pas de géo dans le HTML. | |
| 17 | +# ----------------------------------------------------------------------------- | |
| 18 | +from __future__ import annotations | |
| 19 | + | |
| 20 | +import re | |
| 21 | + | |
| 22 | +from bs4 import BeautifulSoup | |
| 23 | + | |
| 24 | +from ...normalize import strip_accents | |
| 25 | +from ..schema import StListing, parse_price_night | |
| 26 | +from .base import StConnector | |
| 27 | + | |
| 28 | +SITEMAP = "https://memoriachalets.com/cpt_chalets-sitemap.xml" | |
| 29 | + | |
| 30 | +_URL_DETAIL = re.compile( | |
| 31 | + r"https://(?:www\.)?memoriachalets\.com/locations/([\w%-]+)/?$") | |
| 32 | + | |
| 33 | +_SPECS_RE = re.compile( | |
| 34 | + r"([\d.,]+)\s*CHAMBRES?\s*[-·—]\s*([\d.,]+)\s*SALLES?\s+DE\s+BAIN" | |
| 35 | + r"\s*[-·—]\s*([\d.,]+)\s*PERSONNES?", re.I) | |
| 36 | + | |
| 37 | +_CP_RE = re.compile(r"^[A-Za-z]\d[A-Za-z]\s?\d[A-Za-z]\d$") # code postal | |
| 38 | + | |
| 39 | +# mot-clé (titre) → type canonique ; défaut : Chalet | |
| 40 | +_TYPE_HINTS = [("condo", "Condo"), ("appartement", "Appartement"), | |
| 41 | + ("apt", "Appartement"), ("loft", "Loft"), ("studio", "Studio"), | |
| 42 | + ("maison", "Maison")] | |
| 43 | + | |
| 44 | +# indices de région dans le texte (leur parc : Estrie surtout) | |
| 45 | +_REGION_HINTS = [ | |
| 46 | + ("estrie", "Cantons-de-l'Est"), ("cantons-de-l'est", "Cantons-de-l'Est"), | |
| 47 | + ("laurentides", "Laurentides"), ("lanaudiere", "Lanaudière"), | |
| 48 | + ("centre-du-quebec", "Centre-du-Québec"), ("mauricie", "Mauricie"), | |
| 49 | + ("charlevoix", "Charlevoix"), ("monteregie", "Montérégie"), | |
| 50 | +] | |
| 51 | + | |
| 52 | +# villes connues du parc → région touristique | |
| 53 | +_VILLE_REGION = { | |
| 54 | + "magog": "Cantons-de-l'Est", "orford": "Cantons-de-l'Est", | |
| 55 | + "eastman": "Cantons-de-l'Est", "austin": "Cantons-de-l'Est", | |
| 56 | + "ayer's cliff": "Cantons-de-l'Est", "ayers cliff": "Cantons-de-l'Est", | |
| 57 | + "north hatley": "Cantons-de-l'Est", "sherbrooke": "Cantons-de-l'Est", | |
| 58 | + "sainte-catherine-de-hatley": "Cantons-de-l'Est", | |
| 59 | + "canton d'orford": "Cantons-de-l'Est", "bromont": "Cantons-de-l'Est", | |
| 60 | + "sutton": "Cantons-de-l'Est", "stukely-sud": "Cantons-de-l'Est", | |
| 61 | + "bolton-est": "Cantons-de-l'Est", "potton": "Cantons-de-l'Est", | |
| 62 | + "mansonville": "Cantons-de-l'Est", "coaticook": "Cantons-de-l'Est", | |
| 63 | + "piopolis": "Cantons-de-l'Est", "lac-megantic": "Cantons-de-l'Est", | |
| 64 | +} | |
| 65 | + | |
| 66 | +_RUE_RE = re.compile(r"^(rue|chemin|ch\.|avenue|av\.|route|rte|boulevard|" | |
| 67 | + r"boul\.?|montee|impasse|allee|place)\b") | |
| 68 | + | |
| 69 | + | |
| 70 | +def _num(raw: str) -> float | None: | |
| 71 | + try: | |
| 72 | + return float(raw.replace(",", ".")) | |
| 73 | + except (TypeError, ValueError): | |
| 74 | + return None | |
| 75 | + | |
| 76 | + | |
| 77 | +class MemoriaChalets(StConnector): | |
| 78 | + source_id = "memoriachalets" | |
| 79 | + | |
| 80 | + # -- inventaire (sitemap FR + lastmod) ----------------------------------- | |
| 81 | + def _sitemap_urls(self) -> dict[str, tuple[str, str]]: | |
| 82 | + """slug -> (url détail FR, lastmod).""" | |
| 83 | + xml = self.get(SITEMAP).text | |
| 84 | + urls: dict[str, tuple[str, str]] = {} | |
| 85 | + for bloc in re.findall(r"<url>(.*?)</url>", xml, re.S): | |
| 86 | + m = re.search(r"<loc>([^<]+)</loc>", bloc) | |
| 87 | + if not m: | |
| 88 | + continue | |
| 89 | + loc = m.group(1).strip() | |
| 90 | + mu = _URL_DETAIL.match(loc) | |
| 91 | + if not mu or mu.group(1) == "locations": | |
| 92 | + continue | |
| 93 | + lastmod = re.search(r"<lastmod>([^<]+)</lastmod>", bloc) | |
| 94 | + urls.setdefault(mu.group(1), | |
| 95 | + (loc, lastmod.group(1) if lastmod else "")) | |
| 96 | + return urls | |
| 97 | + | |
| 98 | + # -- page détail --------------------------------------------------------- | |
| 99 | + def _detail(self, url: str) -> dict: | |
| 100 | + html = self.get(url).text | |
| 101 | + soup = BeautifulSoup(html, "html.parser") | |
| 102 | + d: dict = {} | |
| 103 | + | |
| 104 | + h2 = soup.select_one("h2.location-title") or soup.find("h2") | |
| 105 | + if h2 is not None: | |
| 106 | + d["title"] = re.sub(r"\s+", " ", h2.get_text(" ", strip=True)) | |
| 107 | + | |
| 108 | + specs = soup.select_one("p.location-specs") | |
| 109 | + if specs is not None: | |
| 110 | + m = _SPECS_RE.search(specs.get_text(" ", strip=True)) | |
| 111 | + if m: | |
| 112 | + d["bedrooms"] = _num(m.group(1)) | |
| 113 | + d["bathrooms"] = _num(m.group(2)) | |
| 114 | + d["capacity"] = _num(m.group(3)) | |
| 115 | + | |
| 116 | + # « 207, Rue des Pruches, Magog, Québec, J1X 0M9, Canada » | |
| 117 | + adresse = soup.select_one("p.location-adress") | |
| 118 | + if adresse is not None: | |
| 119 | + texte = re.sub(r"\s+", " ", adresse.get_text(" ", strip=True)) | |
| 120 | + d["address"] = texte | |
| 121 | + restants = [] | |
| 122 | + for p in texte.split(","): | |
| 123 | + # code postal parfois collé à la ville (« Bromont J2L 2C1 ») | |
| 124 | + p = re.sub(r"\b[A-Za-z]\d[A-Za-z]\s?\d[A-Za-z]\d\b", "", | |
| 125 | + p).strip() | |
| 126 | + k = strip_accents(p).lower() | |
| 127 | + if (not p or _CP_RE.match(p) or k in ("quebec", "qc", "canada") | |
| 128 | + or re.match(r"^(appartement|app|unite)\b", k)): | |
| 129 | + continue | |
| 130 | + restants.append(p) | |
| 131 | + if restants: | |
| 132 | + ville = restants[-1] | |
| 133 | + # adresse sans virgule (« Rue des Noyers Orford ») : | |
| 134 | + # la ville est le dernier mot | |
| 135 | + if _RUE_RE.match(strip_accents(ville).lower()): | |
| 136 | + ville = ville.split()[-1] | |
| 137 | + d["city"] = ville | |
| 138 | + | |
| 139 | + prix = soup.select_one("p.starting-price") | |
| 140 | + if prix is not None: | |
| 141 | + label = re.sub(r"\s+", " ", prix.get_text(" ", strip=True)) | |
| 142 | + label = re.sub(r"^(À partir de\s+)+", "À partir de ", label, flags=re.I) | |
| 143 | + d["price_label"] = label | |
| 144 | + | |
| 145 | + citq = soup.select_one("p.citq") | |
| 146 | + if citq is not None: | |
| 147 | + texte = citq.get_text(" ", strip=True) | |
| 148 | + m = re.search(r"CITQ\s*#?\s*(\d{6})", texte) | |
| 149 | + if m: | |
| 150 | + d["citq"] = m.group(1) | |
| 151 | + m = re.search(r"\(([^)]*[Ll]ocation minimum[^)]*)\)", texte) | |
| 152 | + if m: | |
| 153 | + d["location_minimum"] = m.group(1).strip() | |
| 154 | + | |
| 155 | + # commodités (dont le statut animaux) | |
| 156 | + amen: list[str] = [] | |
| 157 | + for span in soup.select(".accomosdations-filters li span"): | |
| 158 | + t = re.sub(r"\s+", " ", span.get_text(" ", strip=True)) | |
| 159 | + if t and t not in amen: | |
| 160 | + amen.append(t) | |
| 161 | + if amen: | |
| 162 | + d["amenities"] = amen | |
| 163 | + for a in amen: | |
| 164 | + k = strip_accents(a).lower() | |
| 165 | + if "animaux" in k or "chien" in k: | |
| 166 | + d["pets"] = "non" if ("interdit" in k or "refus" in k | |
| 167 | + or "non admis" in k) else "oui" | |
| 168 | + | |
| 169 | + # description : paragraphes de la 1re colonne de la fiche | |
| 170 | + col = soup.select_one('[itemprop="articleBody"] div') | |
| 171 | + if col is not None: | |
| 172 | + morceaux = [] | |
| 173 | + for p in col.find_all("p"): | |
| 174 | + classes = " ".join(p.get("class") or []) | |
| 175 | + if any(c in classes for c in | |
| 176 | + ("location-specs", "location-adress", | |
| 177 | + "starting-price", "citq")): | |
| 178 | + continue | |
| 179 | + t = p.get_text(" ", strip=True) | |
| 180 | + if t: | |
| 181 | + morceaux.append(re.sub(r"\s+", " ", t)) | |
| 182 | + if morceaux: | |
| 183 | + d["description"] = "\n".join(morceaux)[:5000] | |
| 184 | + | |
| 185 | + # photos (uploads, sans logo/icônes ; dédoublonnage sur le nom de | |
| 186 | + # base sans suffixe de taille -WxH) | |
| 187 | + imgs, vus = [], set() | |
| 188 | + for u in re.findall(r'(https://(?:www\.)?memoriachalets\.com/' | |
| 189 | + r'wp-content/uploads/[^"\'\s>]+\.(?:jpe?g|webp))', | |
| 190 | + html): | |
| 191 | + base = re.sub(r"-\d+x\d+(?=\.\w+$)", "", u) | |
| 192 | + nom = strip_accents(base).lower() | |
| 193 | + if base in vus or "logo" in nom or "icon" in nom: | |
| 194 | + continue | |
| 195 | + vus.add(base) | |
| 196 | + imgs.append(re.sub(r"-\d+x\d+(?=\.\w+$)", "", u)) | |
| 197 | + d["images"] = imgs[:20] | |
| 198 | + return d | |
| 199 | + | |
| 200 | + # -- contrat -------------------------------------------------------------- | |
| 201 | + def fetch(self) -> list[StListing]: | |
| 202 | + listings: list[StListing] = [] | |
| 203 | + for slug, (url, lastmod) in self._sitemap_urls().items(): | |
| 204 | + try: | |
| 205 | + d = self.detail(slug, lastmod or "sans-lastmod", | |
| 206 | + lambda u=url: self._detail(u)) | |
| 207 | + except Exception: # une fiche cassée ≠ inventaire perdu | |
| 208 | + d = {} | |
| 209 | + if not d.get("title"): | |
| 210 | + continue | |
| 211 | + | |
| 212 | + hay = strip_accents(d["title"]).lower() | |
| 213 | + ptype = next((c for n, c in _TYPE_HINTS if n in hay), "Chalet") | |
| 214 | + | |
| 215 | + # région : ville connue, sinon indice textuel (« en Estrie »…) | |
| 216 | + ville = d.get("city", "") | |
| 217 | + region = _VILLE_REGION.get(strip_accents(ville).lower(), "") | |
| 218 | + if not region: | |
| 219 | + texte = strip_accents(" ".join( | |
| 220 | + (d.get("address", ""), d.get("description", "")))).lower() | |
| 221 | + region = next((r for n, r in _REGION_HINTS if n in texte), "") | |
| 222 | + | |
| 223 | + details = {} | |
| 224 | + if d.get("location_minimum"): | |
| 225 | + details["location_minimum"] = d["location_minimum"] | |
| 226 | + | |
| 227 | + listings.append(StListing( | |
| 228 | + source=self.source_id, | |
| 229 | + external_id=slug, # slug WP, stable | |
| 230 | + url=url, | |
| 231 | + title=d["title"], | |
| 232 | + property_type=ptype, | |
| 233 | + address=d.get("address", ""), | |
| 234 | + city=ville, | |
| 235 | + region=region, | |
| 236 | + price_night=parse_price_night(d.get("price_label", "")), | |
| 237 | + price_label=d.get("price_label", ""), | |
| 238 | + capacity=d.get("capacity"), | |
| 239 | + bedrooms=d.get("bedrooms"), | |
| 240 | + bathrooms=d.get("bathrooms"), | |
| 241 | + pets=d.get("pets"), | |
| 242 | + citq=d.get("citq", ""), | |
| 243 | + description=d.get("description", ""), | |
| 244 | + amenities=d.get("amenities") or [], | |
| 245 | + details=details, | |
| 246 | + images=d.get("images") or [], | |
| 247 | + )) | |
| 248 | + return listings | |
added
louka/shortterm/connectors/panoraloges.py
+135 −0
@@ -0,0 +1,135 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/panoraloges.py : Panora Loges Fluviales (panoraloges.com) — | |
| 4 | +# 10 loges (mini-chalets) en bord de mer à Sainte-Anne-des-Monts (Gaspésie), | |
| 5 | +# déclinées en DEUX gammes identiques à l'intérieur d'une même gamme : | |
| 6 | +# loges simples (nos 8-10, 2 pers.) et loges doubles (nos 1-7, 4 pers.). | |
| 7 | +# | |
| 8 | +# Méthode : site Squarespace ; l'API ?format=json de la page /fr/loges est | |
| 9 | +# vide (items=0) → on parse les DEUX pages de gamme statiques | |
| 10 | +# /fr/logessimples et /fr/logesdoubles. Chaque page expose en <h2-4> : | |
| 11 | +# prix (« 175 $ » sous « À partir de »), « N invités », « N chambres », | |
| 12 | +# « N lits Queen », « 1 salle de bain », puis la liste des commodités | |
| 13 | +# jusqu'à « Stationnements ». 2 fiches « gamme » (multi_unit) plutôt que | |
| 14 | +# 10 fiches par unité artificiellement identiques — le site lui-même | |
| 15 | +# laisse le libre choix de la loge à l'intérieur d'une gamme. | |
| 16 | +# ----------------------------------------------------------------------------- | |
| 17 | +from __future__ import annotations | |
| 18 | + | |
| 19 | +import html as _html | |
| 20 | +import re | |
| 21 | + | |
| 22 | +from ..schema import StListing | |
| 23 | +from .base import StConnector | |
| 24 | + | |
| 25 | +SITE = "https://www.panoraloges.com" | |
| 26 | + | |
| 27 | +_PAGES = [ | |
| 28 | + # (external_id, chemin, titre, unités) | |
| 29 | + ("loges-doubles", "/fr/logesdoubles", "Loges doubles — Panora Loges " | |
| 30 | + "Fluviales", "Loges 1 à 7"), | |
| 31 | + ("loges-simples", "/fr/logessimples", "Loges simples — Panora Loges " | |
| 32 | + "Fluviales", "Loges 8 à 10"), | |
| 33 | +] | |
| 34 | + | |
| 35 | +_TAG_RE = re.compile(r"<[^>]+>") | |
| 36 | + | |
| 37 | +# images de gabarit à exclure de la galerie | |
| 38 | +_IMG_EXCLUDE = ("favicon", "logo", "icon", "texture", "panora_ecran", | |
| 39 | + "Fond+de+page", "plan") | |
| 40 | + | |
| 41 | + | |
| 42 | +def _text(fragment: str) -> str: | |
| 43 | + return _html.unescape(re.sub(r"\s+", " ", _TAG_RE.sub(" ", fragment))).strip() | |
| 44 | + | |
| 45 | + | |
| 46 | +class PanoraLoges(StConnector): | |
| 47 | + source_id = "panoraloges" | |
| 48 | + request_delay = 1.0 | |
| 49 | + | |
| 50 | + def _parse(self, path: str) -> dict: | |
| 51 | + h = self.get(SITE + path).text | |
| 52 | + d: dict = {} | |
| 53 | + | |
| 54 | + heads = [t for t in (_text(x) for x in re.findall( | |
| 55 | + r"(?s)<h[1-4][^>]*>(.*?)</h[1-4]>", h)) if t] | |
| 56 | + | |
| 57 | + m = re.search(r"(\d+)\s*\$", " ".join(heads)) | |
| 58 | + if m: | |
| 59 | + d["price_night"] = float(m.group(1)) | |
| 60 | + | |
| 61 | + # commodités : entêtes entre « N invités » et « Stationnements » | |
| 62 | + start = next((i for i, t in enumerate(heads) | |
| 63 | + if re.fullmatch(r"\d+\s+invités?", t)), None) | |
| 64 | + end = next((i for i, t in enumerate(heads) | |
| 65 | + if t.lower().startswith("stationnement")), None) | |
| 66 | + amen: list[str] = [] | |
| 67 | + if start is not None and end is not None and start < end: | |
| 68 | + amen = heads[start:end + 1] | |
| 69 | + blob = " | ".join(amen) | |
| 70 | + for pat, key in ((r"(\d+)\s+invités?", "capacity"), | |
| 71 | + (r"(\d+)\s+chambres?", "bedrooms"), | |
| 72 | + (r"(\d+)\s+lits?\s+Queen", "beds"), | |
| 73 | + (r"(\d+)\s+salles?\s+de\s+bain", "bathrooms")): | |
| 74 | + m = re.search(pat, blob, re.I) | |
| 75 | + if m: | |
| 76 | + d[key] = float(m.group(1)) | |
| 77 | + d["amenities"] = [a for a in amen | |
| 78 | + if not re.fullmatch(r"\d+\s+(invités?|chambres?" | |
| 79 | + r"|salles?\s+de\s+bain)", a)] | |
| 80 | + | |
| 81 | + # description : sous-titre de gamme + infos pratiques (les <p> du | |
| 82 | + # gabarit Squarespace contiennent beaucoup de CSS → filtrés) | |
| 83 | + paras = [_text(p) for p in re.findall(r"(?s)<p[^>]*>(.*?)</p>", h)] | |
| 84 | + keep = [p for p in paras | |
| 85 | + if 60 < len(p) < 600 and "{" not in p | |
| 86 | + and not p.startswith(("FR:", "0 ")) | |
| 87 | + and "Infolettre" not in p and "418-" not in p] | |
| 88 | + if keep: | |
| 89 | + d["description"] = "\n".join(keep[:6])[:3000] | |
| 90 | + | |
| 91 | + imgs: list[str] = [] | |
| 92 | + for u in re.findall(r"https://images\.squarespace-cdn\.com/[^\"\s?]+", | |
| 93 | + h): | |
| 94 | + if any(x.lower() in u.lower() for x in _IMG_EXCLUDE): | |
| 95 | + continue | |
| 96 | + u = u.rstrip("\\,&") | |
| 97 | + if not re.search(r"\.(?:jpe?g|png|webp)$", u, re.I): | |
| 98 | + continue | |
| 99 | + if u not in imgs: | |
| 100 | + imgs.append(u) | |
| 101 | + if len(imgs) >= 15: | |
| 102 | + break | |
| 103 | + d["images"] = imgs | |
| 104 | + return d | |
| 105 | + | |
| 106 | + # -- contrat -------------------------------------------------------------- | |
| 107 | + def fetch(self) -> list[StListing]: | |
| 108 | + listings: list[StListing] = [] | |
| 109 | + for ext_id, path, title, units in _PAGES: | |
| 110 | + d = self._parse(path) | |
| 111 | + if not d.get("price_night") and not d.get("capacity"): | |
| 112 | + continue | |
| 113 | + price = d.get("price_night") | |
| 114 | + listings.append(StListing( | |
| 115 | + source=self.source_id, | |
| 116 | + external_id=ext_id, | |
| 117 | + url=SITE + path, | |
| 118 | + title=title, | |
| 119 | + property_type="Chalet", | |
| 120 | + address="610 boul. Sainte-Anne E", | |
| 121 | + city="Sainte-Anne-des-Monts", | |
| 122 | + region="Gaspésie", | |
| 123 | + price_night=price, | |
| 124 | + price_label=f"À partir de {price:g} $ / nuit" if price else "", | |
| 125 | + capacity=d.get("capacity"), | |
| 126 | + bedrooms=d.get("bedrooms"), | |
| 127 | + beds=d.get("beds"), | |
| 128 | + bathrooms=d.get("bathrooms"), | |
| 129 | + description=d.get("description") or "", | |
| 130 | + amenities=d.get("amenities") or [], | |
| 131 | + details={"multi_unit": True, "units": units, | |
| 132 | + "domain": "Panora Loges Fluviales"}, | |
| 133 | + images=d.get("images") or [], | |
| 134 | + )) | |
| 135 | + return listings | |
modified
louka/shortterm/connectors/parcscanada.py
+135 −15
@@ -16,7 +16,14 @@ | ||
| 16 | 16 | # description fr, capacité, photos, catégorie. On ne garde que les |
| 17 | 17 | # catégories « hébergement » (table KEEP ci-dessous) ; |
| 18 | 18 | # 3. prix (cache self.detail, clé mensuelle) : POST /api/resource/feeDetails |
| 19 | −# ?resourceId=…&startDate=<J+35> → feeTotal = tarif d'une nuit. | |
| 19 | +# ?resourceId=…&startDate=<J+n> → feeTotal = tarif d'une nuit. Les unités | |
| 20 | +# fermées à J+35 (saisonnier) sont réessayées à J+95/185/275 ; | |
| 21 | +# 4. commodités : chaque ressource porte des definedAttributes | |
| 22 | +# (attributeDefinitionId + valeur/énums) que GET /api/attribute/filterable | |
| 23 | +# permet de traduire en libellés français (« Foyer sur l'emplacement », | |
| 24 | +# « Chiens permis », « Éclairage fourni : solaire »…) — on ne retient que | |
| 25 | +# les attributs pertinents (table AMEN_DEFS), les dimensions techniques | |
| 26 | +# d'emplacement sont ignorées. | |
| 20 | 27 | # |
| 21 | 28 | # Les ids sont des entiers négatifs (int32 min + n). Aucune géoloc par unité |
| 22 | 29 | # dans l'API : lat/lng et région touristique viennent de la table statique |
@@ -78,6 +85,40 @@ KEEP = { | ||
| 78 | 85 | |
| 79 | 86 | _TAG_RE = re.compile(r"<[^>]+>") |
| 80 | 87 | |
| 88 | +# Attributs « commodités » pertinents (id de définition → retenu). Les autres | |
| 89 | +# (dimensions, pentes, obstructions, ampérage…) sont du bruit technique. | |
| 90 | +AMEN_DEFS = { | |
| 91 | + -32758, # Foyer sur l'emplacement | |
| 92 | + -32571, # Type de foyer | |
| 93 | + -32717, # Feux de camp permis | |
| 94 | + -32721, # Admissible au permis de feu | |
| 95 | + -32753, # Wi-fi | |
| 96 | + -32754, # Couverture cellulaire | |
| 97 | + -32760, # Barbecue fourni | |
| 98 | + -32570, # Fourneau fourni | |
| 99 | + -32762, # Source de chaleur interne disponible | |
| 100 | + -32763, # Éclairage fourni | |
| 101 | + -32764, # Électricité disponible à l'intérieur | |
| 102 | + -32761, # Électricité disponible à l'extérieur | |
| 103 | + -32582, # Électricité | |
| 104 | + -32736, # Service d'eau | |
| 105 | + -32756, # Accessible | |
| 106 | + -32723, # Secteur riverain | |
| 107 | + -32724, # Accès au rivage | |
| 108 | + -32725, # Conditions de baignade | |
| 109 | + -32757, # Tables de pique-nique | |
| 110 | + -32748, # Ombrage à l'emplacement | |
| 111 | + -32759, # Casiers à provisions fournis | |
| 112 | + -32713, # Stationnement | |
| 113 | + -32709, # Permis de pêche inclus | |
| 114 | + -32715, # Vélos permis | |
| 115 | +} | |
| 116 | +PETS_DEFS = {-32718, -32572} # Chiens permis / Animaux de compagnie permis | |
| 117 | +MIN_STAY_DEF = -32697 # Séjour minimum (nuits) | |
| 118 | + | |
| 119 | +# valeurs d'énum « négatives » : l'attribut est alors omis des commodités | |
| 120 | +_VAL_NEGATIVES = {"non", "no", "aucun", "aucune", "none", "n/a"} | |
| 121 | + | |
| 81 | 122 | |
| 82 | 123 | def _fr(localized: list[dict], *keys: str) -> dict: |
| 83 | 124 | by_culture = {v.get("cultureName"): v for v in (localized or [])} |
@@ -105,17 +146,80 @@ class ParcsCanada(StConnector): | ||
| 105 | 146 | |
| 106 | 147 | # -- prix d'une nuit (cache BD, clé mensuelle) ------------------------------- |
| 107 | 148 | def _fetch_fee(self, resource_id: int) -> dict: |
| 108 | − start = (datetime.date.today() | |
| 109 | − + datetime.timedelta(days=35)).isoformat() | |
| 110 | − resp = self.post( | |
| 111 | − BASE + "/api/resource/feeDetails", | |
| 112 | − params={"resourceId": resource_id, "startDate": start, | |
| 113 | − "boatLength": 0, "bookingCategoryId": 0, | |
| 114 | − "entryPointResourceId": 0, "exitPointResourceId": 0}, | |
| 115 | − json=[], headers=HEADERS) | |
| 116 | − fees = (resp.json() or {}).get("resourceFeeDetails") or [] | |
| 117 | − total = sum(f.get("feeTotal") or 0 for f in fees) | |
| 118 | − return {"fee": round(total, 2)} if total > 0 else {} | |
| 149 | + # unités saisonnières : pas de tarif hors saison → on sonde plusieurs | |
| 150 | + # dates réparties sur l'année jusqu'à trouver une nuit tarifée | |
| 151 | + for jours in (35, 95, 185, 275): | |
| 152 | + start = (datetime.date.today() | |
| 153 | + + datetime.timedelta(days=jours)).isoformat() | |
| 154 | + try: | |
| 155 | + resp = self.post( | |
| 156 | + BASE + "/api/resource/feeDetails", | |
| 157 | + params={"resourceId": resource_id, "startDate": start, | |
| 158 | + "boatLength": 0, "bookingCategoryId": 0, | |
| 159 | + "entryPointResourceId": 0, | |
| 160 | + "exitPointResourceId": 0}, | |
| 161 | + json=[], headers=HEADERS) | |
| 162 | + except Exception: # noqa: BLE001 — on tente la date suivante | |
| 163 | + continue | |
| 164 | + fees = (resp.json() or {}).get("resourceFeeDetails") or [] | |
| 165 | + total = sum(f.get("feeTotal") or 0 for f in fees) | |
| 166 | + if total > 0: | |
| 167 | + return {"fee": round(total, 2), "date": start} | |
| 168 | + return {} | |
| 169 | + | |
| 170 | + # -- définitions d'attributs (libellés fr des commodités) -------------------- | |
| 171 | + def _fetch_attr_defs(self) -> dict: | |
| 172 | + data = self._api("/api/attribute/filterable") or {} | |
| 173 | + defs: dict = {} | |
| 174 | + for key, v in data.items(): | |
| 175 | + fr = _fr(v.get("localizedValues") or []) | |
| 176 | + name = fr.get("displayName") or "" | |
| 177 | + if not name: | |
| 178 | + continue | |
| 179 | + values = {} | |
| 180 | + for ev in v.get("values") or []: | |
| 181 | + lab = _fr(ev.get("localizedValues") or []).get("displayName") | |
| 182 | + if lab is not None and ev.get("enumValue") is not None: | |
| 183 | + values[str(ev["enumValue"])] = lab | |
| 184 | + defs[str(key)] = {"name": name, "values": values} | |
| 185 | + return defs | |
| 186 | + | |
| 187 | + @staticmethod | |
| 188 | + def _amenities(res: dict, defs: dict) -> tuple[list[str], str | None, | |
| 189 | + float | None]: | |
| 190 | + """(commodités, animaux oui/non, séjour minimum) d'une ressource.""" | |
| 191 | + amen: list[str] = [] | |
| 192 | + pets: str | None = None | |
| 193 | + min_stay: float | None = None | |
| 194 | + for da in res.get("definedAttributes") or []: | |
| 195 | + did = da.get("attributeDefinitionId") | |
| 196 | + d = defs.get(str(did)) | |
| 197 | + if did == MIN_STAY_DEF and da.get("value") is not None: | |
| 198 | + min_stay = float(da["value"]) | |
| 199 | + continue | |
| 200 | + if d is None: | |
| 201 | + continue | |
| 202 | + labels = [d["values"].get(str(v)) for v in (da.get("values") or [])] | |
| 203 | + labels = [l for l in labels if l] | |
| 204 | + if did in PETS_DEFS and labels: | |
| 205 | + pets = "non" if all( | |
| 206 | + l.lower() in _VAL_NEGATIVES for l in labels) else "oui" | |
| 207 | + continue | |
| 208 | + if did not in AMEN_DEFS: | |
| 209 | + continue | |
| 210 | + positifs = [l for l in labels | |
| 211 | + if l.lower() not in _VAL_NEGATIVES] | |
| 212 | + if not positifs: | |
| 213 | + continue | |
| 214 | + if all(l.lower() in ("oui", "yes") for l in positifs): | |
| 215 | + item = d["name"] # booléen : le nom suffit | |
| 216 | + else: | |
| 217 | + item = f"{d['name']} : {', '.join(positifs)}" | |
| 218 | + if item not in amen: | |
| 219 | + amen.append(item) | |
| 220 | + # l'API renvoie les attributs dans un ordre variable : trier pour | |
| 221 | + # un diff stable dans la base | |
| 222 | + return sorted(amen), pets, min_stay | |
| 119 | 223 | |
| 120 | 224 | # -- contrat -------------------------------------------------------------- |
| 121 | 225 | def fetch(self) -> list[StListing]: |
@@ -128,6 +232,14 @@ class ParcsCanada(StConnector): | ||
| 128 | 232 | cat_names[c.get("resourceCategoryId")] = name |
| 129 | 233 | |
| 130 | 234 | month = datetime.date.today().strftime("%Y-%m") # prix revus au mois |
| 235 | + | |
| 236 | + # libellés fr des attributs (une requête, cache BD mensuel) | |
| 237 | + try: | |
| 238 | + attr_defs = self.detail("_attrdefs", month, self._fetch_attr_defs) | |
| 239 | + except Exception as exc: # noqa: BLE001 — les commodités sont optionnelles | |
| 240 | + print(f"[parcscanada] définitions d'attributs : {exc}", | |
| 241 | + file=sys.stderr) | |
| 242 | + attr_defs = {} | |
| 131 | 243 | listings: list[StListing] = [] |
| 132 | 244 | for loc_id, (loc_name, region, lat, lng) in QC_LOCATIONS.items(): |
| 133 | 245 | try: |
@@ -162,13 +274,20 @@ class ParcsCanada(StConnector): | ||
| 162 | 274 | title = f"{cat_fr} {name} – {loc_name}" |
| 163 | 275 | |
| 164 | 276 | try: |
| 165 | − fee = self.detail(str(rid), month, | |
| 277 | + # « v2 » : sondage multi-dates (unités saisonnières) | |
| 278 | + fee = self.detail(str(rid), month + ":v2", | |
| 166 | 279 | lambda r=rid: self._fetch_fee(r)) |
| 167 | 280 | except Exception as exc: # noqa: BLE001 — le prix est optionnel |
| 168 | 281 | print(f"[parcscanada] tarif {rid} : {exc}", file=sys.stderr) |
| 169 | 282 | fee = {} |
| 170 | 283 | price = fee.get("fee") |
| 171 | 284 | |
| 285 | + amenities, pets_attr, min_stay = self._amenities( | |
| 286 | + res, attr_defs) | |
| 287 | + details = {"parc": loc_name, "categorie": cat_fr} | |
| 288 | + if min_stay: | |
| 289 | + details["sejour_minimum"] = min_stay | |
| 290 | + | |
| 172 | 291 | cap = res.get("maxCapacity") |
| 173 | 292 | listings.append(StListing( |
| 174 | 293 | source=self.source_id, |
@@ -182,9 +301,10 @@ class ParcsCanada(StConnector): | ||
| 182 | 301 | price_night=float(price) if price else None, |
| 183 | 302 | price_label=(f"{price:.2f} $ / nuit" if price else ""), |
| 184 | 303 | capacity=float(cap) if cap else None, |
| 185 | − pets=_pets(desc), | |
| 304 | + pets=pets_attr or _pets(desc), | |
| 186 | 305 | description=desc[:4000], |
| 187 | − details={"parc": loc_name, "categorie": cat_fr}, | |
| 306 | + amenities=amenities, | |
| 307 | + details=details, | |
| 188 | 308 | images=images, |
| 189 | 309 | lat=lat, |
| 190 | 310 | lng=lng, |
added
louka/shortterm/connectors/pourvoiries.py
+288 −0
@@ -0,0 +1,288 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/pourvoiries.py : Fédération des pourvoiries du Québec | |
| 4 | +# (pourvoiries.com) — ~340 pourvoiries avec hébergement (chalets, camps, | |
| 5 | +# pavillons, prêt-à-camper) partout en région. | |
| 6 | +# | |
| 7 | +# Méthode (TYPO3, aucun anti-bot) : | |
| 8 | +# 1. sitemap officiel des établissements (/sitemap/outfitters/sitemap.xml) | |
| 9 | +# → inventaire complet des fiches /pourvoiries/<slug>-<zone>-<permis> ; | |
| 10 | +# 2. fiche établissement (cache self.detail, clé mensuelle) : nom, ville + | |
| 11 | +# région (bandeau), description, coordonnées GPS (« Latitude : 48.805 »), | |
| 12 | +# no d'établissement CITQ, période d'ouverture, type de restauration, | |
| 13 | +# unités d'hébergement de l'onglet Hébergements (Pavillon / Chalet / | |
| 14 | +# Camp / Prêt-à-camper… avec capacité et chambres), photos du carrousel ; | |
| 15 | +# 3. prix : les fiches n'affichent pas de tarif d'hébergement ; le sitemap | |
| 16 | +# des forfaits (/sitemap/packages/sitemap.xml, slug préfixé du no de | |
| 17 | +# permis) donne un prix « par personne / nuit » pour ~80 pourvoiries → | |
| 18 | +# price_label du forfait le plus bas (à défaut d'un vrai prix/nuit). | |
| 19 | +# | |
| 20 | +# Une annonce = un établissement (les unités individuelles ne sont pas | |
| 21 | +# réservables en ligne — la liste des unités va dans details["unites"]). | |
| 22 | +# Seules les fiches avec au moins une unité d'hébergement sont retenues. | |
| 23 | +# Réglage env : LOUKA_POURVOIRIES_LIMIT (nb max de fiches, 0 = tout). | |
| 24 | +# ----------------------------------------------------------------------------- | |
| 25 | +from __future__ import annotations | |
| 26 | + | |
| 27 | +import os | |
| 28 | +import re | |
| 29 | +import sys | |
| 30 | +import time | |
| 31 | + | |
| 32 | +from ..schema import StListing, normalize_region | |
| 33 | +from .base import StConnector | |
| 34 | + | |
| 35 | +BASE = "https://www.pourvoiries.com" | |
| 36 | +SITEMAP_FICHES = f"{BASE}/sitemap/outfitters/sitemap.xml" | |
| 37 | +SITEMAP_FORFAITS = f"{BASE}/sitemap/packages/sitemap.xml" | |
| 38 | +PREFIX_FICHE = f"{BASE}/pourvoiries/" | |
| 39 | +PREFIX_FORFAIT = f"{BASE}/forfaits/" | |
| 40 | + | |
| 41 | +_LOC_RE = re.compile(r"<loc>\s*(https?://[^<\s]+)") | |
| 42 | +_PERMIS_RE = re.compile(r"(\d{2}-\d{3})$") # fin du slug établissement | |
| 43 | +_FORFAIT_PERMIS_RE = re.compile(r"^(\d{2}-\d{3})-") # début du slug forfait | |
| 44 | +_LAT_RE = re.compile(r"Latitude\s*:\s*(-?\d+\.\d+)") | |
| 45 | +_LNG_RE = re.compile(r"Longitude\s*:\s*(-?\d+\.\d+)") | |
| 46 | +_CAP_RE = re.compile(r"Pour\s+(\d+)\s+personne", re.I) | |
| 47 | +_CH_RE = re.compile(r"(\d+)\s+chambre", re.I) | |
| 48 | +_PRIX_FORFAIT_RE = re.compile( | |
| 49 | + r'<strong class="price">\s*([\d\s,. ]+)\s*\$', re.S) | |
| 50 | + | |
| 51 | +# Régions FPQ → région touristique canonique (le reste passe par | |
| 52 | +# normalize_region : Mauricie, Outaouais, Côte-Nord…) | |
| 53 | +_REGIONS_FPQ = { | |
| 54 | + "gaspésie et îles-de-la-madeleine": "Gaspésie", | |
| 55 | + "gaspesie et iles-de-la-madeleine": "Gaspésie", | |
| 56 | + "saguenay-lac-saint-jean": "Saguenay–Lac-Saint-Jean", | |
| 57 | + "nord-du-québec": "Nord-du-Québec", | |
| 58 | + "baie-james": "Eeyou Istchee Baie-James", | |
| 59 | +} | |
| 60 | + | |
| 61 | +# Titre de section d'hébergement → type canonique Lou-Ka | |
| 62 | +_TYPE_UNITE = { | |
| 63 | + "chalet": "Chalet", "pavillon": "Auberge", "auberge": "Auberge", | |
| 64 | + "camp": "Refuge", "refuge": "Refuge", "yourte": "Yourte", | |
| 65 | + "dôme": "Dôme", "tente": "Prêt-à-camper", | |
| 66 | + "prêt-à-camper": "Prêt-à-camper", "camping": "Camping", | |
| 67 | + "chambre": "Chambre", "maison": "Maison", "condo": "Condo", | |
| 68 | +} | |
| 69 | + | |
| 70 | + | |
| 71 | +def _property_type(types_unites: list[str]) -> str: | |
| 72 | + """Type dominant de l'établissement — le chalet prime (offre principale).""" | |
| 73 | + canon = [_TYPE_UNITE.get(t.strip().lower(), "") for t in types_unites] | |
| 74 | + for pref in ("Chalet", "Auberge", "Yourte", "Dôme", "Prêt-à-camper", | |
| 75 | + "Refuge", "Maison", "Condo", "Chambre", "Camping"): | |
| 76 | + if pref in canon: | |
| 77 | + return pref | |
| 78 | + return "Chalet" | |
| 79 | + | |
| 80 | + | |
| 81 | +class Pourvoiries(StConnector): | |
| 82 | + source_id = "pourvoiries" | |
| 83 | + request_delay = 0.8 | |
| 84 | + | |
| 85 | + # -- fiche établissement -------------------------------------------------- | |
| 86 | + def _fetch_fiche(self, slug: str) -> dict: | |
| 87 | + from bs4 import BeautifulSoup | |
| 88 | + html = self.get(PREFIX_FICHE + slug).text | |
| 89 | + soup = BeautifulSoup(html, "html.parser") | |
| 90 | + d: dict = {} | |
| 91 | + | |
| 92 | + h1 = soup.select_one("h1.page-title") | |
| 93 | + if not h1: | |
| 94 | + return {} | |
| 95 | + d["nom"] = h1.get_text(" ", strip=True) | |
| 96 | + | |
| 97 | + # bandeau : « Rivière-Bonjour, Gaspésie et Îles-de-la-Madeleine » | |
| 98 | + banner = soup.select_one(".banner-single .region") | |
| 99 | + if banner: | |
| 100 | + loc = banner.get_text(" ", strip=True) | |
| 101 | + ville, _, region = loc.partition(", ") | |
| 102 | + d["ville"], d["region"] = ville.strip(), region.strip() | |
| 103 | + | |
| 104 | + # description (premier bloc sous le h2 « Description ») | |
| 105 | + for h2 in soup.find_all("h2"): | |
| 106 | + if h2.get_text(strip=True).lower() == "description": | |
| 107 | + paras = [p.get_text(" ", strip=True) | |
| 108 | + for p in h2.find_all_next("p", limit=4)] | |
| 109 | + d["description"] = "\n".join(x for x in paras if x)[:2500] | |
| 110 | + break | |
| 111 | + | |
| 112 | + # onglet Informations : paires h3 → p | |
| 113 | + infos: dict[str, str] = {} | |
| 114 | + for h3 in soup.find_all("h3"): | |
| 115 | + p = h3.find_next_sibling("p") | |
| 116 | + if p is not None: | |
| 117 | + infos[h3.get_text(" ", strip=True).lower()] = \ | |
| 118 | + p.get_text(" ", strip=True) | |
| 119 | + for label, key in (("numéro d'établissement", "citq"), | |
| 120 | + ("période d'ouverture", "ouverture"), | |
| 121 | + ("type de restauration", "restauration"), | |
| 122 | + ("type de pourvoirie", "type_pourvoirie"), | |
| 123 | + ("langue de service", "langues")): | |
| 124 | + for k, v in infos.items(): | |
| 125 | + if k.startswith(label): | |
| 126 | + d[key] = v | |
| 127 | + break | |
| 128 | + | |
| 129 | + m = _LAT_RE.search(html) | |
| 130 | + if m: | |
| 131 | + d["lat"] = float(m.group(1)) | |
| 132 | + m = _LNG_RE.search(html) | |
| 133 | + if m: | |
| 134 | + d["lng"] = float(m.group(1)) | |
| 135 | + | |
| 136 | + # onglet Hébergements : sections (h2.block-title) → cartes d'unités | |
| 137 | + unites: list[dict] = [] | |
| 138 | + heb = soup.find(id="hebergements") | |
| 139 | + if heb is not None: | |
| 140 | + for bloc in heb.select(".block-slides"): | |
| 141 | + t = bloc.select_one("h2.block-title") | |
| 142 | + type_u = t.get_text(" ", strip=True) if t else "" | |
| 143 | + for card in bloc.select(".card"): | |
| 144 | + titre = card.select_one(".card-title") | |
| 145 | + if titre is None: | |
| 146 | + continue | |
| 147 | + txt = card.get_text(" ", strip=True) | |
| 148 | + cap = _CAP_RE.search(txt) | |
| 149 | + ch = _CH_RE.search(txt) | |
| 150 | + unites.append({ | |
| 151 | + "type": type_u, | |
| 152 | + "nom": titre.get_text(" ", strip=True), | |
| 153 | + "capacite": int(cap.group(1)) if cap else None, | |
| 154 | + "chambres": int(ch.group(1)) if ch else None, | |
| 155 | + "etoiles": len(card.select(".card-icons-stars " | |
| 156 | + ".icon-star")) or None, | |
| 157 | + }) | |
| 158 | + d["unites"] = unites | |
| 159 | + | |
| 160 | + # photos du carrousel principal | |
| 161 | + imgs: list[str] = [] | |
| 162 | + slider = soup.select_one(".block-slider-img") | |
| 163 | + if slider is not None: | |
| 164 | + for img in slider.find_all("img"): | |
| 165 | + u = img.get("data-src") or img.get("src") or "" | |
| 166 | + if u.startswith("/"): | |
| 167 | + u = BASE + u | |
| 168 | + if u.startswith("https://") and u not in imgs: | |
| 169 | + imgs.append(u) | |
| 170 | + d["images"] = imgs[:12] | |
| 171 | + return d | |
| 172 | + | |
| 173 | + # -- page forfait (prix « par personne / nuit ») --------------------------- | |
| 174 | + def _fetch_forfait(self, slug: str) -> dict: | |
| 175 | + html = self.get(PREFIX_FORFAIT + slug).text | |
| 176 | + m = _PRIX_FORFAIT_RE.search(html) | |
| 177 | + if not m: | |
| 178 | + return {} | |
| 179 | + try: | |
| 180 | + prix = float(re.sub(r"[\s ]", "", m.group(1)).replace(",", ".")) | |
| 181 | + except ValueError: | |
| 182 | + return {} | |
| 183 | + unite = "" | |
| 184 | + mm = re.search(r'class="card-pricing">.*?<p>([^<]+)</p>', html, re.S) | |
| 185 | + if mm: | |
| 186 | + unite = mm.group(1).strip() | |
| 187 | + return {"prix": prix, "unite": unite} | |
| 188 | + | |
| 189 | + # -- inventaire ------------------------------------------------------------ | |
| 190 | + def fetch(self) -> list[StListing]: | |
| 191 | + limit = int(os.environ.get("LOUKA_POURVOIRIES_LIMIT", "0") or 0) | |
| 192 | + month = time.strftime("%Y-%m") # re-visite mensuelle des fiches | |
| 193 | + | |
| 194 | + xml = self.get(SITEMAP_FICHES).text | |
| 195 | + slugs = sorted({loc[len(PREFIX_FICHE):].strip("/") | |
| 196 | + for loc in _LOC_RE.findall(xml) | |
| 197 | + if loc.startswith(PREFIX_FICHE)}) | |
| 198 | + | |
| 199 | + # forfaits groupés par no de permis (slug « 01-501-… ») | |
| 200 | + forfaits: dict[str, list[str]] = {} | |
| 201 | + try: | |
| 202 | + xmlf = self.get(SITEMAP_FORFAITS).text | |
| 203 | + for loc in _LOC_RE.findall(xmlf): | |
| 204 | + if not loc.startswith(PREFIX_FORFAIT): | |
| 205 | + continue | |
| 206 | + fslug = loc[len(PREFIX_FORFAIT):].strip("/") | |
| 207 | + m = _FORFAIT_PERMIS_RE.match(fslug) | |
| 208 | + if m: | |
| 209 | + forfaits.setdefault(m.group(1), []).append(fslug) | |
| 210 | + except Exception as exc: # noqa: BLE001 — les forfaits sont optionnels | |
| 211 | + print(f"[pourvoiries] sitemap forfaits : {exc}", file=sys.stderr) | |
| 212 | + | |
| 213 | + listings: list[StListing] = [] | |
| 214 | + for slug in slugs: | |
| 215 | + try: | |
| 216 | + d = self.detail(slug, month, lambda s=slug: self._fetch_fiche(s)) | |
| 217 | + except Exception as exc: # noqa: BLE001 | |
| 218 | + print(f"[pourvoiries] fiche {slug} : {exc}", file=sys.stderr) | |
| 219 | + continue | |
| 220 | + unites = d.get("unites") or [] | |
| 221 | + if not d.get("nom") or not unites: # pas d'hébergement → hors sujet | |
| 222 | + continue | |
| 223 | + | |
| 224 | + # prix : forfait le moins cher de la pourvoirie (par pers. / nuit) | |
| 225 | + price_label = "" | |
| 226 | + m = _PERMIS_RE.search(slug) | |
| 227 | + permis = m.group(1) if m else "" | |
| 228 | + best: dict = {} | |
| 229 | + for fslug in forfaits.get(permis, []): | |
| 230 | + try: | |
| 231 | + f = self.detail(f"forfait:{fslug}", month, | |
| 232 | + lambda s=fslug: self._fetch_forfait(s)) | |
| 233 | + except Exception as exc: # noqa: BLE001 | |
| 234 | + print(f"[pourvoiries] forfait {fslug} : {exc}", | |
| 235 | + file=sys.stderr) | |
| 236 | + continue | |
| 237 | + # seuls les forfaits tarifés à la nuit (ou au jour) sont | |
| 238 | + # comparables — un prix « par personne / séjour » fausserait | |
| 239 | + # le prix/nuit dérivé par finalize(). Attention : « séjour » | |
| 240 | + # contient « jour », d'où l'exclusion explicite. | |
| 241 | + unite = f.get("unite", "").lower() | |
| 242 | + if "jour" not in unite and "nuit" not in unite: | |
| 243 | + continue | |
| 244 | + if "séjour" in unite or "sejour" in unite: | |
| 245 | + continue | |
| 246 | + if f.get("prix") and (not best or f["prix"] < best["prix"]): | |
| 247 | + best = f | |
| 248 | + if best: | |
| 249 | + unite = best.get("unite") or "par personne / nuit" | |
| 250 | + price_label = (f"forfait à partir de {best['prix']:.0f} $ " | |
| 251 | + f"{unite}") | |
| 252 | + | |
| 253 | + region = d.get("region", "") | |
| 254 | + region = _REGIONS_FPQ.get(region.lower(), normalize_region(region)) | |
| 255 | + | |
| 256 | + caps = [u["capacite"] for u in unites if u.get("capacite")] | |
| 257 | + chs = [u["chambres"] for u in unites if u.get("chambres")] | |
| 258 | + details = {k: v for k, v in { | |
| 259 | + "permis": permis, | |
| 260 | + "nb_unites": len(unites), | |
| 261 | + "unites": unites[:40], | |
| 262 | + "ouverture": d.get("ouverture", ""), | |
| 263 | + "restauration": d.get("restauration", ""), | |
| 264 | + "type_pourvoirie": d.get("type_pourvoirie", ""), | |
| 265 | + "langues": d.get("langues", ""), | |
| 266 | + }.items() if v} | |
| 267 | + | |
| 268 | + listings.append(StListing( | |
| 269 | + source=self.source_id, | |
| 270 | + external_id=slug, | |
| 271 | + url=PREFIX_FICHE + slug, | |
| 272 | + title=d["nom"], | |
| 273 | + property_type=_property_type([u["type"] for u in unites]), | |
| 274 | + city=d.get("ville", ""), | |
| 275 | + region=region, | |
| 276 | + price_label=price_label, | |
| 277 | + capacity=float(max(caps)) if caps else None, | |
| 278 | + bedrooms=float(max(chs)) if chs else None, | |
| 279 | + citq=d.get("citq", ""), | |
| 280 | + description=d.get("description", ""), | |
| 281 | + details=details, | |
| 282 | + images=d.get("images") or [], | |
| 283 | + lat=d.get("lat"), | |
| 284 | + lng=d.get("lng"), | |
| 285 | + )) | |
| 286 | + if limit and len(listings) >= limit: | |
| 287 | + break | |
| 288 | + return listings | |
modified
louka/shortterm/connectors/qldc.py
+54 −7
@@ -7,10 +7,18 @@ | ||
| 7 | 7 | # Méthode : pagination de la liste globale /chalets-a-louer?page=N (site |
| 8 | 8 | # ASP.NET WebForms, 12 cartes/page, HTML statique — la pagination « infinie » |
| 9 | 9 | # accepte le paramètre ?page). Cartes : id stable (/chalet-a-louer/<id>), |
| 10 | −# titre, région + ville, capacité, chambres, photo. La page détail (via | |
| 11 | −# self.detail, cache BD) en variante ?map=o ajoute lat/lng (champs cachés | |
| 12 | −# InfoLocalisation_hf_lat/long — absents de la page de base), grille de | |
| 13 | −# tarifs, description, no CITQ, sdb/lits, commodités et photos. | |
| 10 | +# titre, région + ville, capacité, chambres, photo, et souvent un prix | |
| 11 | +# « à partir de » (encadré .ListPrix : « Nuit 395$ » ou « Semaine 1030$ »). | |
| 12 | +# La page détail (via self.detail, cache BD) en variante ?map=o ajoute | |
| 13 | +# lat/lng (champs cachés InfoLocalisation_hf_lat/long — absents de la page | |
| 14 | +# de base), grille de tarifs, description, no CITQ, sdb/lits, commodités | |
| 15 | +# et photos. | |
| 16 | +# | |
| 17 | +# Prix : ~40 % des fiches seulement ont la grille de tarifs ; les autres ont | |
| 18 | +# soit un tarif en texte libre (ctl16_lblvchTarif_Terme, parfois avec | |
| 19 | +# montants — attention aux dépôts), soit rien du tout (contact direct). | |
| 20 | +# Ordre de préférence : grille détail > texte libre détail > encadré de la | |
| 21 | +# carte liste. Beaucoup de fiches n'affichent réellement aucun prix. | |
| 14 | 22 | # ----------------------------------------------------------------------------- |
| 15 | 23 | from __future__ import annotations |
| 16 | 24 | |
@@ -88,10 +96,11 @@ class QuebecLocationDeChalets(StConnector): | ||
| 88 | 96 | page += 1 |
| 89 | 97 | |
| 90 | 98 | for lst in listings: |
| 99 | + # « v2 » : tarif en texte libre ajouté au parseur détail | |
| 91 | 100 | cle = hashlib.sha1(("|".join([ |
| 92 | 101 | lst.title, lst.city, lst.region, |
| 93 | 102 | str(lst.capacity), str(lst.bedrooms), |
| 94 | − ]) + time.strftime("|%Y-%m")).encode("utf-8")).hexdigest() | |
| 103 | + ]) + time.strftime("|%Y-%m|v2")).encode("utf-8")).hexdigest() | |
| 95 | 104 | try: |
| 96 | 105 | d = self.detail(lst.external_id, cle, |
| 97 | 106 | lambda u=lst.url: self._detail(u)) |
@@ -99,8 +108,12 @@ class QuebecLocationDeChalets(StConnector): | ||
| 99 | 108 | d = {} |
| 100 | 109 | if not d: |
| 101 | 110 | continue |
| 102 | − lst.price_night = d.get("price_night") | |
| 103 | − lst.price_label = d.get("price_label") or "" | |
| 111 | + # le prix de la page détail prime ; sinon on garde celui de la | |
| 112 | + # carte de liste (« à partir de … ») | |
| 113 | + if d.get("price_night") is not None: | |
| 114 | + lst.price_night = d["price_night"] | |
| 115 | + if d.get("price_label"): | |
| 116 | + lst.price_label = d["price_label"] | |
| 104 | 117 | lst.description = d.get("description") or "" |
| 105 | 118 | lst.citq = d.get("citq") or "" |
| 106 | 119 | lst.amenities = d.get("amenities") or [] |
@@ -159,6 +172,16 @@ class QuebecLocationDeChalets(StConnector): | ||
| 159 | 172 | if img is not None: |
| 160 | 173 | images.append(urljoin(BASE, img["src"].split("?")[0])) |
| 161 | 174 | |
| 175 | + # encadré de prix de la carte (« à partir de / Nuit 395$ » ou | |
| 176 | + # « Semaine 1030$ ») — repli si la page détail n'affiche aucun tarif. | |
| 177 | + # NE PAS remonter plus haut que la carte : on attraperait le prix | |
| 178 | + # d'une carte voisine. | |
| 179 | + prix_label, prix_nuit = "", None | |
| 180 | + bloc = carte.select_one(".ListPrix") if carte is not None else None | |
| 181 | + if bloc is not None: | |
| 182 | + prix_label = re.sub(r"\s+", " ", bloc.get_text(" ", strip=True)) | |
| 183 | + prix_nuit = _prix_nuit(prix_label, prix_label) | |
| 184 | + | |
| 162 | 185 | return StListing( |
| 163 | 186 | source=self.source_id, |
| 164 | 187 | external_id=eid, |
@@ -167,6 +190,8 @@ class QuebecLocationDeChalets(StConnector): | ||
| 167 | 190 | property_type="Chalet", |
| 168 | 191 | city=ville, |
| 169 | 192 | region=_REGIONS.get(region, region), |
| 193 | + price_night=prix_nuit, | |
| 194 | + price_label=prix_label, | |
| 170 | 195 | capacity=capacite, |
| 171 | 196 | bedrooms=chambres, |
| 172 | 197 | images=images, |
@@ -238,6 +263,28 @@ class QuebecLocationDeChalets(StConnector): | ||
| 238 | 263 | d["price_night"] = _prix_nuit(d["price_label"], |
| 239 | 264 | d["price_label"]) |
| 240 | 265 | |
| 266 | + # tarif en texte libre (fiches sans grille) : on ne retient que les | |
| 267 | + # phrases avec un montant ET une période (nuit/jour/semaine), en | |
| 268 | + # ignorant dépôts et cautions | |
| 269 | + if "price_night" not in d: | |
| 270 | + terme = soup.find(id="ctl16_lblvchTarif_Terme") | |
| 271 | + if terme is not None: | |
| 272 | + candidats = [] | |
| 273 | + # split en phrases sans casser les décimales (« 129.00$ ») | |
| 274 | + for phrase in re.split(r"[\n;•]|\.(?!\d)", | |
| 275 | + terme.get_text("\n", strip=True)): | |
| 276 | + if "$" not in phrase \ | |
| 277 | + or re.search(r"(?i)d[ée]p[ôo]t|caution|rabais", phrase) \ | |
| 278 | + or not re.search(r"(?i)nuit|jour|sem", phrase): | |
| 279 | + continue | |
| 280 | + pn = _prix_nuit(phrase, phrase) | |
| 281 | + if pn: | |
| 282 | + candidats.append((pn, phrase.strip())) | |
| 283 | + if candidats: | |
| 284 | + pn, phrase = min(candidats) | |
| 285 | + d["price_night"] = pn | |
| 286 | + d.setdefault("price_label", re.sub(r"\s+", " ", phrase)[:120]) | |
| 287 | + | |
| 241 | 288 | desc = soup.find(id="InfoDescription_pnlDescription") |
| 242 | 289 | if desc is not None: |
| 243 | 290 | texte = desc.get_text("\n", strip=True) |
added
louka/shortterm/connectors/rezerve.py
+116 −0
@@ -0,0 +1,116 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/rezerve.py : Rëzerve / reserver.ca (Charlevoix, Laurentides, | |
| 4 | +# Estrie, Lanaudière, Mauricie, Outaouais…) | |
| 5 | +# | |
| 6 | +# Gestionnaire québécois (~85 chalets) sur moteur Guesty. Le site reserver.ca | |
| 7 | +# est un Next.js (App Router) : le catalogue /chalets est RENDU SERVEUR et le | |
| 8 | +# payload React Flight (`self.__next_f.push([1,"…"])`) contient l'objet | |
| 9 | +# {"listings":[…]} complet — id Guesty, slug, nom, description fr, photos, | |
| 10 | +# chambres, sdb, lits, capacité, prix « à partir de » CAD, ville, adresse, | |
| 11 | +# lat/lng, commodités, numéro CITQ, heures d'arrivée/départ. UNE SEULE requête | |
| 12 | +# suffit, aucune page détail à visiter. | |
| 13 | +# External_id = id Guesty (stable). URL publique : /chalets/<slug>. | |
| 14 | +# ----------------------------------------------------------------------------- | |
| 15 | +from __future__ import annotations | |
| 16 | + | |
| 17 | +import json | |
| 18 | +import re | |
| 19 | + | |
| 20 | +from ..schema import StListing | |
| 21 | +from .base import StConnector | |
| 22 | + | |
| 23 | +SITE = "https://reserver.ca" | |
| 24 | + | |
| 25 | + | |
| 26 | +def _num(v) -> float | None: | |
| 27 | + try: | |
| 28 | + return float(v) if v not in (None, "") else None | |
| 29 | + except (TypeError, ValueError): | |
| 30 | + return None | |
| 31 | + | |
| 32 | + | |
| 33 | +def _flight_blob(html: str) -> str: | |
| 34 | + """Reconstitue le payload React Flight (chunks __next_f concaténés).""" | |
| 35 | + parts = [] | |
| 36 | + for c in re.findall(r'self\.__next_f\.push\(\[1,"((?:[^"\\]|\\.)*)"\]\)', | |
| 37 | + html): | |
| 38 | + try: | |
| 39 | + parts.append(json.loads(f'"{c}"')) | |
| 40 | + except ValueError: | |
| 41 | + continue | |
| 42 | + return "".join(parts) | |
| 43 | + | |
| 44 | + | |
| 45 | +class Rezerve(StConnector): | |
| 46 | + source_id = "rezerve" | |
| 47 | + | |
| 48 | + def _catalog(self) -> list[dict]: | |
| 49 | + html = self.get(f"{SITE}/chalets").text | |
| 50 | + blob = _flight_blob(html) | |
| 51 | + i = blob.find('{"listings":[') | |
| 52 | + if i < 0: | |
| 53 | + return [] | |
| 54 | + obj, _ = json.JSONDecoder().raw_decode(blob[i:]) | |
| 55 | + return obj.get("listings") or [] | |
| 56 | + | |
| 57 | + @staticmethod | |
| 58 | + def _description(it: dict) -> str: | |
| 59 | + desc = str(it.get("description") or "") | |
| 60 | + if not desc or desc.startswith("$"): # référence Flight non résolue | |
| 61 | + desc = str((it.get("descriptions") or {}).get("fr") or "") | |
| 62 | + if desc.startswith("$"): | |
| 63 | + desc = "" | |
| 64 | + return re.sub(r"\s+", " ", desc).strip()[:4000] | |
| 65 | + | |
| 66 | + # -- contrat ---------------------------------------------------------- | |
| 67 | + def fetch(self) -> list[StListing]: | |
| 68 | + listings: list[StListing] = [] | |
| 69 | + for it in self._catalog(): | |
| 70 | + lid = str(it.get("id") or "").strip() | |
| 71 | + slug = (it.get("slug") or "").strip() | |
| 72 | + title = (it.get("name") or "").strip() | |
| 73 | + if not lid or not slug or not title: | |
| 74 | + continue | |
| 75 | + | |
| 76 | + geo = it.get("geo") or {} | |
| 77 | + citq = str((it.get("citq") or {}).get("number") or "").strip() | |
| 78 | + street = (it.get("addressStreet") or "").strip() | |
| 79 | + postal = (it.get("addressPostal") or "").strip() | |
| 80 | + price = _num(it.get("priceFromCAD")) | |
| 81 | + | |
| 82 | + details = {k: v for k, v in { | |
| 83 | + "guesty_id": lid, | |
| 84 | + "area_sqft": it.get("areaSquareFeet"), | |
| 85 | + "check_in": it.get("checkInTime"), | |
| 86 | + "check_out": it.get("checkOutTime"), | |
| 87 | + "tags": it.get("tags") or None, | |
| 88 | + }.items() if v} | |
| 89 | + | |
| 90 | + listings.append(StListing( | |
| 91 | + source=self.source_id, | |
| 92 | + external_id=lid, | |
| 93 | + url=f"{SITE}/chalets/{slug}", | |
| 94 | + title=title, | |
| 95 | + property_type="Chalet", | |
| 96 | + address=" ".join(p for p in (street, postal) if p), | |
| 97 | + city=(it.get("city") or "").strip(), | |
| 98 | + region="", # ville + lat/lng font foi | |
| 99 | + price_night=price, | |
| 100 | + price_label=(f"à partir de {price:.0f} $ / nuit" | |
| 101 | + if price else ""), | |
| 102 | + capacity=_num(it.get("maxGuests")), | |
| 103 | + bedrooms=_num(it.get("bedrooms")), | |
| 104 | + beds=_num(it.get("beds")), | |
| 105 | + bathrooms=_num(it.get("bathrooms")), | |
| 106 | + citq=citq if re.fullmatch(r"\d{6}", citq) else "", | |
| 107 | + description=self._description(it), | |
| 108 | + amenities=[a for a in (it.get("amenities") or []) | |
| 109 | + if isinstance(a, str)][:80], | |
| 110 | + details=details, | |
| 111 | + images=[u for u in (it.get("images") or []) | |
| 112 | + if isinstance(u, str)][:20], | |
| 113 | + lat=_num(geo.get("lat")), | |
| 114 | + lng=_num(geo.get("lng")), | |
| 115 | + )) | |
| 116 | + return listings | |
added
louka/shortterm/connectors/rvmt.py
+215 −0
@@ -0,0 +1,215 @@ | ||
| 1 | +# ----------------------------------------------------------------------------- | |
| 2 | +# Lou-Ka — Location court terme | |
| 3 | +# connectors/rvmt.py : Rendez-vous Mont-Tremblant (rvmt.com) — agence de | |
| 4 | +# gestion locative de la station Mont-Tremblant, ~108 condos, maisons et | |
| 5 | +# maisons de ville (Plateau, Algonquin, Étoile du Matin, etc.). | |
| 6 | +# | |
| 7 | +# Méthode : le site (Nuxt 3) renvoie systématiquement 429 en direct → | |
| 8 | +# UN SEUL appel Scrapfly (sans render_js) sur une page de complexe | |
| 9 | +# (/fr/algonquin — la page liste n'embarque PAS les taxonomies, la page | |
| 10 | +# complexe si). Tout l'inventaire (les 108 unités) vit dans le | |
| 11 | +# <script id="__NUXT_DATA__"> au format devalue : un tableau plat où les | |
| 12 | +# valeurs des dicts/listes sont des INDEX entiers vers d'autres cases | |
| 13 | +# (wrappers ["ShallowReactive", n] à déréférencer). | |
| 14 | +# On y trouve : | |
| 15 | +# - le nœud racine {units-rvmt: [108 unités], features-rvmt, beds-rvmt, | |
| 16 | +# rooms-rvmt, areas-rvmt, amenities-rvmt} — chaque unité est complète : | |
| 17 | +# code (external_id stable, ex. PLA214-03), adresse+lat/lng, chambres, | |
| 18 | +# sdb, occupancy, pets_allowed, rental_license (CITQ), min/max_price | |
| 19 | +# ($/nuit CAD), description fr, galerie Cloudinary, features/amenities | |
| 20 | +# (ids → libellés fr via les taxonomies), lits par pièce ; | |
| 21 | +# - le nœud site-config {pages, nav, slug} dont les pages | |
| 22 | +# data == {"id": <mongo-id du complexe>} donnent l'URL du complexe → | |
| 23 | +# URL détail = /fr/<complexe>/<slug(nom)> (2 unités non mappées | |
| 24 | +# retombent sur la page liste). | |
| 25 | +# | |
| 26 | +# Réglage env : LOUKA_RVMT_LIMIT (nb max d'annonces, 0 = tout). | |
| 27 | +# ----------------------------------------------------------------------------- | |
| 28 | +from __future__ import annotations | |
| 29 | + | |
| 30 | +import html as _html | |
| 31 | +import json | |
| 32 | +import os | |
| 33 | +import re | |
| 34 | +import unicodedata | |
| 35 | + | |
| 36 | +from ..schema import StListing | |
| 37 | +from .base import StConnector | |
| 38 | + | |
| 39 | +LIST_URL = "https://www.rvmt.com/fr/condos-mont-tremblant" | |
| 40 | +PAYLOAD_URL = "https://www.rvmt.com/fr/algonquin" # payload complet (taxos) | |
| 41 | + | |
| 42 | +# tag pt-* → type canonique | |
| 43 | +_TYPES = {"pt-condo": "Condo", "pt-townhome": "Maison", | |
| 44 | + "pt-private-home": "Maison", "pt-hotel": "Condo"} | |
| 45 | + | |
| 46 | +_WRAPPERS = ("ShallowReactive", "Reactive", "Ref", "ShallowRef", "EmptyRef", | |
| 47 | + "EmptyShallowRef") | |
| 48 | + | |
| 49 | + | |
| 50 | +def _slugify(name: str) -> str: | |
| 51 | + s = unicodedata.normalize("NFKD", name).encode("ascii", "ignore").decode() | |
| 52 | + return re.sub(r"-{2,}", "-", re.sub(r"[^a-z0-9]+", "-", s.lower())).strip("-") | |
| 53 | + | |
| 54 | + | |
| 55 | +def _fr(node) -> str: | |
| 56 | + """Nœud {text: {fr, en}} ou {fr, en} → libellé français (repli anglais).""" | |
| 57 | + if isinstance(node, dict): | |
| 58 | + t = node.get("text") if isinstance(node.get("text"), dict) else node | |
| 59 | + if isinstance(t, dict): | |
| 60 | + return _html.unescape(str(t.get("fr") or t.get("en") or "")).strip() | |
| 61 | + return "" | |
| 62 | + | |
| 63 | + | |
| 64 | +class Rvmt(StConnector): | |
| 65 | + source_id = "rvmt" | |
| 66 | + | |
| 67 | + # -- devalue -------------------------------------------------------------- | |
| 68 | + def _resolve(self, arr: list, i, seen: frozenset = frozenset()): | |
| 69 | + v = arr[i] if isinstance(i, int) and 0 <= i < len(arr) else i | |
| 70 | + if isinstance(v, dict): | |
| 71 | + if i in seen: | |
| 72 | + return None | |
| 73 | + return {k: self._resolve(arr, x, seen | {i}) for k, x in v.items()} | |
| 74 | + if isinstance(v, list): | |
| 75 | + if v and v[0] in _WRAPPERS: | |
| 76 | + return self._resolve(arr, v[1], seen) if len(v) > 1 else None | |
| 77 | + if i in seen: | |
| 78 | + return None | |
| 79 | + return [self._resolve(arr, x, seen | {i}) for x in v] | |
| 80 | + return v | |
| 81 | + | |
| 82 | + # -- contrat -------------------------------------------------------------- | |
| 83 | + def fetch(self) -> list[StListing]: | |
| 84 | + limit = int(os.environ.get("LOUKA_RVMT_LIMIT", "0") or 0) | |
| 85 | + h = self.get_scrapfly(PAYLOAD_URL, render_js=False) | |
| 86 | + m = re.search(r'(?s)<script[^>]*id="__NUXT_DATA__"[^>]*>(.*?)</script>', | |
| 87 | + h) | |
| 88 | + if not m: | |
| 89 | + return [] | |
| 90 | + arr = json.loads(m.group(1)) | |
| 91 | + | |
| 92 | + # racine = le dict qui contient unités ET taxonomies (une page liste | |
| 93 | + # n'aurait que units-rvmt : on garde le nœud le plus complet) | |
| 94 | + root = None | |
| 95 | + site_cfg = None | |
| 96 | + fallback_i = None | |
| 97 | + for i, v in enumerate(arr): | |
| 98 | + if not isinstance(v, dict): | |
| 99 | + continue | |
| 100 | + if root is None and "units-rvmt" in v: | |
| 101 | + if "features-rvmt" in v: | |
| 102 | + root = self._resolve(arr, i) | |
| 103 | + elif fallback_i is None: | |
| 104 | + fallback_i = i | |
| 105 | + elif site_cfg is None and {"pages", "nav", "slug"} <= set(v): | |
| 106 | + site_cfg = self._resolve(arr, i) | |
| 107 | + if root is not None and site_cfg is not None: | |
| 108 | + break | |
| 109 | + if root is None and fallback_i is not None: | |
| 110 | + root = self._resolve(arr, fallback_i) | |
| 111 | + if not root: | |
| 112 | + return [] | |
| 113 | + | |
| 114 | + units = root.get("units-rvmt") or [] | |
| 115 | + tax = {name: {t.get("value"): _fr(t) | |
| 116 | + for t in (root.get(f"{name}-rvmt") or []) | |
| 117 | + if isinstance(t, dict)} | |
| 118 | + for name in ("features", "beds", "rooms", "areas", "amenities")} | |
| 119 | + | |
| 120 | + # id de complexe → slug d'URL (pages du site-config avec data={id}) | |
| 121 | + complexes: dict[str, str] = {} | |
| 122 | + for p in (site_cfg or {}).get("pages") or []: | |
| 123 | + data = p.get("data") if isinstance(p, dict) else None | |
| 124 | + if isinstance(data, dict) and set(data) == {"id"} and p.get("url"): | |
| 125 | + complexes[str(data["id"])] = str(p["url"]).strip("/") | |
| 126 | + | |
| 127 | + listings: list[StListing] = [] | |
| 128 | + vus: set[str] = set() | |
| 129 | + for u in units: | |
| 130 | + if not isinstance(u, dict): | |
| 131 | + continue | |
| 132 | + code = str(u.get("code") or "").strip() | |
| 133 | + name = _html.unescape(str(u.get("name") or "")).strip() | |
| 134 | + if not code or code in vus or not name: | |
| 135 | + continue | |
| 136 | + vus.add(code) | |
| 137 | + | |
| 138 | + comp = complexes.get(str(u.get("complex") or "")) | |
| 139 | + url = (f"https://www.rvmt.com/fr/{comp}/{_slugify(name)}" | |
| 140 | + if comp else LIST_URL) | |
| 141 | + | |
| 142 | + addr = u.get("address") or {} | |
| 143 | + tags = u.get("tags") or [] | |
| 144 | + ptype = next((_TYPES[t] for t in tags if t in _TYPES), "Condo") | |
| 145 | + area = tax["areas"].get(u.get("area")) or "" | |
| 146 | + | |
| 147 | + # lits : somme des quantités des pièces déclarées | |
| 148 | + beds = 0 | |
| 149 | + for room in u.get("rooms") or []: | |
| 150 | + for b in (room or {}).get("beds") or []: | |
| 151 | + try: | |
| 152 | + beds += int(b.get("quantity") or 0) | |
| 153 | + except (TypeError, ValueError): | |
| 154 | + pass | |
| 155 | + | |
| 156 | + amen: list[str] = [] | |
| 157 | + for f in u.get("features") or []: | |
| 158 | + t = tax["features"].get(f) | |
| 159 | + if t and t not in amen: | |
| 160 | + amen.append(t) | |
| 161 | + for a in u.get("amenities") or []: | |
| 162 | + t = tax["amenities"].get((a or {}).get("amenity")) | |
| 163 | + if t and t not in amen: | |
| 164 | + amen.append(t) | |
| 165 | + | |
| 166 | + desc = u.get("short_description") or {} | |
| 167 | + description = _html.unescape(str(desc.get("fr") | |
| 168 | + or desc.get("en") or "")).strip() | |
| 169 | + | |
| 170 | + price = u.get("min_price") | |
| 171 | + price = float(price) if isinstance(price, (int, float)) \ | |
| 172 | + and 20 <= price <= 20000 else None | |
| 173 | + price_label = f"À partir de {price:g} $ / nuit" if price else "" | |
| 174 | + | |
| 175 | + gallery = [g for g in (u.get("gallery") or []) | |
| 176 | + if isinstance(g, str) and g.startswith("https://")] | |
| 177 | + | |
| 178 | + details = {k: v for k, v in { | |
| 179 | + "area": area, | |
| 180 | + "complex": comp or "", | |
| 181 | + "max_price_night": u.get("max_price"), | |
| 182 | + "wheelchair_accessible": bool(u.get("wheelchair_accessible")), | |
| 183 | + }.items() if v} | |
| 184 | + | |
| 185 | + line1 = str(addr.get("line_1") or "").strip() | |
| 186 | + apt = str(addr.get("apt") or "").strip() | |
| 187 | + | |
| 188 | + listings.append(StListing( | |
| 189 | + source=self.source_id, | |
| 190 | + external_id=code, | |
| 191 | + url=url, | |
| 192 | + title=name, | |
| 193 | + property_type=ptype, | |
| 194 | + address=f"{line1}, app. {apt}" if line1 and apt else line1, | |
| 195 | + city=str(addr.get("city") or "Mont-Tremblant").strip(), | |
| 196 | + region="Laurentides", | |
| 197 | + price_night=price, | |
| 198 | + price_label=price_label, | |
| 199 | + capacity=float(u["occupancy"]) if u.get("occupancy") else None, | |
| 200 | + bedrooms=float(u["bedrooms"]) if u.get("bedrooms") is not None | |
| 201 | + else None, | |
| 202 | + beds=float(beds) if beds else None, | |
| 203 | + bathrooms=float(u["bathrooms"]) if u.get("bathrooms") else None, | |
| 204 | + pets="oui" if u.get("pets_allowed") else "non", | |
| 205 | + citq=str(u.get("rental_license") or "").strip(), | |
| 206 | + description=description[:5000], | |
| 207 | + amenities=amen, | |
| 208 | + details=details, | |
| 209 | + images=gallery[:20], | |
| 210 | + lat=addr.get("latitude"), | |
| 211 | + lng=addr.get("longitude"), | |
| 212 | + )) | |
| 213 | + if limit and len(listings) >= limit: | |
| 214 | + break | |
| 215 | + return listings | |
modified
louka/shortterm/connectors/sepaq.py
+141 −2
@@ -23,6 +23,14 @@ | ||
| 23 | 23 | # distinctif dans le sitemap (il faudrait crawler chaque boucle de camping). |
| 24 | 24 | # Région touristique déduite de l'établissement (table statique ci-dessous). |
| 25 | 25 | # Le site est derrière Cloudflare : self.get() escalade automatiquement. |
| 26 | +# | |
| 27 | +# Description : les pages unité n'ont AUCUN texte descriptif (que des listes | |
| 28 | +# d'équipements) — la prose vit sur la page éditoriale de l'établissement | |
| 29 | +# (sepaq.com/<code>/, ex. pq/mot) : on la récupère une fois par établissement | |
| 30 | +# (cache self.detail) et on l'applique aux unités. Géolocalisation : l'API | |
| 31 | +# carte échoue souvent (cookie de session perdu derrière l'anti-bot) — repli | |
| 32 | +# sur les coordonnées statiques de l'établissement (fait géographique stable, | |
| 33 | +# précision « parc » signalée dans details.geo_precision). | |
| 26 | 34 | # ----------------------------------------------------------------------------- |
| 27 | 35 | from __future__ import annotations |
| 28 | 36 | |
@@ -77,6 +85,85 @@ REGION_ETAB = { | ||
| 77 | 85 | "auberge-de-montagne-des-chic-chocs": "Gaspésie", |
| 78 | 86 | } |
| 79 | 87 | |
| 88 | +# Coordonnées approximatives de chaque établissement (accueil/centre du | |
| 89 | +# territoire — fait géographique stable). Repli quand l'API carte ne donne | |
| 90 | +# pas les coordonnées précises de l'unité. | |
| 91 | +ETAB_COORDS = { | |
| 92 | + "centre-touristique-du-lac-kenogami": (48.336, -71.433), | |
| 93 | + "centre-touristique-du-lac-simon": (46.008, -75.092), | |
| 94 | + "parc-national-d-aiguebelle": (48.494, -78.712), | |
| 95 | + "parc-national-d-oka": (45.472, -74.023), | |
| 96 | + "parc-national-de-frontenac": (45.940, -71.150), | |
| 97 | + "parc-national-de-la-gaspesie": (48.982, -66.242), | |
| 98 | + "parc-national-de-la-jacques-cartier": (47.174, -71.374), | |
| 99 | + "parc-national-de-la-pointe-taillon": (48.665, -72.080), | |
| 100 | + "parc-national-de-la-yamaska": (45.462, -72.632), | |
| 101 | + "parc-national-de-plaisance": (45.590, -75.070), | |
| 102 | + "parc-national-des-grands-jardins": (47.667, -70.628), | |
| 103 | + "parc-national-des-hautes-gorges-de-la-riviere-malbaie": (47.899, -70.412), | |
| 104 | + "parc-national-des-monts-valin": (48.597, -70.828), | |
| 105 | + "parc-national-du-bic": (48.336, -68.802), | |
| 106 | + "parc-national-du-fjord-du-saguenay": (48.311, -70.324), | |
| 107 | + "parc-national-du-mont-megantic": (45.455, -71.152), | |
| 108 | + "parc-national-du-mont-orford": (45.357, -72.235), | |
| 109 | + "parc-national-du-mont-tremblant": (46.356, -74.462), | |
| 110 | + "reserve-faunique-ashuapmushuan": (48.890, -72.870), | |
| 111 | + "reserve-faunique-de-matane": (48.632, -67.114), | |
| 112 | + "reserve-faunique-de-papineau-labelle": (46.150, -75.300), | |
| 113 | + "reserve-faunique-de-port-cartier-sept-iles": (50.318, -67.083), | |
| 114 | + "reserve-faunique-de-port-daniel": (48.240, -64.970), | |
| 115 | + "reserve-faunique-de-portneuf": (47.100, -72.250), | |
| 116 | + "reserve-faunique-de-rimouski": (48.075, -68.375), | |
| 117 | + "reserve-faunique-des-chic-chocs": (48.732, -66.463), | |
| 118 | + "reserve-faunique-des-laurentides": (47.567, -71.233), | |
| 119 | + "reserve-faunique-du-saint-maurice": (46.900, -73.100), | |
| 120 | + "reserve-faunique-la-verendrye": (47.348, -76.868), | |
| 121 | + "reserve-faunique-mastigouche": (46.600, -73.400), | |
| 122 | + "reserve-faunique-rouge-matawin": (46.723, -74.500), | |
| 123 | + "sepaq-anticosti": (49.500, -63.300), | |
| 124 | + "station-touristique-duchesnay": (46.878, -71.635), | |
| 125 | + "auberge-de-montagne-des-chic-chocs": (48.900, -66.489), | |
| 126 | +} | |
| 127 | + | |
| 128 | +# Page éditoriale de chaque établissement (sepaq.com/<code>/) — codes vérifiés | |
| 129 | +# via les titres des sitemaps etablissement/parc-national/reserve-faunique. | |
| 130 | +ETAB_EDITO = { | |
| 131 | + "centre-touristique-du-lac-kenogami": "ct/ken", | |
| 132 | + "centre-touristique-du-lac-simon": "ct/sim", | |
| 133 | + "parc-national-d-aiguebelle": "pq/aig", | |
| 134 | + "parc-national-d-oka": "pq/oka", | |
| 135 | + "parc-national-de-frontenac": "pq/fro", | |
| 136 | + "parc-national-de-la-gaspesie": "pq/gas", | |
| 137 | + "parc-national-de-la-jacques-cartier": "pq/jac", | |
| 138 | + "parc-national-de-la-pointe-taillon": "pq/pta", | |
| 139 | + "parc-national-de-la-yamaska": "pq/yam", | |
| 140 | + "parc-national-de-plaisance": "pq/pla", | |
| 141 | + "parc-national-des-grands-jardins": "pq/grj", | |
| 142 | + "parc-national-des-hautes-gorges-de-la-riviere-malbaie": "pq/hgo", | |
| 143 | + "parc-national-des-monts-valin": "pq/mva", | |
| 144 | + "parc-national-du-bic": "pq/bic", | |
| 145 | + "parc-national-du-fjord-du-saguenay": "pq/sag", | |
| 146 | + "parc-national-du-mont-megantic": "pq/mme", | |
| 147 | + "parc-national-du-mont-orford": "pq/mor", | |
| 148 | + "parc-national-du-mont-tremblant": "pq/mot", | |
| 149 | + "reserve-faunique-ashuapmushuan": "rf/ash", | |
| 150 | + "reserve-faunique-de-matane": "rf/mat", | |
| 151 | + "reserve-faunique-de-papineau-labelle": "rf/pal", | |
| 152 | + "reserve-faunique-de-port-cartier-sept-iles": "rf/spc", | |
| 153 | + "reserve-faunique-de-port-daniel": "rf/pod", | |
| 154 | + "reserve-faunique-de-portneuf": "rf/por", | |
| 155 | + "reserve-faunique-de-rimouski": "rf/rim", | |
| 156 | + "reserve-faunique-des-chic-chocs": "rf/chc", | |
| 157 | + "reserve-faunique-des-laurentides": "rf/lau", | |
| 158 | + "reserve-faunique-du-saint-maurice": "rf/stm", | |
| 159 | + "reserve-faunique-la-verendrye": "rf/lvy", | |
| 160 | + "reserve-faunique-mastigouche": "rf/mas", | |
| 161 | + "reserve-faunique-rouge-matawin": "rf/rom", | |
| 162 | + "sepaq-anticosti": "sepaq-anticosti", | |
| 163 | + "station-touristique-duchesnay": "ct/duc", | |
| 164 | + "auberge-de-montagne-des-chic-chocs": "ct/amc", | |
| 165 | +} | |
| 166 | + | |
| 80 | 167 | _LOC_RE = re.compile(r"<loc>\s*(https?://[^<\s]+)") |
| 81 | 168 | _COOKIE_RE = re.compile(r"(?:JSESSIONID_TRANSAC|__cf_bm)=[^;,\s]+") |
| 82 | 169 | _JSON_ARRAY_RE = re.compile(r"\[.*\]", re.S) |
@@ -171,6 +258,34 @@ class Sepaq(StConnector): | ||
| 171 | 258 | }) |
| 172 | 259 | return {"items": items} |
| 173 | 260 | |
| 261 | + # -- page éditoriale d'un établissement (description riche) ----------------- | |
| 262 | + _EDITO_BRUIT = ("témoins", "cookies", "javascript", "modalités de réserv", | |
| 263 | + "navigateur", "abonnez-vous", "infolettre") | |
| 264 | + | |
| 265 | + def _fetch_edito(self, code: str) -> dict: | |
| 266 | + from bs4 import BeautifulSoup | |
| 267 | + html = self.get(f"{BASE}/{code}/").text | |
| 268 | + soup = BeautifulSoup(html, "html.parser") | |
| 269 | + if soup.title and "404" in soup.title.get_text(): | |
| 270 | + return {} | |
| 271 | + parts: list[str] = [] | |
| 272 | + for p in soup.find_all("p"): | |
| 273 | + txt = re.sub(r"\s+", " ", p.get_text(" ", strip=True)) | |
| 274 | + low = txt.lower() | |
| 275 | + if len(txt) < 120 or any(b in low for b in self._EDITO_BRUIT): | |
| 276 | + continue | |
| 277 | + if txt not in parts: | |
| 278 | + parts.append(txt) | |
| 279 | + if sum(len(x) for x in parts) > 1200: | |
| 280 | + break | |
| 281 | + desc = "\n\n".join(parts) | |
| 282 | + if not desc: # repli : meta description de la page | |
| 283 | + m = re.search(r'<meta name="description" content="([^"]+)"', html) | |
| 284 | + if m: | |
| 285 | + import html as _h | |
| 286 | + desc = _h.unescape(m.group(1)).strip() | |
| 287 | + return {"description": desc[:2000]} if desc else {} | |
| 288 | + | |
| 174 | 289 | # -- page détail d'une unité -------------------------------------------------- |
| 175 | 290 | def _fetch_unit(self, slug: str) -> dict: |
| 176 | 291 | from bs4 import BeautifulSoup |
@@ -287,6 +402,21 @@ class Sepaq(StConnector): | ||
| 287 | 402 | if path: |
| 288 | 403 | geo[path] = it |
| 289 | 404 | |
| 405 | + # description éditoriale par établissement (une visite, cache BD) | |
| 406 | + editos: dict[str, str] = {} | |
| 407 | + for etab in etabs: | |
| 408 | + code = ETAB_EDITO.get(etab) | |
| 409 | + if not code: | |
| 410 | + continue | |
| 411 | + try: | |
| 412 | + payload = self.detail(f"edito:{etab}", "v1", | |
| 413 | + lambda cd=code: self._fetch_edito(cd)) | |
| 414 | + except Exception as exc: # noqa: BLE001 — la description est optionnelle | |
| 415 | + print(f"[sepaq] édito {etab} : {exc}", file=sys.stderr) | |
| 416 | + continue | |
| 417 | + if payload.get("description"): | |
| 418 | + editos[etab] = payload["description"] | |
| 419 | + | |
| 290 | 420 | month = time.strftime("%Y-%m") # re-visite mensuelle (prix/saison) |
| 291 | 421 | listings: list[StListing] = [] |
| 292 | 422 | for slug in sorted(set(units) | set(geo)): |
@@ -322,6 +452,14 @@ class Sepaq(StConnector): | ||
| 322 | 452 | |
| 323 | 453 | amenities = (d.get("sections") or {}).get("Description") or [] |
| 324 | 454 | |
| 455 | + # géo : coordonnées précises de l'unité (API carte) sinon repli | |
| 456 | + # sur celles de l'établissement (précision « parc ») | |
| 457 | + lat, lng = g.get("lat"), g.get("lng") | |
| 458 | + if lat is None or lng is None: | |
| 459 | + lat, lng = ETAB_COORDS.get(etab, (None, None)) | |
| 460 | + if lat is not None: | |
| 461 | + details["geo_precision"] = "etablissement" | |
| 462 | + | |
| 325 | 463 | lst = StListing( |
| 326 | 464 | source=self.source_id, |
| 327 | 465 | external_id=slug, |
@@ -339,11 +477,12 @@ class Sepaq(StConnector): | ||
| 339 | 477 | beds=float(d["beds"]) if d.get("beds") else None, |
| 340 | 478 | pets=d.get("pets"), |
| 341 | 479 | citq=d.get("citq", ""), |
| 480 | + description=editos.get(etab, ""), | |
| 342 | 481 | amenities=amenities, |
| 343 | 482 | details=details, |
| 344 | 483 | images=d.get("images") or [], |
| 345 | − lat=g.get("lat"), | |
| 346 | − lng=g.get("lng"), | |
| 484 | + lat=lat, | |
| 485 | + lng=lng, | |
| 347 | 486 | ) |
| 348 | 487 | listings.append(lst) |
| 349 | 488 | return listings |
modified
louka/shortterm/connectors/sinistar.py
+4 −0
@@ -17,6 +17,10 @@ | ||
| 17 | 17 | # (self.__next_f.push) : description, commodités, capacité. |
| 18 | 18 | # 3. AUCUN PRIX PUBLIC : le tarif est négocié entre l'hôte et l'assureur |
| 19 | 19 | # (« couvert par l'assurance ») → price_night=None, prix absent assumé. |
| 20 | +# Vérifié 2026-08-25 : ni l'index Algolia, ni le flux React Flight, ni | |
| 21 | +# le document Firestore public (projects/sinistar-13fdf …/housings/<id>, | |
| 22 | +# lisible sans auth) ne contiennent de champ prix ou note — les montants | |
| 23 | +# circulent uniquement dans les soumissions hôte↔assureur (auth requise). | |
| 20 | 24 | # La région touristique est déduite des coordonnées (centroïdes Airbnb). |
| 21 | 25 | # |
| 22 | 26 | # Réglage env : LOUKA_SINISTAR_LIMIT (nb max d'annonces, 0 = tout ; utile |
modified
louka/shortterm/connectors/tremblantliving.py
+63 −2
@@ -10,22 +10,31 @@ | ||
| 10 | 10 | # salles de bain, capacité, note/avis, adresse, lat/lng, photos (galerie |
| 11 | 11 | # streamlinevrs.com). La description longue vient du bloc |
| 12 | 12 | # <div class="description block">, les commodités des <li class="amenity_item">. |
| 13 | −# Pas de prix statique : les tarifs passent par l'API Streamline | |
| 14 | −# (admin-ajax.php), bloquée par Cloudflare en POST — price_night reste vide. | |
| 13 | +# PRIX : pas de prix statique dans le HTML, mais l'API Streamline passe par | |
| 14 | +# admin-ajax.php avec action=streamlinecore-api-request et le corps JSON | |
| 15 | +# {methodName, params} DANS LA QUERY STRING (format du plugin Angular) — | |
| 16 | +# contrairement au POST classique, ce format n'est pas bloqué par | |
| 17 | +# Cloudflare. GetPropertyRatesRawData(unit_id) retourne la grille des | |
| 18 | +# tarifs saisonniers ($/nuit) → price_night = minimum des périodes | |
| 19 | +# courantes/futures (« à partir de »). Rafraîchi à chaque run (37 appels). | |
| 15 | 20 | # Les /monthly-rentals/ (units-sitemap.xml) sont du long terme : ignorés. |
| 16 | 21 | # ----------------------------------------------------------------------------- |
| 17 | 22 | from __future__ import annotations |
| 18 | 23 | |
| 24 | +import datetime as _dt | |
| 19 | 25 | import html as _html |
| 20 | 26 | import json |
| 21 | 27 | import os |
| 22 | 28 | import re |
| 29 | +import sys | |
| 30 | +from urllib.parse import urlencode | |
| 23 | 31 | |
| 24 | 32 | from ..schema import StListing |
| 25 | 33 | from .base import StConnector |
| 26 | 34 | |
| 27 | 35 | SITE = "https://www.tremblantliving.ca" |
| 28 | 36 | SITEMAP = SITE + "/property-sitemap.xml" |
| 37 | +AJAX = SITE + "/wp-admin/admin-ajax.php" | |
| 29 | 38 | |
| 30 | 39 | # type déduit du nom de la fiche (agence ~100 % chalets et condos) |
| 31 | 40 | _TYPE_HINTS = [ |
@@ -53,6 +62,45 @@ def _f(v) -> float | None: | ||
| 53 | 62 | class TremblantLiving(StConnector): |
| 54 | 63 | source_id = "tremblant_living" |
| 55 | 64 | |
| 65 | + # -- API Streamline (via admin-ajax, JSON en query string) ---------------- | |
| 66 | + def _api(self, method: str, params: dict) -> dict: | |
| 67 | + req = json.dumps({"methodName": method, "params": params}, | |
| 68 | + separators=(",", ":")) | |
| 69 | + q = urlencode({"action": "streamlinecore-api-request", "params": req}) | |
| 70 | + resp = self.post(f"{AJAX}?{q}", | |
| 71 | + headers={"Content-Type": "application/json"}) | |
| 72 | + return resp.json() | |
| 73 | + | |
| 74 | + def _price_from_rates(self, unit_id: str) -> tuple[float | None, str]: | |
| 75 | + """Prix « à partir de » = minimum $/nuit des périodes tarifaires | |
| 76 | + courantes et futures (GetPropertyRatesRawData). Jamais mis en cache : | |
| 77 | + les tarifs saisonniers bougent sans que la page change.""" | |
| 78 | + data = self._api("GetPropertyRatesRawData", | |
| 79 | + {"unit_id": int(unit_id)}).get("data") or {} | |
| 80 | + rates = data.get("rates") or [] | |
| 81 | + today = _dt.date.today() | |
| 82 | + prices: list[float] = [] | |
| 83 | + for r in rates if isinstance(rates, list) else [rates]: | |
| 84 | + try: | |
| 85 | + end = _dt.datetime.strptime( | |
| 86 | + str(r.get("period_end") or ""), "%m/%d/%Y").date() | |
| 87 | + except ValueError: | |
| 88 | + end = today # période sans date : on la garde | |
| 89 | + if end < today: | |
| 90 | + continue # saison passée | |
| 91 | + for k in ("daily_first_interval_price", | |
| 92 | + "daily_second_interval_price"): | |
| 93 | + m = re.search(r"(\d[\d,]*(?:\.\d+)?)", str(r.get(k) or "")) | |
| 94 | + if m: | |
| 95 | + v = float(m.group(1).replace(",", "")) | |
| 96 | + if 20 <= v <= 20000: | |
| 97 | + prices.append(v) | |
| 98 | + if not prices: | |
| 99 | + return None, "" | |
| 100 | + mn = min(prices) | |
| 101 | + mn = int(mn) if mn == int(mn) else mn | |
| 102 | + return float(mn), f"à partir de {mn} $ / nuit" | |
| 103 | + | |
| 56 | 104 | # -- page détail -------------------------------------------------------- |
| 57 | 105 | def _detail(self, url: str) -> dict: |
| 58 | 106 | h = self.get(url).text |
@@ -134,6 +182,17 @@ class TremblantLiving(StConnector): | ||
| 134 | 182 | if isinstance(imgs, str): |
| 135 | 183 | imgs = [imgs] |
| 136 | 184 | reviews = agg.get("reviewCount") |
| 185 | + | |
| 186 | + # tarif « à partir de » via l'API Streamline (hors cache détail) | |
| 187 | + price_night, price_label = None, "" | |
| 188 | + unit_id = str(ld.get("identifier") or "") | |
| 189 | + if unit_id.isdigit(): | |
| 190 | + try: | |
| 191 | + price_night, price_label = self._price_from_rates(unit_id) | |
| 192 | + except Exception as exc: # tarif manquant ≠ annonce perdue | |
| 193 | + print(f"[tremblant_living] tarifs {unit_id} : {exc}", | |
| 194 | + file=sys.stderr) | |
| 195 | + | |
| 137 | 196 | listings.append(StListing( |
| 138 | 197 | source=self.source_id, |
| 139 | 198 | external_id=str(ld.get("identifier") or slug), |
@@ -143,6 +202,8 @@ class TremblantLiving(StConnector): | ||
| 143 | 202 | address=_text(str(addr.get("streetAddress") or "")), |
| 144 | 203 | city=_text(str(addr.get("addressLocality") or "Mont-Tremblant")), |
| 145 | 204 | region="Laurentides", |
| 205 | + price_night=price_night, | |
| 206 | + price_label=price_label, | |
| 146 | 207 | capacity=_f(occupancy), |
| 147 | 208 | bedrooms=_f(place.get("numberOfBedrooms")), |
| 148 | 209 | bathrooms=_f(place.get("numberOfBathroomsTotal")), |
modified
louka/shortterm/connectors/vrbo.py
+55 −6
@@ -10,13 +10,21 @@ | ||
| 10 | 10 | # |
| 11 | 11 | # Limites assumées : ~18 cartes rendues par destination (liste virtualisée, |
| 12 | 12 | # le scroll ne persiste pas plus de cartes dans le snapshot DOM), pas de |
| 13 | −# lat/lng ni d'adresse sur les cartes (géocodage aval possible), images | |
| 14 | −# présentes seulement sur les cartes proches du viewport initial. | |
| 13 | +# lat/lng ni d'adresse sur les cartes, images présentes seulement sur les | |
| 14 | +# cartes proches du viewport initial. | |
| 15 | 15 | # Recherche SANS dates : Vrbo affiche alors un prix « à partir de » par nuit |
| 16 | 16 | # sur les prochaines dates disponibles → price_label + price_night plancher. |
| 17 | +# | |
| 18 | +# Enrichissement : la page détail (Scrapfly ASP SANS rendu JS — le SSR suffit) | |
| 19 | +# porte description, commodités, capacité, lat/lng et ~6 photos (voir | |
| 20 | +# _expediadetail.py). ⚠️ certains slugs ont un suffixe (p1234567vb) : l'URL | |
| 21 | +# sans suffixe redirige vers une page région — on conserve le slug complet. | |
| 22 | +# Réglage env : LOUKA_VRBO_DETAIL_LIMIT (fetchs détail par sync, défaut 100 ; | |
| 23 | +# cache permanent dans louka_ct.db, le parc se complète au fil des syncs). | |
| 17 | 24 | # ----------------------------------------------------------------------------- |
| 18 | 25 | from __future__ import annotations |
| 19 | 26 | |
| 27 | +import os | |
| 20 | 28 | import re |
| 21 | 29 | import sys |
| 22 | 30 | from urllib.parse import quote |
@@ -24,8 +32,13 @@ from urllib.parse import quote | ||
| 24 | 32 | from bs4 import BeautifulSoup |
| 25 | 33 | |
| 26 | 34 | from ..schema import StListing |
| 35 | +from . import _expediadetail as _ed | |
| 27 | 36 | from .base import StConnector |
| 28 | 37 | |
| 38 | + | |
| 39 | +class _DetailSkip(Exception): | |
| 40 | + """Fiche détail sautée (budget épuisé / page invalide) — pas de cache.""" | |
| 41 | + | |
| 29 | 42 | # (destination Vrbo, ville affichée, région touristique QC) |
| 30 | 43 | DESTINATIONS = [ |
| 31 | 44 | ("Mont-Tremblant, Québec, Canada", "Mont-Tremblant", "Laurentides"), |
@@ -61,7 +74,8 @@ TYPE_MAP = { | ||
| 61 | 74 | "hébergement": "Autre", |
| 62 | 75 | } |
| 63 | 76 | |
| 64 | −_ID_RE = re.compile(r"/location/p(\d+)") | |
| 77 | +# slug complet (p123vb) ET id numérique — le suffixe est requis dans l'URL | |
| 78 | +_ID_RE = re.compile(r"/location/(p(\d+)[a-z]{0,2})") | |
| 65 | 79 | _TYPELINE_RE = re.compile( |
| 66 | 80 | r"^([A-ZÀ-Ý][\w’' -]{2,30})\s*·", re.UNICODE) |
| 67 | 81 | _BEDROOMS_RE = re.compile(r"(\d+)\s*chambres?") |
@@ -84,8 +98,8 @@ class Vrbo(StConnector): | ||
| 84 | 98 | m = _ID_RE.search(href) |
| 85 | 99 | if not m: |
| 86 | 100 | return None |
| 87 | − external_id = m.group(1) | |
| 88 | − url = f"https://www.vrbo.com/fr-ca/location/p{external_id}" | |
| 101 | + external_id = m.group(2) | |
| 102 | + url = f"https://www.vrbo.com/fr-ca/location/{m.group(1)}" | |
| 89 | 103 | |
| 90 | 104 | title = "" |
| 91 | 105 | for h in card.find_all("h3"): |
@@ -164,6 +178,39 @@ class Vrbo(StConnector): | ||
| 164 | 178 | images=images, |
| 165 | 179 | ) |
| 166 | 180 | |
| 181 | + # -- enrichissement par la page détail --------------------------------------- | |
| 182 | + def _enrich_details(self, listings: list[StListing]) -> None: | |
| 183 | + """Visite les fiches détail via le cache self.detail() sous budget : | |
| 184 | + les hits de cache sont gratuits, seuls les fetchs réseau comptent.""" | |
| 185 | + limit = max(0, int(os.environ.get("LOUKA_VRBO_DETAIL_LIMIT", "100") | |
| 186 | + or 100)) | |
| 187 | + used = enriched = streak = 0 | |
| 188 | + for lst in listings: | |
| 189 | + def fetch_fn(url=lst.url): | |
| 190 | + nonlocal used, streak | |
| 191 | + if used >= limit or streak >= 5: # tempête anti-bot : on coupe | |
| 192 | + raise _DetailSkip | |
| 193 | + used += 1 | |
| 194 | + html = self.get_scrapfly(url, render_js=False, asp=True) | |
| 195 | + payload = _ed.parse_detail(html) | |
| 196 | + if not payload: | |
| 197 | + streak += 1 | |
| 198 | + raise _DetailSkip # blocage/vide : pas de cache | |
| 199 | + streak = 0 | |
| 200 | + return payload | |
| 201 | + | |
| 202 | + try: | |
| 203 | + d = self.detail(lst.external_id, "v1", fetch_fn) | |
| 204 | + except _DetailSkip: | |
| 205 | + continue | |
| 206 | + except Exception: # noqa: BLE001 — une fiche ne bloque pas le run | |
| 207 | + continue | |
| 208 | + if d: | |
| 209 | + _ed.apply_detail(lst, d) | |
| 210 | + enriched += 1 | |
| 211 | + print(f"[vrbo] détail : {enriched} annonces enrichies" | |
| 212 | + f" ({used}/{limit} fetchs réseau)", file=sys.stderr) | |
| 213 | + | |
| 167 | 214 | # -- contrat ----------------------------------------------------------------- |
| 168 | 215 | def fetch(self) -> list[StListing]: |
| 169 | 216 | listings: dict[str, StListing] = {} |
@@ -188,4 +235,6 @@ class Vrbo(StConnector): | ||
| 188 | 235 | continue |
| 189 | 236 | if lst and lst.external_id not in listings: |
| 190 | 237 | listings[lst.external_id] = lst |
| 191 | − return list(listings.values()) | |
| 238 | + out = list(listings.values()) | |
| 239 | + self._enrich_details(out) | |
| 240 | + return out | |
modified
louka/shortterm/web.py
+123 −0
@@ -6,6 +6,7 @@ | ||
| 6 | 6 | from __future__ import annotations |
| 7 | 7 | |
| 8 | 8 | import json |
| 9 | +import math | |
| 9 | 10 | import threading |
| 10 | 11 | from pathlib import Path |
| 11 | 12 | |
@@ -224,6 +225,128 @@ def ct_sources(): | ||
| 224 | 225 | return {"sources": registry} |
| 225 | 226 | |
| 226 | 227 | |
| 228 | +def _percentile(sorted_vals: list[float], q: float) -> float: | |
| 229 | + if not sorted_vals: | |
| 230 | + return 0.0 | |
| 231 | + pos = (len(sorted_vals) - 1) * q | |
| 232 | + lo, hi = int(pos), min(int(pos) + 1, len(sorted_vals) - 1) | |
| 233 | + return sorted_vals[lo] + (sorted_vals[hi] - sorted_vals[lo]) * (pos - lo) | |
| 234 | + | |
| 235 | + | |
| 236 | +def _haversine_km(lat1, lng1, lat2, lng2) -> float: | |
| 237 | + rl1, rl2 = math.radians(lat1), math.radians(lat2) | |
| 238 | + dlat, dlng = rl2 - rl1, math.radians(lng2 - lng1) | |
| 239 | + a = (math.sin(dlat / 2) ** 2 | |
| 240 | + + math.cos(rl1) * math.cos(rl2) * math.sin(dlng / 2) ** 2) | |
| 241 | + return 6371.0 * 2 * math.asin(math.sqrt(a)) | |
| 242 | + | |
| 243 | + | |
| 244 | +@router.get("/listings/{uid}/context") | |
| 245 | +def ct_listing_context(uid: str): | |
| 246 | + """Contexte d'une fiche : analyse du prix/nuit vs segment comparable | |
| 247 | + (région → + type → + capacité) et hébergements similaires à proximité.""" | |
| 248 | + con = db.connect() | |
| 249 | + row = con.execute("SELECT * FROM st_listings WHERE uid=?", | |
| 250 | + (uid,)).fetchone() | |
| 251 | + if row is None: | |
| 252 | + con.close() | |
| 253 | + raise HTTPException(404, "Hébergement introuvable") | |
| 254 | + l = dict(row) | |
| 255 | + | |
| 256 | + # --- analyse de prix : segment le plus précis avec ≥ 12 comparables ------ | |
| 257 | + price_block = None | |
| 258 | + if l["price_night"] is not None and l["region"]: | |
| 259 | + candidates: list[tuple[str, str, list]] = [] | |
| 260 | + base_sql = (" AND region=?") | |
| 261 | + base_args: list = [l["region"]] | |
| 262 | + if l["property_type"] and l["capacity"]: | |
| 263 | + candidates.append(( | |
| 264 | + f"{l['property_type']} · {int(l['capacity'])}±2 pers. · {l['region']}", | |
| 265 | + base_sql + " AND property_type=? AND capacity BETWEEN ? AND ?", | |
| 266 | + base_args + [l["property_type"], l["capacity"] - 2, | |
| 267 | + l["capacity"] + 2])) | |
| 268 | + if l["property_type"]: | |
| 269 | + candidates.append(( | |
| 270 | + f"{l['property_type']} · {l['region']}", | |
| 271 | + base_sql + " AND property_type=?", | |
| 272 | + base_args + [l["property_type"]])) | |
| 273 | + candidates.append((l["region"], base_sql, base_args)) | |
| 274 | + | |
| 275 | + for label, extra, args in candidates: | |
| 276 | + vals = [r["p"] for r in con.execute( | |
| 277 | + "SELECT price_night p FROM st_listings WHERE active=1" | |
| 278 | + " AND price_night IS NOT NULL AND uid<>?" + extra | |
| 279 | + + " ORDER BY price_night", [uid] + args)] | |
| 280 | + if len(vals) < 12: | |
| 281 | + continue | |
| 282 | + med = _percentile(vals, 0.5) | |
| 283 | + deviation = (l["price_night"] - med) / med if med else None | |
| 284 | + verdict = None | |
| 285 | + if deviation is not None: | |
| 286 | + verdict = ("sous" if deviation <= -0.15 | |
| 287 | + else "dans" if deviation < 0.12 else "dessus") | |
| 288 | + # histogramme 12 classes entre p5 et p95 (queues écrasées) | |
| 289 | + lo, hi = _percentile(vals, 0.05), _percentile(vals, 0.95) | |
| 290 | + bins = [] | |
| 291 | + if hi > lo: | |
| 292 | + step = (hi - lo) / 12 | |
| 293 | + edges = [lo + i * step for i in range(13)] | |
| 294 | + counts = [0] * 12 | |
| 295 | + for v in vals: | |
| 296 | + i = min(11, max(0, int((v - lo) / step))) | |
| 297 | + counts[i] += 1 | |
| 298 | + bins = [{"x0": round(edges[i]), "x1": round(edges[i + 1]), | |
| 299 | + "n": counts[i]} for i in range(12)] | |
| 300 | + rank = sum(1 for v in vals if v <= l["price_night"]) | |
| 301 | + price_block = { | |
| 302 | + "segment": label, "n": len(vals), | |
| 303 | + "median": round(med), "p25": round(_percentile(vals, 0.25)), | |
| 304 | + "p75": round(_percentile(vals, 0.75)), | |
| 305 | + "deviation": round(deviation, 3) if deviation is not None else None, | |
| 306 | + "verdict": verdict, | |
| 307 | + "percentile": round(100 * rank / len(vals)), | |
| 308 | + "histogram": bins, | |
| 309 | + } | |
| 310 | + break | |
| 311 | + | |
| 312 | + # --- hébergements similaires --------------------------------------------- | |
| 313 | + similar: list[dict] = [] | |
| 314 | + if l["lat"] is not None and l["lng"] is not None: | |
| 315 | + dlat = 0.45 # ≈ 50 km | |
| 316 | + dlng = dlat / max(0.2, math.cos(math.radians(l["lat"]))) | |
| 317 | + rows = con.execute( | |
| 318 | + "SELECT * FROM st_listings WHERE active=1 AND uid<>?" | |
| 319 | + " AND lat BETWEEN ? AND ? AND lng BETWEEN ? AND ?" | |
| 320 | + " AND images IS NOT NULL AND images<>'[]' LIMIT 400", | |
| 321 | + (uid, l["lat"] - dlat, l["lat"] + dlat, | |
| 322 | + l["lng"] - dlng, l["lng"] + dlng)).fetchall() | |
| 323 | + scored = [] | |
| 324 | + for r in rows: | |
| 325 | + km = _haversine_km(l["lat"], l["lng"], r["lat"], r["lng"]) | |
| 326 | + if km > 50: | |
| 327 | + continue | |
| 328 | + same_type = (l["property_type"] | |
| 329 | + and r["property_type"] == l["property_type"]) | |
| 330 | + scored.append((0 if same_type else 1, km, r)) | |
| 331 | + scored.sort(key=lambda t: (t[0], t[1])) | |
| 332 | + for _, km, r in scored[:8]: | |
| 333 | + d = _row_to_dict(r) | |
| 334 | + d["distance_km"] = round(km, 1) | |
| 335 | + similar.append(d) | |
| 336 | + if not similar and l["region"]: | |
| 337 | + sql = ("SELECT * FROM st_listings WHERE active=1 AND uid<>?" | |
| 338 | + " AND region=? AND images IS NOT NULL AND images<>'[]'") | |
| 339 | + args = [uid, l["region"]] | |
| 340 | + if l["property_type"]: | |
| 341 | + sql += " AND property_type=?" | |
| 342 | + args.append(l["property_type"]) | |
| 343 | + sql += " ORDER BY rating IS NULL, rating DESC, first_seen DESC LIMIT 8" | |
| 344 | + similar = [_row_to_dict(r) for r in con.execute(sql, args)] | |
| 345 | + | |
| 346 | + con.close() | |
| 347 | + return {"price": price_block, "similar": similar} | |
| 348 | + | |
| 349 | + | |
| 227 | 350 | @router.get("/listings/{uid}") |
| 228 | 351 | def ct_listing(uid: str): |
| 229 | 352 | con = db.connect() |
| 230 | 353 | |