// UQO Éval — modélisation hédonique : orchestration (échantillon → design → estimation → diagnostics → rapport). import { buildSample, type Obs } from "./data"; import { buildDesign, type Column, type Design } from "./design"; import { breuschPagan, fitMetrics, histogram, jarqueBera, qqPoints, smearing, toDollars, vif, type FitMetrics } from "./diagnostics"; import { covariance, huber, lad, ols, penalized, sar, sarPredict, type Fit } from "./estimators"; import { hcat, mean, median, quantile, sd, sortedCopy, takeRows, takeVec, type Mat } from "./linalg"; import { KnnIndex, lag, moranI, type Moran } from "./spatial"; import { LIMITS, METHODS, type HedonicSpec } from "./spec"; import { fSf, stars, tPValue, tQuantile } from "./stats"; type Txt = { fr: string; en: string }; export interface Coefficient { name: string; label: Txt; kind: Column["kind"] | "rho"; var?: string; level?: string; base?: string; n?: number; coef: number; se: number | null; t: number | null; p: number | null; stars: string; ciLo: number | null; ciHi: number | null; /** effet économique : % du prix et $ au prix médian, pour le pas indiqué */ effectPct: number | null; effectDollar: number | null; effectLo: number | null; // IC 95 % de l'effet en % effectHi: number | null; step: Txt; } export interface Descriptive { key: string; label: Txt; n: number; mean: number; sd: number; min: number; p25: number; median: number; p75: number; max: number; } export interface HedonicResult { spec: HedonicSpec; sample: { n: number; available: number; trimmed: number; dropped: { reason: Txt; n: number }[]; period: { from: string; to: string }; byType: { type: string; n: number }[]; descriptives: Descriptive[]; medianPrice: number; center: { lat: number; lng: number }; }; model: { method: HedonicSpec["method"]; methodLabel: Txt; depvar: HedonicSpec["depvar"]; equation: string; n: number; k: number; df: number; seType: HedonicSpec["se"]; nClusters: number; coefficients: Coefficient[]; fit: FitMetrics; fstat: { f: number; df1: number; df2: number; p: number } | null; lambda: { value: number; grid: { lambda: number; cvErr: number }[] } | null; lassoSelected: string[] | null; postLasso: Coefficient[] | null; rho: Coefficient | null; robust: { scale: number; downweightedPct: number } | null; notes: Txt[]; }; diagnostics: { bp: { stat: number; df: number; p: number } | null; jb: { stat: number; p: number; skew: number; kurt: number }; moran: Moran | null; vif: { name: string; label: Txt; vif: number }[]; }; cv: { folds: number; metrics: FitMetrics; gapMdape: number } | null; lab: { n: number; lab: { mdape: number; rmse: number; within10: number; within20: number }; ours: { mdape: number; rmse: number; within10: number; within20: number }; label: Txt } | null; timeIndex: { level: string; n: number; coef: number; index: number; lo: number | null; hi: number | null }[] | null; geoEffects: { level: string; n: number; coef: number; pct: number; se: number | null }[] | null; plots: { residFitted: { f: number; e: number }[]; qq: { q: number; e: number }[]; hist: { lo: number; hi: number; n: number }[]; map: { lat: number; lng: number; e: number }[]; actualPred: { a: number; p: number }[]; }; insights: Txt[]; code: { r: string; python: string; stata: string }; timingMs: number; } const T = (fr: string, en: string): Txt => ({ fr, en }); const fmtPct = (x: number, d = 1) => `${x >= 0 ? "+" : "−"}${Math.abs(x).toFixed(d)} %`; const fmtMoney = (x: number) => `${Math.round(x).toLocaleString("fr-CA")} $`; const fmtMoneyEn = (x: number) => `$${Math.round(x).toLocaleString("en-CA")}`; function descriptives(design: Design): Descriptive[] { const { X, cols } = design; const out: Descriptive[] = []; const pushVec = (key: string, label: Txt, v: Float64Array) => { const s = sortedCopy(v); out.push({ key, label, n: v.length, mean: mean(v), sd: sd(v), min: s[0], p25: quantile(s, 0.25), median: quantile(s, 0.5), p75: quantile(s, 0.75), max: s[s.length - 1] }); }; pushVec("price", T("Prix ($)", "Price ($)"), Float64Array.from(design.obs, (o) => o.price)); for (const c of cols) { if (c.kind !== "cont" || c.var === "age2") continue; const v = new Float64Array(X.n); const j = cols.indexOf(c); for (let i = 0; i < X.n; i++) v[i] = c.log ? Math.exp(X.a[i * X.k + j]) : X.a[i * X.k + j]; pushVec(c.var!, c.log ? { fr: c.label.fr.replace(/^ln /, ""), en: c.label.en.replace(/^ln /, "") } : c.label, v); } return out; } function coefTable(cols: Column[], beta: Float64Array, V: Float64Array | null, df: number, logModel: boolean, medP: number, offset = 0): Coefficient[] { const k = cols.length; const tcrit = V ? tQuantile(0.975, df) : NaN; return cols.map((c, j) => { const b = beta[j + offset]; const se = V ? Math.sqrt(Math.max(0, V[(j + offset) * (k + offset) + (j + offset)])) : null; const t = se ? b / se : null; const p = t != null ? tPValue(t, df) : null; let step = T("par unité", "per unit"); // effet économique d'un coefficient b (en % du prix et en $ au prix médian) let eff: (x: number) => { pct: number; dollar: number } | null = () => null; if (c.kind === "intercept") { step = T("—", "—"); } else if (c.kind === "cont" && c.log) { step = T("pour +10 %", "for +10%"); eff = logModel ? (x) => ({ pct: 100 * (Math.pow(1.1, x) - 1), dollar: medP * (Math.pow(1.1, x) - 1) }) : (x) => ({ pct: (100 * x * Math.log(1.1)) / medP, dollar: x * Math.log(1.1) }); } else { if (c.kind !== "cont" && c.kind !== "trend") step = T("vs référence", "vs base"); eff = logModel ? (x) => ({ pct: 100 * (Math.exp(x) - 1), dollar: medP * (Math.exp(x) - 1) }) : (x) => ({ pct: (100 * x) / medP, dollar: x }); } const e0 = eff(b); const eLo = se != null ? eff(b - tcrit * se) : null; const eHi = se != null ? eff(b + tcrit * se) : null; const effectPct = e0?.pct ?? null; const effectDollar = e0?.dollar ?? null; return { name: c.name, label: c.label, kind: c.kind, var: c.var, level: c.level, base: c.base, n: c.n, coef: b, se, t, p, stars: p != null ? stars(p) : "", ciLo: se != null ? b - tcrit * se : null, ciHi: se != null ? b + tcrit * se : null, effectPct, effectDollar, effectLo: eLo?.pct ?? null, effectHi: eHi?.pct ?? null, step, }; }); } function equationOf(spec: HedonicSpec, cols: Column[]): string { const lhs = spec.depvar === "lnprice" ? "ln(Prix)" : "Prix"; const terms = cols.filter((c) => c.kind === "cont" || c.kind === "trend").map((c) => c.name); const cats = [...new Set(cols.filter((c) => c.kind === "dummy").map((c) => c.var))]; const fe: string[] = []; if (spec.fe.time !== "none") fe.push(`δ_${spec.fe.time === "year" ? "année" : spec.fe.time === "quarter" ? "trimestre" : "mois"}`); if (spec.fe.geo !== "none") fe.push(`γ_zone`); const rho = spec.method === "sar" ? "ρ·W·" + lhs + " + " : ""; return `${lhs} = ${rho}β₀ + ${terms.map((t, i) => `β${i + 1}·${t}`).join(" + ")}${cats.length ? " + " + cats.map((c) => `Σ β·1[${c}]`).join(" + ") : ""}${fe.length ? " + " + fe.join(" + ") : ""} + ε`; } function codeSnippets(spec: HedonicSpec, cols: Column[]): HedonicResult["code"] { const cont = cols.filter((c) => c.kind === "cont").map((c) => (c.var === "age2" ? "I(age^2/100)" : c.log ? `log(${c.var})` : c.var!)); const trend = cols.some((c) => c.kind === "trend") ? ["poly(x_km, 2)", "poly(y_km, 2)", "x_km:y_km"] : []; const cats = [...new Set(cols.filter((c) => c.kind === "dummy").map((c) => c.var!))]; const fe: string[] = []; if (spec.fe.time !== "none") fe.push(spec.fe.time); if (spec.fe.geo !== "none") fe.push("zone"); const y = spec.depvar === "lnprice" ? "log(price)" : "price"; const rhs = [...cont, ...trend, ...cats].join(" + ") || "1"; const vcovR = spec.se === "hc1" ? 'vcov = "hetero"' : spec.se === "cluster" ? "vcov = ~zone" : 'vcov = "iid"'; const r = spec.method === "ols" || spec.method === "huber" || spec.method === "lad" ? `library(fixest)${spec.method !== "ols" ? "\nlibrary(MASS); library(quantreg)" : ""} d <- read.csv("echantillon_uqo_eval.csv") ${spec.method === "ols" ? `m <- feols(${y} ~ ${rhs}${fe.length ? " | " + fe.join(" + ") : ""}, data = d, ${vcovR})\netable(m)` : spec.method === "huber" ? `m <- rlm(${y} ~ ${rhs}${fe.map((f) => ` + factor(${f})`).join("")}, data = d, psi = psi.huber)\nsummary(m)` : `m <- rq(${y} ~ ${rhs}${fe.map((f) => ` + factor(${f})`).join("")}, tau = 0.5, data = d)\nsummary(m, se = "boot")`}` : spec.method === "sar" ? `library(spatialreg); library(spdep) d <- read.csv("echantillon_uqo_eval.csv") nb <- knn2nb(knearneigh(cbind(d$lng, d$lat), k = 8)); W <- nb2listw(nb, style = "W") m <- stsls(${y} ~ ${rhs}${fe.map((f) => ` + factor(${f})`).join("")}, data = d, listw = W) # SAR par 2SLS (Kelejian-Prucha) summary(m)` : `library(glmnet) d <- read.csv("echantillon_uqo_eval.csv") X <- model.matrix(~ ${rhs}${fe.map((f) => ` + factor(${f})`).join("")}, d)[, -1] cv <- cv.glmnet(X, ${y.replace("price", "d$price")}, alpha = ${spec.method === "lasso" ? 1 : 0}, nfolds = 5) coef(cv, s = "lambda.min")`; const pyRhs = [...cols.filter((c) => c.kind === "cont").map((c) => (c.var === "age2" ? "I(age**2/100)" : c.log ? `np.log(${c.var})` : c.var!)), ...cats.map((c) => `C(${c})`), ...fe.map((f) => `C(${f})`)].join(" + ") || "1"; const pyY = spec.depvar === "lnprice" ? "np.log(price)" : "price"; const python = spec.method === "ridge" || spec.method === "lasso" ? `import pandas as pd, numpy as np from sklearn.linear_model import ${spec.method === "lasso" ? "LassoCV" : "RidgeCV"} d = pd.read_csv("echantillon_uqo_eval.csv") X = pd.get_dummies(d[[${[...cols.filter((c) => c.kind === "cont" && c.var !== "age2").map((c) => `"${c.var}"`), ...cats.map((c) => `"${c}"`), ...fe.map((f) => `"${f}"`)].join(", ")}]], drop_first=True) y = ${pyY.replace("price", "d.price")} m = ${spec.method === "lasso" ? "LassoCV(cv=5)" : "RidgeCV(alphas=np.logspace(-4, 1, 24))"}.fit((X - X.mean()) / X.std(), y) print(dict(zip(X.columns, m.coef_)))` : `import pandas as pd, numpy as np import statsmodels.formula.api as smf d = pd.read_csv("echantillon_uqo_eval.csv") m = smf.${spec.method === "lad" ? "quantreg" : "ols"}("${pyY} ~ ${pyRhs}", data=d).fit(${spec.method === "lad" ? "q=0.5" : spec.se === "hc1" ? 'cov_type="HC1"' : spec.se === "cluster" ? 'cov_type="cluster", cov_kwds={"groups": d["zone"]}' : ""}) print(m.summary())`; const stY = spec.depvar === "lnprice" ? "lnprice" : "price"; const stX = [...cols.filter((c) => c.kind === "cont").map((c) => (c.var === "age2" ? "c.age#c.age" : c.log ? `ln_${c.var}` : c.var!)), ...cats.map((c) => `i.${c}`)].join(" "); const stata = spec.method === "ols" ? `import delimited echantillon_uqo_eval.csv, clear ${spec.depvar === "lnprice" ? "gen lnprice = ln(price)\n" : ""}${cols.filter((c) => c.kind === "cont" && c.log).map((c) => `gen ln_${c.var} = ln(${c.var})`).join("\n")} ${fe.length ? `reghdfe ${stY} ${stX}, absorb(${fe.join(" ")}) vce(${spec.se === "cluster" ? "cluster zone" : "robust"})` : `regress ${stY} ${stX}, vce(${spec.se === "cluster" ? "cluster zone" : "robust"})`}` : spec.method === "lad" ? `import delimited echantillon_uqo_eval.csv, clear\nqreg ${stY} ${stX} ${fe.map((f) => `i.${f}`).join(" ")}, quantile(.5)` : spec.method === "huber" ? `import delimited echantillon_uqo_eval.csv, clear\nrreg ${stY} ${stX} ${fe.map((f) => `i.${f}`).join(" ")}` : spec.method === "sar" ? `import delimited echantillon_uqo_eval.csv, clear\nspset, modify coordsys(latlong, kilometers)\nspmatrix create idistance W, knn(8) normalize(row)\nspivregress ${stY} ${stX} ${fe.map((f) => `i.${f}`).join(" ")}, dvarlag(W)` : `import delimited echantillon_uqo_eval.csv, clear\n${spec.method === "lasso" ? "lasso" : "elasticnet"} linear ${stY} ${stX} ${fe.map((f) => `i.${f}`).join(" ")}, selection(cv, folds(5))${spec.method === "ridge" ? " alpha(0)" : ""}`; return { r, python, stata }; } /* ------------------------------ estimation ------------------------------ */ interface Est { fit: Fit; V: Float64Array | null; cols: Column[]; // colonnes correspondant à fit.beta (SAR : ρ en tête) offsetX: number; // 1 si ρ en tête lambda: HedonicResult["model"]["lambda"]; lassoSelected: string[] | null; postLasso: Coefficient[] | null; rho: Coefficient | null; robust: HedonicResult["model"]["robust"]; nb: import("./spatial").KnnIndex | null; sarRes: ReturnType | null; notes: Txt[]; } function estimateAll(design: Design, spec: HedonicSpec, medP: number): Est { const { X, y, cols, clusters, nClusters } = design; const logModel = spec.depvar === "lnprice"; const notes: Txt[] = []; const lat = Float64Array.from(design.obs, (o) => o.lat); const lng = Float64Array.from(design.obs, (o) => o.lng); if (spec.method === "ols" || spec.method === "huber") { const fit = spec.method === "ols" ? ols(X, y) : huber(X, y); const V = covariance(fit, spec.se, clusters, nClusters); if (spec.method === "huber") notes.push(T("Huber : écarts-types de type sandwich sur la dernière itération pondérée (approximation usuelle).", "Huber: sandwich-type standard errors on the last weighted iteration (usual approximation).")); return { fit, V, cols, offsetX: 0, lambda: null, lassoSelected: null, postLasso: null, rho: null, robust: spec.method === "huber" ? { scale: fit.extra.scale, downweightedPct: fit.extra.downweightedPct } : null, nb: null, sarRes: null, notes }; } if (spec.method === "lad") { const fit = lad(X, y); notes.push(T("Régression médiane : coefficients par IRLS ; les écarts-types exigeraient un bootstrap (non calculé ici).", "Median regression: IRLS coefficients; standard errors would require a bootstrap (not computed here).")); return { fit, V: null, cols, offsetX: 0, lambda: null, lassoSelected: null, postLasso: null, rho: null, robust: null, nb: null, sarRes: null, notes }; } if (spec.method === "ridge" || spec.method === "lasso") { const res = penalized(X, y, spec.method); notes.push(T(`λ choisi par validation croisée à 5 blocs sur une grille de ${res.grid.length} valeurs (échelle standardisée).`, `λ chosen by 5-fold cross-validation over a ${res.grid.length}-value grid (standardised scale).`)); let postLasso: Coefficient[] | null = null; let selected: string[] | null = null; if (spec.method === "lasso") { selected = res.selected.filter((j) => j > 0).map((j) => cols[j].name); if (res.postFit) { const pcols = res.selected.map((j) => cols[j]); const Vp = covariance(res.postFit, spec.se, clusters, nClusters); postLasso = coefTable(pcols, res.postFit.beta, Vp, res.postFit.df, logModel, medP); } notes.push(T("Coefficients LASSO biaisés vers zéro ; l'inférence post-LASSO (MCO sur les variables retenues) est indicative.", "LASSO coefficients are shrunk toward zero; post-LASSO inference (OLS on kept variables) is indicative.")); } else notes.push(T("Ridge : pas d'écarts-types (estimateur biaisé) ; lire les coefficients comme des effets stabilisés.", "Ridge: no standard errors (biased estimator); read coefficients as stabilised effects.")); return { fit: res.fit, V: null, cols, offsetX: 0, lambda: { value: res.lambda, grid: res.grid }, lassoSelected: selected, postLasso, rho: null, robust: null, nb: null, sarRes: null, notes }; } // SAR if (X.n > LIMITS.maxN) throw new Error("SAR : échantillon trop grand"); const res = sar(X, y, lat, lng, 8); const V = covariance(res.fit, spec.se === "cluster" ? "cluster" : spec.se, clusters, nClusters); const k = res.fit.k; const se = V ? Math.sqrt(Math.max(0, V[0])) : null; const t = se ? res.rho / se : null; const p = t != null ? tPValue(t, res.fit.df) : null; const rho: Coefficient = { name: "rho", label: T("ρ — autocorrélation spatiale (W·y, 8 voisins)", "ρ — spatial autocorrelation (W·y, 8 neighbours)"), kind: "rho", coef: res.rho, se, t, p, stars: p != null ? stars(p) : "", ciLo: se != null ? res.rho - 1.96 * se : null, ciHi: se != null ? res.rho + 1.96 * se : null, effectPct: null, effectDollar: null, effectLo: null, effectHi: null, step: T("—", "—") }; void k; notes.push(T("SAR estimé par doubles moindres carrés (instruments WX, W²X) : les β sont des effets directs ; l'effet total ≈ β/(1−ρ).", "SAR estimated by two-stage least squares (WX, W²X instruments): β are direct effects; total effect ≈ β/(1−ρ).")); return { fit: res.fit, V, cols, offsetX: 1, lambda: null, lassoSelected: null, postLasso: null, rho, robust: null, nb: res.index, sarRes: res, notes }; } /* ------------------------------ validation croisée ------------------------------ */ function crossValidate(design: Design, spec: HedonicSpec, folds: number, medP: number): FitMetrics { const { X, y } = design; const n = X.n; const logModel = spec.depvar === "lnprice"; const prices = Float64Array.from(design.obs, (o) => o.price); const lat = Float64Array.from(design.obs, (o) => o.lat); const lng = Float64Array.from(design.obs, (o) => o.lng); const predAll = new Float64Array(n); const fittedAll = new Float64Array(n); for (let f = 0; f < folds; f++) { const tr: number[] = []; const te: number[] = []; for (let i = 0; i < n; i++) (i % folds === f ? te : tr).push(i); const Xtr = takeRows(X, tr); const ytr = takeVec(y, tr); const Xte = takeRows(X, te); let predTe: Float64Array; let smear = 1; if (spec.method === "sar") { const r = sar(Xtr, ytr, takeVec(lat, tr), takeVec(lng, tr), 8); smear = smearing(r.fit.resid, logModel); predTe = sarPredict(r, Xte, takeVec(lat, te), takeVec(lng, te), ytr, r.fit.beta); } else { let beta: Float64Array; let resid: Float64Array; if (spec.method === "ols") ({ beta, resid } = ols(Xtr, ytr)); else if (spec.method === "huber") ({ beta, resid } = huber(Xtr, ytr, 15)); else if (spec.method === "lad") ({ beta, resid } = lad(Xtr, ytr, 25)); else ({ beta, resid } = penalized(Xtr, ytr, spec.method, 3).fit); smear = smearing(resid, logModel); predTe = new Float64Array(te.length); for (let r = 0; r < te.length; r++) { let s = 0; for (let j = 0; j < Xte.k; j++) s += Xte.a[r * Xte.k + j] * beta[j]; predTe[r] = s; } } for (let r = 0; r < te.length; r++) { fittedAll[te[r]] = predTe[r]; predAll[te[r]] = logModel ? Math.exp(predTe[r]) * smear : predTe[r]; } } const resid = new Float64Array(n); for (let i = 0; i < n; i++) resid[i] = y[i] - fittedAll[i]; void medP; // métriques hors échantillon : on passe des prédictions déjà en $ via smear=1 et fitted=ln(pred) const fittedForMetrics = logModel ? Float64Array.from(predAll, Math.log) : predAll; return fitMetrics(y, fittedForMetrics, resid, prices, logModel, X.k, 1); } /* --------------------------------- rapport --------------------------------- */ export function runHedonic(spec: HedonicSpec): HedonicResult { const t0 = Date.now(); const sample = buildSample(spec); if (sample.obs.length < LIMITS.minN) throw new Error(`échantillon insuffisant : ${sample.obs.length} observations (min ${LIMITS.minN}). Élargissez la zone, la période ou les types.`); const design = buildDesign(sample.obs, spec); const { X, y, cols, obs } = design; const n = X.n; const logModel = spec.depvar === "lnprice"; const prices = Float64Array.from(obs, (o) => o.price); const medP = median(prices); const est = estimateAll(design, spec, medP); const fit = est.fit; const coefficients = coefTable(cols, fit.beta, est.V, fit.df, logModel, medP, est.offsetX); const metrics = fitMetrics(y, fit.fitted, fit.resid, prices, logModel, fit.k); // F global (MCO/Huber seulement) let fstat: HedonicResult["model"]["fstat"] = null; if ((spec.method === "ols" || spec.method === "huber") && fit.k > 1) { const df1 = fit.k - 1; const df2 = fit.df; const f = (metrics.r2 / df1) / ((1 - metrics.r2) / df2); fstat = { f, df1, df2, p: fSf(f, df1, df2) }; } // diagnostics let bp: HedonicResult["diagnostics"]["bp"] = null; try { bp = breuschPagan(X, fit.resid); } catch {} const jb = jarqueBera(fit.resid); const lat = Float64Array.from(obs, (o) => o.lat); const lng = Float64Array.from(obs, (o) => o.lng); let moran: Moran | null = null; try { const index = est.nb ?? new KnnIndex(lat, lng); const nb = est.sarRes ? est.sarRes.nb : index.selfNeighbours(8); moran = moranI(nb, fit.resid); } catch {} let vifs: HedonicResult["diagnostics"]["vif"] = []; try { const contCols = cols.map((c, j) => ({ c, j })).filter(({ c }) => c.kind === "cont" || c.kind === "trend"); const olsFit = spec.method === "ols" ? fit : ols(X, y); if (olsFit.bread && contCols.length) { const v = vif(X, olsFit.bread, contCols.map(({ j }) => j)); vifs = contCols.map(({ c }, i) => ({ name: c.name, label: c.label, vif: v[i] })); } } catch {} // validation croisée let cv: HedonicResult["cv"] = null; if (spec.cv) { try { const m = crossValidate(design, spec, 5, medP); cv = { folds: 5, metrics: m, gapMdape: m.mdape - metrics.mdape }; } catch {} } // comparaison au laboratoire let lab: HedonicResult["lab"] = null; if (spec.compareLab) { const idx: number[] = []; for (let i = 0; i < n; i++) if (obs[i].lab && obs[i].lab! > 0) idx.push(i); if (idx.length >= 30) { const pred = toDollars(fit.fitted, logModel, metrics.smear); const m = (get: (i: number) => number) => { const ape: number[] = []; let se = 0; let w10 = 0; let w20 = 0; for (const i of idx) { const a = Math.abs(get(i) - prices[i]) / prices[i]; ape.push(a); se += (get(i) - prices[i]) ** 2; if (a <= 0.1) w10++; if (a <= 0.2) w20++; } return { mdape: median(ape), rmse: Math.sqrt(se / idx.length), within10: w10 / idx.length, within20: w20 / idx.length }; }; lab = { n: idx.length, lab: m((i) => obs[i].lab!), ours: m((i) => pred[i]), label: spec.source === "sales" ? T("LightGBM du laboratoire (est_2026 ramenée à la date de vente par l'indice)", "Lab LightGBM (est_2026 deflated to the sale date with the index)") : T("Mesure UQO Éval de l'annonce (hybride 65/35)", "UQO Éval measure of the listing (65/35 hybrid)"), }; } } // indice temporel et effets géo const V = est.V; const kk = fit.k; const idxOf = (name: string) => cols.findIndex((c) => c.name === name); let timeIndex: HedonicResult["timeIndex"] = null; if (spec.fe.time !== "none" && design.timeLevels.length) { timeIndex = design.timeLevels.map((l) => { if (l.base) return { level: l.level, n: l.n, coef: 0, index: 100, lo: null, hi: null }; const j = idxOf(`t=${l.level}`) + est.offsetX; const b = fit.beta[j]; const se = V ? Math.sqrt(Math.max(0, V[j * kk + j])) : null; const toIdx = (x: number) => (logModel ? 100 * Math.exp(x) : 100 * (1 + x / medP)); return { level: l.level, n: l.n, coef: b, index: toIdx(b), lo: se != null ? toIdx(b - 1.96 * se) : null, hi: se != null ? toIdx(b + 1.96 * se) : null }; }); } let geoEffects: HedonicResult["geoEffects"] = null; if (spec.fe.geo !== "none" && design.geoLevels.length) { geoEffects = design.geoLevels .map((l) => { if (l.base) return { level: l.level, n: l.n, coef: 0, pct: 0, se: null }; const j = idxOf(`g=${l.level}`) + est.offsetX; const b = fit.beta[j]; const se = V ? Math.sqrt(Math.max(0, V[j * kk + j])) : null; return { level: l.level, n: l.n, coef: b, pct: logModel ? 100 * (Math.exp(b) - 1) : (100 * b) / medP, se }; }) .sort((a, b) => b.pct - a.pct); } // graphiques const stride = Math.max(1, Math.floor(n / 1200)); const residFitted: { f: number; e: number }[] = []; const actualPred: { a: number; p: number }[] = []; const predD = toDollars(fit.fitted, logModel, metrics.smear); for (let i = 0; i < n; i += stride) { residFitted.push({ f: fit.fitted[i], e: fit.resid[i] / metrics.sigma }); actualPred.push({ a: prices[i], p: predD[i] }); } const strideMap = Math.max(1, Math.floor(n / 2500)); const map: { lat: number; lng: number; e: number }[] = []; for (let i = 0; i < n; i += strideMap) map.push({ lat: obs[i].lat, lng: obs[i].lng, e: fit.resid[i] / metrics.sigma }); // insights // typographie française : virgule décimale const insights = buildInsights(spec, design, coefficients, metrics, cv, lab, bp, jb, moran, vifs, timeIndex, geoEffects, est, medP).map((s) => ({ fr: s.fr.replace(/(\d)\.(\d)/g, "$1,$2"), en: s.en, })); const byTypeMap = new Map(); for (const o of obs) byTypeMap.set(o.type, (byTypeMap.get(o.type) ?? 0) + 1); const dates = obs.map((o) => o.date).sort(); const methodDef = METHODS.find((m) => m.key === spec.method)!; return { spec, sample: { n, available: sample.available, trimmed: sample.trimmed, dropped: design.dropped, period: { from: dates[0], to: dates[dates.length - 1] }, byType: [...byTypeMap.entries()].map(([type, n2]) => ({ type, n: n2 })).sort((a, b) => b.n - a.n), descriptives: descriptives(design), medianPrice: medP, center: design.center, }, model: { method: spec.method, methodLabel: T(methodDef.fr, methodDef.en), depvar: spec.depvar, equation: equationOf(spec, cols), n, k: fit.k, df: fit.df, seType: spec.se, nClusters: design.nClusters, coefficients, fit: metrics, fstat, lambda: est.lambda, lassoSelected: est.lassoSelected, postLasso: est.postLasso, rho: est.rho, robust: est.robust, notes: est.notes, }, diagnostics: { bp, jb, moran, vif: vifs }, cv, lab, timeIndex, geoEffects, plots: { residFitted, qq: qqPoints(fit.resid, metrics.sigma), hist: histogram(Float64Array.from(fit.resid, (e) => e / metrics.sigma)), map, actualPred }, insights, code: codeSnippets(spec, cols), timingMs: Date.now() - t0, }; } /* --------------------------------- lecture --------------------------------- */ function buildInsights( spec: HedonicSpec, design: Design, coefs: Coefficient[], m: FitMetrics, cv: HedonicResult["cv"], lab: HedonicResult["lab"], bp: HedonicResult["diagnostics"]["bp"], jb: HedonicResult["diagnostics"]["jb"], moran: Moran | null, vifs: HedonicResult["diagnostics"]["vif"], timeIndex: HedonicResult["timeIndex"], geo: HedonicResult["geoEffects"], est: Est, medP: number ): Txt[] { const out: Txt[] = []; const pc = (x: number) => (100 * x).toFixed(1); // 1. qualité out.push( T( `Le modèle explique ${pc(m.r2)} % de la variance ${spec.depvar === "lnprice" ? "du log du prix" : "du prix"} (R² ajusté ${pc(m.adjR2)} %) sur ${m.n.toLocaleString("fr-CA")} observations. Erreur médiane absolue : ${pc(m.mdape)} % ; ${pc(m.within10)} % des prédictions à ±10 %, ${pc(m.within20)} % à ±20 %.`, `The model explains ${pc(m.r2)}% of the variance ${spec.depvar === "lnprice" ? "of log price" : "of price"} (adjusted R² ${pc(m.adjR2)}%) over ${m.n.toLocaleString("en-CA")} observations. Median absolute error: ${pc(m.mdape)}%; ${pc(m.within10)}% of predictions within ±10%, ${pc(m.within20)}% within ±20%.` ) ); if (cv) { const gap = cv.gapMdape * 100; out.push( T( `Hors échantillon (validation croisée à ${cv.folds} blocs), l'erreur médiane passe à ${pc(cv.metrics.mdape)} % (${gap >= 0 ? "+" : "−"}${Math.abs(gap).toFixed(1)} pt) — ${gap > 3 ? "écart notable : le modèle sur-apprend (trop d'effets fixes fins pour l'échantillon ?)" : "écart faible : le modèle généralise bien"}.`, `Out of sample (${cv.folds}-fold cross-validation), the median error moves to ${pc(cv.metrics.mdape)}% (${gap >= 0 ? "+" : "−"}${Math.abs(gap).toFixed(1)} pt) — ${gap > 3 ? "notable gap: the model overfits (too many fine fixed effects for the sample?)" : "small gap: the model generalises well"}.` ) ); } // 2. effets clés const key = (v: string) => coefs.find((c) => c.var === v && c.kind === "cont"); const a = key("aire"); if (a && a.effectPct != null) out.push( a.step.fr === "pour +10 %" ? T(`Élasticité prix-superficie : ${a.coef.toFixed(3)} — +10 % d'aire d'étages ⇒ ${fmtPct(a.effectPct)} du prix (≈ ${fmtMoney(a.effectDollar!)} au prix médian de ${fmtMoney(medP)})${a.p != null ? `, p ${a.p < 0.001 ? "< 0,001" : "= " + a.p.toFixed(3)}` : ""}.`, `Price-area elasticity: ${a.coef.toFixed(3)} — +10% floor area ⇒ ${fmtPct(a.effectPct)} of price (≈ ${fmtMoneyEn(a.effectDollar!)} at the median price of ${fmtMoneyEn(medP)})${a.p != null ? `, p ${a.p < 0.001 ? "< 0.001" : "= " + a.p.toFixed(3)}` : ""}.`) : T(`Chaque m² d'aire d'étages supplémentaire vaut ${fmtPct(a.effectPct, 2)} du prix, soit ≈ ${fmtMoney(a.effectDollar!)} au prix médian.`, `Each extra m² of floor area is worth ${fmtPct(a.effectPct, 2)} of price, ≈ ${fmtMoneyEn(a.effectDollar!)} at the median price.`) ); const te = key("terrain"); if (te && te.effectPct != null) out.push(T(`Terrain : ${te.step.fr === "pour +10 %" ? `élasticité ${te.coef.toFixed(3)} (+10 % ⇒ ${fmtPct(te.effectPct)})` : `${fmtMoney(te.effectDollar!)} par m²`} — ${Math.abs(te.coef) < (te.step.fr === "pour +10 %" ? 0.3 : 50) ? "rendement décroissant marqué, typique de la rente foncière résidentielle" : "contribution forte du sol"}.`, `Lot: ${te.step.en === "for +10%" ? `elasticity ${te.coef.toFixed(3)} (+10% ⇒ ${fmtPct(te.effectPct)})` : `${fmtMoneyEn(te.effectDollar!)} per m²`} — ${Math.abs(te.coef) < (te.step.en === "for +10%" ? 0.3 : 50) ? "strong diminishing returns, typical of residential land rent" : "strong land contribution"}.`)); const ag = key("age"); const ag2 = coefs.find((c) => c.var === "age2"); if (ag && ag.effectPct != null) { if (ag2) { // sommet de la parabole : d/dage = b1 + 2·b2·age/100 = 0 → age* = −50·b1/b2 const turn = ag2.coef !== 0 ? (-50 * ag.coef) / ag2.coef : NaN; out.push(T(`Âge : ${fmtPct(ag.effectPct, 2)} par année au départ, courbure ${ag2.coef >= 0 ? "convexe" : "concave"}${Number.isFinite(turn) && turn > 0 && turn < 200 ? ` — la dépréciation ${ag2.coef > 0 ? "s'inverse (effet vintage) vers" : "s'accélère au-delà de"} ${turn.toFixed(0)} ans` : ""}.`, `Age: ${fmtPct(ag.effectPct, 2)} per year initially, ${ag2.coef >= 0 ? "convex" : "concave"} curvature${Number.isFinite(turn) && turn > 0 && turn < 200 ? ` — depreciation ${ag2.coef > 0 ? "reverses (vintage effect) around" : "accelerates beyond"} ${turn.toFixed(0)} years` : ""}.`)); } else out.push(T(`Dépréciation : ${fmtPct(ag.effectPct, 2)} du prix par année d'âge (≈ ${fmtMoney(ag.effectDollar!)} au prix médian).`, `Depreciation: ${fmtPct(ag.effectPct, 2)} of price per year of age (≈ ${fmtMoneyEn(ag.effectDollar!)} at the median price).`)); } const dc = key("dist_centre"); if (dc && dc.effectPct != null) out.push(T(`Gradient de rente : ${fmtPct(dc.effectPct, 2)} du prix ${dc.step.fr} de distance au centre${dc.effectPct < 0 ? " — la centralité est valorisée" : " — la périphérie est ici valorisée (grands terrains, secteurs cossus ?)"}.`, `Rent gradient: ${fmtPct(dc.effectPct, 2)} of price ${dc.step.en} of distance to the centre${dc.effectPct < 0 ? " — centrality is valued" : " — the periphery is valued here (large lots, affluent sectors?)"}.`)); const dummies = coefs.filter((c) => c.kind === "dummy" && c.p != null && c.p < 0.05 && c.effectPct != null).sort((x, y2) => Math.abs(y2.effectPct!) - Math.abs(x.effectPct!)); if (dummies.length) { const d = dummies[0]; out.push(T(`Attribut le plus discriminant : « ${d.label.fr} » ⇒ ${fmtPct(d.effectPct!)} vs ${d.base} (significatif à 5 %).`, `Most discriminating attribute: “${d.label.en}” ⇒ ${fmtPct(d.effectPct!)} vs ${d.base} (significant at 5%).`)); } // 3. temps / géo if (timeIndex && timeIndex.length > 1) { const last = timeIndex[timeIndex.length - 1]; const peak = timeIndex.reduce((p, c) => (c.index > p.index ? c : p), timeIndex[0]); out.push(T(`Indice hédonique (qualité constante) : ${last.index.toFixed(1)} en ${last.level} (base 100 = ${timeIndex[0].level}), sommet ${peak.index.toFixed(1)} en ${peak.level}. C'est l'appréciation nette des changements de composition des ventes.`, `Hedonic (constant-quality) index: ${last.index.toFixed(1)} in ${last.level} (base 100 = ${timeIndex[0].level}), peak ${peak.index.toFixed(1)} in ${peak.level}. This is appreciation net of changes in the sales mix.`)); } if (geo && geo.length > 2) { const top = geo[0]; const bottom = geo[geo.length - 1]; out.push(T(`Prime de localisation : ${top.level} ${fmtPct(top.pct)} vs ${bottom.level} ${fmtPct(bottom.pct)} (référence : ${design.geoLevels.find((l) => l.base)?.level}) — l'écart de ${(top.pct - bottom.pct).toFixed(0)} pts est l'effet « quartier » à caractéristiques égales.`, `Location premium: ${top.level} ${fmtPct(top.pct)} vs ${bottom.level} ${fmtPct(bottom.pct)} (base: ${design.geoLevels.find((l) => l.base)?.level}) — the ${(top.pct - bottom.pct).toFixed(0)}-pt gap is the “neighbourhood” effect holding characteristics constant.`)); } // 4. diagnostics if (bp) out.push(bp.p < 0.05 ? T(`Hétéroscédasticité détectée (Breusch-Pagan p ${bp.p < 0.001 ? "< 0,001" : "= " + bp.p.toFixed(3)}) : ${spec.se === "classic" ? "passez aux écarts-types robustes (HC1) — les p-valeurs classiques sont trompeuses" : "vos écarts-types robustes sont donc de mise"}.`, `Heteroskedasticity detected (Breusch-Pagan p ${bp.p < 0.001 ? "< 0.001" : "= " + bp.p.toFixed(3)}): ${spec.se === "classic" ? "switch to robust (HC1) standard errors — classical p-values are misleading" : "your robust standard errors are therefore warranted"}.`) : T(`Pas d'hétéroscédasticité détectable (Breusch-Pagan p = ${bp.p.toFixed(3)}).`, `No detectable heteroskedasticity (Breusch-Pagan p = ${bp.p.toFixed(3)}).`)); out.push(jb.p < 0.05 ? T(`Résidus non normaux (Jarque-Bera, asymétrie ${jb.skew.toFixed(2)}, aplatissement ${jb.kurt.toFixed(1)}) — sans conséquence sur les coefficients en grand échantillon, mais ${jb.kurt > 5 ? "les queues épaisses suggèrent des ventes atypiques : essayez la régression robuste ou quantile" : "les intervalles de prédiction individuels sont approximatifs"}.`, `Non-normal residuals (Jarque-Bera, skewness ${jb.skew.toFixed(2)}, kurtosis ${jb.kurt.toFixed(1)}) — harmless for coefficients in large samples, but ${jb.kurt > 5 ? "fat tails suggest atypical sales: try robust or quantile regression" : "individual prediction intervals are approximate"}.`) : T("Résidus compatibles avec la normalité (Jarque-Bera).", "Residuals compatible with normality (Jarque-Bera).")); if (moran && Number.isFinite(moran.z)) out.push(moran.p < 0.05 ? T(`Autocorrélation spatiale des résidus (I de Moran = ${moran.I.toFixed(3)}, z = ${moran.z.toFixed(1)}) : les erreurs voisines se ressemblent — il manque de la localisation au modèle (${spec.method === "sar" ? "malgré le terme ρWy" : spec.fe.geo === "none" ? "ajoutez des effets fixes géo ou passez au SAR" : "affinez la grille ou passez au SAR"}).`, `Spatial autocorrelation of residuals (Moran's I = ${moran.I.toFixed(3)}, z = ${moran.z.toFixed(1)}): neighbouring errors look alike — the model lacks location (${spec.method === "sar" ? "despite the ρWy term" : spec.fe.geo === "none" ? "add geo fixed effects or switch to SAR" : "refine the grid or switch to SAR"}).`) : T(`Pas d'autocorrélation spatiale résiduelle notable (I de Moran = ${moran.I.toFixed(3)}, p = ${moran.p.toFixed(2)}) — la localisation est bien captée.`, `No notable residual spatial autocorrelation (Moran's I = ${moran.I.toFixed(3)}, p = ${moran.p.toFixed(2)}) — location is well captured.`)); const bad = vifs.filter((v) => v.vif > 10 && v.name !== "age2" && v.name !== "age" && !v.name.startsWith("trend_")); const mech = vifs.some((v) => v.vif > 10 && (v.name === "age2" || v.name === "age" || v.name.startsWith("trend_"))); if (mech && !bad.length) out.push(T("Les VIF élevés de l'âge² ou de la surface de tendance sont mécaniques (termes polynomiaux) — sans gravité pour l'interprétation des effets marginaux.", "High VIFs for age² or the trend surface are mechanical (polynomial terms) — harmless for interpreting marginal effects.")); if (bad.length) out.push(T(`Colinéarité : VIF > 10 pour ${bad.map((b) => b.label.fr).join(", ")} — coefficients instables ; ${spec.method === "ridge" ? "le ridge l'atténue" : "envisagez le ridge ou retirez une variable redondante"}.`, `Collinearity: VIF > 10 for ${bad.map((b) => b.label.en).join(", ")} — unstable coefficients; ${spec.method === "ridge" ? "ridge mitigates it" : "consider ridge or drop a redundant variable"}.`)); // 5. méthode if (est.rho) out.push(T(`ρ = ${est.rho.coef.toFixed(3)}${est.rho.p != null ? ` (p ${est.rho.p < 0.001 ? "< 0,001" : "= " + est.rho.p.toFixed(3)})` : ""} : ${Math.abs(est.rho.coef) > 0.2 ? "forte dépendance spatiale — le prix des 8 voisins « explique » une part importante du prix ; multiplicateur spatial ≈ " + (1 / (1 - est.rho.coef)).toFixed(2) : "dépendance spatiale modérée"}.`, `ρ = ${est.rho.coef.toFixed(3)}${est.rho.p != null ? ` (p ${est.rho.p < 0.001 ? "< 0.001" : "= " + est.rho.p.toFixed(3)})` : ""}: ${Math.abs(est.rho.coef) > 0.2 ? "strong spatial dependence — neighbours' prices “explain” a large share of price; spatial multiplier ≈ " + (1 / (1 - est.rho.coef)).toFixed(2) : "moderate spatial dependence"}.`)); if (est.robust) out.push(T(`Huber : ${est.robust.downweightedPct.toFixed(1)} % des observations sous-pondérées (résidus > 1,345·σ̂).`, `Huber: ${est.robust.downweightedPct.toFixed(1)}% of observations downweighted (residuals > 1.345·σ̂).`)); if (est.lassoSelected) out.push(T(`LASSO : ${est.lassoSelected.length} variables retenues sur ${coefs.length - 1} (λ = ${est.lambda!.value.toExponential(2)}).`, `LASSO: ${est.lassoSelected.length} variables kept out of ${coefs.length - 1} (λ = ${est.lambda!.value.toExponential(2)}).`)); // 6. laboratoire if (lab) { const better = lab.ours.mdape < lab.lab.mdape; out.push(T(`Face au laboratoire (${lab.n.toLocaleString("fr-CA")} observations communes) : votre modèle ${pc(lab.ours.mdape)} % d'erreur médiane vs ${pc(lab.lab.mdape)} % pour ${lab.label.fr} — ${better ? "vous faites mieux en échantillon (attention : le laboratoire est évalué hors échantillon, comparez avec votre validation croisée)" : "le laboratoire reste devant : il exploite 690 000 ventes et des interactions non linéaires (arbres)"}.`, `Against the lab (${lab.n.toLocaleString("en-CA")} common observations): your model ${pc(lab.ours.mdape)}% median error vs ${pc(lab.lab.mdape)}% for ${lab.label.en} — ${better ? "you do better in-sample (caution: the lab is evaluated out of sample; compare with your cross-validation)" : "the lab stays ahead: it uses 690,000 sales and non-linear interactions (trees)"}.`)); } return out; } export type { Obs }; export { hcat, lag, type Mat };