#!/usr/bin/env python3 # ============================================================================= # Project : modelmap # File : benchmarks/promptsets/make_promptsets.py # Purpose : Generate versioned, checksummed probing corpora (v1, template) # Author : Simon-Pierre Boucher # Contact : contact@spboucher.ai # Website : https://modelmap.io # Created : 2026-08-12 # Modified : 2026-08-12 # Platform : macOS / Apple Silicon (arm64) # License : All rights reserved (research code) # ============================================================================= """Probing corpora v1 (charter §0.3: versioned artifacts with checksums). Three binary properties, TWO deliberately different template families per property (set A / set B) so that dataset sensitivity is measurable: lang_id : French vs English sentences code_prose : code snippets vs English prose arith : arithmetic-context vs non-arithmetic sentences Deterministic given SEED. v1 limitation, declared: template-generated text, not natural corpora — lexical diversity is bounded; natural-corpus v2 is a registered follow-up. Output: .jsonl + manifest.json with sha256. """ from __future__ import annotations import hashlib import itertools import json import random from pathlib import Path SEED = 12345 N_PER_CLASS = 120 VERSION = "v1" OUT = Path(__file__).resolve().parent TOPICS_EN = ["the harbor", "the library", "the orchard", "the workshop", "the observatory", "the market", "the glacier", "the archive", "the vineyard", "the lighthouse"] TOPICS_FR = ["le port", "la bibliothèque", "le verger", "l'atelier", "l'observatoire", "le marché", "le glacier", "les archives", "le vignoble", "le phare"] ADJ_EN = ["quiet", "ancient", "crowded", "restored", "abandoned", "famous", "modest", "vast"] ADJ_FR = ["calme", "ancien", "bondé", "restauré", "abandonné", "célèbre", "modeste", "vaste"] VERB_EN = ["remained open despite the storm", "attracted visitors from the region", "was documented in the survey", "changed hands twice last century", "stood at the edge of town", "required constant maintenance"] VERB_FR = ["est resté ouvert malgré la tempête", "a attiré des visiteurs de la région", "a été documenté dans l'enquête", "a changé de mains deux fois au siècle dernier", "se trouvait à la limite de la ville", "exigeait un entretien constant"] Q_EN = ["Could you tell me whether", "Do you happen to know if", "I was wondering whether"] Q_FR = ["Pourrais-tu me dire si", "Sais-tu par hasard si", "Je me demandais si"] PY_FUNCS = ["total", "scale", "merge", "clip", "score", "rank", "fold", "trim"] JS_VARS = ["items", "nodes", "queue", "cache", "rows", "edges", "bins", "keys"] PROSE_SUBJ = ["The committee", "A local historian", "The lead engineer", "Her assistant", "The night watchman", "An early visitor", "The town council", "The apprentice"] PROSE_TAIL = ["reviewed the plans before the meeting.", "kept detailed notes for years.", "questioned the original estimate.", "preferred the older method.", "described the process in a letter.", "returned before the first frost."] def gen(rng: random.Random): sets: dict[str, list[dict]] = {} # -------- lang_id: set A = descriptive statements, set B = questions a, b = [], [] for t, adj, v in itertools.product(TOPICS_EN, ADJ_EN, VERB_EN): a.append({"text": f"The {adj} site near {t} {v}.", "label": "en"}) for t, adj, v in itertools.product(TOPICS_FR, ADJ_FR, VERB_FR): a.append({"text": f"Le site {adj} près de {t} {v}.", "label": "fr"}) for q, t, v in itertools.product(Q_EN, TOPICS_EN, VERB_EN): b.append({"text": f"{q} the place near {t} {v}?", "label": "en"}) for q, t, v in itertools.product(Q_FR, TOPICS_FR, VERB_FR): b.append({"text": f"{q} l'endroit près de {t} {v} ?", "label": "fr"}) sets["lang_id_A"], sets["lang_id_B"] = a, b # -------- code_prose: set A = python vs prose, set B = js vs prose a, b = [], [] for f, op, k in itertools.product(PY_FUNCS, ["+", "-", "*"], [1, 2, 3, 5, 7]): a.append({"text": f"def {f}(xs):\n return [x {op} {k} for x in xs if x > {k + 1}]", "label": "code"}) for s, t, adv in itertools.product(PROSE_SUBJ, PROSE_TAIL, ["eventually", "reluctantly", "quietly"]): a.append({"text": f"{s} {adv} {t.lower()}", "label": "prose"}) for v, meth, k in itertools.product(JS_VARS, ["filter", "map", "find", "some"], [0, 1, 4, 9]): b.append({"text": f"const out = {v}.{meth}(x => x.size > {k}).length;", "label": "code"}) for s, t, adj in itertools.product(PROSE_SUBJ, PROSE_TAIL[:3], ADJ_EN[:5]): b.append({"text": f"{s}, though {adj}, {t.lower()}", "label": "prose"}) sets["code_prose_A"], sets["code_prose_B"] = a, b # -------- arith: set A = imperative computations, set B = embedded quantities a, b = [], [] for _ in range(4 * N_PER_CLASS): x, y = rng.randint(11, 97), rng.randint(11, 97) a.append({"text": f"Calculate {x} + {y} and report the result.", "label": "arith"}) s, t = rng.choice(PROSE_SUBJ), rng.choice(PROSE_TAIL) adv = rng.choice(["eventually", "reluctantly", "quietly", "finally", "later"]) a.append({"text": f"{s} {adv} {t.lower()}", "label": "plain"}) p, q = rng.randint(12, 89), rng.randint(12, 89) b.append({"text": f"If the crate holds {p} jars and {q} more arrive, how many jars are there in total?", "label": "arith"}) topic, verb, adj = rng.choice(TOPICS_EN), rng.choice(VERB_EN), rng.choice(ADJ_EN) b.append({"text": f"According to the {adj} report, {topic} {verb}.", "label": "plain"}) sets["arith_A"], sets["arith_B"] = a, b # balance + subsample every set to N_PER_CLASS per label, deterministic final = {} for name, items in sets.items(): by = {} for it in items: by.setdefault(it["label"], []).append(it) chosen = [] for label, pool in sorted(by.items()): rng.shuffle(pool) # dedupe by text before sampling seen, uniq = set(), [] for it in pool: if it["text"] not in seen: seen.add(it["text"]) uniq.append(it) if len(uniq) < N_PER_CLASS: raise SystemExit(f"{name}/{label}: only {len(uniq)} unique items") chosen += uniq[:N_PER_CLASS] rng.shuffle(chosen) final[name] = chosen return final def main() -> int: rng = random.Random(SEED) sets = gen(rng) manifest = {"version": VERSION, "seed": SEED, "n_per_class": N_PER_CLASS, "author": "Simon-Pierre Boucher", "contact": "contact@spboucher.ai", "website": "https://modelmap.io", "limitation": "template-generated v1; natural-corpus v2 registered", "files": {}} for name, items in sorted(sets.items()): path = OUT / f"{name}.jsonl" payload = "\n".join(json.dumps(it, ensure_ascii=False) for it in items) + "\n" path.write_text(payload) manifest["files"][path.name] = { "sha256": hashlib.sha256(payload.encode()).hexdigest(), "n": len(items), } print(f"{path.name:22s} n={len(items)} sha256={manifest['files'][path.name]['sha256'][:12]}…") (OUT / "manifest.json").write_text(json.dumps(manifest, indent=2, ensure_ascii=False) + "\n") print("manifest.json written") return 0 if __name__ == "__main__": raise SystemExit(main())