{
  "map_id": "atlas/qwen3-0.6b-4bit/probes/v1",
  "map_type": "probes",
  "model_id": "mlx-community/Qwen3-0.6B-4bit",
  "model_hash": "392e8d466d56100ada00eb82031fb854297fc9e389b7d303eba3af114e87bce2",
  "quantization": "q4 (mlx)",
  "commit": "3935e7933294b9c1293cd31b887126402fc53115",
  "config": "results/expA_probe_reliability/20260812T062605Z/results.json",
  "created": "2026-08-12",
  "hardware_manifest": {
    "author": "Simon-Pierre Boucher",
    "contact": "contact@spboucher.ai",
    "website": "https://modelmap.io",
    "chip": {
      "brand": "Apple M5 Max",
      "cores_total": 18,
      "cores_performance": 6,
      "cores_efficiency": 12
    },
    "memory": {
      "unified_gb": 48.0,
      "pagesize": 16384
    },
    "os": {
      "system": "Darwin",
      "version": "27.0",
      "arch": "arm64"
    },
    "software": {
      "python": "3.14.4",
      "numpy": "2.5.2",
      "mlx": "0.32.0",
      "torch": "2.13.0",
      "safetensors": "0.8.0"
    }
  },
  "confidence_level": 1,
  "regenerate_command": ".venv/bin/python benchmarks/promptsets/make_promptsets.py && .venv/bin/python experiments/micro/expA_probe_reliability/implementation/benchmark.py && .venv/bin/python experiments/micro/expA_probe_reliability/implementation/make_mapcard.py",
  "seeds": [
    0,
    1,
    2,
    3,
    4
  ],
  "prompt_sets": [
    "arith_A.jsonl#2fd80600d8a1b4ad",
    "arith_B.jsonl#04bb0eceb4b9262e",
    "code_prose_A.jsonl#dfbfb13dade0fd0b",
    "code_prose_B.jsonl#e9c3f78d8754b0ad",
    "lang_id_A.jsonl#43bd7ed12d2d9f95",
    "lang_id_B.jsonl#833435dae9a61ae0"
  ],
  "controls": [
    "shuffled-label (every probe)",
    "random-init architecture twin",
    "BH-FDR q=0.05 across layer scans"
  ],
  "methods_in_agreement": [],
  "interventions": [],
  "replication_rate": 1.0,
  "per_dataset_agreement": null,
  "ablation_schemes": [],
  "featurizer_class": "natural-basis (mean-pooled residual)",
  "intervention_protocol": "none (observational map — Level 1 by design)",
  "negative_result": true,
  "notes": "NEGATIVE RESULT: the random-init architecture twin reaches task accuracy 1.00 at every layer for every property — the probe map is indistinguishable from the architecture+tokenizer null on these template promptsets. This map is published as evidence that probe maps on lexically separable classes are uninformative about trained structure. See analysis.md.",
  "author": "Simon-Pierre Boucher",
  "contact": "contact@spboucher.ai",
  "website": "https://modelmap.io",
  "schema_version": "0.1"
}
