Results / atlas/qwen3-0.6b-4bit/probes/v2/mapcard.json
{
"map_id": "atlas/qwen3-0.6b-4bit/probes/v2",
"map_type": "probes",
"model_id": "mlx-community/Qwen3-0.6B-4bit",
"model_hash": "392e8d466d56100ada00eb82031fb854297fc9e389b7d303eba3af114e87bce2",
"quantization": "q4 (mlx)",
"commit": "1dcd820126ca96fe99c1b473947f3c7e54038500",
"config": "results/expA_probe_reliability/20260812T063856Z/results.json",
"created": "2026-08-12",
"hardware_manifest": {
"author": "Simon-Pierre Boucher",
"contact": "contact@spboucher.ai",
"website": "https://modelmap.io",
"chip": {
"brand": "Apple M5 Max",
"cores_total": 18,
"cores_performance": 6,
"cores_efficiency": 12
},
"memory": {
"unified_gb": 48,
"pagesize": 16384
},
"os": {
"system": "Darwin",
"version": "27.0",
"arch": "arm64"
},
"software": {
"python": "3.14.4",
"numpy": "2.5.2",
"mlx": "0.32.0",
"torch": "2.13.0",
"safetensors": "0.8.0"
}
},
"confidence_level": 1,
"regenerate_command": ".venv/bin/python benchmarks/promptsets/make_promptsets_v2.py && .venv/bin/python experiments/micro/expA_probe_reliability/implementation/benchmark_v2.py && .venv/bin/python experiments/micro/expA_probe_reliability/implementation/make_mapcard_v2.py",
"seeds": [
0,
1,
2,
3,
4
],
"prompt_sets": [
"agreement_A.jsonl#622c5e0966d2b8e8",
"agreement_B.jsonl#243b85e2c207be6c",
"arith_valid_A.jsonl#a4bb4c645b3f1797",
"arith_valid_B.jsonl#adc68ef876a823ec",
"word_order_A.jsonl#9052b924930aa5cd",
"word_order_B.jsonl#3012db3b5053ba5d"
],
"controls": [
"shuffled-label (every probe)",
"random-init architecture twin",
"BH-FDR q=0.05",
"v1 positive control (ceiling check)",
"class token-overlap certificates in promptset manifest"
],
"methods_in_agreement": [],
"interventions": [
"layer-skip ablation (expC run #1, 2026-08-12): top-5 differential layers NOT confirmed — damage below random-5 mean; see experiments/micro/expC_causal_verification/analysis.md"
],
"replication_rate": 0.5366,
"per_dataset_agreement": null,
"ablation_schemes": [],
"featurizer_class": "natural-basis (mean-pooled + last-token residual)",
"intervention_protocol": "none (observational map — Level 1 by design)",
"negative_result": false,
"notes": "DIFFERENTIAL map (real minus random-init twin), per the doctrine adopted after v1. Mixed outcome by property: agreement and arith_valid carry trained-model signal above the architecture prior; word_order is null-dominated and flagged as such. Strict twin gate (<0.05) still fails on word_order/agreement — only differential claims are published. CAUSAL CHECK: expC run #1 layer-skip ablation did NOT confirm the top differential layers (survival 0/1); map remains Level 1 and its layer ranking must not be read as causal.",
"author": "Simon-Pierre Boucher",
"contact": "contact@spboucher.ai",
"website": "https://modelmap.io",
"schema_version": "0.1"
}