{
 "schema": 1,
 "suite": "parity",
 "source": "compiled from docs/families/*.md on 2026-10-01, plus laya and cross-family rows from docs/decisions/0001-mlx-engine.md, docs/decisions/0004-cuda-onnxruntime-builds.md and docs/distribution.md; every number is quoted from the cited line(s)",
 "results": [
  {
   "family": "winnow",
   "model": "winnow:12b",
   "device": "cuda",
   "device_detail": "RTX 4090 (Linux)",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": 505,
   "decisions": "505/505",
   "max_logit_diff": 0.0000113,
   "max_prob_diff": 0.0000026,
   "date": "2026-09-26",
   "pass": true,
   "cite": "docs/families/winnow.md:122"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "cuda",
   "device_detail": "RTX 4090 (Linux)",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": 505,
   "decisions": "505/505",
   "max_logit_diff": 0.0000129,
   "max_prob_diff": 0.000003,
   "date": "2026-09-26",
   "pass": true,
   "cite": "docs/families/winnow.md:123"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "cpu",
   "device_detail": "x86-64 Linux",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": 505,
   "decisions": "505/505",
   "max_logit_diff": 0.0000114,
   "max_prob_diff": 0.0000029,
   "date": "2026-09-26",
   "pass": true,
   "cite": "docs/families/winnow.md:124"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "cpu",
   "device_detail": "Windows x86-64",
   "reference": "llama-server.exe win-cpu",
   "kind": "runtime",
   "questions": 15,
   "decisions": "15/15",
   "max_logit_diff": 0.0000076,
   "max_prob_diff": 0.0000011,
   "date": null,
   "pass": true,
   "cite": "docs/families/winnow.md:125",
   "note": "partial: the golden replay stopped on a cp1252 text-encoding error, so only the first 3 requests were checked (winnow.md:128-129)"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "cuda",
   "device_detail": "RTX 4090, Windows 11 (ggml-cuda.dll in the 0.7.4 CUDA pack)",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": 505,
   "decisions": "505/505",
   "max_logit_diff": 0.0000129,
   "max_prob_diff": 0.000003,
   "date": "2026-09-28",
   "pass": true,
   "cite": "docs/families/winnow.md:126"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "vulkan",
   "device_detail": "RTX 4090, Windows, llama.cpp win-vulkan-x64 of the same build (not shipped)",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": 505,
   "decisions": "501 of 505",
   "max_logit_diff": 0.32,
   "max_prob_diff": 0.055,
   "date": null,
   "pass": false,
   "cite": "docs/families/winnow.md:133-135",
   "note": "fails the gate (1e-3 on logits, every decision the same); Ollaya does not use Vulkan (#27). Values are 'within 0.32' / 'within 0.055'"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "device": "cuda",
   "device_detail": "RTX 4090 with a CUDA 13 driver, llama.cpp's CUDA 12 backend (CUDA 12 pack)",
   "reference": "stock llama-server of the pinned build (b11146), loaded with the author's GGUF on the device under test",
   "kind": "runtime",
   "questions": null,
   "decisions": "505/505",
   "max_logit_diff": 0.000013,
   "max_prob_diff": null,
   "date": "2026-09-27",
   "pass": true,
   "cite": "docs/distribution.md:123-127"
  },
  {
   "family": "decision",
   "model": "decision:eos",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 CPU (Python)",
   "reference": "the author's code in fp32 (DecisionModel.from_checkpoint(dtype=float32), no autocast, CUDA with TF32 off, SDPA exact math kernel)",
   "kind": "export",
   "questions": 616,
   "decisions": "100 %",
   "max_logit_diff": 0.000038,
   "max_prob_diff": 0.000003,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:176-180",
   "note": "logit max is choice 3.8e-5 (noul 1.3e-5, score 1.3e-5); prob max is choice 3.0e-6"
  },
  {
   "family": "decision",
   "model": "decision:eos",
   "device": "cpu",
   "device_detail": "24 cores; ort 2.0.0-rc.13, ONNX Runtime 1.28",
   "reference": "goldens from the author's code in fp32 (families/decision/goldens.py)",
   "kind": "runtime",
   "questions": 466,
   "decisions": "100 %",
   "max_logit_diff": 0.000036,
   "max_prob_diff": 0.0000024,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:211-214"
  },
  {
   "family": "decision",
   "model": "decision:eos",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "goldens from the author's code in fp32 (families/decision/goldens.py)",
   "kind": "runtime",
   "questions": 466,
   "decisions": "100 %",
   "max_logit_diff": 0.000016,
   "max_prob_diff": 0.0000029,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:211-214"
  },
  {
   "family": "decision",
   "model": "Decision-1.0-Nox-4B",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 Python, CPU",
   "reference": "goldens from the author's code in fp32 (families/decision/goldens.py)",
   "kind": "export",
   "questions": null,
   "decisions": "100 %",
   "max_logit_diff": 0.00017,
   "max_prob_diff": 0.0000068,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:252-257",
   "note": "not in the library (memory); doc: 'parity passes everywhere it runs' (decision.md:250)"
  },
  {
   "family": "decision",
   "model": "Decision-1.0-Nox-4B",
   "device": "cpu",
   "device_detail": "Rust, CPU",
   "reference": "goldens from the author's code in fp32 (families/decision/goldens.py)",
   "kind": "runtime",
   "questions": 466,
   "decisions": "100 %",
   "max_logit_diff": 0.00017,
   "max_prob_diff": 0.0000055,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:252-257"
  },
  {
   "family": "decision",
   "model": "Decision-1.0-Nox-4B",
   "device": "cuda",
   "device_detail": "Rust, CUDA, RTX 4090",
   "reference": "goldens from the author's code in fp32 (families/decision/goldens.py)",
   "kind": "runtime",
   "questions": 466,
   "decisions": "100 %",
   "max_logit_diff": 0.000091,
   "max_prob_diff": 0.0000094,
   "date": null,
   "pass": true,
   "cite": "docs/families/decision.md:252-257",
   "note": "24.0 of 24.5 GB GPU; WSL paging gives 93 s outliers"
  },
  {
   "family": "gliclass",
   "model": "gliclass:large",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU EP, one padded batch per request",
   "reference": "fp32 reference on GPU with TF32 off, one unpadded row per question",
   "kind": "export",
   "questions": 882,
   "decisions": "100.00%",
   "max_logit_diff": 0.000052,
   "max_prob_diff": 0.0000034,
   "date": null,
   "pass": true,
   "cite": "docs/families/gliclass.md:166-171"
  },
  {
   "family": "gliclass",
   "model": "gliclass-instruct-edge-v1.0",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU EP, one padded batch per request",
   "reference": "fp32 reference on GPU with TF32 off, one unpadded row per question",
   "kind": "export",
   "questions": 882,
   "decisions": "100.00%",
   "max_logit_diff": 0.000086,
   "max_prob_diff": 0.0000075,
   "date": null,
   "pass": true,
   "cite": "docs/families/gliclass.md:166-171",
   "note": "not in the library"
  },
  {
   "family": "gliclass",
   "model": "gliclass-instruct-edge-v1.0",
   "device": "mlx",
   "device_detail": "Apple M4 Pro, macOS 27, MLX on Metal",
   "reference": "goldens-gliclass-instruct-edge.jsonl, gate: label logits within 1e-3, same decisions",
   "kind": "runtime",
   "questions": 482,
   "decisions": "100 %",
   "max_logit_diff": 0.000077,
   "max_prob_diff": 0.000014,
   "date": null,
   "pass": true,
   "cite": "docs/families/gliclass.md:191-198",
   "note": "verifies the MLX head; edge is not in the library"
  },
  {
   "family": "gliclass",
   "model": "gliclass-instruct-edge-v1.0",
   "device": "cpu",
   "device_detail": "Apple M4 Pro, macOS 27, ONNX Runtime CPU",
   "reference": "goldens-gliclass-instruct-edge.jsonl, gate: label logits within 1e-3, same decisions",
   "kind": "runtime",
   "questions": 482,
   "decisions": "100 %",
   "max_logit_diff": 0.000091,
   "max_prob_diff": 0.0000069,
   "date": null,
   "pass": true,
   "cite": "docs/families/gliclass.md:191-198"
  },
  {
   "family": "nli",
   "model": "nli:deberta-v3-large",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU EP, one padded batch per request",
   "reference": "fp32 reference on GPU with TF32 off, one unpadded pair at a time",
   "kind": "export",
   "questions": 883,
   "decisions": "100.00%",
   "max_logit_diff": 0.000041,
   "max_prob_diff": 0.0000054,
   "date": null,
   "pass": true,
   "cite": "docs/families/nli.md:159-165",
   "note": "logit = entailment logit"
  },
  {
   "family": "nli",
   "model": "nli:modernbert-large",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU EP, one padded batch per request",
   "reference": "fp32 reference on GPU with TF32 off, one unpadded pair at a time",
   "kind": "export",
   "questions": 883,
   "decisions": "100.00%",
   "max_logit_diff": 0.00015,
   "max_prob_diff": 0.00001,
   "date": null,
   "pass": true,
   "cite": "docs/families/nli.md:159-165",
   "note": "logit = entailment logit"
  },
  {
   "family": "nli",
   "model": "nli:modernbert-large",
   "device": "mlx",
   "device_detail": "Apple M4 Pro, macOS 27, MLX on Metal",
   "reference": "goldens-modernbert-large-zeroshot-v2.0.jsonl, gate: row scores and option logits within 1e-3, same decisions",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100 %",
   "max_logit_diff": 0.0007,
   "max_prob_diff": 0.000059,
   "date": null,
   "pass": true,
   "cite": "docs/families/nli.md:191-199",
   "note": "option-logit max 7.0e-4; row-score max 7.2e-4"
  },
  {
   "family": "nli",
   "model": "nli:modernbert-large",
   "device": "cpu",
   "device_detail": "Apple M4 Pro, macOS 27, ONNX Runtime CPU",
   "reference": "goldens-modernbert-large-zeroshot-v2.0.jsonl, same gate",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100 %",
   "max_logit_diff": 0.00041,
   "max_prob_diff": 0.000052,
   "date": null,
   "pass": true,
   "cite": "docs/families/nli.md:191-199"
  },
  {
   "family": "nli",
   "model": "nli:deberta-v3-large",
   "device": "cuda",
   "device_detail": "RTX 4090 with a CUDA 13 driver, CUDA 12 pack",
   "reference": "goldens (tolerances unchanged)",
   "kind": "runtime",
   "questions": null,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0000037,
   "date": "2026-09-27",
   "pass": true,
   "cite": "docs/distribution.md:123-126",
   "note": "3.7e-6 listed after 'prob max' in the same sentence; read as probability"
  },
  {
   "family": "qwen3guard",
   "model": "qwen3guard:0.6b",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 CPU, weightless graph",
   "reference": "transformers fp32 (CPU)",
   "kind": "export",
   "questions": 296,
   "decisions": "100 %",
   "max_logit_diff": 0.000036,
   "max_prob_diff": 0.0000092,
   "date": null,
   "pass": true,
   "cite": "docs/families/qwen3guard.md:115-122",
   "note": "candidate-logit max; prob max per question: safety 3.8e-6, unsafe 8.9e-7, unsafe_strict 3.8e-6, category 9.2e-6"
  },
  {
   "family": "qwen3guard",
   "model": "qwen3guard:0.6b",
   "device": "cpu",
   "device_detail": "24 cores; ONNX Runtime 1.28",
   "reference": "goldens-qwen3guard-gen-0.6b.jsonl (transformers fp32), option logits tolerance 1e-3",
   "kind": "runtime",
   "questions": 296,
   "decisions": "100 % (296 / 296)",
   "max_logit_diff": 0.000057,
   "max_prob_diff": 0.000014,
   "date": null,
   "pass": true,
   "cite": "docs/families/qwen3guard.md:157-163"
  },
  {
   "family": "qwen3guard",
   "model": "qwen3guard:0.6b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "goldens-qwen3guard-gen-0.6b.jsonl (transformers fp32), option logits tolerance 1e-3",
   "kind": "runtime",
   "questions": 296,
   "decisions": "100 % (296 / 296)",
   "max_logit_diff": 0.000078,
   "max_prob_diff": 0.0000091,
   "date": null,
   "pass": true,
   "cite": "docs/families/qwen3guard.md:157-163"
  },
  {
   "family": "schema-scorer",
   "model": "mobarmg/jev-schema-scorer-deberta-v3-large",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU EP, one padded batch per request",
   "reference": "upstream LocalSystemOne.score_pairs, fp32 on GPU, TF32 off",
   "kind": "export",
   "questions": 861,
   "decisions": "100.00%",
   "max_logit_diff": 0.000074,
   "max_prob_diff": 0.000011,
   "date": null,
   "pass": true,
   "cite": "docs/families/schema-scorer.md:106-111",
   "note": "22 questions rejected as upstream does; prob max per type: choice 6.9e-6, score 4.6e-6, noul 1.1e-5"
  },
  {
   "family": "von",
   "model": "von:1.1",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 CPU EP, one padded batch per request",
   "reference": "upstream's network in float64, one unpadded row at a time (ref.Exact)",
   "kind": "export",
   "questions": 883,
   "decisions": "100.00%",
   "max_logit_diff": 0.00035,
   "max_prob_diff": 0.000038,
   "date": null,
   "pass": true,
   "cite": "docs/families/von.md:301-306",
   "note": "row-logit max; prob max per type: choice 3.8e-5, score 8.9e-6, noul 6.1e-6"
  },
  {
   "family": "von",
   "model": "von:1.1",
   "device": "cpu",
   "device_detail": "x86-64, 24 cores; ONNX Runtime 1.28",
   "reference": "goldens-von.jsonl (network in float64), row and option logits within 1e-3",
   "kind": "runtime",
   "questions": 485,
   "decisions": "100 %",
   "max_logit_diff": 0.00044,
   "max_prob_diff": 0.000034,
   "date": null,
   "pass": true,
   "cite": "docs/families/von.md:329-334"
  },
  {
   "family": "von",
   "model": "von:1.1",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "goldens-von.jsonl (network in float64), row and option logits within 1e-3",
   "kind": "runtime",
   "questions": 485,
   "decisions": "100 %",
   "max_logit_diff": 0.00039,
   "max_prob_diff": 0.000047,
   "date": null,
   "pass": true,
   "cite": "docs/families/von.md:329-334"
  },
  {
   "family": "von",
   "model": "von:1.1",
   "device": "mlx",
   "device_detail": "Apple M4 Pro, macOS 27, MLX on Metal",
   "reference": "goldens-von.jsonl (network in float64), row and option logits within 1e-3",
   "kind": "runtime",
   "questions": 485,
   "decisions": "485/485",
   "max_logit_diff": 0.002,
   "max_prob_diff": 0.000031,
   "date": null,
   "pass": false,
   "cite": "docs/families/von.md:396-398",
   "note": "one row over the 1e-3 gate (2.0e-3); p99 row logits 3.3e-4; von stays on ONNX Runtime on Macs. Doc wording: 'same decision on all 485 questions'"
  },
  {
   "family": "von",
   "model": "von:1.1",
   "device": "cpu",
   "device_detail": "Apple silicon (M4 Pro), ONNX Runtime CPU",
   "reference": "goldens-von.jsonl (network in float64), row and option logits within 1e-3",
   "kind": "runtime",
   "questions": null,
   "decisions": "100 %",
   "max_logit_diff": 0.0011,
   "max_prob_diff": null,
   "date": null,
   "pass": false,
   "cite": "docs/families/von.md:404-405",
   "note": "over the 1e-3 gate on one row (p99 2.4e-4); also von.md:22"
  },
  {
   "family": "kev",
   "model": "kev:0.8b",
   "device": "cpu",
   "device_detail": "i9-13900K; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CUDA",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.000046,
   "max_prob_diff": 0.0000028,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:238",
   "note": "checkpoint: round 15; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "kev:0.8b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CUDA",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.000034,
   "max_prob_diff": 0.0000019,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:239",
   "note": "checkpoint: round 15; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "kev:4b",
   "device": "cpu",
   "device_detail": "i9-13900K; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CUDA",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.00023,
   "max_prob_diff": 0.000031,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:240",
   "note": "checkpoint: round 10; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "kev:4b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CUDA",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.00022,
   "max_prob_diff": 0.00003,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:241",
   "note": "checkpoint: round 10; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "kev:9b",
   "device": "cpu",
   "device_detail": "i9-13900K; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CPU",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.000036,
   "max_prob_diff": 0.0000038,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:242",
   "note": "checkpoint: 2629c06a; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "kev:9b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "upstream kev at the pinned commit in fp32 (Checkpoint.load(dtype=fp32, merge=True), row form, TF32 off); option logits within 1e-3; goldens' reference ran on CPU",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.000075,
   "max_prob_diff": 0.0000042,
   "date": null,
   "pass": true,
   "cite": "docs/families/kev.md:243",
   "note": "checkpoint: 2629c06a; summary of every run at kev.md:231-234"
  },
  {
   "family": "kev",
   "model": "jaredpalmer/kev-0.8b round 7 (54f4f877)",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 CPU",
   "reference": "upstream fp32",
   "kind": "export",
   "questions": null,
   "decisions": null,
   "max_logit_diff": null,
   "max_prob_diff": 0.0000019,
   "date": null,
   "pass": null,
   "cite": "docs/families/kev.md:201-202",
   "note": "131 requests; superseded checkpoint (Ollaya now pins round 15)"
  },
  {
   "family": "decider",
   "model": "decider:0.8b",
   "device": "cpu",
   "device_detail": "ONNX Runtime 1.30 CPU, weightless graph",
   "reference": "upstream fp32 on CUDA, TF32 off",
   "kind": "export",
   "questions": 632,
   "decisions": "100 %",
   "max_logit_diff": 0.000032,
   "max_prob_diff": 0.0000074,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider.md:246-254",
   "note": "logit max over all 255 labels; prob max per type: choice 3.5e-6, noul 7.4e-6, score 2.9e-6"
  },
  {
   "family": "decider",
   "model": "decider:2b",
   "device": "cpu",
   "device_detail": "ONNX Runtime CPU, 12 threads",
   "reference": "upstream fp32 on CUDA",
   "kind": "export",
   "questions": 632,
   "decisions": "100 % / 100 %",
   "max_logit_diff": 0.000034,
   "max_prob_diff": 0.0000059,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider.md:263-268",
   "note": "prob max per type: choice 5.5e-6, noul 4.0e-6, score 5.9e-6"
  },
  {
   "family": "decider",
   "model": "decider:4b",
   "device": "cpu",
   "device_detail": "i9-13900K; ONNX Runtime 1.28; weights_in_memory bf16",
   "reference": "goldens from upstream decider in fp32 on CUDA (TF32 off); label and option logits within 1e-3",
   "kind": "runtime",
   "questions": 479,
   "decisions": "100 %",
   "max_logit_diff": 0.00003,
   "max_prob_diff": 0.0000048,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider.md:298-303"
  },
  {
   "family": "decider",
   "model": "decider:4b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "goldens from upstream decider in fp32 on CUDA (TF32 off); label and option logits within 1e-3",
   "kind": "runtime",
   "questions": 479,
   "decisions": "100 %",
   "max_logit_diff": 0.000027,
   "max_prob_diff": 0.0000061,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider.md:298-303"
  },
  {
   "family": "decider",
   "model": "decider:2b",
   "device": "cuda",
   "device_detail": "RTX 4090, CUDA 13; ONNX Runtime 1.28",
   "reference": "goldens from upstream decider in fp32 on CUDA (TF32 off); label and option logits within 1e-3",
   "kind": "runtime",
   "questions": 479,
   "decisions": "100 %",
   "max_logit_diff": 0.000052,
   "max_prob_diff": 0.0000063,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider.md:298-303",
   "note": "re-run after exact arena growth: numbers unchanged (decider.md:305-306)"
  },
  {
   "family": "decider",
   "model": "decider:4b",
   "device": "cuda",
   "device_detail": "RTX 4090 with a CUDA 13 driver, CUDA 12 pack",
   "reference": "goldens (tolerances unchanged)",
   "kind": "runtime",
   "questions": null,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0000061,
   "date": "2026-09-27",
   "pass": true,
   "cite": "docs/distribution.md:123-127"
  },
  {
   "family": "decider-vision",
   "model": "decider:2b-vision",
   "device": "cpu",
   "device_detail": "CPU (machine not stated)",
   "reference": "upstream VisionDecisionModel in fp32 on synthetic PNGs plus the shared text cases (103 requests)",
   "kind": "runtime",
   "questions": 404,
   "decisions": "100 % of 404 questions",
   "max_logit_diff": 0.00011,
   "max_prob_diff": 0.000027,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider-vision.md:89-95"
  },
  {
   "family": "decider-vision",
   "model": "decider:2b-vision",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "upstream VisionDecisionModel in fp32 on synthetic PNGs plus the shared text cases (103 requests)",
   "kind": "runtime",
   "questions": 404,
   "decisions": "100 %",
   "max_logit_diff": 0.00016,
   "max_prob_diff": 0.000036,
   "date": null,
   "pass": true,
   "cite": "docs/families/decider-vision.md:89-95",
   "note": "question count stated in the CPU column; same goldens"
  },
  {
   "family": "decider-vision",
   "model": "decider:2b-vision",
   "device": "other",
   "device_detail": "ONNX Runtime (device not stated)",
   "reference": "upstream",
   "kind": "export",
   "questions": null,
   "decisions": null,
   "max_logit_diff": null,
   "max_prob_diff": null,
   "date": null,
   "pass": null,
   "cite": "docs/families/decider-vision.md:98",
   "note": "'matches upstream to 1.6e-5 (with an image) and 6.7e-6 (text only)'; the doc does not say whether logits or probabilities"
  },
  {
   "family": "clm",
   "model": "clm:8b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "goldens from families/clm/ref.py: the model in fp32 on the BF16 weights (upstream serves BF16 on vLLM)",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.00013,
   "max_prob_diff": 0.000029,
   "date": "2026-09-28",
   "pass": true,
   "cite": "docs/families/clm.md:55"
  },
  {
   "family": "clm",
   "model": "clm:8b",
   "device": "cpu",
   "device_detail": "24 cores",
   "reference": "goldens from families/clm/ref.py: the model in fp32 on the BF16 weights (upstream serves BF16 on vLLM)",
   "kind": "runtime",
   "questions": 480,
   "decisions": "100 %",
   "max_logit_diff": 0.00012,
   "max_prob_diff": 0.000021,
   "date": "2026-09-28",
   "pass": true,
   "cite": "docs/families/clm.md:56"
  },
  {
   "family": "jevk5",
   "model": "jevk5:4b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "stock llama-server of the pinned build (b11146) with the author's GGUF, one cold pass per question, bias-trick readout",
   "kind": "runtime",
   "questions": 593,
   "decisions": "593/593",
   "max_logit_diff": 0.0000077,
   "max_prob_diff": 0.0000016,
   "date": "2026-09-26",
   "pass": true,
   "cite": "docs/families/jevk5.md:105"
  },
  {
   "family": "jebadiah",
   "model": "jeb:4b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "stock llama-server of the pinned build (b11146) with the authors' file, one cold pass per question",
   "kind": "runtime",
   "questions": 494,
   "decisions": "494/494",
   "max_logit_diff": 0.0000076,
   "max_prob_diff": 0.0000021,
   "date": "2026-09-30",
   "pass": true,
   "cite": "docs/families/jebadiah.md:61"
  },
  {
   "family": "jebadiah",
   "model": "jeb:9b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "stock llama-server of the pinned build (b11146) with the authors' file, one cold pass per question",
   "kind": "runtime",
   "questions": 494,
   "decisions": "494/494",
   "max_logit_diff": 0.0000077,
   "max_prob_diff": 0.0000021,
   "date": "2026-09-30",
   "pass": true,
   "cite": "docs/families/jebadiah.md:62"
  },
  {
   "family": "jebadiah",
   "model": "jeb:27b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "stock llama-server of the pinned build (b11146) with the authors' file, one cold pass per question",
   "kind": "runtime",
   "questions": 494,
   "decisions": "494/494",
   "max_logit_diff": 0.0000077,
   "max_prob_diff": 0.0000024,
   "date": "2026-09-30",
   "pass": true,
   "cite": "docs/families/jebadiah.md:63"
  },
  {
   "family": "cygnet",
   "model": "cygnet:12b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "stock llama-server of the pinned build (b11146) with the GGUF, one cold pass per question",
   "kind": "runtime",
   "questions": 502,
   "decisions": "502/502",
   "max_logit_diff": 0.0000077,
   "max_prob_diff": 4.1e-7,
   "date": "2026-09-30",
   "pass": true,
   "cite": "docs/families/cygnet.md:59"
  },
  {
   "family": "nimble",
   "model": "nimble:9b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "the author's code (serving_schema.prepare_prompts, inference.candidate_logits, LoRA unmerged) in fp32, TF32 off",
   "kind": "runtime",
   "questions": 492,
   "decisions": "100 %",
   "max_logit_diff": 0.00011,
   "max_prob_diff": 0.0000065,
   "date": null,
   "pass": true,
   "cite": "docs/families/nimble.md:72"
  },
  {
   "family": "jeeves",
   "model": "jeeves:9b",
   "device": "cuda",
   "device_detail": "RTX 4090",
   "reference": "the authors' own model code (export.load_export) in fp32 on the CPU, with their Encoder",
   "kind": "runtime",
   "questions": 430,
   "decisions": "100 %",
   "max_logit_diff": 0.00019,
   "max_prob_diff": 0.000014,
   "date": "2026-09-30",
   "pass": true,
   "cite": "docs/families/jeeves.md:61",
   "note": "logit column is 'scores max' (pointer-head scores)"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "mlx",
   "device_detail": "Apple M4 Pro (16-core GPU), macOS 27, MLX on Metal",
   "reference": "laya goldens, 97 cases; gate: ids and markers identical, same decisions, laya's answers within rounding",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.000025,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0001-mlx-engine.md:148",
   "note": "0/483 answers off"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cpu",
   "device_detail": "Apple M4 Pro, macOS 27, ONNX Runtime CPU",
   "reference": "laya goldens, 97 cases; gate: ids and markers identical, same decisions, laya's answers within rounding",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00019,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0001-mlx-engine.md:148",
   "note": "0/483 answers off"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "mlx",
   "device_detail": "Apple M4 Pro (16-core GPU), macOS 27, MLX on Metal",
   "reference": "laya goldens-all; gate: ids and markers identical, same decisions, laya's answers within rounding",
   "kind": "runtime",
   "questions": 2383,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00014,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0001-mlx-engine.md:149",
   "note": "p99 8.5e-6; max is one question (preset/guard/email_dict jailbreak)"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cpu",
   "device_detail": "Apple M4 Pro, macOS 27, ONNX Runtime CPU",
   "reference": "laya goldens-all; gate: ids and markers identical, same decisions, laya's answers within rounding",
   "kind": "runtime",
   "questions": 2383,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00008,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0001-mlx-engine.md:149"
  },
  {
   "family": "laya",
   "model": "laya:multilingual",
   "device": "mlx",
   "device_detail": "Apple M4 Pro (16-core GPU), macOS 27, MLX on Metal",
   "reference": "laya goldens-all; gate: ids and markers identical, same decisions, laya's answers within rounding",
   "kind": "runtime",
   "questions": 2383,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00001,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0001-mlx-engine.md:150",
   "note": "p99 6.2e-6"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "other",
   "device_detail": "ONNX Runtime WebGPU provider on a Mac (rejected)",
   "reference": "laya goldens",
   "kind": "runtime",
   "questions": 483,
   "decisions": "1 of 483 differ",
   "max_logit_diff": null,
   "max_prob_diff": 0.13,
   "date": null,
   "pass": false,
   "cite": "docs/decisions/0001-mlx-engine.md:13-14",
   "note": "'probabilities up to 0.13 apart'; doc says the WebGPU provider fails parity"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "other",
   "device_detail": "Python ONNX export (device not stated)",
   "reference": "PyTorch",
   "kind": "export",
   "questions": null,
   "decisions": null,
   "max_logit_diff": null,
   "max_prob_diff": 0.00011,
   "date": null,
   "pass": null,
   "cite": "docs/decisions/0001-mlx-engine.md:156",
   "note": "the FAQ's figure, quoted in the ADR"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cuda",
   "device_detail": "RTX 4090 (sm_89), Microsoft ORT 1.28.2 cuda13 + CUDA 13.x/cuDNN 9 libs, fp32 graph",
   "reference": "goldens-all, tolerances unchanged",
   "kind": "runtime",
   "questions": null,
   "decisions": "100.00%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0001,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0004-cuda-onnxruntime-builds.md:73",
   "note": "477 cases"
  },
  {
   "family": "laya",
   "model": "laya:multilingual",
   "device": "cuda",
   "device_detail": "RTX 4090 (sm_89), Microsoft ORT 1.28.2 cuda13, fp32 graph",
   "reference": "goldens-all, tolerances unchanged",
   "kind": "runtime",
   "questions": null,
   "decisions": "100.00%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0000065,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0004-cuda-onnxruntime-builds.md:74",
   "note": "477 cases"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cuda",
   "device_detail": "RTX 4090 (sm_89), Microsoft ORT 1.28.2 cuda12 + CUDA 12.8 wheels, cuDNN 9.26, fp32 graph",
   "reference": "goldens-all, tolerances unchanged",
   "kind": "runtime",
   "questions": null,
   "decisions": "100.00%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0001,
   "date": "2026-09-27",
   "pass": true,
   "cite": "docs/decisions/0004-cuda-onnxruntime-builds.md:75",
   "note": "477 cases; also docs/distribution.md:123-125 (dated 2026-09-27; the opset-23 graph 3.7e-5)"
  },
  {
   "family": "laya",
   "model": "laya:multilingual",
   "device": "cuda",
   "device_detail": "RTX 4090 (sm_89), Microsoft ORT 1.28.2 cuda12, fp32 graph",
   "reference": "goldens-all, tolerances unchanged",
   "kind": "runtime",
   "questions": null,
   "decisions": "100.00%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0000065,
   "date": null,
   "pass": true,
   "cite": "docs/decisions/0004-cuda-onnxruntime-builds.md:76",
   "note": "477 cases"
  },
  {
   "family": "laya",
   "model": "laya:en-fp16",
   "device": "cuda",
   "device_detail": "RTX 4090 with a CUDA 13 driver, CUDA 12 pack",
   "reference": "fp32 reference goldens",
   "kind": "runtime",
   "questions": null,
   "decisions": "99.62%",
   "max_logit_diff": null,
   "max_prob_diff": null,
   "date": "2026-09-27",
   "pass": null,
   "cite": "docs/distribution.md:124-126",
   "note": "fp16 differs from the fp32 reference on near-ties; doc reports it as expected, not as a failure"
  },
  {
   "family": "laya",
   "model": "laya:multilingual-fp16",
   "device": "cuda",
   "device_detail": "RTX 4090 with a CUDA 13 driver, CUDA 12 pack",
   "reference": "fp32 reference goldens",
   "kind": "runtime",
   "questions": null,
   "decisions": "99.12%",
   "max_logit_diff": null,
   "max_prob_diff": null,
   "date": "2026-09-27",
   "pass": null,
   "cite": "docs/distribution.md:125-126",
   "note": "fp16 near-tie differences, same as with CUDA 13"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cuda",
   "device_detail": "RTX 4090 under WSL 2, installed cuda_v13 pack (install.sh layout)",
   "reference": "laya goldens, 97 cases",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00022,
   "date": null,
   "pass": true,
   "cite": "docs/distribution.md:377"
  },
  {
   "family": "laya",
   "model": "laya:multilingual",
   "device": "cuda",
   "device_detail": "RTX 4090 under WSL 2, installed cuda_v13 pack (install.sh layout)",
   "reference": "laya goldens-all, 477 cases",
   "kind": "runtime",
   "questions": 2383,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.0000065,
   "date": null,
   "pass": true,
   "cite": "docs/distribution.md:377"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cuda",
   "device_detail": "RTX 4090, Windows 11 (driver 616.92, CUDA UMD 13.4), release.yml dry-run archives",
   "reference": "laya goldens, 97 cases",
   "kind": "runtime",
   "questions": 483,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.00022,
   "date": null,
   "pass": true,
   "cite": "docs/distribution.md:391",
   "note": "0 answers outside rounding"
  },
  {
   "family": "laya",
   "model": "laya:en-fp16",
   "device": "cuda",
   "device_detail": "RTX 4090, Windows 11, release.yml dry-run archives",
   "reference": "fp32 reference goldens",
   "kind": "runtime",
   "questions": 483,
   "decisions": "1 of 483 differs (99.79%)",
   "max_logit_diff": null,
   "max_prob_diff": 0.04,
   "date": null,
   "pass": null,
   "cite": "docs/distribution.md:392",
   "note": "fp16 near-tie difference; doc reports it as expected (docs/api.md:413-415: 99.1-99.6% agreement)"
  },
  {
   "family": "laya",
   "model": "laya:en",
   "device": "cpu",
   "device_detail": "Windows 11 (the new Windows build)",
   "reference": "laya goldens",
   "kind": "runtime",
   "questions": null,
   "decisions": "100%",
   "max_logit_diff": null,
   "max_prob_diff": 0.000074,
   "date": null,
   "pass": true,
   "cite": "docs/distribution.md:393"
  }
 ],
 "not_run": [
  {
   "family": "winnow",
   "what": "Metal and the Apple CPU, linux-arm64, and 12b on the CPU",
   "cite": "docs/families/winnow.md:136"
  },
  {
   "family": "jevk5",
   "what": "Metal, Windows, and a comparison with the author's transformers runtime (bf16); CPU parity was queued",
   "cite": "docs/families/jevk5.md:113-114"
  },
  {
   "family": "decision",
   "what": "Decision-1.0-Lux-9B: goldens did not finish (GPU overflow, rerun queued); no manifest, no library entry",
   "cite": "docs/families/decision.md:265-267"
  },
  {
   "family": "gliclass",
   "what": "gliclass:large (instruct-large) on Metal/MLX: stays on ONNX Runtime until the DeBERTa backbone (phase 3)",
   "cite": "docs/families/gliclass.md:184-185"
  },
  {
   "family": "nli",
   "what": "nli:deberta-v3-large on Metal/MLX: stays on ONNX Runtime until the DeBERTa backbone (phase 3)",
   "cite": "docs/families/nli.md:184-185"
  },
  {
   "family": "qwen3guard/kev/decider",
   "what": "graphs re-exported with OLLAYA_FOLD_LIMIT=64 have not been through full parity",
   "cite": "docs/families/qwen3guard.md:134-137"
  },
  {
   "family": "von",
   "what": "von 1.2 (von-option-marker-v2): needs a fresh export and parity run",
   "cite": "docs/families/von.md:458-460"
  },
  {
   "family": "laya",
   "what": "ONNX Runtime Core ML provider on a Mac: fails to initialise in MLProgram form; slower with wrong outputs in NeuralNetwork form (no numbers given)",
   "cite": "docs/decisions/0001-mlx-engine.md:11-13"
  },
  {
   "family": "all ONNX (CUDA)",
   "what": "parity on GPUs other than sm_89 (RTX 4090); RTX 50-series only confirmed to run by a reporter",
   "cite": "docs/decisions/0004-cuda-onnxruntime-builds.md:118-120"
  }
 ],
 "other_checks": [
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "kind": "reference-vs-author-server",
   "device": null,
   "what": "ref.py/llama-server reference vs winnow-inference's own server, reference mode",
   "questions": 503,
   "decisions": "all 503 agree",
   "max_prob_diff": 0.0028,
   "p99_prob_diff": 0.000003,
   "cite": "docs/families/winnow.md:130-132"
  },
  {
   "family": "winnow",
   "model": "winnow:e4b",
   "kind": "reference-vs-author-server",
   "device": null,
   "what": "same, author's server in its default mode",
   "questions": 503,
   "decisions": "501 of 503 agree",
   "max_prob_diff": 0.052,
   "p99_prob_diff": 0.03,
   "cite": "docs/families/winnow.md:130-132"
  },
  {
   "family": "llm-logits",
   "model": "Qwen3-4B-Instruct-2507 Q8_0",
   "kind": "reference-readout",
   "device": "cuda",
   "what": "bias-trick readout vs unbiased full-vocabulary distribution, 200 prompts (llama.cpp b11149 CUDA build)",
   "max_logprob_diff": 0.000016,
   "max_prob_diff": 0.0000019,
   "cite": "docs/families/llm-logits.md:128-130"
  },
  {
   "family": "llm-logits",
   "model": "Gemma 4 E2B-it Q8_0",
   "kind": "reference-readout",
   "device": "cuda",
   "what": "bias-trick readout vs unbiased full-vocabulary distribution, 200 prompts (llama.cpp b11149 CUDA build)",
   "max_logprob_diff": 0.0000078,
   "max_prob_diff": 0.0000015,
   "cite": "docs/families/llm-logits.md:128-130"
  },
  {
   "family": "qwen3guard",
   "model": "qwen3guard:0.6b",
   "kind": "cuda-vs-cpu",
   "device": "cuda",
   "what": "ORT 1.30 CUDA EP with enable_mem_reuse=false vs CPU",
   "max_diff": 0.00001,
   "cite": "docs/families/qwen3guard.md:133"
  },
  {
   "family": "qwen3guard",
   "model": "qwen3guard:0.6b",
   "kind": "cuda-vs-cpu",
   "device": "cuda",
   "what": "OLLAYA_FOLD_LIMIT=64 re-export, CUDA EP default options vs CPU",
   "max_diff": 0.000013,
   "cite": "docs/families/qwen3guard.md:134-136"
  }
 ]
}
