{
 "meta": {
  "date_utc": "2026-09-24T07:42:28Z",
  "git_sha": "8028a10",
  "host": "m3-max-metal",
  "profile": "release",
  "laya_feature": true,
  "laya_max_questions": 0,
  "corpus_cap_mode": "registry defaults",
  "corpus_protocol": "one engine domain per label; corpus = train-split docs of that label, capped per label (registry), self-doc fallback for labels absent from the fetched train rows",
  "calibration_protocol": "sigmoid-gate fit on a train-tail calibration slice (same builder over the train rows); fused-gate thresholds fitted per suite at the cal-slice 30th percentile (the T1.6 arena posture rho=30%; the birth constants do not transfer \u2014 measured: 100% abstain on several suites at defaults); raw-vs-calibrated readout ECE compared per the G1 gate; no calibration claim when the calibrator never moved",
  "floor_definition": "conformal-naive floor = split-conformal recalibration of the raw readout confidence, c'(s) = (1 + #{cal s_i <= s}) / (n_cal + 1) \u2014 the exchangeability-valid baseline (G1 fails unless the calibrated ECE beats it)",
  "determinism_scoping": "the bit-identity claim is the MODELLESS lane's (plan caveat 3); the laya lane gets the same observed repeat check and it is reported, not claimed",
  "laya_device": "metal (LAYA_DEVICE or the build's macOS default)",
  "laya_python_lane": "on \u2014 the ORIGINAL torch reference as a JSONL subprocess oracle: same cases, the reference's own rounded-4 probabilities, latency = subprocess round-trip (IPC included)",
  "box_state": {
   "start": {
    "power": "AC Power",
    "power_mode": "high",
    "load_1m": 6.31,
    "swap_used_mb": 2563.94,
    "latency_quotable": false,
    "refusals": [
     "load 6.31 > 6 \u2014 a sibling job is on the box"
    ]
   },
   "end": {
    "power": "AC Power",
    "power_mode": "high",
    "load_1m": 6.97,
    "swap_used_mb": 2563.94,
    "latency_quotable": false,
    "refusals": [
     "load 6.97 > 6 \u2014 a sibling job is on the box"
    ]
   }
  },
  "divergences": [
   "massive option sampling: SplitMix64 (fixed seed), not CPython MT19937 \u2014 both lanes see byte-identical questions, which is the integrity that matters",
   "banking77: mteb/banking77 mirror (the reference's own bench_apps variant); PolyAI/banking77 is script-based and unservable",
   "engine context = state + prompt (wire criteria None) \u2014 the modelless serving path"
  ],
  "hosts": [
   {
    "host": "m3-max-metal",
    "date_utc": "2026-09-24T07:42:28Z",
    "git_sha": "8028a10",
    "profile": "release",
    "laya_feature": true,
    "laya_device": "metal (LAYA_DEVICE or the build's macOS default)",
    "laya_python_lane": "on \u2014 the ORIGINAL torch reference as a JSONL subprocess oracle: same cases, the reference's own rounded-4 probabilities, latency = subprocess round-trip (IPC included)",
    "laya_max_questions": 0,
    "corpus_cap_mode": "registry defaults",
    "lane_sources": {
     "modelless": {
      "git_sha": "d009604",
      "date_utc": "2026-10-03T08:35:02Z"
     },
     "laya:english": {
      "git_sha": "8ca8770",
      "date_utc": "2026-10-02T12:01:18Z"
     },
     "laya:multilingual": {
      "git_sha": "5ada17a",
      "date_utc": "2026-09-27T08:00:55Z"
     },
     "laya:py/english": {
      "git_sha": "8ca8770",
      "date_utc": "2026-10-02T12:01:18Z"
     },
     "laya:py/multilingual": {
      "git_sha": "5ada17a",
      "date_utc": "2026-09-27T08:00:55Z"
     },
     "laya:py/typed": {
      "git_sha": "5ada17a",
      "date_utc": "2026-09-27T08:00:55Z"
     },
     "laya:typed": {
      "git_sha": "5ada17a",
      "date_utc": "2026-09-27T08:00:55Z"
     },
     "hybrid": {
      "git_sha": "b6425bf",
      "date_utc": "2026-10-03T09:03:51Z"
     },
     "paw": {
      "git_sha": "9dc33da",
      "date_utc": "2026-09-29T02:12:55Z"
     },
     "openthai": {
      "git_sha": "00bddd1",
      "date_utc": "2026-09-28T16:27:36Z"
     },
     "encoder": {
      "git_sha": "e0440e3",
      "date_utc": "2026-10-01T19:20:00Z"
     },
     "bekko": {
      "git_sha": "8ca8770",
      "date_utc": "2026-10-02T12:08:12Z"
     }
    },
    "head_posture": "OFF (head_scale 0 \u2014 the published baseline posture)"
   },
   {
    "host": "m3-max-ane",
    "date_utc": "2026-09-24T07:16:02Z",
    "git_sha": "afacc3a",
    "profile": "release",
    "laya_feature": true,
    "laya_device": "ane (LAYA_DEVICE=ane \u2192 load_ane; whole-graph CoreML encoder, Plan 002)",
    "laya_python_lane": "off (pass --laya-python to add the reference lane)",
    "laya_max_questions": 0,
    "corpus_cap_mode": "registry defaults",
    "lane_sources": {
     "laya:english": {
      "git_sha": "2f6b58c",
      "date_utc": "2026-09-30T10:35:35Z"
     }
    }
   },
   {
    "host": "4090-win",
    "date_utc": "2026-09-24T07:11:03Z",
    "git_sha": "afacc3a",
    "profile": "release",
    "laya_feature": true,
    "laya_device": "cuda (LAYA_DEVICE or the build's non-macOS CUDA default)",
    "laya_python_lane": "off (pass --laya-python to add the reference lane)",
    "laya_max_questions": 0,
    "corpus_cap_mode": "registry defaults",
    "lane_sources": {
     "modelless": {
      "git_sha": "3cad008",
      "date_utc": "2026-10-02T20:23:08Z"
     },
     "laya:english": {
      "git_sha": "3cad008",
      "date_utc": "2026-10-02T20:23:08Z"
     },
     "laya:multilingual": {
      "git_sha": "97c2b1f",
      "date_utc": "2026-09-28T07:13:04Z"
     },
     "laya:typed": {
      "git_sha": "97c2b1f",
      "date_utc": "2026-09-28T07:13:04Z"
     },
     "clm": {
      "git_sha": "967415c",
      "date_utc": "2026-09-24T21:57:22Z"
     },
     "gliner": {
      "git_sha": "d69f0c7",
      "date_utc": "2026-09-25T01:53:35Z"
     },
     "agentjev": {
      "git_sha": "a2c03aa",
      "date_utc": "2026-09-25T03:09:08Z"
     },
     "paw_local": {
      "git_sha": "634093f",
      "date_utc": "2026-09-26T20:05:28Z"
     },
     "openthai": {
      "git_sha": "bee92d3",
      "date_utc": "2026-09-28T14:02:42Z"
     }
    },
    "head_posture": "OFF (head_scale 0 \u2014 the published baseline posture)"
   }
  ],
  "head_posture": "OFF (head_scale 0 \u2014 the published baseline posture)",
  "edition": "2026-10"
 },
 "suites": [
  {
   "name": "typed_decisions",
   "n_cases": 400,
   "n_questions": 2000,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 2000,
     "accuracy": 0.5725,
     "macro_f1": 0.5428588466644556,
     "ece": 0.09930298611894249,
     "brier": 0.5634444755044175,
     "nll": 0.9993372554767699,
     "aurc": 0.2967333943470369,
     "mean_confidence": 0.4731970138810575,
     "acc_at_50_coverage": 0.676,
     "acc_at_80_coverage": 0.613125
    },
    "readout_ece": 0.04959756752848618,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.717,
     "selective_accuracy": 0.6537102473498233,
     "selective_n": 566
    },
    "readout_ece_raw": 0.48374495857954025,
    "readout_ece_calibrated": 0.04959756752848618,
    "floor_ece": 0.18175548902195604,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": {
     "choice": {
      "n": 600,
      "accuracy": 0.5483333333333333,
      "macro_f1": 0.432556704787651,
      "ece": 0.14819766720136004,
      "brier": 0.6304564691715672,
      "nll": 1.1799754866711054,
      "aurc": 0.3777896524930003,
      "mean_confidence": 0.40013566613197327,
      "acc_at_50_coverage": 0.59,
      "acc_at_80_coverage": 0.5291666666666667
     },
     "noul": {
      "n": 600,
      "accuracy": 0.73,
      "macro_f1": 0.7272727272727273,
      "ece": 0.06820527624338865,
      "brier": 0.3752274001772096,
      "nll": 0.5569624129029627,
      "aurc": 0.16418833870677263,
      "mean_confidence": 0.6617947237566113,
      "acc_at_50_coverage": 0.8266666666666667,
      "acc_at_80_coverage": 0.7770833333333333
     },
     "score": {
      "n": 800,
      "accuracy": 0.4725,
      "macro_f1": 0.37283935453238826,
      "ece": 0.09739159643650055,
      "brier": 0.6543482867494519,
      "nll": 1.1956397140113715,
      "aurc": 0.4607954011325077,
      "mean_confidence": 0.3865447422862053,
      "acc_at_50_coverage": 0.5275,
      "acc_at_80_coverage": 0.51875
     }
    },
    "soft_acc": 0.3826607784062584,
    "brier_soft": 0.16762865443923508,
    "score_mae": 0.5717842324853881,
    "within_1": 0.81625,
    "latency_p50_ms": 0.782,
    "latency_p99_ms": 1.69,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 0.22,
     "max_ms": 1.701,
     "argmax_case": 150
    },
    "determinism_ok": true,
    "seconds": 6.8922451670000004,
    "n_cases": 400,
    "n_questions": 2000,
    "score_threshold": 0.57852864,
    "distance_threshold": 0.91222644,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.57852864,
      "status": "fitted",
      "targetAccuracy": 0.45714286,
      "support": {
       "n": 100,
       "nPass": 70,
       "nAbstain": 30
      }
     },
     "distance": {
      "threshold": 0.91222644,
      "status": "fitted",
      "targetAccuracy": 0.5285714,
      "support": {
       "n": 100,
       "nPass": 70,
       "nAbstain": 30
      }
     }
    },
    "corpus_cap": {
     "effective": 48,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 0.0,
     "selected_alpha": "off",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      }
     ]
    },
    "oc_selection": {
     "selected_scale": 4.0,
     "candidates": [
      {
       "scale": 0.0,
       "cal_acc": 0.36
      },
      {
       "scale": 0.25,
       "cal_acc": 0.452
      },
      {
       "scale": 0.5,
       "cal_acc": 0.496
      },
      {
       "scale": 1.0,
       "cal_acc": 0.536
      },
      {
       "scale": 2.0,
       "cal_acc": 0.54
      },
      {
       "scale": 4.0,
       "cal_acc": 0.544
      },
      {
       "scale": 8.0,
       "cal_acc": 0.542
      }
     ]
    },
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.544
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.542
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.554
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.516
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.508
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.466
      }
     ]
    },
    "genome_selection": null,
    "transductive": null,
    "confusion": [
     {
      "gold": "Elevated; should be handled within the same week.",
      "pred": "Critical; requires action within the same day.",
      "count": 63,
      "share_of_errors": 0.09090909090909091
     },
     {
      "gold": "success",
      "pred": "partial",
      "count": 50,
      "share_of_errors": 0.07215007215007214
     },
     {
      "gold": "No time pressure; can wait indefinitely.",
      "pred": "Routine; handle within the normal queue.",
      "count": 36,
      "share_of_errors": 0.05194805194805195
     },
     {
      "gold": "Material: a large or unexplained difference.",
      "pred": "None: everything reconciles.",
      "count": 35,
      "share_of_errors": 0.050505050505050504
     },
     {
      "gold": "Elevated; should be handled within the same week.",
      "pred": "Routine; handle within the normal queue.",
      "count": 34,
      "share_of_errors": 0.049062049062049064
     },
     {
      "gold": "Benign: read-only or clearly safe actions.",
      "pred": "Low: routine writes within scope.",
      "count": 33,
      "share_of_errors": 0.047619047619047616
     },
     {
      "gold": "Moderate: access to internal systems or non-public data.",
      "pred": "High: access to production, secrets or customer data.",
      "count": 30,
      "share_of_errors": 0.04329004329004329
     },
     {
      "gold": "approve",
      "pred": "manual_review",
      "count": 27,
      "share_of_errors": 0.03896103896103896
     },
     {
      "gold": "Routine; handle within the normal queue.",
      "pred": "Critical; requires action within the same day.",
      "count": 25,
      "share_of_errors": 0.03607503607503607
     },
     {
      "gold": "request_information",
      "pred": "escalate_to_human",
      "count": 24,
      "share_of_errors": 0.03463203463203463
     },
     {
      "gold": "Mild frustration, but the relationship is intact.",
      "pred": "Clearly unhappy; repeat problems or explicit complaints.",
      "count": 23,
      "share_of_errors": 0.03318903318903319
     },
     {
      "gold": "observe",
      "pred": "human_review",
      "count": 22,
      "share_of_errors": 0.031746031746031744
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.482341635107994
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.1155914064347744
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.482341635107994
      }
     ]
    },
    "corpus_digest": "fnv1a64-2fe9ec3132dbc924",
    "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "19cb043",
     "date_utc": "2026-10-01T08:34:03Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 2000,
      "accuracy": 0.3575,
      "macro_f1": 0.32620639534238516,
      "ece": 0.2112706999999999,
      "brier": 0.7684491694499974,
      "nll": 1.3593418376936985,
      "aurc": 0.5626661372266888,
      "mean_confidence": 0.5687707000000009,
      "acc_at_50_coverage": 0.429,
      "acc_at_80_coverage": 0.38125
     },
     "readout_ece": 0.19182040000000003,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.28833333333333333,
       "macro_f1": 0.2958479995436556,
       "ece": 0.16805866666666666,
       "brier": 0.8079409903499993,
       "nll": 1.5603174844492513,
       "aurc": 0.6171637396940312,
       "mean_confidence": 0.4527049999999996,
       "acc_at_50_coverage": 0.38666666666666666,
       "acc_at_80_coverage": 0.325
      },
      "noul": {
       "n": 600,
       "accuracy": 0.47333333333333333,
       "macro_f1": 0.46994095544820186,
       "ece": 0.2808480000000001,
       "brier": 0.6693728044000007,
       "nll": 0.9693957725723233,
       "aurc": 0.4926269983870734,
       "mean_confidence": 0.7541813333333328,
       "acc_at_50_coverage": 0.4866666666666667,
       "acc_at_80_coverage": 0.48333333333333334
      },
      "score": {
       "n": 800,
       "accuracy": 0.3225,
       "macro_f1": 0.24129818605373515,
       "ece": 0.2064690000000001,
       "brier": 0.8131375775625017,
       "nll": 1.5010696514680573,
       "aurc": 0.6560603649790228,
       "mean_confidence": 0.5167620000000003,
       "acc_at_50_coverage": 0.335,
       "acc_at_80_coverage": 0.3125
      }
     },
     "soft_acc": 0.33458230168279346,
     "brier_soft": 0.3348088366904328,
     "score_mae": 0.6937235449999997,
     "within_1": 0.7525,
     "latency_p50_ms": 195.0,
     "latency_p99_ms": 358.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 92.0,
      "max_ms": 401.0,
      "argmax_case": 120
     },
     "determinism_ok": true,
     "seconds": 81.877849,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    },
    "multilingual": {
     "lane": "laya (rust)",
     "model": "multilingual",
     "hard": {
      "n": 2000,
      "accuracy": 0.349,
      "macro_f1": 0.3498994829046305,
      "ece": 0.32280229999999993,
      "brier": 0.9056679379849982,
      "nll": 1.9436809234449117,
      "aurc": 0.554820549344135,
      "mean_confidence": 0.6687793000000012,
      "acc_at_50_coverage": 0.404,
      "acc_at_80_coverage": 0.36375
     },
     "readout_ece": 0.25666635,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.295,
       "macro_f1": 0.3242803546952691,
       "ece": 0.362346,
       "brier": 0.9884960357500008,
       "nll": 2.5096392870563395,
       "aurc": 0.5968789601913673,
       "mean_confidence": 0.6559423333333334,
       "acc_at_50_coverage": 0.31666666666666665,
       "acc_at_80_coverage": 0.28541666666666665
      },
      "noul": {
       "n": 600,
       "accuracy": 0.4866666666666667,
       "macro_f1": 0.4534096824570536,
       "ece": 0.40840199999999977,
       "brier": 0.874118264733335,
       "nll": 1.6946655329027263,
       "aurc": 0.4613874178933604,
       "mean_confidence": 0.8919399999999996,
       "acc_at_50_coverage": 0.45,
       "acc_at_80_coverage": 0.48333333333333334
      },
      "score": {
       "n": 800,
       "accuracy": 0.28625,
       "macro_f1": 0.2554689567204614,
       "ece": 0.23129124999999998,
       "brier": 0.8672091196000032,
       "nll": 1.7059736936429828,
       "aurc": 0.7179156295538103,
       "mean_confidence": 0.5110365000000006,
       "acc_at_50_coverage": 0.26,
       "acc_at_80_coverage": 0.2765625
      }
     },
     "soft_acc": 0.3271328198665314,
     "brier_soft": 0.4737295325604735,
     "score_mae": 0.7603941325000002,
     "within_1": 0.7,
     "latency_p50_ms": 92.0,
     "latency_p99_ms": 163.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 40.0,
      "max_ms": 174.0,
      "argmax_case": 151
     },
     "determinism_ok": true,
     "seconds": 38.118442625,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    },
    "typed": {
     "lane": "laya (rust)",
     "model": "typed",
     "hard": {
      "n": 2000,
      "accuracy": 0.7445,
      "macro_f1": 0.7349433525103708,
      "ece": 0.1923918,
      "brier": 0.40988110527999805,
      "nll": 0.7184640377885018,
      "aurc": 0.12634017215900445,
      "mean_confidence": 0.5521081999999994,
      "acc_at_50_coverage": 0.859,
      "acc_at_80_coverage": 0.793125
     },
     "readout_ece": 0.3952849499999999,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.7333333333333333,
       "macro_f1": 0.6943336435123875,
       "ece": 0.25484566666666664,
       "brier": 0.46509025648333213,
       "nll": 0.8649966941522675,
       "aurc": 0.11299371741797679,
       "mean_confidence": 0.47848766666666676,
       "acc_at_50_coverage": 0.8966666666666666,
       "acc_at_80_coverage": 0.8
      },
      "noul": {
       "n": 600,
       "accuracy": 0.785,
       "macro_f1": 0.7833092099185031,
       "ece": 0.12942600000000015,
       "brier": 0.3215147912666666,
       "nll": 0.4946194012650314,
       "aurc": 0.08280543014765192,
       "mean_confidence": 0.6624203333333335,
       "acc_at_50_coverage": 0.93,
       "acc_at_80_coverage": 0.8541666666666666
      },
      "score": {
       "n": 800,
       "accuracy": 0.7225,
       "macro_f1": 0.7150339821484726,
       "ece": 0.19867975000000002,
       "brier": 0.4347489773875002,
       "nll": 0.7764480229082796,
       "aurc": 0.1370651831938734,
       "mean_confidence": 0.5245894999999998,
       "acc_at_50_coverage": 0.87,
       "acc_at_80_coverage": 0.7671875
      }
     },
     "soft_acc": 0.4668227307218911,
     "brier_soft": 0.06772113657369036,
     "score_mae": 0.24240826250000005,
     "within_1": 0.995,
     "latency_p50_ms": 286.0,
     "latency_p99_ms": 484.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 93.0,
      "max_ms": 532.0,
      "argmax_case": 150
     },
     "determinism_ok": true,
     "seconds": 109.645121,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 2000,
      "accuracy": 0.3575,
      "macro_f1": 0.32620639534238516,
      "ece": 0.2112708999999999,
      "brier": 0.7684491959949972,
      "nll": 1.3593406048220138,
      "aurc": 0.5626626389078578,
      "mean_confidence": 0.5687709000000007,
      "acc_at_50_coverage": 0.429,
      "acc_at_80_coverage": 0.38125
     },
     "readout_ece": 0.19182075000000007,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.28833333333333333,
       "macro_f1": 0.2958479995436556,
       "ece": 0.16805883333333335,
       "brier": 0.8079407131999993,
       "nll": 1.5603165852368217,
       "aurc": 0.6171637396940312,
       "mean_confidence": 0.4527051666666662,
       "acc_at_50_coverage": 0.38666666666666666,
       "acc_at_80_coverage": 0.325
      },
      "noul": {
       "n": 600,
       "accuracy": 0.47333333333333333,
       "macro_f1": 0.46994095544820186,
       "ece": 0.28084866666666675,
       "brier": 0.6693738674000009,
       "nll": 0.9693971360440156,
       "aurc": 0.49261854857569193,
       "mean_confidence": 0.7541819999999997,
       "acc_at_50_coverage": 0.4866666666666667,
       "acc_at_80_coverage": 0.48333333333333334
      },
      "score": {
       "n": 800,
       "accuracy": 0.3225,
       "macro_f1": 0.24129818605373515,
       "ece": 0.20646887500000002,
       "brier": 0.8131370545375017,
       "nll": 1.5010662210943988,
       "aurc": 0.6560603649790228,
       "mean_confidence": 0.5167618750000003,
       "acc_at_50_coverage": 0.335,
       "acc_at_80_coverage": 0.3125
      }
     },
     "soft_acc": 0.3345823158039128,
     "brier_soft": 0.334808769875873,
     "score_mae": 0.6937230449999998,
     "within_1": 0.7525,
     "latency_p50_ms": 328.0,
     "latency_p99_ms": 557.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 104.0,
      "max_ms": 693.0,
      "argmax_case": 183
     },
     "determinism_ok": true,
     "seconds": 141.410336666,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    },
    "py/multilingual": {
     "lane": "laya (python)",
     "model": "multilingual",
     "hard": {
      "n": 2000,
      "accuracy": 0.349,
      "macro_f1": 0.3498994829046305,
      "ece": 0.3228019999999999,
      "brier": 0.9056681466699985,
      "nll": 1.9436844062731538,
      "aurc": 0.5548199956342791,
      "mean_confidence": 0.6687790000000011,
      "acc_at_50_coverage": 0.404,
      "acc_at_80_coverage": 0.36375
     },
     "readout_ece": 0.2566663,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.295,
       "macro_f1": 0.3242803546952691,
       "ece": 0.3623458333333333,
       "brier": 0.9884961356000006,
       "nll": 2.509649991474549,
       "aurc": 0.5968789601913673,
       "mean_confidence": 0.6559421666666665,
       "acc_at_50_coverage": 0.31666666666666665,
       "acc_at_80_coverage": 0.28541666666666665
      },
      "noul": {
       "n": 600,
       "accuracy": 0.4866666666666667,
       "macro_f1": 0.4534096824570536,
       "ece": 0.40840183333333313,
       "brier": 0.8741184182333351,
       "nll": 1.6946657494232418,
       "aurc": 0.4613874178933604,
       "mean_confidence": 0.8919398333333328,
       "acc_at_50_coverage": 0.45,
       "acc_at_80_coverage": 0.48333333333333334
      },
      "score": {
       "n": 800,
       "accuracy": 0.28625,
       "macro_f1": 0.2554689567204614,
       "ece": 0.23129074999999996,
       "brier": 0.8672094513000029,
       "nll": 1.7059742100095436,
       "aurc": 0.7179156295538103,
       "mean_confidence": 0.5110360000000005,
       "acc_at_50_coverage": 0.26,
       "acc_at_80_coverage": 0.2765625
      }
     },
     "soft_acc": 0.32713275531640584,
     "brier_soft": 0.47372954021453173,
     "score_mae": 0.7603942575000003,
     "within_1": 0.7,
     "latency_p50_ms": 173.0,
     "latency_p99_ms": 327.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 50.0,
      "max_ms": 393.0,
      "argmax_case": 144
     },
     "determinism_ok": true,
     "seconds": 73.405661625,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    },
    "py/typed": {
     "lane": "laya (python)",
     "model": "typed",
     "hard": {
      "n": 2000,
      "accuracy": 0.7445,
      "macro_f1": 0.7349433525103708,
      "ece": 0.19239184999999998,
      "brier": 0.4098811812149981,
      "nll": 0.7184641706793969,
      "aurc": 0.12633869093638475,
      "mean_confidence": 0.5521081499999994,
      "acc_at_50_coverage": 0.859,
      "acc_at_80_coverage": 0.793125
     },
     "readout_ece": 0.39528484999999997,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": {
      "choice": {
       "n": 600,
       "accuracy": 0.7333333333333333,
       "macro_f1": 0.6943336435123875,
       "ece": 0.2548458333333333,
       "brier": 0.46509071859999873,
       "nll": 0.864997551289102,
       "aurc": 0.11299371741797679,
       "mean_confidence": 0.47848750000000007,
       "acc_at_50_coverage": 0.8966666666666666,
       "acc_at_80_coverage": 0.8
      },
      "noul": {
       "n": 600,
       "accuracy": 0.785,
       "macro_f1": 0.7833092099185031,
       "ece": 0.1294261666666668,
       "brier": 0.32151443403333324,
       "nll": 0.49461904218570935,
       "aurc": 0.08280543014765192,
       "mean_confidence": 0.6624201666666668,
       "acc_at_50_coverage": 0.93,
       "acc_at_80_coverage": 0.8541666666666666
      },
      "score": {
       "n": 800,
       "accuracy": 0.7225,
       "macro_f1": 0.7150339821484726,
       "ece": 0.198679625,
       "brier": 0.4347490885625001,
       "nll": 0.776447981592383,
       "aurc": 0.1370651831938734,
       "mean_confidence": 0.5245896249999998,
       "acc_at_50_coverage": 0.87,
       "acc_at_80_coverage": 0.7671875
      }
     },
     "soft_acc": 0.46682272509426403,
     "brier_soft": 0.06772116115371934,
     "score_mae": 0.2424075125000001,
     "within_1": 0.995,
     "latency_p50_ms": 352.0,
     "latency_p99_ms": 774.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 105.0,
      "max_ms": 822.0,
      "argmax_case": 119
     },
     "determinism_ok": true,
     "seconds": 148.247935,
     "n_cases": 400,
     "n_questions": 2000,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {},
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 2000,
       "accuracy": 0.5725,
       "macro_f1": 0.5428588466644556,
       "ece": 0.09930298610404133,
       "brier": 0.5634444754243869,
       "nll": 0.9993372552981049,
       "aurc": 0.2967333943470369,
       "mean_confidence": 0.47319701389595864,
       "acc_at_50_coverage": 0.676,
       "acc_at_80_coverage": 0.613125
      },
      "readout_ece": 0.04959756752848618,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.717,
       "selective_accuracy": 0.6537102473498233,
       "selective_n": 566
      },
      "readout_ece_raw": 0.48374495854973787,
      "readout_ece_calibrated": 0.04959756752848618,
      "floor_ece": 0.18175548902195604,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": {
       "choice": {
        "n": 600,
        "accuracy": 0.5483333333333333,
        "macro_f1": 0.432556704787651,
        "ece": 0.14819766715168953,
        "brier": 0.6304564691368547,
        "nll": 1.1799754864068186,
        "aurc": 0.3777896524930003,
        "mean_confidence": 0.4001356661816438,
        "acc_at_50_coverage": 0.59,
        "acc_at_80_coverage": 0.5291666666666667
       },
       "noul": {
        "n": 600,
        "accuracy": 0.73,
        "macro_f1": 0.7272727272727273,
        "ece": 0.06820527629305917,
        "brier": 0.37522739989001525,
        "nll": 0.5569624126123545,
        "aurc": 0.16418833870677263,
        "mean_confidence": 0.6617947237069408,
        "acc_at_50_coverage": 0.8266666666666667,
        "acc_at_80_coverage": 0.7770833333333333
       },
       "score": {
        "n": 800,
        "accuracy": 0.4725,
        "macro_f1": 0.37283935453238826,
        "ece": 0.09739159647375348,
        "brier": 0.6543482867908058,
        "nll": 1.1956397139808805,
        "aurc": 0.4607954011325077,
        "mean_confidence": 0.3865447423234582,
        "acc_at_50_coverage": 0.5275,
        "acc_at_80_coverage": 0.51875
       }
      },
      "soft_acc": 0.38266077842183205,
      "brier_soft": 0.16762865440042116,
      "score_mae": 0.5717842323177501,
      "within_1": 0.81625,
      "latency_p50_ms": 0.917,
      "latency_p99_ms": 2.732,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 0.205,
       "max_ms": 2.772,
       "argmax_case": 161
      },
      "determinism_ok": true,
      "seconds": 9.3050037,
      "n_cases": 400,
      "n_questions": 2000,
      "score_threshold": 0.57852864,
      "distance_threshold": 0.91222644,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.57852864,
        "status": "fitted",
        "targetAccuracy": 0.45714286,
        "support": {
         "n": 100,
         "nPass": 70,
         "nAbstain": 30
        }
       },
       "distance": {
        "threshold": 0.91222644,
        "status": "fitted",
        "targetAccuracy": 0.5285714,
        "support": {
         "n": 100,
         "nPass": 70,
         "nAbstain": 30
        }
       }
      },
      "corpus_cap": {
       "effective": 48,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 0.0,
       "selected_alpha": "off",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        }
       ]
      },
      "oc_selection": {
       "selected_scale": 4.0,
       "candidates": [
        {
         "scale": 0.0,
         "cal_acc": 0.36
        },
        {
         "scale": 0.25,
         "cal_acc": 0.452
        },
        {
         "scale": 0.5,
         "cal_acc": 0.496
        },
        {
         "scale": 1.0,
         "cal_acc": 0.536
        },
        {
         "scale": 2.0,
         "cal_acc": 0.54
        },
        {
         "scale": 4.0,
         "cal_acc": 0.544
        },
        {
         "scale": 8.0,
         "cal_acc": 0.542
        }
       ]
      },
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.544
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.542
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.554
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.516
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.508
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.466
        }
       ]
      },
      "genome_selection": null,
      "transductive": null,
      "confusion": [
       {
        "gold": "Elevated; should be handled within the same week.",
        "pred": "Critical; requires action within the same day.",
        "count": 63,
        "share_of_errors": 0.09090909090909091
       },
       {
        "gold": "success",
        "pred": "partial",
        "count": 50,
        "share_of_errors": 0.07215007215007214
       },
       {
        "gold": "No time pressure; can wait indefinitely.",
        "pred": "Routine; handle within the normal queue.",
        "count": 36,
        "share_of_errors": 0.05194805194805195
       },
       {
        "gold": "Material: a large or unexplained difference.",
        "pred": "None: everything reconciles.",
        "count": 35,
        "share_of_errors": 0.050505050505050504
       },
       {
        "gold": "Elevated; should be handled within the same week.",
        "pred": "Routine; handle within the normal queue.",
        "count": 34,
        "share_of_errors": 0.049062049062049064
       },
       {
        "gold": "Benign: read-only or clearly safe actions.",
        "pred": "Low: routine writes within scope.",
        "count": 33,
        "share_of_errors": 0.047619047619047616
       },
       {
        "gold": "Moderate: access to internal systems or non-public data.",
        "pred": "High: access to production, secrets or customer data.",
        "count": 30,
        "share_of_errors": 0.04329004329004329
       },
       {
        "gold": "approve",
        "pred": "manual_review",
        "count": 27,
        "share_of_errors": 0.03896103896103896
       },
       {
        "gold": "Routine; handle within the normal queue.",
        "pred": "Critical; requires action within the same day.",
        "count": 25,
        "share_of_errors": 0.03607503607503607
       },
       {
        "gold": "request_information",
        "pred": "escalate_to_human",
        "count": 24,
        "share_of_errors": 0.03463203463203463
       },
       {
        "gold": "Mild frustration, but the relationship is intact.",
        "pred": "Clearly unhappy; repeat problems or explicit complaints.",
        "count": 23,
        "share_of_errors": 0.03318903318903319
       },
       {
        "gold": "observe",
        "pred": "human_review",
        "count": 22,
        "share_of_errors": 0.031746031746031744
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.482341635107994
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.11559140637516975
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.482341635107994
        }
       ]
      },
      "corpus_digest": "fnv1a64-2fe9ec3132dbc924",
      "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      },
      "latency_provenance": {
       "note": "latency cells carried from the host's incumbent run; accuracy is this lane's own (LANE-CARRY, Issue 032)"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 2000,
        "accuracy": 0.3575,
        "macro_f1": 0.32620639534238516,
        "ece": 0.21127114999999994,
        "brier": 0.7684488626149973,
        "nll": 1.3593395233945826,
        "aurc": 0.5626661372266888,
        "mean_confidence": 0.5687711500000007,
        "acc_at_50_coverage": 0.429,
        "acc_at_80_coverage": 0.38125
       },
       "readout_ece": 0.19182065,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": {
        "choice": {
         "n": 600,
         "accuracy": 0.28833333333333333,
         "macro_f1": 0.2958479995436556,
         "ece": 0.16805983333333332,
         "brier": 0.8079412361333324,
         "nll": 1.5603165852368217,
         "aurc": 0.6171637396940312,
         "mean_confidence": 0.45270616666666624,
         "acc_at_50_coverage": 0.38666666666666666,
         "acc_at_80_coverage": 0.325
        },
        "noul": {
         "n": 600,
         "accuracy": 0.47333333333333333,
         "macro_f1": 0.46994095544820186,
         "ece": 0.2808481666666668,
         "brier": 0.6693727361666675,
         "nll": 0.9693957264213799,
         "aurc": 0.4926269983870734,
         "mean_confidence": 0.7541814999999997,
         "acc_at_50_coverage": 0.4866666666666667,
         "acc_at_80_coverage": 0.48333333333333334
        },
        "score": {
         "n": 800,
         "accuracy": 0.3225,
         "macro_f1": 0.24129818605373515,
         "ece": 0.20646887500000005,
         "brier": 0.8131366773125016,
         "nll": 1.501064574742798,
         "aurc": 0.6560603649790228,
         "mean_confidence": 0.5167621250000003,
         "acc_at_50_coverage": 0.335,
         "acc_at_80_coverage": 0.3125
        }
       },
       "soft_acc": 0.3345824092731849,
       "brier_soft": 0.33480864483416156,
       "score_mae": 0.6937211699999999,
       "within_1": 0.7525,
       "latency_p50_ms": 168.0,
       "latency_p99_ms": 252.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 56.0,
        "max_ms": 284.0,
        "argmax_case": 184
       },
       "determinism_ok": true,
       "seconds": 61.4124903,
       "n_cases": 400,
       "n_questions": 2000,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      },
      "multilingual": {
       "lane": "laya (rust)",
       "model": "multilingual",
       "hard": {
        "n": 2000,
        "accuracy": 0.349,
        "macro_f1": 0.3498994829046305,
        "ece": 0.3228023499999999,
        "brier": 0.9056686497349985,
        "nll": 1.9436850897748563,
        "aurc": 0.5548208277405716,
        "mean_confidence": 0.6687793500000011,
        "acc_at_50_coverage": 0.404,
        "acc_at_80_coverage": 0.36375
       },
       "readout_ece": 0.2566663,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": {
        "choice": {
         "n": 600,
         "accuracy": 0.295,
         "macro_f1": 0.3242803546952691,
         "ece": 0.36234616666666664,
         "brier": 0.988496027433334,
         "nll": 2.5096494088603847,
         "aurc": 0.5968789601913673,
         "mean_confidence": 0.6559424999999999,
         "acc_at_50_coverage": 0.31666666666666665,
         "acc_at_80_coverage": 0.28541666666666665
        },
        "noul": {
         "n": 600,
         "accuracy": 0.4866666666666667,
         "macro_f1": 0.4534096824570536,
         "ece": 0.4084023333333331,
         "brier": 0.8741193240000018,
         "nll": 1.6946673304708622,
         "aurc": 0.4613874178933604,
         "mean_confidence": 0.8919403333333329,
         "acc_at_50_coverage": 0.45,
         "acc_at_80_coverage": 0.48333333333333334
        },
        "score": {
         "n": 800,
         "accuracy": 0.28625,
         "macro_f1": 0.2554689567204614,
         "ece": 0.23129099999999994,
         "brier": 0.8672101107625029,
         "nll": 1.7059751699387093,
         "aurc": 0.7179176231742249,
         "mean_confidence": 0.5110362500000005,
         "acc_at_50_coverage": 0.26,
         "acc_at_80_coverage": 0.2765625
        }
       },
       "soft_acc": 0.32713270906562,
       "brier_soft": 0.47372988499163493,
       "score_mae": 0.7603950075000003,
       "within_1": 0.7,
       "latency_p50_ms": 82.0,
       "latency_p99_ms": 126.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 32.0,
        "max_ms": 139.0,
        "argmax_case": 183
       },
       "determinism_ok": true,
       "seconds": 33.5396411,
       "n_cases": 400,
       "n_questions": 2000,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      },
      "typed": {
       "lane": "laya (rust)",
       "model": "typed",
       "hard": {
        "n": 2000,
        "accuracy": 0.7445,
        "macro_f1": 0.7349433525103708,
        "ece": 0.1923919,
        "brier": 0.40988108352999814,
        "nll": 0.7184638360675057,
        "aurc": 0.12634017215900445,
        "mean_confidence": 0.5521080999999993,
        "acc_at_50_coverage": 0.859,
        "acc_at_80_coverage": 0.793125
       },
       "readout_ece": 0.39528484999999997,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": {
        "choice": {
         "n": 600,
         "accuracy": 0.7333333333333333,
         "macro_f1": 0.6943336435123875,
         "ece": 0.25484549999999995,
         "brier": 0.4650906017499988,
         "nll": 0.8649971870301854,
         "aurc": 0.11299371741797679,
         "mean_confidence": 0.4784878333333334,
         "acc_at_50_coverage": 0.8966666666666666,
         "acc_at_80_coverage": 0.8
        },
        "noul": {
         "n": 600,
         "accuracy": 0.785,
         "macro_f1": 0.7833092099185031,
         "ece": 0.1294261666666668,
         "brier": 0.32151493469999987,
         "nll": 0.4946195947319354,
         "aurc": 0.08280543014765192,
         "mean_confidence": 0.6624198333333333,
         "acc_at_50_coverage": 0.93,
         "acc_at_80_coverage": 0.8541666666666666
        },
        "score": {
         "n": 800,
         "accuracy": 0.7225,
         "macro_f1": 0.7150339821484726,
         "ece": 0.19867975000000002,
         "brier": 0.4347485564875002,
         "nll": 0.7764470038471731,
         "aurc": 0.1370651831938734,
         "mean_confidence": 0.5245894999999998,
         "acc_at_50_coverage": 0.87,
         "acc_at_80_coverage": 0.7671875
        }
       },
       "soft_acc": 0.46682272260714175,
       "brier_soft": 0.06772116157482795,
       "score_mae": 0.2424063875000001,
       "within_1": 0.995,
       "latency_p50_ms": 164.0,
       "latency_p99_ms": 294.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 57.0,
        "max_ms": 340.0,
        "argmax_case": 120
       },
       "determinism_ok": true,
       "seconds": 63.1842924,
       "n_cases": 400,
       "n_questions": 2000,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 2000,
       "accuracy": 0.3465,
       "macro_f1": 0.34536555317011297,
       "ece": 0.46574270132929085,
       "brier": 1.008599356699486,
       "nll": 2.428297600987895,
       "aurc": 0.5976801724358101,
       "mean_confidence": 0.8122427013292909,
       "acc_at_50_coverage": 0.382,
       "acc_at_80_coverage": 0.37625
      },
      "readout_ece": 0.3846414018794895,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": {
       "choice": {
        "n": 600,
        "accuracy": 0.25166666666666665,
        "macro_f1": 0.23354303995222062,
        "ece": 0.5765019276241461,
        "brier": 1.2175378343259564,
        "nll": 3.0143047336087263,
        "aurc": 0.7211419563541321,
        "mean_confidence": 0.8281685942908128,
        "acc_at_50_coverage": 0.29333333333333333,
        "acc_at_80_coverage": 0.25625
       },
       "noul": {
        "n": 600,
        "accuracy": 0.465,
        "macro_f1": 0.45240376550598627,
        "ece": 0.28695926743249095,
        "brier": 0.6416298567659945,
        "nll": 0.9179163240054247,
        "aurc": 0.4648953483503599,
        "mean_confidence": 0.751959267432491,
        "acc_at_50_coverage": 0.5433333333333333,
        "acc_at_80_coverage": 0.5208333333333334
       },
       "score": {
        "n": 800,
        "accuracy": 0.32875,
        "macro_f1": 0.26370818138161845,
        "ece": 0.5167608570307494,
        "brier": 1.1271226234297316,
        "nll": 3.1215782092591295,
        "aurc": 0.5544591918300347,
        "mean_confidence": 0.8455108570307494,
        "acc_at_50_coverage": 0.43,
        "acc_at_80_coverage": 0.359375
       }
      },
      "soft_acc": 0.35578530334475555,
      "brier_soft": 0.5844037750078159,
      "score_mae": 0.9175394711867856,
      "within_1": 0.60875,
      "latency_p50_ms": 66.0,
      "latency_p99_ms": 109.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 123.0,
       "max_ms": 144.0,
       "argmax_case": 300
      },
      "determinism_ok": true,
      "seconds": 27.5684839,
      "n_cases": 400,
      "n_questions": 2000,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 2000,
       "accuracy": 0.528,
       "macro_f1": 0.5198316028118526,
       "ece": 0.09588082381940428,
       "brier": 0.6057403150456773,
       "nll": 1.0662487183760088,
       "aurc": 0.36563423961725566,
       "mean_confidence": 0.4379060066996901,
       "acc_at_50_coverage": 0.634,
       "acc_at_80_coverage": 0.571875
      },
      "readout_ece": 0.09588082381940428,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": {
       "choice": {
        "n": 600,
        "accuracy": 0.49166666666666664,
        "macro_f1": 0.49646777652104745,
        "ece": 0.09388112721498428,
        "brier": 0.6416265634210919,
        "nll": 1.2105778934133675,
        "aurc": 0.3608241499380246,
        "mean_confidence": 0.40650598964619045,
        "acc_at_50_coverage": 0.6,
        "acc_at_80_coverage": 0.5375
       },
       "noul": {
        "n": 600,
        "accuracy": 0.6183333333333333,
        "macro_f1": 0.6177716700373604,
        "ece": 0.019429831672370917,
        "brier": 0.4718547816588068,
        "nll": 0.6648422845239664,
        "aurc": 0.34594609102918034,
        "mean_confidence": 0.6082711974065109,
        "acc_at_50_coverage": 0.6566666666666666,
        "acc_at_80_coverage": 0.6354166666666666
       },
       "score": {
        "n": 800,
        "accuracy": 0.4875,
        "macro_f1": 0.4032685049617461,
        "ece": 0.159997498668753,
        "brier": 0.6792397788042708,
        "nll": 1.259056662487016,
        "aurc": 0.3783267125021683,
        "mean_confidence": 0.3336821264596977,
        "acc_at_50_coverage": 0.58,
        "acc_at_80_coverage": 0.5234375
       }
      },
      "soft_acc": 0.36544210737156896,
      "brier_soft": 0.18658760669736948,
      "score_mae": 0.5665355150813695,
      "within_1": 0.87625,
      "latency_p50_ms": 30.0,
      "latency_p99_ms": 48.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 102.0,
       "max_ms": 102.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 24.1390113,
      "n_cases": 400,
      "n_questions": 2000,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 2000,
       "accuracy": 0.7715,
       "macro_f1": 0.7696099670768246,
       "ece": 0.1082401276661549,
       "brier": 0.33718911882889857,
       "nll": 0.5771258248095354,
       "aurc": 0.10057597049321372,
       "mean_confidence": 0.6644951696859207,
       "acc_at_50_coverage": 0.903,
       "acc_at_80_coverage": 0.81625
      },
      "readout_ece": 0.1087401276661549,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": {
       "choice": {
        "n": 600,
        "accuracy": 0.7766666666666666,
        "macro_f1": 0.747756823648196,
        "ece": 0.1364873769382636,
        "brier": 0.3455498767772061,
        "nll": 0.6239821937966089,
        "aurc": 0.08302548353736164,
        "mean_confidence": 0.644296947568655,
        "acc_at_50_coverage": 0.9466666666666667,
        "acc_at_80_coverage": 0.8375
       },
       "noul": {
        "n": 600,
        "accuracy": 0.76,
        "macro_f1": 0.7560865441076833,
        "ece": 0.04444630547814691,
        "brier": 0.2946578993128625,
        "nll": 0.43886999146926137,
        "aurc": 0.08616842111878155,
        "mean_confidence": 0.7624967417671966,
        "acc_at_50_coverage": 0.9266666666666666,
        "acc_at_80_coverage": 0.8291666666666667
       },
       "score": {
        "n": 800,
        "accuracy": 0.77625,
        "macro_f1": 0.7726608596817798,
        "ece": 0.17010734278708697,
        "brier": 0.3628169650046908,
        "nll": 0.6456754230744373,
        "aurc": 0.09794889145843973,
        "mean_confidence": 0.606142657212913,
        "acc_at_50_coverage": 0.9225,
        "acc_at_80_coverage": 0.828125
       }
      },
      "soft_acc": 0.527161014059792,
      "brier_soft": 0.06134425403672728,
      "score_mae": 0.21673142401790066,
      "within_1": 0.9925,
      "latency_p50_ms": 89.0,
      "latency_p99_ms": 183.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 183.0,
       "max_ms": 190.0,
       "argmax_case": 151
      },
      "determinism_ok": false,
      "seconds": 38.6836678,
      "n_cases": 400,
      "n_questions": 2000,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 2000,
       "accuracy": 0.536,
       "macro_f1": 0.539188499881466,
       "ece": 0.2274510137247853,
       "brier": 0.6464800785384385,
       "nll": 1.176883793206239,
       "aurc": 0.32452831651455416,
       "mean_confidence": 0.7620469068097881,
       "acc_at_50_coverage": 0.634,
       "acc_at_80_coverage": 0.56
      },
      "readout_ece": 0.15663834208015734,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": {
       "choice": {
        "n": 600,
        "accuracy": 0.545,
        "macro_f1": 0.5550637131854588,
        "ece": 0.2832088556389014,
        "brier": 0.726042368799983,
        "nll": 1.4632473185020318,
        "aurc": 0.28980483340473656,
        "mean_confidence": 0.8276060942312081,
        "acc_at_50_coverage": 0.6433333333333333,
        "acc_at_80_coverage": 0.5708333333333333
       },
       "noul": {
        "n": 600,
        "accuracy": 0.5166666666666667,
        "macro_f1": 0.5152879301123194,
        "ece": 0.2366313620765383,
        "brier": 0.5719942126358104,
        "nll": 0.8068832841381363,
        "aurc": 0.3256950147452612,
        "mean_confidence": 0.7457535545388236,
        "acc_at_50_coverage": 0.6333333333333333,
        "acc_at_80_coverage": 0.5354166666666667
       },
       "score": {
        "n": 800,
        "accuracy": 0.54375,
        "macro_f1": 0.5207694222850818,
        "ece": 0.18621573891490698,
        "brier": 0.6426727602692369,
        "nll": 1.2396115310354738,
        "aurc": 0.33882020236382976,
        "mean_confidence": 0.7250975304469466,
        "acc_at_50_coverage": 0.65,
        "acc_at_80_coverage": 0.5609375
       }
      },
      "soft_acc": 0.4559382317932149,
      "brier_soft": 0.31618158912991967,
      "score_mae": 0.4994295016957298,
      "within_1": 0.89375,
      "latency_p50_ms": 79.0,
      "latency_p99_ms": 187.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 361.0,
       "max_ms": 361.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 38.6678185,
      "n_cases": 400,
      "n_questions": 2000,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-6e37760ee2a5b6c9"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-6e37760ee2a5b6c9"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-6e37760ee2a5b6c9"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-6e37760ee2a5b6c9"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "H2(\u03b2=0.5,nmin=2,\u03c4=2)",
    "hard": {
     "n": 2000,
     "accuracy": 0.6475,
     "ece": 0.18986412296281885,
     "mean_confidence": 0.8235891979113439,
     "acc_at_50_coverage": 0.754
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.003166,
    "latency_p99_ms": 0.012083,
    "latency_tail_support": 21,
    "serves": "H2(\u03b2=0.5,nmin=2,\u03c4=2)",
    "gate": "certified (paired LB95 +0.0580, mean +0.0750)",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 2000,
     "accuracy": 0.5345,
     "macro_f1": 0.5371199035294525,
     "ece": 0.2274083137717098,
     "brier": 0.6460861957061594,
     "nll": 1.1753930534378259,
     "aurc": 0.3246053071014397,
     "mean_confidence": 0.7618385489489883,
     "acc_at_50_coverage": 0.633,
     "acc_at_80_coverage": 0.559375
    },
    "readout_ece": 0.15472393500125628,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": {
     "choice": {
      "n": 600,
      "accuracy": 0.545,
      "macro_f1": 0.554520813706687,
      "ece": 0.28237459932764375,
      "brier": 0.7253539905037218,
      "nll": 1.4589747245828801,
      "aurc": 0.29013083854246025,
      "mean_confidence": 0.8268162489434083,
      "acc_at_50_coverage": 0.6466666666666666,
      "acc_at_80_coverage": 0.5666666666666667
     },
     "noul": {
      "n": 600,
      "accuracy": 0.5166666666666667,
      "macro_f1": 0.5151096298112829,
      "ece": 0.234377175240467,
      "brier": 0.5712280944901422,
      "nll": 0.805847085584037,
      "aurc": 0.32524087769305254,
      "mean_confidence": 0.7453727274822692,
      "acc_at_50_coverage": 0.6333333333333333,
      "acc_at_80_coverage": 0.5375
     },
     "score": {
      "n": 800,
      "accuracy": 0.54,
      "macro_f1": 0.5138024878101111,
      "ece": 0.185966830663383,
      "brier": 0.6427789255199906,
      "nll": 1.2398662759693813,
      "aurc": 0.33940640735851063,
      "mean_confidence": 0.7254546400532127,
      "acc_at_50_coverage": 0.6525,
      "acc_at_80_coverage": 0.5578125
     }
    },
    "soft_acc": 0.45594216445138996,
    "brier_soft": 0.31585399196464964,
    "score_mae": 0.4991223548573117,
    "within_1": 0.895,
    "latency_p50_ms": 497.0,
    "latency_p99_ms": 779.0,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 392.0,
     "max_ms": 787.0,
     "argmax_case": 151
    },
    "determinism_ok": true,
    "seconds": 208.132836417,
    "n_cases": 400,
    "n_questions": 2000,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "73c963ebfd40a8d25288,f1d08b30f8c84c2fec6c,04a3b05698d9f7afc8cd,3f8d28006aa102c9df04,0189d0989f5222454bea,5c74a4fca552e2471669,0c366580689869c1f689,0522b1a4023281b00c19,c4bf7c8af208cf4374af,19114692867411036ed5,3466ab136a4aa4ae6bdf,84d32e31a96917d47a0b,eeecbb0dbfd3bba96bdb,d8564cd360417f43590c,492c5669e60b7b07a7db",
    "spec_file": "scripts/paw_specs/typed_decisions.action.txt,scripts/paw_specs/typed_decisions.needs_review.txt,scripts/paw_specs/typed_decisions.outcome.txt,scripts/paw_specs/typed_decisions.risk.txt,scripts/paw_specs/typed_decisions.urgency.txt,scripts/paw_specs/typed_decisions.category.txt,scripts/paw_specs/typed_decisions.churn_risk.txt,scripts/paw_specs/typed_decisions.needs_human.txt,scripts/paw_specs/typed_decisions.discrepancy_severity.txt,scripts/paw_specs/typed_decisions.disposition.txt,scripts/paw_specs/typed_decisions.duplicate.txt,scripts/paw_specs/typed_decisions.matches_order.txt,scripts/paw_specs/typed_decisions.credential_compromise.txt,scripts/paw_specs/typed_decisions.severity.txt,scripts/paw_specs/typed_decisions.true_positive.txt",
    "spec_blake3": "72055396b7348aa43c7fcc2f56d7617e9d6448defdcd1a37fe46680a29c3973c,c7a474c539477d43c9b1de0d464f695836ecf7744fd3021c64dbf166a482451b,b18771699ebb4d643ef2a95de0890fc3f735e090e58bda0afb4eee4115ee4cd5,8d5ae953d07c1ba75f325e3cee87eec54cccf4a3a2adb05221e287b7c6852a46,dadfd9808e4aa4edbcdcb1ca8445b5070ff132d48e5c6a1d0ce082a80b2e68ea,1e91ffffbd7ecc71492da3dc24f9552cd08d48e162ab1b7923b7cdf1ba42bf66,b3d2713c4a8bea1a38b46ed2c26188ac506c54712b06736daa90240516cad297,0ac1f08efb9a0e0c4bfe8d80cadf12b90221ad20ad3965d5e99685aff3f7c75b,6a6ed8cbcc154f47435e9258a84d560d8fd1292a51edc639be5218d3dcd84016,eb7abb02fb227a9ccd04136c7b46ac9ffd72553780da57e608ddd232cf92212d,bdcd453d1d3e321e39d3ab77685d1e0f00b959f0bdc9c1df59b031f90b91eaf6,175eb2d46596884958f2bafc6786a0945aa8982e08270b81dcb50e05471a9f19,b0edaed6fff9b21a2d8d94406acc82fb7b2d2fda05751eb62346f6e105ff1c61,2173eeb74a3af2ea95857ad6ad8d5963cdab904921d1d42abc96b9784bebe6d9,87c8f166174f846c75f6ec2170e031e3e34dff569e9edb99112c2ecd6da3ca06",
    "compile_cache_hit": false,
    "compile_wall_s": 2645.5270762899995,
    "n_cases": 400,
    "n_questions": 2000,
    "n_answered": 1998,
    "refusals": 2,
    "quote_stripped": 0,
    "accuracy": 0.5925,
    "answered_accuracy": 0.5930930930930931,
    "refusal_rate": 0.001,
    "score_mae_answered": 0.6484559450000004,
    "within_1_answered": 0.79375,
    "refusal_samples": [
     "data",
     "observe"
    ],
    "server_latency_p50_ms": 96.6,
    "determinism_ok": true,
    "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "9dc33da",
     "date_utc": "2026-09-29T02:12:55Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-typed_encoder_v2 (laya-typed encoder + NLEH v2 per-option head)",
    "hard": {
     "n": 2000,
     "accuracy": 0.755
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 364.486,
    "latency_p99_ms": 782.684,
    "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
    "gate": "acc 0.7550 (1510/2000 question rows) \u00b7 paired vs the incumbent A1: mean +0.1250 \u00b7 LB95 +0.1032 \u00b7 per-row p50 364.486 ms (metal device, laya-typed). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017). Train-side (riir-train t608 / Bench 040): NLEH v2 per-option head over the typed-checkpoint TRAIN cache (4000 gold-only rows, widths 2/4/5), holdout 0.7975 vs the reference 0.7762 (EARN); artifact blake3 c2e9f346\u2026; trainer-side frozen read 0.7550 (1510/2000) \u2014 this arena read is the third-posture cell-identity witness.",
    "device": "metal",
    "head_kind": "v2 per-option",
    "ckpt": "typed",
    "shape_desc": "d 1024 \u00b7 hidden 128 \u00b7 widths 2/4/5",
    "record_only": true,
    "latency_quotable": false,
    "source_run": {
     "git_sha": "c2c5dc3",
     "date_utc": "2026-10-01T14:59:02Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 2000,
     "accuracy": 0.6235,
     "macro_f1": 0.6118980306296695,
     "ece": 0.07355827322602271,
     "brier": 0.5079956884791216,
     "nll": 0.8702929329235691,
     "aurc": 0.27857286503863055,
     "mean_confidence": 0.6615492779910565,
     "acc_at_50_coverage": 0.722,
     "acc_at_80_coverage": 0.6575
    },
    "readout_ece": 0.07355827322602271,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": {
     "choice": {
      "n": 600,
      "accuracy": 0.6466666666666666,
      "macro_f1": 0.6369820906714498,
      "ece": 0.06022603397568067,
      "brier": 0.4684623287214902,
      "nll": 0.8409224754314247,
      "aurc": 0.1903552858511572,
      "mean_confidence": 0.6070189823210239,
      "acc_at_50_coverage": 0.8066666666666666,
      "acc_at_80_coverage": 0.7020833333333333
     },
     "noul": {
      "n": 600,
      "accuracy": 0.6583333333333333,
      "macro_f1": 0.6571667479618145,
      "ece": 0.14293939212958015,
      "brier": 0.49694519340378807,
      "nll": 0.7986957730217075,
      "aurc": 0.28931944597954057,
      "mean_confidence": 0.7990485644340515,
      "acc_at_50_coverage": 0.6866666666666666,
      "acc_at_80_coverage": 0.6833333333333333
     },
     "score": {
      "n": 800,
      "accuracy": 0.58,
      "macro_f1": 0.5806196861293669,
      "ece": 0.06745047811418772,
      "brier": 0.5459335796038359,
      "nll": 0.9460186459690771,
      "aurc": 0.35564505152085907,
      "mean_confidence": 0.5993225349113345,
      "acc_at_50_coverage": 0.645,
      "acc_at_80_coverage": 0.6078125
     }
    },
    "soft_acc": 0.4717161859231123,
    "brier_soft": 0.17086411334936705,
    "score_mae": 0.3657786200978856,
    "within_1": 0.95375,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 400,
    "n_questions": 2000,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.2925,
    "jdi_skill": 0.46784452296819795,
    "cases_digest": "fnv1a64-6e37760ee2a5b6c9",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:08:12Z"
    }
   }
  },
  {
   "name": "ag_news",
   "n_cases": 400,
   "n_questions": 400,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 400,
     "accuracy": 0.8825,
     "macro_f1": 0.8741933646432797,
     "ece": 0.47827894262969495,
     "brier": 0.5053158266072506,
     "nll": 0.9629571322547602,
     "aurc": 0.04689348589556988,
     "mean_confidence": 0.40422105737030506,
     "acc_at_50_coverage": 0.94,
     "acc_at_80_coverage": 0.9375
    },
    "readout_ece": 0.04585413459688425,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.49,
     "selective_accuracy": 0.9509803921568627,
     "selective_n": 204
    },
    "readout_ece_raw": 0.8304792892932893,
    "readout_ece_calibrated": 0.04585413459688425,
    "floor_ece": 0.24822139303482577,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.149,
    "latency_p99_ms": 0.262,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 0.251,
     "max_ms": 0.306,
     "argmax_case": 238
    },
    "determinism_ok": true,
    "seconds": 4.760110666,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": 0.8358875,
    "distance_threshold": 0.41053298,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.8358875,
      "status": "fitted",
      "targetAccuracy": 0.9357143,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.41053298,
      "status": "fitted",
      "targetAccuracy": 0.8857143,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 4.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.7
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.7
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.745
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.745
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.735
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.735
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.74
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.735
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.74
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.8825,
     "honest_accuracy": 0.8825,
     "n_pseudo": 400,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "business",
      "pred": "sci_tech",
      "count": 10,
      "share_of_errors": 0.2127659574468085
     },
     {
      "gold": "sci_tech",
      "pred": "business",
      "count": 8,
      "share_of_errors": 0.1702127659574468
     },
     {
      "gold": "world",
      "pred": "business",
      "count": 7,
      "share_of_errors": 0.14893617021276595
     },
     {
      "gold": "world",
      "pred": "sports",
      "count": 6,
      "share_of_errors": 0.1276595744680851
     },
     {
      "gold": "sports",
      "pred": "sci_tech",
      "count": 3,
      "share_of_errors": 0.06382978723404255
     },
     {
      "gold": "sports",
      "pred": "world",
      "count": 3,
      "share_of_errors": 0.06382978723404255
     },
     {
      "gold": "world",
      "pred": "sci_tech",
      "count": 3,
      "share_of_errors": 0.06382978723404255
     },
     {
      "gold": "business",
      "pred": "world",
      "count": 2,
      "share_of_errors": 0.0425531914893617
     },
     {
      "gold": "sci_tech",
      "pred": "sports",
      "count": 2,
      "share_of_errors": 0.0425531914893617
     },
     {
      "gold": "business",
      "pred": "sports",
      "count": 1,
      "share_of_errors": 0.02127659574468085
     },
     {
      "gold": "sci_tech",
      "pred": "world",
      "count": 1,
      "share_of_errors": 0.02127659574468085
     },
     {
      "gold": "sports",
      "pred": "business",
      "count": 1,
      "share_of_errors": 0.02127659574468085
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.8204048293828965
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.48570768773555756
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.8204048293828965
      }
     ]
    },
    "corpus_digest": "fnv1a64-a14927832fc48da9",
    "cases_digest": "fnv1a64-07eff23c06505584",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 400,
      "accuracy": 0.95,
      "macro_f1": 0.9439373207318056,
      "ece": 0.03155950000000057,
      "brier": 0.08078989162499997,
      "nll": 0.16374224530654963,
      "aurc": 0.006681571222221135,
      "mean_confidence": 0.9220469999999991,
      "acc_at_50_coverage": 1.0,
      "acc_at_80_coverage": 0.990625
     },
     "readout_ece": 0.14119374999999995,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 23.0,
     "latency_p99_ms": 76.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 39.0,
      "max_ms": 95.0,
      "argmax_case": 333
     },
     "determinism_ok": true,
     "seconds": 18.214984708,
     "n_cases": 400,
     "n_questions": 400,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-07eff23c06505584",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 400,
      "accuracy": 0.95,
      "macro_f1": 0.9439373207318056,
      "ece": 0.03155950000000057,
      "brier": 0.08078967889999995,
      "nll": 0.16374224530654963,
      "aurc": 0.006681571222221135,
      "mean_confidence": 0.9220469999999991,
      "acc_at_50_coverage": 1.0,
      "acc_at_80_coverage": 0.990625
     },
     "readout_ece": 0.14119374999999995,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 34.0,
     "latency_p99_ms": 84.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 20.0,
      "max_ms": 98.0,
      "argmax_case": 33
     },
     "determinism_ok": true,
     "seconds": 22.166054834,
     "n_cases": 400,
     "n_questions": 400,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-07eff23c06505584",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 365,
        "accuracy": 0.9452054794520548,
        "macro_f1": 0.9373782486886029,
        "ece": 0.038674246575342346,
        "brier": 0.08643471446575335,
        "nll": 0.17340547255438327,
        "aurc": 0.007544982421972155,
        "mean_confidence": 0.9196917808219178,
        "acc_at_50_coverage": 1.0,
        "acc_at_80_coverage": 0.9897260273972602
       },
       "readout_ece": 0.14208958904109587,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 28.0,
       "latency_p99_ms": 34.0,
       "latency_tail_support": 4,
       "latency_extremes": {
        "first_ms": 27.0,
        "max_ms": 36.0,
        "argmax_case": 50
       },
       "determinism_ok": true,
       "seconds": 48.386979667,
       "n_cases": 365,
       "n_questions": 365,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-07eff23c06505584",
       "latency_quotable": true,
       "source_run": {
        "git_sha": "2f6b58c",
        "date_utc": "2026-09-30T10:35:35Z"
       }
      }
     }
    },
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 400,
       "accuracy": 0.8825,
       "macro_f1": 0.8741933646432797,
       "ece": 0.4782789427042008,
       "brier": 0.5053158268175713,
       "nll": 0.9629571323951134,
       "aurc": 0.04689348589556988,
       "mean_confidence": 0.40422105729579927,
       "acc_at_50_coverage": 0.94,
       "acc_at_80_coverage": 0.9375
      },
      "readout_ece": 0.04585413459688427,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.49,
       "selective_accuracy": 0.9509803921568627,
       "selective_n": 204
      },
      "readout_ece_raw": 0.8304792888462544,
      "readout_ece_calibrated": 0.04585413459688427,
      "floor_ece": 0.24822139303482577,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.303,
      "latency_p99_ms": 1.919,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 0.578,
       "max_ms": 5.182,
       "argmax_case": 94
      },
      "determinism_ok": true,
      "seconds": 9.3921419,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": 0.8358875,
      "distance_threshold": 0.41053298,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.8358875,
        "status": "fitted",
        "targetAccuracy": 0.9357143,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.41053298,
        "status": "fitted",
        "targetAccuracy": 0.8857143,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 4.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.7
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.7
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.745
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.745
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.735
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.735
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.74
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.735
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.74
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.8825,
       "honest_accuracy": 0.8825,
       "n_pseudo": 400,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "business",
        "pred": "sci_tech",
        "count": 10,
        "share_of_errors": 0.2127659574468085
       },
       {
        "gold": "sci_tech",
        "pred": "business",
        "count": 8,
        "share_of_errors": 0.1702127659574468
       },
       {
        "gold": "world",
        "pred": "business",
        "count": 7,
        "share_of_errors": 0.14893617021276595
       },
       {
        "gold": "world",
        "pred": "sports",
        "count": 6,
        "share_of_errors": 0.1276595744680851
       },
       {
        "gold": "sports",
        "pred": "sci_tech",
        "count": 3,
        "share_of_errors": 0.06382978723404255
       },
       {
        "gold": "sports",
        "pred": "world",
        "count": 3,
        "share_of_errors": 0.06382978723404255
       },
       {
        "gold": "world",
        "pred": "sci_tech",
        "count": 3,
        "share_of_errors": 0.06382978723404255
       },
       {
        "gold": "business",
        "pred": "world",
        "count": 2,
        "share_of_errors": 0.0425531914893617
       },
       {
        "gold": "sci_tech",
        "pred": "sports",
        "count": 2,
        "share_of_errors": 0.0425531914893617
       },
       {
        "gold": "business",
        "pred": "sports",
        "count": 1,
        "share_of_errors": 0.02127659574468085
       },
       {
        "gold": "sci_tech",
        "pred": "world",
        "count": 1,
        "share_of_errors": 0.02127659574468085
       },
       {
        "gold": "sports",
        "pred": "business",
        "count": 1,
        "share_of_errors": 0.02127659574468085
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.8204048293828965
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.48570768788456914
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.8204048293828965
        }
       ]
      },
      "corpus_digest": "fnv1a64-a14927832fc48da9",
      "cases_digest": "fnv1a64-07eff23c06505584",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 400,
        "accuracy": 0.95,
        "macro_f1": 0.9439373207318056,
        "ece": 0.03155925000000057,
        "brier": 0.08078943217499995,
        "nll": 0.16374175177185255,
        "aurc": 0.006681571222221135,
        "mean_confidence": 0.9220472499999991,
        "acc_at_50_coverage": 1.0,
        "acc_at_80_coverage": 0.990625
       },
       "readout_ece": 0.14119374999999995,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 13.0,
       "latency_p99_ms": 18.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 13.0,
        "max_ms": 20.0,
        "argmax_case": 301
       },
       "determinism_ok": true,
       "seconds": 7.852086,
       "n_cases": 400,
       "n_questions": 400,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-07eff23c06505584",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 400,
       "accuracy": 0.4025,
       "macro_f1": 0.3588414509467141,
       "ece": 0.2913862624764442,
       "brier": 0.8586495668977957,
       "nll": 2.1584452673637795,
       "aurc": 0.4328917190412105,
       "mean_confidence": 0.6938862624764442,
       "acc_at_50_coverage": 0.54,
       "acc_at_80_coverage": 0.44375
      },
      "readout_ece": 0.1983789935708046,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 33.0,
      "latency_p99_ms": 39.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 80.0,
       "max_ms": 80.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 13.4941302,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.3941018766756032,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 400,
       "accuracy": 0.7025,
       "macro_f1": 0.6590344066076197,
       "ece": 0.12537422353601715,
       "brier": 0.48717057102560624,
       "nll": 1.0569244178165471,
       "aurc": 0.20165422979334033,
       "mean_confidence": 0.8141866097185432,
       "acc_at_50_coverage": 0.77,
       "acc_at_80_coverage": 0.725
      },
      "readout_ece": 0.12537422353601715,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 23.0,
      "latency_p99_ms": 39.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 105.0,
       "max_ms": 105.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 21.320197,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 400,
       "accuracy": 0.8,
       "macro_f1": 0.7803662202944874,
       "ece": 0.13818189471960068,
       "brier": 0.31533994326004366,
       "nll": 0.6129086605259458,
       "aurc": 0.08276539931737022,
       "mean_confidence": 0.6618181052803993,
       "acc_at_50_coverage": 0.945,
       "acc_at_80_coverage": 0.853125
      },
      "readout_ece": 0.13818189471960068,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 39.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 38.0,
       "max_ms": 49.0,
       "argmax_case": 14
      },
      "determinism_ok": false,
      "seconds": 13.579293,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "paw_local": {
      "lane": "paw (local)",
      "model": "paw-ft-bs48-20260530",
      "posture": "local-subprocess",
      "program_id": "35a7963d2d18aac54616",
      "spec_file": "scripts/paw_specs\\ag_news.txt",
      "spec_blake3": "c5d5b475b88da357364a5f3ae3d87b02424a72bb4b16331fb9b59b0d1d543063",
      "compile_cache_hit": true,
      "compile_wall_s": 11.7595367,
      "n_cases": 400,
      "n_questions": 400,
      "n_answered": 400,
      "refusals": 0,
      "quote_stripped": 0,
      "accuracy": 0.8,
      "answered_accuracy": 0.8,
      "refusal_rate": 0.0,
      "score_mae_answered": null,
      "within_1_answered": null,
      "refusal_samples": [],
      "latency_p50_ms": 264.0,
      "latency_p99_ms": 1563.774,
      "latency_tail_support": 5,
      "server_latency_p50_ms": null,
      "determinism_ok": true,
      "seconds": 241.0684133,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 400,
       "accuracy": 0.89,
       "macro_f1": 0.882644318569296,
       "ece": 0.047420404031872684,
       "brier": 0.1550014348564587,
       "nll": 0.28137258555480354,
       "aurc": 0.02023301346851728,
       "mean_confidence": 0.9207828643172979,
       "acc_at_50_coverage": 0.985,
       "acc_at_80_coverage": 0.96875
      },
      "readout_ece": 0.06789136253893188,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 56.0,
      "latency_p99_ms": 174.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 66.0,
       "max_ms": 193.0,
       "argmax_case": 44
      },
      "determinism_ok": true,
      "seconds": 30.4535398,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-07eff23c06505584",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 400,
    "exact": 1,
    "near": 26
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "35a7963d2d18aac54616",
    "spec_file": "scripts/paw_specs/ag_news.txt",
    "spec_blake3": "c5d5b475b88da357364a5f3ae3d87b02424a72bb4b16331fb9b59b0d1d543063",
    "compile_cache_hit": false,
    "compile_wall_s": 1.72005275,
    "n_cases": 400,
    "n_questions": 400,
    "n_answered": 400,
    "refusals": 0,
    "quote_stripped": 0,
    "accuracy": 0.79,
    "answered_accuracy": 0.79,
    "refusal_rate": 0.0,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [],
    "server_latency_p50_ms": 60.7,
    "determinism_ok": true
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-07eff23c06505584"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-07eff23c06505584"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-07eff23c06505584"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-07eff23c06505584"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "H2(\u03b2=0.25,nmin=2,\u03c4=2)",
    "hard": {
     "n": 400,
     "accuracy": 0.8975,
     "ece": 0.013315694555954022,
     "mean_confidence": 0.888912138562494,
     "acc_at_50_coverage": 0.99
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.0015,
    "latency_p99_ms": 0.003125,
    "latency_tail_support": 5,
    "serves": "H2(\u03b2=0.25,nmin=2,\u03c4=2)",
    "gate": "best measured, T2-uncertified (paired LB95 -0.0100) \u2014 certification is more questions, not a posture rollback",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 400,
     "accuracy": 0.89,
     "macro_f1": 0.882644318569296,
     "ece": 0.051162975877523394,
     "brier": 0.15519215674552342,
     "nll": 0.2817015835268838,
     "aurc": 0.020399605241929217,
     "mean_confidence": 0.9206447131931782,
     "acc_at_50_coverage": 0.985,
     "acc_at_80_coverage": 0.96875
    },
    "readout_ece": 0.07099155767576987,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 113.0,
    "latency_p99_ms": 226.0,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 127.0,
     "max_ms": 237.0,
     "argmax_case": 5
    },
    "determinism_ok": true,
    "seconds": 51.154000542,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-07eff23c06505584",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-ag_news_encoder_v1 (laya-english encoder + NLEH v1 head)",
    "hard": {
     "n": 400,
     "accuracy": 0.9475
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 16.983,
    "latency_p99_ms": 27.313,
    "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
    "gate": "acc 0.9475 (379/400 rows) \u00b7 paired vs the incumbent A1: mean +0.0600 \u00b7 LB95 +0.0320 \u00b7 per-row p50 16.983 ms (metal device). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017). The head reads one question under its own reference (riir-train 600 T1).",
    "device": "metal",
    "record_only": true,
    "latency_quotable": true,
    "source_run": {
     "git_sha": "0f81b54",
     "date_utc": "2026-10-01T11:20:42Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 400,
     "accuracy": 0.915,
     "macro_f1": 0.9088637583052136,
     "ece": 0.035219853520393384,
     "brier": 0.1230511733458803,
     "nll": 0.24119856360330336,
     "aurc": 0.01498919760080796,
     "mean_confidence": 0.9212963885068893,
     "acc_at_50_coverage": 0.99,
     "acc_at_80_coverage": 0.96875
    },
    "readout_ece": 0.035219853520393384,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.3075,
    "jdi_skill": 0.8772563176895307,
    "cases_digest": "fnv1a64-07eff23c06505584",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   }
  },
  {
   "name": "emotion",
   "n_cases": 400,
   "n_questions": 400,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 400,
     "accuracy": 0.885,
     "macro_f1": 0.8014284327939634,
     "ece": 0.6182363981008531,
     "brier": 0.6593071434500918,
     "nll": 1.3642793452306783,
     "aurc": 0.035998860173968965,
     "mean_confidence": 0.26676360189914705,
     "acc_at_50_coverage": 0.975,
     "acc_at_80_coverage": 0.940625
    },
    "readout_ece": 0.04880640662275256,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.4625,
     "selective_accuracy": 0.9348837209302325,
     "selective_n": 215
    },
    "readout_ece_raw": 0.8621306203305722,
    "readout_ece_calibrated": 0.04880640662275256,
    "floor_ece": 0.27810945273631843,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.115,
    "latency_p99_ms": 0.129,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 0.129,
     "max_ms": 0.135,
     "argmax_case": 211
    },
    "determinism_ok": true,
    "seconds": 11.231493041,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": 0.7631245,
    "distance_threshold": 0.4265421,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.7631245,
      "status": "fitted",
      "targetAccuracy": 0.89285713,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.4265421,
      "status": "fitted",
      "targetAccuracy": 0.73571426,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 4.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.195
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.535
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.55
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.55
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.55
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.55
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.465
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.485
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.485
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.485
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.485
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 8.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.55
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.58
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.64
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.69
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.735
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.76
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.8875,
     "honest_accuracy": 0.885,
     "n_pseudo": 400,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "love",
      "pred": "joy",
      "count": 15,
      "share_of_errors": 0.32608695652173914
     },
     {
      "gold": "anger",
      "pred": "sadness",
      "count": 5,
      "share_of_errors": 0.10869565217391304
     },
     {
      "gold": "fear",
      "pred": "joy",
      "count": 5,
      "share_of_errors": 0.10869565217391304
     },
     {
      "gold": "fear",
      "pred": "sadness",
      "count": 4,
      "share_of_errors": 0.08695652173913043
     },
     {
      "gold": "surprise",
      "pred": "joy",
      "count": 4,
      "share_of_errors": 0.08695652173913043
     },
     {
      "gold": "fear",
      "pred": "anger",
      "count": 2,
      "share_of_errors": 0.043478260869565216
     },
     {
      "gold": "joy",
      "pred": "sadness",
      "count": 2,
      "share_of_errors": 0.043478260869565216
     },
     {
      "gold": "joy",
      "pred": "surprise",
      "count": 2,
      "share_of_errors": 0.043478260869565216
     },
     {
      "gold": "sadness",
      "pred": "joy",
      "count": 2,
      "share_of_errors": 0.043478260869565216
     },
     {
      "gold": "surprise",
      "pred": "fear",
      "count": 2,
      "share_of_errors": 0.043478260869565216
     },
     {
      "gold": "anger",
      "pred": "joy",
      "count": 1,
      "share_of_errors": 0.021739130434782608
     },
     {
      "gold": "anger",
      "pred": "surprise",
      "count": 1,
      "share_of_errors": 0.021739130434782608
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.7589061281085013
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.527420337125659
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.7589061281085013
      }
     ]
    },
    "corpus_digest": "fnv1a64-d8c5a7db33864fb4",
    "cases_digest": "fnv1a64-6feefd1c947dd008",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 400,
      "accuracy": 0.5925,
      "macro_f1": 0.47086603548842415,
      "ece": 0.308786,
      "brier": 0.6962648099500026,
      "nll": 2.184045176210792,
      "aurc": 0.2657102861338114,
      "mean_confidence": 0.9012859999999996,
      "acc_at_50_coverage": 0.715,
      "acc_at_80_coverage": 0.6375
     },
     "readout_ece": 0.2570897499999998,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 14.0,
     "latency_p99_ms": 20.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 12.0,
      "max_ms": 23.0,
      "argmax_case": 22
     },
     "determinism_ok": true,
     "seconds": 9.871729916,
     "n_cases": 400,
     "n_questions": 400,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6feefd1c947dd008",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 400,
      "accuracy": 0.5925,
      "macro_f1": 0.47086603548842415,
      "ece": 0.308786,
      "brier": 0.6962642192000027,
      "nll": 2.184042555956118,
      "aurc": 0.2657102861338114,
      "mean_confidence": 0.9012859999999996,
      "acc_at_50_coverage": 0.715,
      "acc_at_80_coverage": 0.6375
     },
     "readout_ece": 0.2570897499999998,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 24.0,
     "latency_p99_ms": 58.0,
     "latency_tail_support": 5,
     "latency_extremes": {
      "first_ms": 17.0,
      "max_ms": 67.0,
      "argmax_case": 143
     },
     "determinism_ok": true,
     "seconds": 16.902667542,
     "n_cases": 400,
     "n_questions": 400,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-6feefd1c947dd008",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 400,
        "accuracy": 0.5975,
        "macro_f1": 0.47767899336751946,
        "ece": 0.3040499999999997,
        "brier": 0.6963723941000033,
        "nll": 2.1820143189369885,
        "aurc": 0.2658936866009599,
        "mean_confidence": 0.9015499999999996,
        "acc_at_50_coverage": 0.715,
        "acc_at_80_coverage": 0.6375
       },
       "readout_ece": 0.2546720000000001,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 19.0,
       "latency_p99_ms": 24.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 21.0,
        "max_ms": 25.0,
        "argmax_case": 119
       },
       "determinism_ok": true,
       "seconds": 11.97681375,
       "n_cases": 400,
       "n_questions": 400,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-6feefd1c947dd008",
       "latency_quotable": true,
       "source_run": {
        "git_sha": "2f6b58c",
        "date_utc": "2026-09-30T10:35:35Z"
       }
      }
     }
    },
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 400,
       "accuracy": 0.885,
       "macro_f1": 0.8014284327939634,
       "ece": 0.6182363714277744,
       "brier": 0.6593071014744065,
       "nll": 1.3642792551882197,
       "aurc": 0.035998860173968965,
       "mean_confidence": 0.2667636285722256,
       "acc_at_50_coverage": 0.975,
       "acc_at_80_coverage": 0.940625
      },
      "readout_ece": 0.048806075789034384,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.4625,
       "selective_accuracy": 0.9348837209302325,
       "selective_n": 215
      },
      "readout_ece_raw": 0.8621306075155736,
      "readout_ece_calibrated": 0.048806075789034384,
      "floor_ece": 0.27812189054726366,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.171,
      "latency_p99_ms": 0.388,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 0.18,
       "max_ms": 2.333,
       "argmax_case": 138
      },
      "determinism_ok": true,
      "seconds": 19.7032119,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": 0.76312464,
      "distance_threshold": 0.4265421,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.76312464,
        "status": "fitted",
        "targetAccuracy": 0.89285713,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.4265421,
        "status": "fitted",
        "targetAccuracy": 0.73571426,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 4.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.195
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.535
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.55
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.55
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.55
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.55
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.465
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.485
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.485
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.485
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.485
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 8.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.55
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.58
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.64
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.69
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.735
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.76
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.8875,
       "honest_accuracy": 0.885,
       "n_pseudo": 400,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "love",
        "pred": "joy",
        "count": 15,
        "share_of_errors": 0.32608695652173914
       },
       {
        "gold": "anger",
        "pred": "sadness",
        "count": 5,
        "share_of_errors": 0.10869565217391304
       },
       {
        "gold": "fear",
        "pred": "joy",
        "count": 5,
        "share_of_errors": 0.10869565217391304
       },
       {
        "gold": "fear",
        "pred": "sadness",
        "count": 4,
        "share_of_errors": 0.08695652173913043
       },
       {
        "gold": "surprise",
        "pred": "joy",
        "count": 4,
        "share_of_errors": 0.08695652173913043
       },
       {
        "gold": "fear",
        "pred": "anger",
        "count": 2,
        "share_of_errors": 0.043478260869565216
       },
       {
        "gold": "joy",
        "pred": "sadness",
        "count": 2,
        "share_of_errors": 0.043478260869565216
       },
       {
        "gold": "joy",
        "pred": "surprise",
        "count": 2,
        "share_of_errors": 0.043478260869565216
       },
       {
        "gold": "sadness",
        "pred": "joy",
        "count": 2,
        "share_of_errors": 0.043478260869565216
       },
       {
        "gold": "surprise",
        "pred": "fear",
        "count": 2,
        "share_of_errors": 0.043478260869565216
       },
       {
        "gold": "anger",
        "pred": "joy",
        "count": 1,
        "share_of_errors": 0.021739130434782608
       },
       {
        "gold": "anger",
        "pred": "surprise",
        "count": 1,
        "share_of_errors": 0.021739130434782608
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.7589061298966407
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.5274203329533338
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.7589061298966407
        }
       ]
      },
      "corpus_digest": "fnv1a64-d8c5a7db33864fb4",
      "cases_digest": "fnv1a64-6feefd1c947dd008",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 400,
        "accuracy": 0.5925,
        "macro_f1": 0.47086603548842415,
        "ece": 0.3087865,
        "brier": 0.6962649922250027,
        "nll": 2.1840444269226102,
        "aurc": 0.2657102861338114,
        "mean_confidence": 0.9012864999999995,
        "acc_at_50_coverage": 0.715,
        "acc_at_80_coverage": 0.6375
       },
       "readout_ece": 0.25708974999999984,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 11.0,
       "latency_p99_ms": 14.0,
       "latency_tail_support": 5,
       "latency_extremes": {
        "first_ms": 11.0,
        "max_ms": 14.0,
        "argmax_case": 7
       },
       "determinism_ok": true,
       "seconds": 6.9131029,
       "n_cases": 400,
       "n_questions": 400,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-6feefd1c947dd008",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 400,
       "accuracy": 0.295,
       "macro_f1": 0.09440569730143315,
       "ece": 0.37592108093202115,
       "brier": 1.0979565177781476,
       "nll": 4.844648455500639,
       "aurc": 0.6125397107064762,
       "mean_confidence": 0.6706122183054686,
       "acc_at_50_coverage": 0.375,
       "acc_at_80_coverage": 0.325
      },
      "readout_ece": 0.30973466217517853,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 35.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 81.0,
       "max_ms": 81.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 12.8509741,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.295,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 400,
       "accuracy": 0.565,
       "macro_f1": 0.5279099010442294,
       "ece": 0.14034680133039706,
       "brier": 0.629630034182988,
       "nll": 1.4156947177261885,
       "aurc": 0.33930307457825565,
       "mean_confidence": 0.6356044120168679,
       "acc_at_50_coverage": 0.665,
       "acc_at_80_coverage": 0.609375
      },
      "readout_ece": 0.14034680133039706,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 22.0,
      "latency_p99_ms": 26.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 100.0,
       "max_ms": 100.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 20.4590679,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 400,
       "accuracy": 0.4225,
       "macro_f1": 0.38441935811362554,
       "ece": 0.05342503041028977,
       "brier": 0.6891708084362554,
       "nll": 1.4363122193401237,
       "aurc": 0.4332617475189119,
       "mean_confidence": 0.43925078079104424,
       "acc_at_50_coverage": 0.56,
       "acc_at_80_coverage": 0.478125
      },
      "readout_ece": 0.05342503041028977,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 28.0,
      "latency_p99_ms": 36.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 40.0,
       "max_ms": 47.0,
       "argmax_case": 1
      },
      "determinism_ok": false,
      "seconds": 12.1988251,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "paw_local": {
      "lane": "paw (local)",
      "model": "paw-ft-bs48-20260530",
      "posture": "local-subprocess",
      "program_id": "703c4b3c3acf668f8ad8",
      "spec_file": "scripts/paw_specs\\emotion.txt",
      "spec_blake3": "ab8a01c6e1d96ca06ce9d40c5c57bd491e7ab3d53b18a8b3e7d51ac8a794baf4",
      "compile_cache_hit": true,
      "compile_wall_s": 11.6640664,
      "n_cases": 400,
      "n_questions": 400,
      "n_answered": 399,
      "refusals": 1,
      "quote_stripped": 0,
      "accuracy": 0.4875,
      "answered_accuracy": 0.48872180451127817,
      "refusal_rate": 0.0025,
      "score_mae_answered": null,
      "within_1_answered": null,
      "refusal_samples": [
       "sleepiness"
      ],
      "latency_p50_ms": 113.736,
      "latency_p99_ms": 213.371,
      "latency_tail_support": 5,
      "server_latency_p50_ms": null,
      "determinism_ok": true,
      "seconds": 54.6759834,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 400,
       "accuracy": 0.595,
       "macro_f1": 0.48767525539925655,
       "ece": 0.2124154042452574,
       "brier": 0.6033856957178485,
       "nll": 1.3950646286483295,
       "aurc": 0.22055689638200907,
       "mean_confidence": 0.8071955532580614,
       "acc_at_50_coverage": 0.775,
       "acc_at_80_coverage": 0.646875
      },
      "readout_ece": 0.11384338745045525,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 50.0,
      "latency_p99_ms": 148.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 50.0,
       "max_ms": 183.0,
       "argmax_case": 247
      },
      "determinism_ok": true,
      "seconds": 26.1094616,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-6feefd1c947dd008",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 400,
    "exact": 0,
    "near": 0
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "703c4b3c3acf668f8ad8",
    "spec_file": "scripts/paw_specs/emotion.txt",
    "spec_blake3": "ab8a01c6e1d96ca06ce9d40c5c57bd491e7ab3d53b18a8b3e7d51ac8a794baf4",
    "compile_cache_hit": false,
    "compile_wall_s": 0.924536416,
    "n_cases": 400,
    "n_questions": 400,
    "n_answered": 400,
    "refusals": 0,
    "quote_stripped": 0,
    "accuracy": 0.5,
    "answered_accuracy": 0.5,
    "refusal_rate": 0.0,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [],
    "server_latency_p50_ms": 63.3,
    "determinism_ok": true
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-6feefd1c947dd008"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-6feefd1c947dd008"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-6feefd1c947dd008"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-6feefd1c947dd008"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A0",
    "hard": {
     "n": 400,
     "accuracy": 0.885,
     "ece": 0.8622394363582135,
     "mean_confidence": 0.022760563641786576,
     "acc_at_50_coverage": 0.97
    },
    "consult_rate": 0.0,
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": 0.115,
    "latency_p99_ms": 0.129,
    "latency_tail_support": 5,
    "serves": "A0",
    "gate": "served by the reflex half \u2014 the best measured arm on this suite",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 400,
     "accuracy": 0.59,
     "macro_f1": 0.4844213484626461,
     "ece": 0.21647185690701007,
     "brier": 0.602934554245623,
     "nll": 1.3914321551837918,
     "aurc": 0.21994493595638656,
     "mean_confidence": 0.8064718569070101,
     "acc_at_50_coverage": 0.775,
     "acc_at_80_coverage": 0.65
    },
    "readout_ece": 0.111127159642895,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 78.0,
    "latency_p99_ms": 149.0,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 75.0,
     "max_ms": 163.0,
     "argmax_case": 378
    },
    "determinism_ok": true,
    "seconds": 38.124298167,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-6feefd1c947dd008",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 400,
     "accuracy": 0.5825,
     "macro_f1": 0.4664338387697607,
     "ece": 0.15611889574676752,
     "brier": 0.6065817658058879,
     "nll": 1.4556839894437805,
     "aurc": 0.23935113232761343,
     "mean_confidence": 0.7376930480077862,
     "acc_at_50_coverage": 0.725,
     "acc_at_80_coverage": 0.64375
    },
    "readout_ece": 0.15611889574676752,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.2975,
    "jdi_skill": 0.40569395017793597,
    "cases_digest": "fnv1a64-6feefd1c947dd008",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A0",
    "hard": {
     "n": 400,
     "accuracy": 0.885,
     "ece": 0.8622394363582135,
     "mean_confidence": 0.022760563641786576,
     "acc_at_50_coverage": 0.97
    },
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": 0.141,
    "latency_p99_ms": 0.281,
    "latency_tail_support": 5,
    "serves": "tier-fallback",
    "served_by": "Instinct (A0)",
    "fallback_note": "screened \u2014 no head earned: 6 gold-only fits (the 5-seed T7 sweep + 1 fresh seed) all refused on holdout; the encoder reference reads 0.5950 vs the incumbent 0.8850 (instinct issue 016 T7/T8, riir-train t599/t7 + t8 record)",
    "derived": true,
    "source_run": {
     "git_sha": "8bbff09",
     "date_utc": "2026-09-28T10:13:40Z"
    }
   }
  },
  {
   "name": "sst5",
   "n_cases": 600,
   "n_questions": 600,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 600,
     "accuracy": 0.39666666666666667,
     "macro_f1": 0.32101890185812104,
     "ece": 0.13891353478034335,
     "brier": 0.7604214524982581,
     "nll": 1.5156292710507784,
     "aurc": 0.5364704218034313,
     "mean_confidence": 0.2577531318863233,
     "acc_at_50_coverage": 0.47,
     "acc_at_80_coverage": 0.42083333333333334
    },
    "readout_ece": 0.10858877512315908,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.5316666666666666,
     "selective_accuracy": 0.40569395017793597,
     "selective_n": 281
    },
    "readout_ece_raw": 0.3849011408289273,
    "readout_ece_calibrated": 0.10858877512315908,
    "floor_ece": 0.22199004975124378,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": 1.1170251806701224,
    "within_1": 0.54,
    "latency_p50_ms": 0.094,
    "latency_p99_ms": 0.11,
    "latency_tail_support": 7,
    "latency_extremes": {
     "first_ms": 0.102,
     "max_ms": 0.135,
     "argmax_case": 482
    },
    "determinism_ok": true,
    "seconds": 4.271112708,
    "n_cases": 600,
    "n_questions": 600,
    "score_threshold": 0.28131148,
    "distance_threshold": 0.41663086,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.28131148,
      "status": "fitted",
      "targetAccuracy": 0.27142859,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.41663086,
      "status": "fitted",
      "targetAccuracy": 0.30714285,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 16.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.24
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.33
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.355
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.36
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.315
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.34
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.34
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.335
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.335
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.36
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.355
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.36
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.37
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.39
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.4
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.39666666666666667,
     "honest_accuracy": 0.39666666666666667,
     "n_pseudo": 0,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "very positive",
      "pred": "positive",
      "count": 62,
      "share_of_errors": 0.1712707182320442
     },
     {
      "gold": "very negative",
      "pred": "negative",
      "count": 49,
      "share_of_errors": 0.13535911602209943
     },
     {
      "gold": "neutral",
      "pred": "negative",
      "count": 47,
      "share_of_errors": 0.1298342541436464
     },
     {
      "gold": "positive",
      "pred": "negative",
      "count": 36,
      "share_of_errors": 0.09944751381215469
     },
     {
      "gold": "negative",
      "pred": "positive",
      "count": 28,
      "share_of_errors": 0.07734806629834254
     },
     {
      "gold": "neutral",
      "pred": "positive",
      "count": 28,
      "share_of_errors": 0.07734806629834254
     },
     {
      "gold": "very positive",
      "pred": "negative",
      "count": 24,
      "share_of_errors": 0.06629834254143646
     },
     {
      "gold": "negative",
      "pred": "neutral",
      "count": 17,
      "share_of_errors": 0.04696132596685083
     },
     {
      "gold": "positive",
      "pred": "very positive",
      "count": 17,
      "share_of_errors": 0.04696132596685083
     },
     {
      "gold": "very negative",
      "pred": "positive",
      "count": 11,
      "share_of_errors": 0.03038674033149171
     },
     {
      "gold": "very negative",
      "pred": "neutral",
      "count": 10,
      "share_of_errors": 0.027624309392265192
     },
     {
      "gold": "very positive",
      "pred": "neutral",
      "count": 9,
      "share_of_errors": 0.024861878453038673
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.2754612743854523
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.04647017426788808
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.2754612743854523
      }
     ]
    },
    "corpus_digest": "fnv1a64-726f717634ca24fb",
    "cases_digest": "fnv1a64-ad64a6cb0039e916",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 600,
      "accuracy": 0.37166666666666665,
      "macro_f1": 0.32915341205482057,
      "ece": 0.2477898333333333,
      "brier": 0.8171809703500024,
      "nll": 1.729144921851138,
      "aurc": 0.5279207298357831,
      "mean_confidence": 0.6194564999999997,
      "acc_at_50_coverage": 0.45666666666666667,
      "acc_at_80_coverage": 0.4041666666666667
     },
     "readout_ece": 0.08109283333333335,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 18.0,
     "latency_p99_ms": 24.0,
     "latency_tail_support": 7,
     "latency_extremes": {
      "first_ms": 14.0,
      "max_ms": 26.0,
      "argmax_case": 19
     },
     "determinism_ok": true,
     "seconds": 14.787980458,
     "n_cases": 600,
     "n_questions": 600,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-ad64a6cb0039e916",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 600,
      "accuracy": 0.37166666666666665,
      "macro_f1": 0.32915341205482057,
      "ece": 0.24778999999999993,
      "brier": 0.8171821949666691,
      "nll": 1.7291473733884497,
      "aurc": 0.5279207298357831,
      "mean_confidence": 0.6194566666666664,
      "acc_at_50_coverage": 0.45666666666666667,
      "acc_at_80_coverage": 0.4041666666666667
     },
     "readout_ece": 0.0810926666666667,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 27.0,
     "latency_p99_ms": 63.0,
     "latency_tail_support": 7,
     "latency_extremes": {
      "first_ms": 16.0,
      "max_ms": 70.0,
      "argmax_case": 85
     },
     "determinism_ok": true,
     "seconds": 23.306305667,
     "n_cases": 600,
     "n_questions": 600,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-ad64a6cb0039e916",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 600,
        "accuracy": 0.37333333333333335,
        "macro_f1": 0.33034260369116886,
        "ece": 0.24612783333333332,
        "brier": 0.8179473067833364,
        "nll": 1.7314241878126222,
        "aurc": 0.5289789146967602,
        "mean_confidence": 0.6194611666666664,
        "acc_at_50_coverage": 0.4533333333333333,
        "acc_at_80_coverage": 0.4041666666666667
       },
       "readout_ece": 0.08072583333333337,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 23.0,
       "latency_p99_ms": 27.0,
       "latency_tail_support": 7,
       "latency_extremes": {
        "first_ms": 23.0,
        "max_ms": 27.0,
        "argmax_case": 1
       },
       "determinism_ok": true,
       "seconds": 17.972210625,
       "n_cases": 600,
       "n_questions": 600,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-ad64a6cb0039e916",
       "latency_quotable": true,
       "source_run": {
        "git_sha": "2f6b58c",
        "date_utc": "2026-09-30T10:35:35Z"
       }
      }
     }
    },
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 600,
       "accuracy": 0.39666666666666667,
       "macro_f1": 0.32101890185812104,
       "ece": 0.13891353500386078,
       "brier": 0.7604214525872166,
       "nll": 1.515629271034501,
       "aurc": 0.5364704218034313,
       "mean_confidence": 0.2577531316628059,
       "acc_at_50_coverage": 0.47,
       "acc_at_80_coverage": 0.42083333333333334
      },
      "readout_ece": 0.10858877954383692,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.5316666666666666,
       "selective_accuracy": 0.40569395017793597,
       "selective_n": 281
      },
      "readout_ece_raw": 0.38490114152431476,
      "readout_ece_calibrated": 0.10858877954383692,
      "floor_ece": 0.22199004975124378,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": 1.1170251805707812,
      "within_1": 0.54,
      "latency_p50_ms": 0.146,
      "latency_p99_ms": 0.401,
      "latency_tail_support": 7,
      "latency_extremes": {
       "first_ms": 0.156,
       "max_ms": 1.34,
       "argmax_case": 381
      },
      "determinism_ok": true,
      "seconds": 7.703255,
      "n_cases": 600,
      "n_questions": 600,
      "score_threshold": 0.2813115,
      "distance_threshold": 0.41663086,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.2813115,
        "status": "fitted",
        "targetAccuracy": 0.27142859,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.41663086,
        "status": "fitted",
        "targetAccuracy": 0.30714285,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 16.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.24
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.33
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.355
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.36
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.315
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.34
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.34
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.335
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.335
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.36
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.355
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.36
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.37
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.39
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.4
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.39666666666666667,
       "honest_accuracy": 0.39666666666666667,
       "n_pseudo": 0,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "very positive",
        "pred": "positive",
        "count": 62,
        "share_of_errors": 0.1712707182320442
       },
       {
        "gold": "very negative",
        "pred": "negative",
        "count": 49,
        "share_of_errors": 0.13535911602209943
       },
       {
        "gold": "neutral",
        "pred": "negative",
        "count": 47,
        "share_of_errors": 0.1298342541436464
       },
       {
        "gold": "positive",
        "pred": "negative",
        "count": 36,
        "share_of_errors": 0.09944751381215469
       },
       {
        "gold": "negative",
        "pred": "positive",
        "count": 28,
        "share_of_errors": 0.07734806629834254
       },
       {
        "gold": "neutral",
        "pred": "positive",
        "count": 28,
        "share_of_errors": 0.07734806629834254
       },
       {
        "gold": "very positive",
        "pred": "negative",
        "count": 24,
        "share_of_errors": 0.06629834254143646
       },
       {
        "gold": "negative",
        "pred": "neutral",
        "count": 17,
        "share_of_errors": 0.04696132596685083
       },
       {
        "gold": "positive",
        "pred": "very positive",
        "count": 17,
        "share_of_errors": 0.04696132596685083
       },
       {
        "gold": "very negative",
        "pred": "positive",
        "count": 11,
        "share_of_errors": 0.03038674033149171
       },
       {
        "gold": "very negative",
        "pred": "neutral",
        "count": 10,
        "share_of_errors": 0.027624309392265192
       },
       {
        "gold": "very positive",
        "pred": "neutral",
        "count": 9,
        "share_of_errors": 0.024861878453038673
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.27546127498149875
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.046470174267888076
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.27546127498149875
        }
       ]
      },
      "corpus_digest": "fnv1a64-726f717634ca24fb",
      "cases_digest": "fnv1a64-ad64a6cb0039e916",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 600,
        "accuracy": 0.37166666666666665,
        "macro_f1": 0.32915341205482057,
        "ece": 0.24779033333333325,
        "brier": 0.8171819570333358,
        "nll": 1.7291456252344222,
        "aurc": 0.5279207298357831,
        "mean_confidence": 0.6194569999999998,
        "acc_at_50_coverage": 0.45666666666666667,
        "acc_at_80_coverage": 0.4041666666666667
       },
       "readout_ece": 0.08109316666666669,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 12.0,
       "latency_p99_ms": 17.0,
       "latency_tail_support": 7,
       "latency_extremes": {
        "first_ms": 11.0,
        "max_ms": 20.0,
        "argmax_case": 80
       },
       "determinism_ok": true,
       "seconds": 10.109242,
       "n_cases": 600,
       "n_questions": 600,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-ad64a6cb0039e916",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 600,
       "accuracy": 0.235,
       "macro_f1": 0.16563654685925427,
       "ece": 0.5211544505258401,
       "brier": 1.2170341258480324,
       "nll": 3.957467651071437,
       "aurc": 0.75573835807425,
       "mean_confidence": 0.7561544505258401,
       "acc_at_50_coverage": 0.22666666666666666,
       "acc_at_80_coverage": 0.21875
      },
      "readout_ece": 0.46188401788473127,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 36.0,
      "latency_tail_support": 7,
      "latency_extremes": {
       "first_ms": 80.0,
       "max_ms": 80.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 19.3249548,
      "n_cases": 600,
      "n_questions": 600,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.2353923205342237,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 600,
       "accuracy": 0.43833333333333335,
       "macro_f1": 0.4179638590590876,
       "ece": 0.07646240903830624,
       "brier": 0.6840256496462226,
       "nll": 1.2902129384345649,
       "aurc": 0.4884865918387397,
       "mean_confidence": 0.49827637203100417,
       "acc_at_50_coverage": 0.5333333333333333,
       "acc_at_80_coverage": 0.47291666666666665
      },
      "readout_ece": 0.07646240903830624,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 22.0,
      "latency_p99_ms": 27.0,
      "latency_tail_support": 7,
      "latency_extremes": {
       "first_ms": 104.0,
       "max_ms": 104.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 27.1875299,
      "n_cases": 600,
      "n_questions": 600,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 600,
       "accuracy": 0.43833333333333335,
       "macro_f1": 0.33147787742701995,
       "ece": 0.012682029257218045,
       "brier": 0.6749968198070849,
       "nll": 1.288419340297777,
       "aurc": 0.4928660617132435,
       "mean_confidence": 0.43408097142974533,
       "acc_at_50_coverage": 0.5,
       "acc_at_80_coverage": 0.4625
      },
      "readout_ece": 0.012682029257218045,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 38.0,
      "latency_tail_support": 7,
      "latency_extremes": {
       "first_ms": 30.0,
       "max_ms": 53.0,
       "argmax_case": 1
      },
      "determinism_ok": false,
      "seconds": 19.880092,
      "n_cases": 600,
      "n_questions": 600,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "paw_local": {
      "lane": "paw (local)",
      "model": "paw-ft-bs48-20260530",
      "posture": "local-subprocess",
      "program_id": "a0fb83a0afc58a1ce2b9",
      "spec_file": "scripts/paw_specs\\sst5.txt",
      "spec_blake3": "03ec007949af34227f6e88bc5d008ac7c6825f6fde9f6aed344aa643493fa71f",
      "compile_cache_hit": true,
      "compile_wall_s": 11.6851636,
      "n_cases": 600,
      "n_questions": 600,
      "n_answered": 600,
      "refusals": 0,
      "quote_stripped": 0,
      "accuracy": 0.4166666666666667,
      "answered_accuracy": 0.4166666666666667,
      "refusal_rate": 0.0,
      "score_mae_answered": 0.86,
      "within_1_answered": 0.8166666666666667,
      "refusal_samples": [],
      "latency_p50_ms": 118.293,
      "latency_p99_ms": 210.725,
      "latency_tail_support": 7,
      "server_latency_p50_ms": null,
      "determinism_ok": true,
      "seconds": 78.9310791,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 600,
       "accuracy": 0.43333333333333335,
       "macro_f1": 0.41443478670275447,
       "ece": 0.17377723559737204,
       "brier": 0.6929953722067029,
       "nll": 1.3198731713128236,
       "aurc": 0.4601873382005463,
       "mean_confidence": 0.60707742040356,
       "acc_at_50_coverage": 0.5233333333333333,
       "acc_at_80_coverage": 0.4625
      },
      "readout_ece": 0.08253375576529452,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 59.0,
      "latency_p99_ms": 158.0,
      "latency_tail_support": 7,
      "latency_extremes": {
       "first_ms": 167.0,
       "max_ms": 281.0,
       "argmax_case": 208
      },
      "determinism_ok": true,
      "seconds": 45.1860115,
      "n_cases": 600,
      "n_questions": 600,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-ad64a6cb0039e916",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 600,
    "exact": 1,
    "near": 0
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "a0fb83a0afc58a1ce2b9",
    "spec_file": "scripts/paw_specs/sst5.txt",
    "spec_blake3": "03ec007949af34227f6e88bc5d008ac7c6825f6fde9f6aed344aa643493fa71f",
    "compile_cache_hit": false,
    "compile_wall_s": 0.942867667,
    "n_cases": 600,
    "n_questions": 600,
    "n_answered": 600,
    "refusals": 0,
    "quote_stripped": 0,
    "accuracy": 0.395,
    "answered_accuracy": 0.395,
    "refusal_rate": 0.0,
    "score_mae_answered": 0.8833333333333333,
    "within_1_answered": 0.8183333333333334,
    "refusal_samples": [],
    "server_latency_p50_ms": 60.9,
    "determinism_ok": true
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-ad64a6cb0039e916"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-ad64a6cb0039e916"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-ad64a6cb0039e916"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-ad64a6cb0039e916"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A1",
    "hard": {
     "n": 600,
     "accuracy": 0.4216666666666667,
     "ece": 0.10076249030729133,
     "mean_confidence": 0.4711805712431669,
     "acc_at_50_coverage": 0.5
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.000833,
    "latency_p99_ms": 0.001792,
    "latency_tail_support": 8,
    "serves": "A1",
    "gate": "best measured, T2-uncertified (paired LB95 -0.0129) \u2014 certification is more questions, not a posture rollback",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 600,
     "accuracy": 0.43,
     "macro_f1": 0.41049715078874643,
     "ece": 0.17805990467468896,
     "brier": 0.6935595157268799,
     "nll": 1.321830326901058,
     "aurc": 0.4612050619400724,
     "mean_confidence": 0.608059904674689,
     "acc_at_50_coverage": 0.53,
     "acc_at_80_coverage": 0.46458333333333335
    },
    "readout_ece": 0.09205816870891721,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 203.0,
    "latency_p99_ms": 283.0,
    "latency_tail_support": 7,
    "latency_extremes": {
     "first_ms": 114.0,
     "max_ms": 300.0,
     "argmax_case": 359
    },
    "determinism_ok": true,
    "seconds": 124.272431334,
    "n_cases": 600,
    "n_questions": 600,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-ad64a6cb0039e916",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-t6s0 (laya-english encoder + NLEH v1 head)",
    "hard": {
     "n": 600,
     "accuracy": 0.5266666666666666
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 14.708917,
    "latency_p99_ms": 17.452375,
    "serves": "\u2717 (encoder class refused at serve \u2014 A1 serves; instinct issue 014 C1)",
    "gate": "T2-certified above the incumbent A1 (paired LB95 +0.0535, mean +0.1050) \u2014 serve REFUSED on the latency class (14.7 ms/row vs the ~0.3 ms provisional bar; issue 014 decision 1)",
    "device": "metal",
    "record_only": true,
    "source_run": {
     "git_sha": "211c68d",
     "date_utc": "2026-09-30T02:01:58Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 600,
     "accuracy": 0.455,
     "macro_f1": 0.30583133122619166,
     "ece": 0.1911103808383147,
     "brier": 0.6737527244270393,
     "nll": 1.1754999466411433,
     "aurc": 0.48222009446122455,
     "mean_confidence": 0.639093052794536,
     "acc_at_50_coverage": 0.51,
     "acc_at_80_coverage": 0.46458333333333335
    },
    "readout_ece": 0.1911103808383147,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 600,
    "n_questions": 600,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.2733333333333333,
    "jdi_skill": 0.25000000000000006,
    "cases_digest": "fnv1a64-ad64a6cb0039e916",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   }
  },
  {
   "name": "prompt_injections",
   "n_cases": 116,
   "n_questions": 116,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 116,
     "accuracy": 0.7672413793103449,
     "macro_f1": 0.7642808760442538,
     "ece": 0.13358817981748744,
     "brier": 0.33647229928899053,
     "nll": 0.5194112460029578,
     "aurc": 0.09487560056959717,
     "mean_confidence": 0.6544157581339622,
     "acc_at_50_coverage": 0.896551724137931,
     "acc_at_80_coverage": 0.8369565217391305
    },
    "readout_ece": 0.09491884348721336,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.6810344827586207,
     "selective_accuracy": 0.9459459459459459,
     "selective_n": 37
    },
    "readout_ece_raw": 0.674452191282963,
    "readout_ece_calibrated": 0.09491884348721336,
    "floor_ece": 0.3622396722430863,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.076,
    "latency_p99_ms": 0.086,
    "latency_tail_support": 2,
    "latency_extremes": {
     "first_ms": 0.084,
     "max_ms": 0.087,
     "argmax_case": 107
    },
    "determinism_ok": true,
    "seconds": 1.566039125,
    "n_cases": 116,
    "n_questions": 116,
    "score_threshold": 0.7991401,
    "distance_threshold": 0.46189553,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.7991401,
      "status": "fitted",
      "targetAccuracy": 0.8,
      "support": {
       "n": 100,
       "nPass": 70,
       "nAbstain": 30
      }
     },
     "distance": {
      "threshold": 0.46189553,
      "status": "fitted",
      "targetAccuracy": 0.78571427,
      "support": {
       "n": 100,
       "nPass": 70,
       "nAbstain": 30
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 1.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": 1,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.5
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.39
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.5
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.49
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.49
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.49
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": 0,
       "view": "bag",
       "cal_acc": 0.49
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.61
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.61
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.61
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.61
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.61
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.51
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.51
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.51
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.51
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": 1,
       "view": "bag",
       "cal_acc": 0.51
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.7672413793103449,
     "honest_accuracy": 0.7672413793103449,
     "n_pseudo": 0,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.6830641734600067
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.1164907401800156
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.6830641734600067
      }
     ]
    },
    "corpus_digest": "fnv1a64-91324bedb87d226f",
    "cases_digest": "fnv1a64-5080803ec29f2423",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    },
    "latency_provenance": {
     "note": "latency cells carried from the host's incumbent run; accuracy is this lane's own (LANE-CARRY, Issue 032)"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 116,
      "accuracy": 0.6982758620689655,
      "macro_f1": 0.6750700280112045,
      "ece": 0.2620396551724138,
      "brier": 0.5259102800000001,
      "nll": 3.1496688651163907,
      "aurc": 0.10727341980747937,
      "mean_confidence": 0.9603155172413793,
      "acc_at_50_coverage": 0.9655172413793104,
      "acc_at_80_coverage": 0.7717391304347826
     },
     "readout_ece": 0.2620396551724138,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 18.0,
     "latency_p99_ms": 37.0,
     "latency_tail_support": 2,
     "latency_extremes": {
      "first_ms": 22.0,
      "max_ms": 38.0,
      "argmax_case": 37
     },
     "determinism_ok": true,
     "seconds": 6.029472583,
     "n_cases": 116,
     "n_questions": 116,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-5080803ec29f2423",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 116,
      "accuracy": 0.6982758620689655,
      "macro_f1": 0.6750700280112045,
      "ece": 0.2620387931034483,
      "brier": 0.5259077681034483,
      "nll": 3.149665690493024,
      "aurc": 0.10727341980747937,
      "mean_confidence": 0.9603146551724138,
      "acc_at_50_coverage": 0.9655172413793104,
      "acc_at_80_coverage": 0.7717391304347826
     },
     "readout_ece": 0.2620387931034483,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 30.0,
     "latency_p99_ms": 73.0,
     "latency_tail_support": 2,
     "latency_extremes": {
      "first_ms": 30.0,
      "max_ms": 84.0,
      "argmax_case": 84
     },
     "determinism_ok": true,
     "seconds": 10.115332209,
     "n_cases": 116,
     "n_questions": 116,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-5080803ec29f2423",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 107,
        "accuracy": 0.6915887850467289,
        "macro_f1": 0.6522210184182016,
        "ece": 0.2668280373831776,
        "brier": 0.5192283248598133,
        "nll": 2.8819367108393252,
        "aurc": 0.10464502378900169,
        "mean_confidence": 0.9584168224299067,
        "acc_at_50_coverage": 0.9811320754716981,
        "acc_at_80_coverage": 0.7764705882352941
       },
       "readout_ece": 0.26682803738317756,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 23.0,
       "latency_p99_ms": 30.0,
       "latency_tail_support": 2,
       "latency_extremes": {
        "first_ms": 23.0,
        "max_ms": 31.0,
        "argmax_case": 40
       },
       "determinism_ok": true,
       "seconds": 6.534780375,
       "n_cases": 107,
       "n_questions": 107,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-5080803ec29f2423",
       "latency_quotable": true,
       "source_run": {
        "git_sha": "2f6b58c",
        "date_utc": "2026-09-30T10:35:35Z"
       }
      }
     }
    },
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 116,
       "accuracy": 0.7672413793103449,
       "macro_f1": 0.7642808760442538,
       "ece": 0.13358817981748744,
       "brier": 0.33647229928899053,
       "nll": 0.5194112460029578,
       "aurc": 0.09487560056959717,
       "mean_confidence": 0.6544157581339622,
       "acc_at_50_coverage": 0.896551724137931,
       "acc_at_80_coverage": 0.8369565217391305
      },
      "readout_ece": 0.09491884348721336,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.6810344827586207,
       "selective_accuracy": 0.9459459459459459,
       "selective_n": 37
      },
      "readout_ece_raw": 0.674452191282963,
      "readout_ece_calibrated": 0.09491884348721336,
      "floor_ece": 0.3622396722430863,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.048,
      "latency_p99_ms": 0.089,
      "latency_tail_support": 2,
      "latency_extremes": {
       "first_ms": 0.033,
       "max_ms": 0.089,
       "argmax_case": 45
      },
      "determinism_ok": true,
      "seconds": 0.0193963,
      "n_cases": 116,
      "n_questions": 116,
      "score_threshold": 0.7991401,
      "distance_threshold": 0.46189553,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.7991401,
        "status": "fitted",
        "targetAccuracy": 0.8,
        "support": {
         "n": 100,
         "nPass": 70,
         "nAbstain": 30
        }
       },
       "distance": {
        "threshold": 0.46189553,
        "status": "fitted",
        "targetAccuracy": 0.78571427,
        "support": {
         "n": 100,
         "nPass": 70,
         "nAbstain": 30
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 1.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": 1,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.5
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.39
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.5
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.49
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.49
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.49
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": 0,
         "view": "bag",
         "cal_acc": 0.49
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.61
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.61
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.61
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.61
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.61
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.51
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.51
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.51
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.51
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": 1,
         "view": "bag",
         "cal_acc": 0.51
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.7672413793103449,
       "honest_accuracy": 0.7672413793103449,
       "n_pseudo": 0,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.6830641734600067
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.1164907401800156
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.6830641734600067
        }
       ]
      },
      "corpus_digest": "fnv1a64-91324bedb87d226f",
      "cases_digest": "fnv1a64-5080803ec29f2423",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      },
      "latency_provenance": {
       "note": "latency cells carried from the host's incumbent run; accuracy is this lane's own (LANE-CARRY, Issue 032)"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 116,
        "accuracy": 0.6982758620689655,
        "macro_f1": 0.6750700280112045,
        "ece": 0.26203793103448275,
        "brier": 0.5259087506896553,
        "nll": 3.1496668960995784,
        "aurc": 0.10727341980747937,
        "mean_confidence": 0.9603137931034482,
        "acc_at_50_coverage": 0.9655172413793104,
        "acc_at_80_coverage": 0.7717391304347826
       },
       "readout_ece": 0.26203793103448275,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 12.0,
       "latency_p99_ms": 18.0,
       "latency_tail_support": 2,
       "latency_extremes": {
        "first_ms": 16.0,
        "max_ms": 37.0,
        "argmax_case": 37
       },
       "determinism_ok": true,
       "seconds": 3.6467929999999997,
       "n_cases": 116,
       "n_questions": 116,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-5080803ec29f2423",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 116,
       "accuracy": 0.5344827586206896,
       "macro_f1": 0.37931034482758624,
       "ece": 0.24657633129892678,
       "brier": 0.5952651188638927,
       "nll": 0.8098890507578412,
       "aurc": 0.32781787446407185,
       "mean_confidence": 0.7686865329742432,
       "acc_at_50_coverage": 0.6379310344827587,
       "acc_at_80_coverage": 0.5543478260869565
      },
      "readout_ece": 0.15592285322731939,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 32.0,
      "latency_p99_ms": 38.0,
      "latency_tail_support": 2,
      "latency_extremes": {
       "first_ms": 85.0,
       "max_ms": 85.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 3.8986152,
      "n_cases": 116,
      "n_questions": 116,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.543859649122807,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 116,
       "accuracy": 0.6810344827586207,
       "macro_f1": 0.6529473599094364,
       "ece": 0.11716827177463737,
       "brier": 0.4537946656520319,
       "nll": 0.698787636455864,
       "aurc": 0.3443143397382513,
       "mean_confidence": 0.7570098172019883,
       "acc_at_50_coverage": 0.7586206896551724,
       "acc_at_80_coverage": 0.717391304347826
      },
      "readout_ece": 0.11716827177463737,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 22.0,
      "latency_p99_ms": 25.0,
      "latency_tail_support": 2,
      "latency_extremes": {
       "first_ms": 104.0,
       "max_ms": 104.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 17.8940805,
      "n_cases": 116,
      "n_questions": 116,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 116,
       "accuracy": 0.4827586206896552,
       "macro_f1": 0.32558139534883723,
       "ece": 0.3830812059278632,
       "brier": 0.7977241566862362,
       "nll": 1.1769246956335173,
       "aurc": 0.5471498253488163,
       "mean_confidence": 0.8658398266175183,
       "acc_at_50_coverage": 0.5172413793103449,
       "acc_at_80_coverage": 0.5
      },
      "readout_ece": 0.3830812059278632,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 28.0,
      "latency_p99_ms": 33.0,
      "latency_tail_support": 2,
      "latency_extremes": {
       "first_ms": 28.0,
       "max_ms": 52.0,
       "argmax_case": 37
      },
      "determinism_ok": false,
      "seconds": 4.0002558,
      "n_cases": 116,
      "n_questions": 116,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 116,
       "accuracy": 0.6293103448275862,
       "macro_f1": 0.5820695433598659,
       "ece": 0.21178160022125295,
       "brier": 0.4788486284339183,
       "nll": 0.6802795495321566,
       "aurc": 0.15917213216918957,
       "mean_confidence": 0.8337148148446055,
       "acc_at_50_coverage": 0.8620689655172413,
       "acc_at_80_coverage": 0.6847826086956522
      },
      "readout_ece": 0.21178160022125295,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 50.0,
      "latency_p99_ms": 118.0,
      "latency_tail_support": 2,
      "latency_extremes": {
       "first_ms": 74.0,
       "max_ms": 119.0,
       "argmax_case": 64
      },
      "determinism_ok": true,
      "seconds": 8.0035117,
      "n_cases": 116,
      "n_questions": 116,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-5080803ec29f2423",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 546,
    "n_eval": 116,
    "exact": 0,
    "near": 2
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-5080803ec29f2423"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-5080803ec29f2423"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-5080803ec29f2423"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-5080803ec29f2423"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A1",
    "hard": {
     "n": 116,
     "accuracy": 0.853448275862069,
     "ece": 0.0823830797754485,
     "mean_confidence": 0.8039373015535289,
     "acc_at_50_coverage": 0.9655172413793104
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.000417,
    "latency_p99_ms": 0.003,
    "latency_tail_support": 2,
    "serves": "A1",
    "gate": "certified (paired LB95 +0.0082, mean +0.0862)",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 116,
     "accuracy": 0.6293103448275862,
     "macro_f1": 0.5820695433598659,
     "ece": 0.20458659990530076,
     "brier": 0.47757539131823673,
     "nll": 0.6793071528875683,
     "aurc": 0.15964234121544751,
     "mean_confidence": 0.833896944732887,
     "acc_at_50_coverage": 0.8620689655172413,
     "acc_at_80_coverage": 0.6847826086956522
    },
    "readout_ece": 0.20458659990530076,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 101.0,
    "latency_p99_ms": 277.0,
    "latency_tail_support": 2,
    "latency_extremes": {
     "first_ms": 277.0,
     "max_ms": 284.0,
     "argmax_case": 8
    },
    "determinism_ok": true,
    "seconds": 17.660868166,
    "n_cases": 116,
    "n_questions": 116,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-5080803ec29f2423",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "d9cfba100af1e46f97cd",
    "spec_file": "scripts/paw_specs/prompt_injections.injection.txt",
    "spec_blake3": "13ef1026c9cecf47fc9b17a084855d57125af8862018ab6892ceda48275bd51a",
    "compile_cache_hit": false,
    "compile_wall_s": 77.575057042,
    "n_cases": 116,
    "n_questions": 116,
    "n_answered": 116,
    "refusals": 0,
    "quote_stripped": 0,
    "accuracy": 0.6379310344827587,
    "answered_accuracy": 0.6379310344827587,
    "refusal_rate": 0.0,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [],
    "server_latency_p50_ms": 76.6,
    "determinism_ok": true,
    "cases_digest": "fnv1a64-5080803ec29f2423",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "9dc33da",
     "date_utc": "2026-09-29T00:50:51Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 116,
     "accuracy": 0.5086206896551724,
     "macro_f1": 0.3789799943646098,
     "ece": 0.3121914329199955,
     "brier": 0.6778031947429162,
     "nll": 0.9820978261541736,
     "aurc": 0.40268594067501284,
     "mean_confidence": 0.820812122575168,
     "acc_at_50_coverage": 0.5862068965517241,
     "acc_at_80_coverage": 0.5543478260869565
    },
    "readout_ece": 0.3121914329199955,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 116,
    "n_questions": 116,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.5172413793103449,
    "jdi_skill": 0.0,
    "cases_digest": "fnv1a64-5080803ec29f2423",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A1",
    "hard": {
     "n": 116,
     "accuracy": 0.853448275862069,
     "ece": 0.0823830797754485,
     "mean_confidence": 0.8039373015535289,
     "acc_at_50_coverage": 0.9655172413793104
    },
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.0005420000000000001,
    "latency_p99_ms": 0.0039169999999999995,
    "latency_tail_support": 2,
    "serves": "tier-fallback",
    "served_by": "Instinct (A1)",
    "fallback_note": "best-of-family (owner 2026-10-02): the encoder arm's own measured read is 0.8017, 0.0517 under the served arm \u2014 refused at serve, so the served answer is shown; the refused arm's full record rides the cell (displaced_record)",
    "derived": true,
    "source_run": {
     "git_sha": "8bbff09",
     "date_utc": "2026-09-28T10:13:40Z"
    },
    "displaced_record": {
     "lane": "Rethink",
     "model": "ENC-prompt_encoder_v1 (laya-english encoder + NLEH v1 head)",
     "hard": {
      "n": 116,
      "accuracy": 0.8017241379310345
     },
     "consult_rate": 1.0,
     "latency_scope": "arm-only",
     "latency_rows": "questions",
     "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
     "gate": "acc 0.8017 (93/116 question rows) \u00b7 paired vs the incumbent A1: mean -0.0517 \u00b7 LB95 -0.1343 \u00b7 per-row p50 13.873 ms (metal device, laya-english). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017).",
     "device": "metal",
     "head_kind": "v1",
     "ckpt": "english",
     "shape_desc": "2 classes \u00b7 feat 3072",
     "record_only": true,
     "latency_quotable": true,
     "source_run": {
      "git_sha": "e0440e3",
      "date_utc": "2026-10-01T18:51:14Z"
     }
    }
   }
  },
  {
   "name": "xnli_en",
   "n_cases": 300,
   "n_questions": 300,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 300,
     "accuracy": 0.5233333333333333,
     "macro_f1": 0.5061129338341462,
     "ece": 0.13834670235713323,
     "brier": 0.6178344532049044,
     "nll": 1.0286514608348634,
     "aurc": 0.3119835082997881,
     "mean_confidence": 0.3849866309762001,
     "acc_at_50_coverage": 0.64,
     "acc_at_80_coverage": 0.5791666666666667
    },
    "readout_ece": 0.11191444128751758,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.54,
     "selective_accuracy": 0.6086956521739131,
     "selective_n": 138
    },
    "readout_ece_raw": 0.5133298230171204,
    "readout_ece_calibrated": 0.11191444128751758,
    "floor_ece": 0.14800995024875624,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.087,
    "latency_p99_ms": 0.103,
    "latency_tail_support": 4,
    "latency_extremes": {
     "first_ms": 0.102,
     "max_ms": 0.105,
     "argmax_case": 58
    },
    "determinism_ok": true,
    "seconds": 3.87077025,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": 0.55691224,
    "distance_threshold": 0.51104105,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.55691224,
      "status": "fitted",
      "targetAccuracy": 0.6357143,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.51104105,
      "status": "fitted",
      "targetAccuracy": 0.62142855,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 4.0,
     "selected_alpha": "fixed-1",
     "selected_noul_domain": null,
     "selected_view": "pair",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.365
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.31
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.34
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.325
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.325
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.32
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.31
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.33
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.325
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.33
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.33
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.56
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.55
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.55
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.55
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.55
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.56
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.58
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.57
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.565
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "pair",
       "cal_acc": 0.555
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.58
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.555
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.54
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.505
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.455
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.44
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.5033333333333333,
     "honest_accuracy": 0.5233333333333333,
     "n_pseudo": 300,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "entailment",
      "pred": "neutral",
      "count": 46,
      "share_of_errors": 0.32167832167832167
     },
     {
      "gold": "neutral",
      "pred": "contradiction",
      "count": 27,
      "share_of_errors": 0.1888111888111888
     },
     {
      "gold": "entailment",
      "pred": "contradiction",
      "count": 26,
      "share_of_errors": 0.18181818181818182
     },
     {
      "gold": "contradiction",
      "pred": "neutral",
      "count": 21,
      "share_of_errors": 0.14685314685314685
     },
     {
      "gold": "neutral",
      "pred": "entailment",
      "count": 18,
      "share_of_errors": 0.1258741258741259
     },
     {
      "gold": "contradiction",
      "pred": "entailment",
      "count": 5,
      "share_of_errors": 0.03496503496503497
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.5868099159002303
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.21351082310080524
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.5868099159002303
      }
     ]
    },
    "corpus_digest": "fnv1a64-7fb156d256a353c4",
    "cases_digest": "fnv1a64-b136750e5af1d412",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 300,
      "accuracy": 0.86,
      "macro_f1": 0.8611876188602533,
      "ece": 0.06853533333333349,
      "brier": 0.22044518360000007,
      "nll": 0.39874945916910204,
      "aurc": 0.0380700868516878,
      "mean_confidence": 0.920768666666667,
      "acc_at_50_coverage": 0.98,
      "acc_at_80_coverage": 0.9208333333333333
     },
     "readout_ece": 0.10435199999999993,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 19.0,
     "latency_p99_ms": 25.0,
     "latency_tail_support": 4,
     "latency_extremes": {
      "first_ms": 17.0,
      "max_ms": 27.0,
      "argmax_case": 13
     },
     "determinism_ok": true,
     "seconds": 9.95162,
     "n_cases": 300,
     "n_questions": 300,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-b136750e5af1d412",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 300,
      "accuracy": 0.86,
      "macro_f1": 0.8611876188602533,
      "ece": 0.06853533333333349,
      "brier": 0.2204451122333334,
      "nll": 0.39874945916910204,
      "aurc": 0.0380700868516878,
      "mean_confidence": 0.920768666666667,
      "acc_at_50_coverage": 0.98,
      "acc_at_80_coverage": 0.9208333333333333
     },
     "readout_ece": 0.10435099999999992,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 32.0,
     "latency_p99_ms": 71.0,
     "latency_tail_support": 4,
     "latency_extremes": {
      "first_ms": 23.0,
      "max_ms": 83.0,
      "argmax_case": 209
     },
     "determinism_ok": true,
     "seconds": 16.605432417,
     "n_cases": 300,
     "n_questions": 300,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-b136750e5af1d412",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 298,
        "accuracy": 0.8590604026845637,
        "macro_f1": 0.860324679891885,
        "ece": 0.06803288590604022,
        "brier": 0.22218071946308734,
        "nll": 0.4016795391289157,
        "aurc": 0.03854688249403715,
        "mean_confidence": 0.9204087248322141,
        "acc_at_50_coverage": 0.9798657718120806,
        "acc_at_80_coverage": 0.9159663865546218
       },
       "readout_ece": 0.10385805369127518,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 28.0,
       "latency_p99_ms": 32.0,
       "latency_tail_support": 3,
       "latency_extremes": {
        "first_ms": 31.0,
        "max_ms": 32.0,
        "argmax_case": 3
       },
       "determinism_ok": true,
       "seconds": 12.437068125,
       "n_cases": 298,
       "n_questions": 298,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-b136750e5af1d412",
       "latency_quotable": true,
       "source_run": {
        "git_sha": "2f6b58c",
        "date_utc": "2026-09-30T10:35:35Z"
       }
      }
     }
    },
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 300,
       "accuracy": 0.5233333333333333,
       "macro_f1": 0.5061129338341462,
       "ece": 0.13834670225779216,
       "brier": 0.6178344533893235,
       "nll": 1.0286514608348634,
       "aurc": 0.3119835082997881,
       "mean_confidence": 0.38498663107554115,
       "acc_at_50_coverage": 0.64,
       "acc_at_80_coverage": 0.5791666666666667
      },
      "readout_ece": 0.11191444128751758,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.54,
       "selective_accuracy": 0.6086956521739131,
       "selective_n": 138
      },
      "readout_ece_raw": 0.5133298230171204,
      "readout_ece_calibrated": 0.11191444128751758,
      "floor_ece": 0.14800995024875624,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.14,
      "latency_p99_ms": 0.393,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 0.282,
       "max_ms": 1.893,
       "argmax_case": 31
      },
      "determinism_ok": true,
      "seconds": 7.2329696,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": 0.55691224,
      "distance_threshold": 0.51104105,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.55691224,
        "status": "fitted",
        "targetAccuracy": 0.6357143,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.51104105,
        "status": "fitted",
        "targetAccuracy": 0.62142855,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 4.0,
       "selected_alpha": "fixed-1",
       "selected_noul_domain": null,
       "selected_view": "pair",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.365
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.31
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.34
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.325
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.325
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.32
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.31
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.33
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.325
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.33
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.33
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.56
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.55
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.55
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.55
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.55
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.56
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.58
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.57
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.565
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "pair",
         "cal_acc": 0.555
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.58
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.555
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.54
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.505
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.455
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.44
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.5033333333333333,
       "honest_accuracy": 0.5233333333333333,
       "n_pseudo": 300,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "entailment",
        "pred": "neutral",
        "count": 46,
        "share_of_errors": 0.32167832167832167
       },
       {
        "gold": "neutral",
        "pred": "contradiction",
        "count": 27,
        "share_of_errors": 0.1888111888111888
       },
       {
        "gold": "entailment",
        "pred": "contradiction",
        "count": 26,
        "share_of_errors": 0.18181818181818182
       },
       {
        "gold": "contradiction",
        "pred": "neutral",
        "count": 21,
        "share_of_errors": 0.14685314685314685
       },
       {
        "gold": "neutral",
        "pred": "entailment",
        "count": 18,
        "share_of_errors": 0.1258741258741259
       },
       {
        "gold": "contradiction",
        "pred": "entailment",
        "count": 5,
        "share_of_errors": 0.03496503496503497
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.5868099159002303
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.213510822802782
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.5868099159002303
        }
       ]
      },
      "corpus_digest": "fnv1a64-7fb156d256a353c4",
      "cases_digest": "fnv1a64-b136750e5af1d412",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 300,
        "accuracy": 0.86,
        "macro_f1": 0.8611876188602533,
        "ece": 0.06853533333333349,
        "brier": 0.22044494230000009,
        "nll": 0.39874945916910204,
        "aurc": 0.0380700868516878,
        "mean_confidence": 0.920768666666667,
        "acc_at_50_coverage": 0.98,
        "acc_at_80_coverage": 0.9208333333333333
       },
       "readout_ece": 0.10435199999999992,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 13.0,
       "latency_p99_ms": 16.0,
       "latency_tail_support": 4,
       "latency_extremes": {
        "first_ms": 13.0,
        "max_ms": 17.0,
        "argmax_case": 90
       },
       "determinism_ok": true,
       "seconds": 6.2867043,
       "n_cases": 300,
       "n_questions": 300,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-b136750e5af1d412",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 300,
       "accuracy": 0.6166666666666667,
       "macro_f1": 0.5558575733898873,
       "ece": 0.07564014782508215,
       "brier": 0.5401282603686581,
       "nll": 0.9888498366458933,
       "aurc": 0.25827792038822533,
       "mean_confidence": 0.6170066524545351,
       "acc_at_50_coverage": 0.74,
       "acc_at_80_coverage": 0.6916666666666667
      },
      "readout_ece": 0.1960698040326436,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 32.0,
      "latency_p99_ms": 37.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 62.0,
       "max_ms": 62.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 9.736046,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.6166666666666667,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 300,
       "accuracy": 0.4766666666666667,
       "macro_f1": 0.40968489334530006,
       "ece": 0.05632179826268047,
       "brier": 0.6175729560703562,
       "nll": 1.0251905097192242,
       "aurc": 0.44561388263656226,
       "mean_confidence": 0.46186815969647116,
       "acc_at_50_coverage": 0.56,
       "acc_at_80_coverage": 0.5166666666666667
      },
      "readout_ece": 0.05632179826268047,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 22.0,
      "latency_p99_ms": 28.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 93.0,
       "max_ms": 93.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 22.3418456,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 300,
       "accuracy": 0.45666666666666667,
       "macro_f1": 0.44032921810699593,
       "ece": 0.14558631380399067,
       "brier": 0.6493632578698356,
       "nll": 1.0945308667485165,
       "aurc": 0.4554917872946374,
       "mean_confidence": 0.6022529804706573,
       "acc_at_50_coverage": 0.5733333333333334,
       "acc_at_80_coverage": 0.49166666666666664
      },
      "readout_ece": 0.14558631380399067,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 47.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 43.0,
       "max_ms": 62.0,
       "argmax_case": 249
      },
      "determinism_ok": false,
      "seconds": 10.4416617,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 300,
       "accuracy": 0.9,
       "macro_f1": 0.8999672827033166,
       "ece": 0.04532527506351472,
       "brier": 0.16604580727055215,
       "nll": 0.3076956385611185,
       "aurc": 0.023331961958249815,
       "mean_confidence": 0.8594962954521179,
       "acc_at_50_coverage": 0.9866666666666667,
       "acc_at_80_coverage": 0.9458333333333333
      },
      "readout_ece": 0.2503718494966851,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 52.0,
      "latency_p99_ms": 143.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 54.0,
       "max_ms": 191.0,
       "argmax_case": 125
      },
      "determinism_ok": true,
      "seconds": 19.5976125,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-b136750e5af1d412",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 300,
    "exact": 0,
    "near": 0
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-b136750e5af1d412"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-b136750e5af1d412"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-b136750e5af1d412"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-b136750e5af1d412"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A0",
    "hard": {
     "n": 300,
     "accuracy": 0.5233333333333333,
     "ece": 0.5026737833023072,
     "mean_confidence": 0.020659550031026205,
     "acc_at_50_coverage": 0.6466666666666666
    },
    "consult_rate": 0.0,
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": 0.087,
    "latency_p99_ms": 0.102,
    "latency_tail_support": 5,
    "serves": "A0",
    "gate": "served by the reflex half \u2014 the best measured arm on this suite",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 300,
     "accuracy": 0.8966666666666666,
     "macro_f1": 0.8966195977749251,
     "ece": 0.044656349023183224,
     "brier": 0.1663415748887174,
     "nll": 0.30846237893560924,
     "aurc": 0.023442450106287704,
     "mean_confidence": 0.8592826906840006,
     "acc_at_50_coverage": 0.9866666666666667,
     "acc_at_80_coverage": 0.9458333333333333
    },
    "readout_ece": 0.24790328881071622,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 111.0,
    "latency_p99_ms": 164.0,
    "latency_tail_support": 4,
    "latency_extremes": {
     "first_ms": 125.0,
     "max_ms": 178.0,
     "argmax_case": 4
    },
    "determinism_ok": true,
    "seconds": 41.818258584,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-b136750e5af1d412",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "3a7be30",
     "date_utc": "2026-09-28T15:36:40Z"
    }
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "1949b801cf9746028d56",
    "spec_file": "scripts/paw_specs/xnli_en.relation.txt",
    "spec_blake3": "ead96a846bbe50430b7d9c6c635c17ecea8c59a2bd43e36210e93e7574d17c31",
    "compile_cache_hit": false,
    "compile_wall_s": 120.965073917,
    "n_cases": 300,
    "n_questions": 300,
    "n_answered": 300,
    "refusals": 0,
    "quote_stripped": 0,
    "accuracy": 0.72,
    "answered_accuracy": 0.72,
    "refusal_rate": 0.0,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [],
    "server_latency_p50_ms": 96.0,
    "determinism_ok": true,
    "cases_digest": "fnv1a64-b136750e5af1d412",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "9dc33da",
     "date_utc": "2026-09-29T00:50:51Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-xnli_en_encoder_v1 (laya-english encoder + NLEH v1 head)",
    "hard": {
     "n": 300,
     "accuracy": 0.86
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 16.513,
    "latency_p99_ms": 18.634,
    "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
    "gate": "acc 0.8600 (258/300 rows) \u00b7 paired vs the incumbent A1: mean +0.4500 \u00b7 LB95 +0.3832 \u00b7 per-row p50 16.513 ms (metal device). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017). Reference-identified: the head ties the laya-english forward 300/300 picks (riir-train 599 T5a) \u2014 the cell is the encoder lane's own frozen read, not the trainer's lift.",
    "device": "metal",
    "record_only": true,
    "latency_quotable": true,
    "source_run": {
     "git_sha": "0f81b54",
     "date_utc": "2026-10-01T11:20:17Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 300,
     "accuracy": 0.86,
     "macro_f1": 0.858750405020872,
     "ece": 0.054230172932148026,
     "brier": 0.20355339462624883,
     "nll": 0.36150648236447586,
     "aurc": 0.04480278953978929,
     "mean_confidence": 0.8634677937626839,
     "acc_at_50_coverage": 0.96,
     "acc_at_80_coverage": 0.9333333333333333
    },
    "readout_ece": 0.054230172932148026,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.3333333333333333,
    "jdi_skill": 0.7899999999999998,
    "cases_digest": "fnv1a64-b136750e5af1d412",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   }
  },
  {
   "name": "massive_intent_en",
   "n_cases": 300,
   "n_questions": 300,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 300,
     "accuracy": 0.78,
     "macro_f1": 0.7698039912157559,
     "ece": 0.635231661486129,
     "brier": 0.7856433271131125,
     "nll": 2.059894100161934,
     "aurc": 0.055327402112393974,
     "mean_confidence": 0.14476833851387103,
     "acc_at_50_coverage": 0.9666666666666667,
     "acc_at_80_coverage": 0.8916666666666667
    },
    "readout_ece": 0.07959517017006876,
    "raw_abstain": {
     "abstain_rate": 0.9366666666666666,
     "selective_accuracy": 1.0,
     "selective_n": 19
    },
    "calibrated_abstain": {
     "abstain_rate": 0.31333333333333335,
     "selective_accuracy": 0.7669902912621359,
     "selective_n": 206
    },
    "readout_ece_raw": 0.635231661486129,
    "readout_ece_calibrated": 0.07959517017006876,
    "floor_ece": 0.12091210613598673,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.112,
    "latency_p99_ms": 0.136,
    "latency_tail_support": 4,
    "latency_extremes": {
     "first_ms": 0.125,
     "max_ms": 0.139,
     "argmax_case": 94
    },
    "determinism_ok": true,
    "seconds": 42.596955125,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": 0.20078932,
    "distance_threshold": 0.5259382,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.20078932,
      "status": "fitted",
      "targetAccuracy": 0.75,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.5259382,
      "status": "fitted",
      "targetAccuracy": 0.57857144,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 48,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 4.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.45
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.585
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.605
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.595
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.595
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.595
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.55
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.585
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.565
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.565
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.565
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.605
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.605
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.605
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.61
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.63
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.62
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.7833333333333333,
     "honest_accuracy": 0.78,
     "n_pseudo": 300,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "music_dislikeness",
      "pred": "play_music",
      "count": 3,
      "share_of_errors": 0.045454545454545456
     },
     {
      "gold": "audio_volume_other",
      "pred": "audio_volume_up",
      "count": 2,
      "share_of_errors": 0.030303030303030304
     },
     {
      "gold": "datetime_convert",
      "pred": "datetime_query",
      "count": 2,
      "share_of_errors": 0.030303030303030304
     },
     {
      "gold": "music_likeness",
      "pred": "play_music",
      "count": 2,
      "share_of_errors": 0.030303030303030304
     },
     {
      "gold": "music_settings",
      "pred": "play_music",
      "count": 2,
      "share_of_errors": 0.030303030303030304
     },
     {
      "gold": "alarm_query",
      "pred": "alarm_set",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_down",
      "pred": "audio_volume_up",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_mute",
      "pred": "calendar_query",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_mute",
      "pred": "general_quirky",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_other",
      "pred": "alarm_set",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_other",
      "pred": "qa_maths",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     },
     {
      "gold": "audio_volume_up",
      "pred": "play_audiobook",
      "count": 1,
      "share_of_errors": 0.015151515151515152
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "dispatch",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.4618278734199702
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.4618278734199702
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.5519066819548607
      }
     ]
    },
    "corpus_digest": "fnv1a64-8e7cc94319af1108",
    "cases_digest": "fnv1a64-49390c8b80475849",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 300,
      "accuracy": 0.6933333333333334,
      "macro_f1": 0.6936125577989888,
      "ece": 0.24830266666666645,
      "brier": 0.540529478666667,
      "nll": 3.5499559002853465,
      "aurc": 0.14436891325283494,
      "mean_confidence": 0.9300260000000002,
      "acc_at_50_coverage": 0.9133333333333333,
      "acc_at_80_coverage": 0.775
     },
     "readout_ece": 0.24406833333333328,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 31.0,
     "latency_p99_ms": 38.0,
     "latency_tail_support": 4,
     "latency_extremes": {
      "first_ms": 25.0,
      "max_ms": 40.0,
      "argmax_case": 6
     },
     "determinism_ok": true,
     "seconds": 13.264258166,
     "n_cases": 300,
     "n_questions": 300,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-49390c8b80475849",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 300,
      "accuracy": 0.6933333333333334,
      "macro_f1": 0.6936125577989888,
      "ece": 0.24830299999999972,
      "brier": 0.5405290377000004,
      "nll": 3.5499541563128743,
      "aurc": 0.14436891325283494,
      "mean_confidence": 0.9300263333333335,
      "acc_at_50_coverage": 0.9133333333333333,
      "acc_at_80_coverage": 0.775
     },
     "readout_ece": 0.2440686666666666,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 48.0,
     "latency_p99_ms": 81.0,
     "latency_tail_support": 4,
     "latency_extremes": {
      "first_ms": 33.0,
      "max_ms": 89.0,
      "argmax_case": 42
     },
     "determinism_ok": true,
     "seconds": 21.45423025,
     "n_cases": 300,
     "n_questions": 300,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-49390c8b80475849",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {},
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 300,
       "accuracy": 0.78,
       "macro_f1": 0.7698039912157559,
       "ece": 0.6352316615109643,
       "brier": 0.7856433271606119,
       "nll": 2.0598941003679703,
       "aurc": 0.055327402112393974,
       "mean_confidence": 0.14476833848903575,
       "acc_at_50_coverage": 0.9666666666666667,
       "acc_at_80_coverage": 0.8916666666666667
      },
      "readout_ece": 0.07959517017006876,
      "raw_abstain": {
       "abstain_rate": 0.9366666666666666,
       "selective_accuracy": 1.0,
       "selective_n": 19
      },
      "calibrated_abstain": {
       "abstain_rate": 0.31333333333333335,
       "selective_accuracy": 0.7669902912621359,
       "selective_n": 206
      },
      "readout_ece_raw": 0.6352316615109643,
      "readout_ece_calibrated": 0.07959517017006876,
      "floor_ece": 0.12091210613598673,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.174,
      "latency_p99_ms": 0.318,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 0.196,
       "max_ms": 0.574,
       "argmax_case": 151
      },
      "determinism_ok": true,
      "seconds": 75.1491404,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": 0.20078932,
      "distance_threshold": 0.5259382,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.20078932,
        "status": "fitted",
        "targetAccuracy": 0.75,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.5259382,
        "status": "fitted",
        "targetAccuracy": 0.57857144,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 48,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 4.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.45
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.585
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.605
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.595
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.595
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.595
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.55
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.585
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.565
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.565
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.565
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.605
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.605
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.605
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.61
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.63
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.62
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.7833333333333333,
       "honest_accuracy": 0.78,
       "n_pseudo": 300,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "music_dislikeness",
        "pred": "play_music",
        "count": 3,
        "share_of_errors": 0.045454545454545456
       },
       {
        "gold": "audio_volume_other",
        "pred": "audio_volume_up",
        "count": 2,
        "share_of_errors": 0.030303030303030304
       },
       {
        "gold": "datetime_convert",
        "pred": "datetime_query",
        "count": 2,
        "share_of_errors": 0.030303030303030304
       },
       {
        "gold": "music_likeness",
        "pred": "play_music",
        "count": 2,
        "share_of_errors": 0.030303030303030304
       },
       {
        "gold": "music_settings",
        "pred": "play_music",
        "count": 2,
        "share_of_errors": 0.030303030303030304
       },
       {
        "gold": "alarm_query",
        "pred": "alarm_set",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_down",
        "pred": "audio_volume_up",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_mute",
        "pred": "calendar_query",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_mute",
        "pred": "general_quirky",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_other",
        "pred": "alarm_set",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_other",
        "pred": "qa_maths",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       },
       {
        "gold": "audio_volume_up",
        "pred": "play_audiobook",
        "count": 1,
        "share_of_errors": 0.015151515151515152
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "dispatch",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.4618278732709586
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.4618278732709586
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.5519066819548607
        }
       ]
      },
      "corpus_digest": "fnv1a64-8e7cc94319af1108",
      "cases_digest": "fnv1a64-49390c8b80475849",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 300,
        "accuracy": 0.6933333333333334,
        "macro_f1": 0.6936125577989888,
        "ece": 0.24830266666666645,
        "brier": 0.540529420866667,
        "nll": 3.549952987360154,
        "aurc": 0.14436891325283494,
        "mean_confidence": 0.9300260000000002,
        "acc_at_50_coverage": 0.9133333333333333,
        "acc_at_80_coverage": 0.775
       },
       "readout_ece": 0.24406899999999992,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 16.0,
       "latency_p99_ms": 19.0,
       "latency_tail_support": 4,
       "latency_extremes": {
        "first_ms": 17.0,
        "max_ms": 21.0,
        "argmax_case": 267
       },
       "determinism_ok": true,
       "seconds": 7.3354609,
       "n_cases": 300,
       "n_questions": 300,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-49390c8b80475849",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 300,
       "accuracy": 0.26666666666666666,
       "macro_f1": 0.2616359616723602,
       "ece": 0.14943921824296316,
       "brier": 0.8766237048460563,
       "nll": 2.51650569824331,
       "aurc": 0.5855085352810563,
       "mean_confidence": 0.4161058849096298,
       "acc_at_50_coverage": 0.36666666666666664,
       "acc_at_80_coverage": 0.2875
      },
      "readout_ece": 0.123150135849913,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 80.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 90.0,
       "max_ms": 90.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 9.8769189,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.27956989247311825,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 300,
       "accuracy": 0.8233333333333334,
       "macro_f1": 0.8196121981135134,
       "ece": 0.20483493914587297,
       "brier": 0.2987530048789672,
       "nll": 0.8081630767910054,
       "aurc": 0.0455102450307677,
       "mean_confidence": 0.6221885743769106,
       "acc_at_50_coverage": 0.98,
       "acc_at_80_coverage": 0.9333333333333333
      },
      "readout_ece": 0.20483493914587297,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 22.0,
      "latency_p99_ms": 29.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 93.0,
       "max_ms": 93.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 18.1345739,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 300,
       "accuracy": 0.6233333333333333,
       "macro_f1": 0.6071436264423247,
       "ece": 0.2885734117031098,
       "brier": 0.6020865265398592,
       "nll": 1.4595161360606754,
       "aurc": 0.17055899673373176,
       "mean_confidence": 0.3347599216302236,
       "acc_at_50_coverage": 0.8133333333333334,
       "acc_at_80_coverage": 0.6916666666666667
      },
      "readout_ece": 0.2885734117031098,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 61.0,
      "latency_p99_ms": 71.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 79.0,
       "max_ms": 79.0,
       "argmax_case": 0
      },
      "determinism_ok": false,
      "seconds": 19.7203989,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 300,
       "accuracy": 0.92,
       "macro_f1": 0.9142354668289988,
       "ece": 0.04438563970228031,
       "brier": 0.10702258525710472,
       "nll": 0.28892249486105676,
       "aurc": 0.008952395232507604,
       "mean_confidence": 0.9417286148418983,
       "acc_at_50_coverage": 1.0,
       "acc_at_80_coverage": 0.9875
      },
      "readout_ece": 0.038264626647573526,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 95.0,
      "latency_p99_ms": 213.0,
      "latency_tail_support": 4,
      "latency_extremes": {
       "first_ms": 272.0,
       "max_ms": 272.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 35.1416285,
      "n_cases": 300,
      "n_questions": 300,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-49390c8b80475849",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 300,
    "exact": 4,
    "near": 17
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "H2(\u03b2=1,nmin=2,\u03c4=4)",
    "hard": {
     "n": 300,
     "accuracy": 0.84,
     "ece": 0.06870501239437647,
     "mean_confidence": 0.8517494499514889,
     "acc_at_50_coverage": 0.9933333333333333
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.001083,
    "latency_p99_ms": 0.001958,
    "latency_tail_support": 4,
    "serves": "H2(\u03b2=1,nmin=2,\u03c4=4)",
    "gate": "best measured, T2-uncertified (paired LB95 -0.0039) \u2014 certification is more questions, not a posture rollback",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-49390c8b80475849"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-49390c8b80475849"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-49390c8b80475849"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-49390c8b80475849"
     },
     "status": "same"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 300,
     "accuracy": 0.92,
     "macro_f1": 0.9142354668289988,
     "ece": 0.04578136064112186,
     "brier": 0.1068547925077969,
     "nll": 0.2889260038103776,
     "aurc": 0.009152540920493028,
     "mean_confidence": 0.941057560518384,
     "acc_at_50_coverage": 1.0,
     "acc_at_80_coverage": 0.9875
    },
    "readout_ece": 0.038352326947816914,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 1710.0,
    "latency_p99_ms": 2384.0,
    "latency_tail_support": 4,
    "latency_extremes": {
     "first_ms": 1705.0,
     "max_ms": 2410.0,
     "argmax_case": 182
    },
    "determinism_ok": true,
    "seconds": 567.472480792,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-49390c8b80475849",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "3a7be30",
     "date_utc": "2026-09-28T15:36:40Z"
    }
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "1b5d4b654f60b508602c",
    "spec_file": "scripts/paw_specs/massive_intent_en.intent.txt",
    "spec_blake3": "10b2865efdaaeff8129e625e18a36b141e07fd7f5fea6407c759487c31096f17",
    "compile_cache_hit": false,
    "compile_wall_s": 88.633114917,
    "n_cases": 300,
    "n_questions": 300,
    "n_answered": 181,
    "refusals": 119,
    "quote_stripped": 0,
    "accuracy": 0.51,
    "answered_accuracy": 0.8453038674033149,
    "refusal_rate": 0.39666666666666667,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [
     "qa_factoid",
     "iot_hue_lightchange",
     "iot_hue_sound",
     "screen_query",
     "iot_hue_lightdown",
     "iot_hue_lightoff",
     "iot_hub_power",
     "toaster_query"
    ],
    "server_latency_p50_ms": 91.6,
    "determinism_ok": true,
    "cases_digest": "fnv1a64-49390c8b80475849",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "9dc33da",
     "date_utc": "2026-09-29T00:50:51Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 300,
     "accuracy": 0.91,
     "macro_f1": 0.9056898356680406,
     "ece": 0.03461575806140901,
     "brier": 0.13930689146055505,
     "nll": 0.31711430182399747,
     "aurc": 0.011505582712503468,
     "mean_confidence": 0.9334321061770121,
     "acc_at_50_coverage": 1.0,
     "acc_at_80_coverage": 0.9833333333333333
    },
    "readout_ece": 0.03461575806140901,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 300,
    "n_questions": 300,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.08666666666666667,
    "jdi_skill": 0.9014598540145986,
    "cases_digest": "fnv1a64-49390c8b80475849",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "H2(\u03b2=1,nmin=2,\u03c4=8)",
    "hard": {
     "n": 300,
     "accuracy": 0.8266666666666667,
     "ece": 0.06070700170483837,
     "mean_confidence": 0.8355173326634961,
     "acc_at_50_coverage": 0.9933333333333333
    },
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.001042,
    "latency_p99_ms": 0.002125,
    "latency_tail_support": 4,
    "serves": "tier-fallback",
    "served_by": "Instinct (H2(\u03b2=1,nmin=2,\u03c4=8))",
    "fallback_note": "best-of-family (owner 2026-10-02): the encoder arm's own measured read is 0.6567, 0.1700 under the served arm \u2014 refused at serve, so the served answer is shown; the refused arm's full record rides the cell (displaced_record)",
    "derived": true,
    "source_run": {
     "git_sha": "8bbff09",
     "date_utc": "2026-09-28T10:13:40Z"
    },
    "displaced_record": {
     "lane": "Rethink",
     "model": "ENC-massive_encoder_v1_probe (laya-english encoder + NLEH v1 head)",
     "hard": {
      "n": 300,
      "accuracy": 0.6566666666666666
     },
     "consult_rate": 1.0,
     "latency_scope": "arm-only",
     "latency_rows": "questions",
     "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
     "gate": "acc 0.6567 (197/300 question rows) \u00b7 paired vs the incumbent A1: mean -0.1600 \u00b7 LB95 -0.2194 \u00b7 per-row p50 31.332 ms (metal device, laya-english). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017).",
     "device": "metal",
     "head_kind": "v1",
     "ckpt": "english",
     "shape_desc": "20 classes \u00b7 feat 21504",
     "record_only": true,
     "latency_quotable": false,
     "source_run": {
      "git_sha": "e0440e3",
      "date_utc": "2026-10-01T17:58:51Z"
     }
    }
   }
  },
  {
   "name": "banking77",
   "n_cases": 500,
   "n_questions": 500,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 500,
     "accuracy": 0.842,
     "macro_f1": 0.8362254543690752,
     "ece": 0.8193658229298889,
     "brier": 0.9687678887240521,
     "nll": 3.821497518480265,
     "aurc": 0.05987299049258608,
     "mean_confidence": 0.022634177070111037,
     "acc_at_50_coverage": 0.972,
     "acc_at_80_coverage": 0.9075
    },
    "readout_ece": 0.051719533622264856,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.478,
     "selective_accuracy": 0.9348659003831418,
     "selective_n": 261
    },
    "readout_ece_raw": 0.8193658229298889,
    "readout_ece_calibrated": 0.051719533622264856,
    "floor_ece": 0.29331343283582084,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.376,
    "latency_p99_ms": 0.698,
    "latency_tail_support": 6,
    "latency_extremes": {
     "first_ms": 0.335,
     "max_ms": 0.734,
     "argmax_case": 16
    },
    "determinism_ok": true,
    "seconds": 57.220160833,
    "n_cases": 500,
    "n_questions": 500,
    "score_threshold": 0.74778104,
    "distance_threshold": 0.55494976,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.74778104,
      "status": "fitted",
      "targetAccuracy": 0.9,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.55494976,
      "status": "fitted",
      "targetAccuracy": 0.8142857,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 40,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": {
     "selected_scale": 1.0,
     "selected_alpha": "observed-laplace",
     "selected_noul_domain": null,
     "selected_view": "bag",
     "candidates": [
      {
       "scale": 0.0,
       "alpha": "off",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.465
      },
      {
       "scale": 1.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.81
      },
      {
       "scale": 4.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.805
      },
      {
       "scale": 16.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.805
      },
      {
       "scale": 32.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.805
      },
      {
       "scale": 64.0,
       "alpha": "observed-laplace",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.8
      },
      {
       "scale": 1.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.795
      },
      {
       "scale": 4.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.775
      },
      {
       "scale": 16.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.77
      },
      {
       "scale": 32.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.76
      },
      {
       "scale": 64.0,
       "alpha": "fixed-1",
       "noul_domain": null,
       "view": "bag",
       "cal_acc": 0.76
      }
     ]
    },
    "oc_selection": null,
    "ridge_selection": {
     "selected_scale": 0.0,
     "selected_lambda": 10.0,
     "candidates": [
      {
       "scale": 0.0,
       "lambda": 10.0,
       "cal_acc": 0.81
      },
      {
       "scale": 0.5,
       "lambda": 10.0,
       "cal_acc": 0.845
      },
      {
       "scale": 1.0,
       "lambda": 10.0,
       "cal_acc": 0.855
      },
      {
       "scale": 2.0,
       "lambda": 10.0,
       "cal_acc": 0.855
      },
      {
       "scale": 4.0,
       "lambda": 10.0,
       "cal_acc": 0.845
      },
      {
       "scale": 8.0,
       "lambda": 10.0,
       "cal_acc": 0.825
      }
     ]
    },
    "genome_selection": null,
    "transductive": {
     "accuracy": 0.836,
     "honest_accuracy": 0.842,
     "n_pseudo": 500,
     "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
    },
    "confusion": [
     {
      "gold": "exchange charge",
      "pred": "exchange rate",
      "count": 3,
      "share_of_errors": 0.0379746835443038
     },
     {
      "gold": "card delivery estimate",
      "pred": "activate my card",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "card delivery estimate",
      "pred": "card arrival",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "card delivery estimate",
      "pred": "transfer timing",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "contactless not working",
      "pred": "declined transfer",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "fiat currency support",
      "pred": "supported cards and currencies",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "pending top up",
      "pred": "top up reverted",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "top up reverted",
      "pred": "Refund not showing up",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "wrong exchange rate for cash withdrawal",
      "pred": "wrong amount of cash received",
      "count": 2,
      "share_of_errors": 0.02531645569620253
     },
     {
      "gold": "balance not updated after bank transfer",
      "pred": "transfer timing",
      "count": 1,
      "share_of_errors": 0.012658227848101266
     },
     {
      "gold": "beneficiary not allowed",
      "pred": "declined transfer",
      "count": 1,
      "share_of_errors": 0.012658227848101266
     },
     {
      "gold": "card acceptance",
      "pred": "atm support",
      "count": 1,
      "share_of_errors": 0.012658227848101266
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "dispatch",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.7729256937466562
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.7729256937466562
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.7924418762326241
      }
     ]
    },
    "corpus_digest": "fnv1a64-07e3c974b5ae1bda",
    "cases_digest": "fnv1a64-dd8ab35333abb82a",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 500,
      "accuracy": 0.422,
      "macro_f1": 0.37652146392023184,
      "ece": 0.38134340000000017,
      "brier": 0.9326880786800371,
      "nll": 6.121876279293705,
      "aurc": 0.4066392903778559,
      "mean_confidence": 0.7951093999999999,
      "acc_at_50_coverage": 0.6,
      "acc_at_80_coverage": 0.475
     },
     "readout_ece": 0.3940097999999995,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 48.0,
     "latency_p99_ms": 58.0,
     "latency_tail_support": 6,
     "latency_extremes": {
      "first_ms": 37.0,
      "max_ms": 66.0,
      "argmax_case": 4
     },
     "determinism_ok": true,
     "seconds": 28.820647875,
     "n_cases": 500,
     "n_questions": 500,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-dd8ab35333abb82a",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 500,
      "accuracy": 0.422,
      "macro_f1": 0.37652146392023184,
      "ece": 0.3813438000000001,
      "brier": 0.9326873171600369,
      "nll": 6.121874967045045,
      "aurc": 0.4066392903778559,
      "mean_confidence": 0.7951097999999999,
      "acc_at_50_coverage": 0.6,
      "acc_at_80_coverage": 0.475
     },
     "readout_ece": 0.3940095999999995,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 76.0,
     "latency_p99_ms": 110.0,
     "latency_tail_support": 6,
     "latency_extremes": {
      "first_ms": 52.0,
      "max_ms": 172.0,
      "argmax_case": 120
     },
     "determinism_ok": true,
     "seconds": 46.196261833,
     "n_cases": 500,
     "n_questions": 500,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-dd8ab35333abb82a",
     "latency_quotable": true
    }
   },
   "extra_host_lanes": {
    "m3-max-ane": {},
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 500,
       "accuracy": 0.842,
       "macro_f1": 0.8362254543690752,
       "ece": 0.8193658229261637,
       "brier": 0.9687678887201593,
       "nll": 3.821497518287094,
       "aurc": 0.05987299049258608,
       "mean_confidence": 0.022634177073836328,
       "acc_at_50_coverage": 0.972,
       "acc_at_80_coverage": 0.9075
      },
      "readout_ece": 0.05171953397989272,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.478,
       "selective_accuracy": 0.9348659003831418,
       "selective_n": 261
      },
      "readout_ece_raw": 0.8193658229261637,
      "readout_ece_calibrated": 0.05171953397989272,
      "floor_ece": 0.29331343283582084,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.54,
      "latency_p99_ms": 2.506,
      "latency_tail_support": 6,
      "latency_extremes": {
       "first_ms": 0.468,
       "max_ms": 4.838,
       "argmax_case": 129
      },
      "determinism_ok": true,
      "seconds": 99.7780475,
      "n_cases": 500,
      "n_questions": 500,
      "score_threshold": 0.74778104,
      "distance_threshold": 0.55494976,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.74778104,
        "status": "fitted",
        "targetAccuracy": 0.9,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.55494976,
        "status": "fitted",
        "targetAccuracy": 0.8142857,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 40,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": {
       "selected_scale": 1.0,
       "selected_alpha": "observed-laplace",
       "selected_noul_domain": null,
       "selected_view": "bag",
       "candidates": [
        {
         "scale": 0.0,
         "alpha": "off",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.465
        },
        {
         "scale": 1.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.81
        },
        {
         "scale": 4.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.805
        },
        {
         "scale": 16.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.805
        },
        {
         "scale": 32.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.805
        },
        {
         "scale": 64.0,
         "alpha": "observed-laplace",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.8
        },
        {
         "scale": 1.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.795
        },
        {
         "scale": 4.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.775
        },
        {
         "scale": 16.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.77
        },
        {
         "scale": 32.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.76
        },
        {
         "scale": 64.0,
         "alpha": "fixed-1",
         "noul_domain": null,
         "view": "bag",
         "cal_acc": 0.76
        }
       ]
      },
      "oc_selection": null,
      "ridge_selection": {
       "selected_scale": 0.0,
       "selected_lambda": 10.0,
       "candidates": [
        {
         "scale": 0.0,
         "lambda": 10.0,
         "cal_acc": 0.81
        },
        {
         "scale": 0.5,
         "lambda": 10.0,
         "cal_acc": 0.845
        },
        {
         "scale": 1.0,
         "lambda": 10.0,
         "cal_acc": 0.855
        },
        {
         "scale": 2.0,
         "lambda": 10.0,
         "cal_acc": 0.855
        },
        {
         "scale": 4.0,
         "lambda": 10.0,
         "cal_acc": 0.845
        },
        {
         "scale": 8.0,
         "lambda": 10.0,
         "cal_acc": 0.825
        }
       ]
      },
      "genome_selection": null,
      "transductive": {
       "accuracy": 0.836,
       "honest_accuracy": 0.842,
       "n_pseudo": 500,
       "protocol": "TRANSDUCTIVE \u2014 unlabeled TEST text joins the count tables, labelled by the honest engine's own forced picks (gold never read); 2-fold cross-fit (half A's pseudo-docs re-score half B and vice versa, so no case scores against its own text); drafter corpora, route, heads and posture unchanged. Not comparable to the headline acc"
      },
      "confusion": [
       {
        "gold": "exchange charge",
        "pred": "exchange rate",
        "count": 3,
        "share_of_errors": 0.0379746835443038
       },
       {
        "gold": "card delivery estimate",
        "pred": "activate my card",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "card delivery estimate",
        "pred": "card arrival",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "card delivery estimate",
        "pred": "transfer timing",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "contactless not working",
        "pred": "declined transfer",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "fiat currency support",
        "pred": "supported cards and currencies",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "pending top up",
        "pred": "top up reverted",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "top up reverted",
        "pred": "Refund not showing up",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "wrong exchange rate for cash withdrawal",
        "pred": "wrong amount of cash received",
        "count": 2,
        "share_of_errors": 0.02531645569620253
       },
       {
        "gold": "balance not updated after bank transfer",
        "pred": "transfer timing",
        "count": 1,
        "share_of_errors": 0.012658227848101266
       },
       {
        "gold": "beneficiary not allowed",
        "pred": "declined transfer",
        "count": 1,
        "share_of_errors": 0.012658227848101266
       },
       {
        "gold": "card acceptance",
        "pred": "atm support",
        "count": 1,
        "share_of_errors": 0.012658227848101266
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "dispatch",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.7729256937466562
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.7729256937466562
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.7924418762326241
        }
       ]
      },
      "corpus_digest": "fnv1a64-07e3c974b5ae1bda",
      "cases_digest": "fnv1a64-dd8ab35333abb82a",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 500,
        "accuracy": 0.422,
        "macro_f1": 0.37652146392023184,
        "ece": 0.38134320000000016,
        "brier": 0.9326872987800371,
        "nll": 6.1218761064175515,
        "aurc": 0.4066392903778559,
        "mean_confidence": 0.7951087999999998,
        "acc_at_50_coverage": 0.6,
        "acc_at_80_coverage": 0.475
       },
       "readout_ece": 0.3940095999999995,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 32.0,
       "latency_p99_ms": 47.0,
       "latency_tail_support": 6,
       "latency_extremes": {
        "first_ms": 26.0,
        "max_ms": 48.0,
        "argmax_case": 290
       },
       "determinism_ok": true,
       "seconds": 19.9027001,
       "n_cases": 500,
       "n_questions": 500,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-dd8ab35333abb82a",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "clm": {
      "lane": "clm (reference)",
      "model": "clm-latest",
      "hard": {
       "n": 500,
       "accuracy": 0.01,
       "macro_f1": 0.010601643254704479,
       "ece": 0.2654645331054925,
       "brier": 1.1069505329122276,
       "nll": 5.797951440220234,
       "aurc": 0.9966510785284138,
       "mean_confidence": 0.27546453310549257,
       "acc_at_50_coverage": 0.004,
       "acc_at_80_coverage": 0.0075
      },
      "readout_ece": 0.2559311717152596,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 31.0,
      "latency_p99_ms": 37.0,
      "latency_tail_support": 6,
      "latency_extremes": {
       "first_ms": 220.0,
       "max_ms": 220.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 16.2248854,
      "n_cases": 500,
      "n_questions": 500,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "acc_deleaked": 0.010351966873706004,
      "latency_quotable": null
     },
     "gliner": {
      "lane": "gliner (reference)",
      "model": "fastino/GLiNER2.5-Decide",
      "hard": {
       "n": 500,
       "accuracy": 0.706,
       "macro_f1": 0.26347993901957295,
       "ece": 0.13325036729006468,
       "brier": 0.4525747554889313,
       "nll": 1.2698150043381327,
       "aurc": 0.13819902613508514,
       "mean_confidence": 0.7902530786522173,
       "acc_at_50_coverage": 0.868,
       "acc_at_80_coverage": 0.7825
      },
      "readout_ece": 0.13325036729006468,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 24.0,
      "latency_p99_ms": 28.0,
      "latency_tail_support": 6,
      "latency_extremes": {
       "first_ms": 94.0,
       "max_ms": 94.0,
       "argmax_case": 0
      },
      "determinism_ok": true,
      "seconds": 23.6673312,
      "n_cases": 500,
      "n_questions": 500,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "agentjev": {
      "lane": "agentjev (reference)",
      "model": "AgentJev-0.6B@9d9b5fc3",
      "hard": {
       "n": 500,
       "accuracy": 0.546,
       "macro_f1": 0.18577943016340512,
       "ece": 0.34468430907279257,
       "brier": 0.7382003610976728,
       "nll": 2.1759095494256497,
       "aurc": 0.20784801125986535,
       "mean_confidence": 0.20131569092720747,
       "acc_at_50_coverage": 0.784,
       "acc_at_80_coverage": 0.62
      },
      "readout_ece": 0.34468430907279257,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 149.0,
      "latency_p99_ms": 162.0,
      "latency_tail_support": 6,
      "latency_extremes": {
       "first_ms": 186.0,
       "max_ms": 187.0,
       "argmax_case": 236
      },
      "determinism_ok": false,
      "seconds": 78.1698197,
      "n_cases": 500,
      "n_questions": 500,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "confusion": null,
      "pair_head_ab": null,
      "latency_quotable": null
     },
     "paw_local": {
      "lane": "paw (local)",
      "model": "paw-ft-bs48-20260530",
      "posture": "local-subprocess",
      "program_id": "ebbfe5a01ae044052526",
      "spec_file": "scripts/paw_specs\\banking77.txt",
      "spec_blake3": "387c823cb86efe63d4cbba2e8b299f213f5a2487f3b7b049fe07315348ff2d43",
      "compile_cache_hit": true,
      "compile_wall_s": 11.9880512,
      "n_cases": 500,
      "n_questions": 500,
      "n_answered": 323,
      "refusals": 177,
      "quote_stripped": 0,
      "accuracy": 0.412,
      "answered_accuracy": 0.6377708978328174,
      "refusal_rate": 0.354,
      "score_mae_answered": null,
      "within_1_answered": null,
      "refusal_samples": [
       "locate my card",
       "wrong exchange rate for foreign exchange",
       "top up by day",
       "exchange currency",
       "contactless",
       "pending gas",
       "cancel transaction",
       "top up limit"
      ],
      "latency_p50_ms": 131.696,
      "latency_p99_ms": 236.735,
      "latency_tail_support": 6,
      "server_latency_p50_ms": null,
      "determinism_ok": true,
      "seconds": 72.0437115,
      "latency_quotable": null
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 500,
       "accuracy": 0.654,
       "macro_f1": 0.6330925363531608,
       "ece": 0.08950261621177197,
       "brier": 0.5155436575799948,
       "nll": 1.9273258252069845,
       "aurc": 0.16820807721410694,
       "mean_confidence": 0.5716962376385927,
       "acc_at_50_coverage": 0.884,
       "acc_at_80_coverage": 0.74
      },
      "readout_ece": 0.11409038517030952,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 178.0,
      "latency_p99_ms": 306.0,
      "latency_tail_support": 6,
      "latency_extremes": {
       "first_ms": 306.0,
       "max_ms": 370.0,
       "argmax_case": 77
      },
      "determinism_ok": true,
      "seconds": 99.4668885,
      "n_cases": 500,
      "n_questions": 500,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-dd8ab35333abb82a",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "leak": {
    "threshold": 0.8,
    "n_reference": 4000,
    "n_eval": 500,
    "exact": 0,
    "near": 17
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "ebbfe5a01ae044052526",
    "spec_file": "scripts/paw_specs/banking77.txt",
    "spec_blake3": "387c823cb86efe63d4cbba2e8b299f213f5a2487f3b7b049fe07315348ff2d43",
    "compile_cache_hit": false,
    "compile_wall_s": 0.975116792,
    "n_cases": 500,
    "n_questions": 500,
    "n_answered": 327,
    "refusals": 173,
    "quote_stripped": 0,
    "accuracy": 0.42,
    "answered_accuracy": 0.6422018348623854,
    "refusal_rate": 0.346,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [
     "locate my card",
     "wrong exchange rate for foreign exchange",
     "top up by day",
     "exchange currency",
     "contactless",
     "pending gas",
     "cancel transaction",
     "top up limit"
    ],
    "server_latency_p50_ms": 96.0,
    "determinism_ok": true
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-dd8ab35333abb82a"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-dd8ab35333abb82a"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-dd8ab35333abb82a"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-dd8ab35333abb82a"
     },
     "status": "same"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "H2(\u03b2=2,nmin=8,\u03c4=8)",
    "hard": {
     "n": 500,
     "accuracy": 0.854,
     "ece": 0.08412793359878214,
     "mean_confidence": 0.9374190933679326,
     "acc_at_50_coverage": 0.996
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.007292,
    "latency_p99_ms": 0.026709,
    "latency_tail_support": 6,
    "serves": "H2(\u03b2=2,nmin=8,\u03c4=8)",
    "gate": "best measured, T2-uncertified (paired LB95 -0.0013) \u2014 certification is more questions, not a posture rollback",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    }
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 500,
     "accuracy": 0.656,
     "macro_f1": 0.6355435167453176,
     "ece": 0.09753488361835477,
     "brier": 0.515381702952365,
     "nll": 1.925635515579029,
     "aurc": 0.16795993855145483,
     "mean_confidence": 0.5719192961454391,
     "acc_at_50_coverage": 0.884,
     "acc_at_80_coverage": 0.74
    },
    "readout_ece": 0.12044315707662828,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 3056.0,
    "latency_p99_ms": 4084.0,
    "latency_tail_support": 6,
    "latency_extremes": {
     "first_ms": 4535.0,
     "max_ms": 4535.0,
     "argmax_case": 0
    },
    "determinism_ok": true,
    "seconds": 1650.295442791,
    "n_cases": 500,
    "n_questions": 500,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-dd8ab35333abb82a",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 500,
     "accuracy": 0.792,
     "macro_f1": 0.7805128981026606,
     "ece": 0.04694363492727279,
     "brier": 0.30066090506936505,
     "nll": 0.765191381825856,
     "aurc": 0.05976304691192087,
     "mean_confidence": 0.7859104135632515,
     "acc_at_50_coverage": 0.98,
     "acc_at_80_coverage": 0.87
    },
    "readout_ece": 0.04694363492727279,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 500,
    "n_questions": 500,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.014,
    "jdi_skill": 0.7890466531440162,
    "cases_digest": "fnv1a64-dd8ab35333abb82a",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:01:48Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "H2(\u03b2=2,nmin=8,\u03c4=8)",
    "hard": {
     "n": 500,
     "accuracy": 0.854,
     "ece": 0.08412793359878214,
     "mean_confidence": 0.9374190933679326,
     "acc_at_50_coverage": 0.996
    },
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.008541,
    "latency_p99_ms": 0.06775,
    "latency_tail_support": 6,
    "serves": "tier-fallback",
    "served_by": "Instinct (H2(\u03b2=2,nmin=8,\u03c4=8))",
    "fallback_note": "best-of-family (owner 2026-10-02): the encoder arm's own measured read is 0.4420, 0.4120 under the served arm \u2014 refused at serve, so the served answer is shown; the refused arm's full record rides the cell (displaced_record)",
    "derived": true,
    "source_run": {
     "git_sha": "8bbff09",
     "date_utc": "2026-09-28T10:13:40Z"
    },
    "displaced_record": {
     "lane": "Rethink",
     "model": "ENC-banking77_encoder_v1 (laya-english encoder + NLEH v1 head)",
     "hard": {
      "n": 500,
      "accuracy": 0.442
     },
     "consult_rate": 1.0,
     "latency_scope": "arm-only",
     "latency_rows": "questions",
     "serves": "\u2717 (encoder class refused at serve \u2014 the incumbent arm serves; instinct issue 014 C1, seating per issue 017)",
     "gate": "acc 0.4420 (221/500 question rows) \u00b7 paired vs the incumbent A1: mean -0.3860 \u00b7 LB95 -0.4363 \u00b7 per-row p50 42.983 ms (metal device, laya-english). Serve REFUSED (014 class-wide latency class); record-only cell, seated per the owner call 2026-10-01 (instinct issue 017).",
     "device": "metal",
     "head_kind": "v1",
     "ckpt": "english",
     "shape_desc": "77 classes \u00b7 feat 79872",
     "record_only": true,
     "latency_quotable": true,
     "source_run": {
      "git_sha": "e0440e3",
      "date_utc": "2026-10-01T18:50:28Z"
     }
    }
   }
  },
  {
   "name": "code_fixtures",
   "n_cases": 16,
   "n_questions": 32,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 32,
     "accuracy": 0.375,
     "macro_f1": 0.16398467432950192,
     "ece": 0.09478947846218944,
     "brier": 0.6858308207025954,
     "nll": 1.3870096269732066,
     "aurc": 0.40726383867930005,
     "mean_confidence": 0.3234607386402786,
     "acc_at_50_coverage": 0.625,
     "acc_at_80_coverage": 0.4
    },
    "readout_ece": 0.09384669177234173,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.34375,
     "selective_accuracy": 0.2857142857142857,
     "selective_n": 21
    },
    "readout_ece_raw": 0.3742090817540884,
    "readout_ece_calibrated": 0.09384669177234173,
    "floor_ece": 0.35101744186046513,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 0.226,
    "latency_p99_ms": 0.386,
    "latency_tail_support": 1,
    "latency_extremes": {
     "first_ms": 0.196,
     "max_ms": 0.386,
     "argmax_case": 9
    },
    "determinism_ok": true,
    "seconds": 0.05484925,
    "n_cases": 16,
    "n_questions": 32,
    "score_threshold": 0.4606726,
    "distance_threshold": 0.3365353,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.4606726,
      "status": "fitted",
      "targetAccuracy": 0.703125,
      "support": {
       "n": 64,
       "nPass": 64,
       "nAbstain": 0
      }
     },
     "distance": {
      "threshold": 0.3365353,
      "status": "fitted",
      "targetAccuracy": 0.62222224,
      "support": {
       "n": 64,
       "nPass": 45,
       "nAbstain": 19
      }
     }
    },
    "corpus_cap": {
     "effective": 18446744073709551615,
     "source": "registry (selection n/a: self-corpora)",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": [
     {
      "gold": "game_heads",
      "pred": "engine",
      "count": 2,
      "share_of_errors": 0.14285714285714285
     },
     {
      "gold": "engine",
      "pred": "lanes::clm",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::runner",
      "pred": "engine",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::runner",
      "pred": "harness::suites",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::slice_leak",
      "pred": "engine",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::slice_leak",
      "pred": "lanes::clm",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::suites",
      "pred": "lanes::clm",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "harness::suites",
      "pred": "lanes::paw",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "lanes::clm",
      "pred": "harness::runner",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "lanes::clm",
      "pred": "lanes::paw",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "lanes::paw",
      "pred": "engine",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     },
     {
      "gold": "serve",
      "pred": "engine",
      "count": 1,
      "share_of_errors": 0.07142857142857142
     }
    ],
    "pair_head_ab": null,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.4601553222164512
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.14789915131404996
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.4601553222164512
      }
     ]
    },
    "corpus_digest": "fnv1a64-58c275cd1d2ec9c0",
    "cases_digest": "fnv1a64-93583fda4fc64946",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "25fd257",
     "date_utc": "2026-10-01T00:33:33Z"
    },
    "latency_provenance": {
     "note": "latency cells carried from the host's incumbent run; accuracy is this lane's own (LANE-CARRY, Issue 032)"
    }
   },
   "laya": {
    "english": {
     "lane": "laya (rust)",
     "model": "english",
     "hard": {
      "n": 32,
      "accuracy": 0.40625,
      "macro_f1": 0.15849282296650719,
      "ece": 0.26474687499999994,
      "brier": 0.723710036875,
      "nll": 1.5985284301479366,
      "aurc": 0.35501885135770367,
      "mean_confidence": 0.578540625,
      "acc_at_50_coverage": 0.5625,
      "acc_at_80_coverage": 0.52
     },
     "readout_ece": 0.1708375,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 35.0,
     "latency_p99_ms": 109.0,
     "latency_tail_support": 1,
     "latency_extremes": {
      "first_ms": 25.0,
      "max_ms": 109.0,
      "argmax_case": 9
     },
     "determinism_ok": true,
     "seconds": 4.637546125,
     "n_cases": 16,
     "n_questions": 32,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "cases_digest": "fnv1a64-93583fda4fc64946",
     "latency_quotable": true
    },
    "py/english": {
     "lane": "laya (python)",
     "model": "english",
     "hard": {
      "n": 32,
      "accuracy": 0.40625,
      "macro_f1": 0.15849282296650719,
      "ece": 0.26474687499999994,
      "brier": 0.723710810625,
      "nll": 1.5985284301479366,
      "aurc": 0.35501885135770367,
      "mean_confidence": 0.578540625,
      "acc_at_50_coverage": 0.5625,
      "acc_at_80_coverage": 0.52
     },
     "readout_ece": 0.1708375,
     "raw_abstain": null,
     "calibrated_abstain": null,
     "abstain_causes": null,
     "readout_ece_raw": null,
     "readout_ece_calibrated": null,
     "floor_ece": null,
     "g1_pass": null,
     "g1_verdict": null,
     "by_question_type": null,
     "soft_acc": null,
     "brier_soft": null,
     "score_mae": null,
     "within_1": null,
     "latency_p50_ms": 75.0,
     "latency_p99_ms": 166.0,
     "latency_tail_support": 1,
     "latency_extremes": {
      "first_ms": 32.0,
      "max_ms": 166.0,
      "argmax_case": 9
     },
     "determinism_ok": true,
     "determinism_n": 10,
     "seconds": 7.906893875,
     "n_cases": 16,
     "n_questions": 32,
     "score_threshold": null,
     "distance_threshold": null,
     "threshold_recommendation": null,
     "corpus_cap": null,
     "head_selection": null,
     "nb_selection": null,
     "oc_selection": null,
     "ridge_selection": null,
     "genome_selection": null,
     "transductive": null,
     "confusion": null,
     "pair_head_ab": null,
     "jdi_chance": 0.375,
     "jdi_skill": 0.05,
     "cases_digest": "fnv1a64-93583fda4fc64946",
     "latency_quotable": true,
     "source_run": {
      "git_sha": "450a81c",
      "date_utc": "2026-10-02T09:03:17Z"
     }
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 32,
       "accuracy": 0.375,
       "macro_f1": 0.16398467432950192,
       "ece": 0.09478947846218944,
       "brier": 0.6858308207025954,
       "nll": 1.3870096269732066,
       "aurc": 0.40726383867930005,
       "mean_confidence": 0.3234607386402786,
       "acc_at_50_coverage": 0.625,
       "acc_at_80_coverage": 0.4
      },
      "readout_ece": 0.09384669177234173,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.34375,
       "selective_accuracy": 0.2857142857142857,
       "selective_n": 21
      },
      "readout_ece_raw": 0.3742090817540884,
      "readout_ece_calibrated": 0.09384669177234173,
      "floor_ece": 0.35101744186046513,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 0.425,
      "latency_p99_ms": 0.761,
      "latency_tail_support": 1,
      "latency_extremes": {
       "first_ms": 0.395,
       "max_ms": 0.761,
       "argmax_case": 5
      },
      "determinism_ok": true,
      "seconds": 0.0973279,
      "n_cases": 16,
      "n_questions": 32,
      "score_threshold": 0.4606726,
      "distance_threshold": 0.3365353,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.4606726,
        "status": "fitted",
        "targetAccuracy": 0.703125,
        "support": {
         "n": 64,
         "nPass": 64,
         "nAbstain": 0
        }
       },
       "distance": {
        "threshold": 0.3365353,
        "status": "fitted",
        "targetAccuracy": 0.62222224,
        "support": {
         "n": 64,
         "nPass": 45,
         "nAbstain": 19
        }
       }
      },
      "corpus_cap": {
       "effective": 18446744073709551615,
       "source": "registry (selection n/a: self-corpora)",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": [
       {
        "gold": "game_heads",
        "pred": "engine",
        "count": 2,
        "share_of_errors": 0.14285714285714285
       },
       {
        "gold": "engine",
        "pred": "lanes::clm",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::runner",
        "pred": "engine",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::runner",
        "pred": "harness::suites",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::slice_leak",
        "pred": "engine",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::slice_leak",
        "pred": "lanes::clm",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::suites",
        "pred": "lanes::clm",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "harness::suites",
        "pred": "lanes::paw",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "lanes::clm",
        "pred": "harness::runner",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "lanes::clm",
        "pred": "lanes::paw",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "lanes::paw",
        "pred": "engine",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       },
       {
        "gold": "serve",
        "pred": "engine",
        "count": 1,
        "share_of_errors": 0.07142857142857142
       }
      ],
      "pair_head_ab": null,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.4601553222164512
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.14789915131404996
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.4601553222164512
        }
       ]
      },
      "corpus_digest": "fnv1a64-58c275cd1d2ec9c0",
      "cases_digest": "fnv1a64-93583fda4fc64946",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "ad94345",
       "date_utc": "2026-10-01T02:07:23Z"
      },
      "latency_provenance": {
       "note": "latency cells carried from the host's incumbent run; accuracy is this lane's own (LANE-CARRY, Issue 032)"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 32,
        "accuracy": 0.40625,
        "macro_f1": 0.15849282296650719,
        "ece": 0.26474687499999994,
        "brier": 0.723710810625,
        "nll": 1.5985284301479366,
        "aurc": 0.35501885135770367,
        "mean_confidence": 0.578540625,
        "acc_at_50_coverage": 0.5625,
        "acc_at_80_coverage": 0.52
       },
       "readout_ece": 0.1708375,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 27.0,
       "latency_p99_ms": 58.0,
       "latency_tail_support": 1,
       "latency_extremes": {
        "first_ms": 17.0,
        "max_ms": 58.0,
        "argmax_case": 9
       },
       "determinism_ok": true,
       "seconds": 3.0493855,
       "n_cases": 16,
       "n_questions": 32,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "cases_digest": "fnv1a64-93583fda4fc64946",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "97c2b1f",
        "date_utc": "2026-09-28T07:13:04Z"
       }
      }
     },
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 32,
       "accuracy": 0.59375,
       "macro_f1": 0.31459790209790206,
       "ece": 0.19326931959949434,
       "brier": 0.57716082247066,
       "nll": 1.4915702284057306,
       "aurc": 0.15755860109027417,
       "mean_confidence": 0.7605854256544262,
       "acc_at_50_coverage": 0.875,
       "acc_at_80_coverage": 0.72
      },
      "readout_ece": 0.16317508867949532,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 86.0,
      "latency_p99_ms": 125.0,
      "latency_tail_support": 1,
      "latency_extremes": {
       "first_ms": 57.0,
       "max_ms": 125.0,
       "argmax_case": 13
      },
      "determinism_ok": true,
      "seconds": 2.9700611,
      "n_cases": 16,
      "n_questions": 32,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-93583fda4fc64946",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T13:59:51Z"
      }
     }
    }
   },
   "pairing": {
    "rust_vs_py": {
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-93583fda4fc64946"
     },
     "py": {
      "kind": "digest",
      "id": "fnv1a64-93583fda4fc64946"
     },
     "status": "same"
    },
    "km_vs_laya": {
     "modelless": {
      "kind": "digest",
      "id": "fnv1a64-93583fda4fc64946"
     },
     "rust": {
      "kind": "digest",
      "id": "fnv1a64-93583fda4fc64946"
     },
     "status": "same"
    }
   },
   "cases_digest": "fnv1a64-93583fda4fc64946",
   "hybrid": {
    "lane": "Instinct",
    "model": "A1",
    "hard": {
     "n": 32,
     "accuracy": 0.5625,
     "ece": 0.2524139366578311,
     "mean_confidence": 0.6191526229958981,
     "acc_at_50_coverage": 0.75
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.000625,
    "latency_p99_ms": 0.007041,
    "latency_tail_support": 1,
    "serves": "A1",
    "gate": "certified (paired LB95 +0.0021, mean +0.1875)",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "eef77d1",
     "date_utc": "2026-10-02T10:01:40Z"
    },
    "cases_digest": "fnv1a64-93583fda4fc64946"
   },
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 32,
     "accuracy": 0.59375,
     "macro_f1": 0.31459790209790206,
     "ece": 0.19328238326124847,
     "brier": 0.5773680473828969,
     "nll": 1.4904762207928075,
     "aurc": 0.1574584408338639,
     "mean_confidence": 0.7601527173537761,
     "acc_at_50_coverage": 0.875,
     "acc_at_80_coverage": 0.72
    },
    "readout_ece": 0.1635526721290408,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 171.0,
    "latency_p99_ms": 538.0,
    "latency_tail_support": 1,
    "latency_extremes": {
     "first_ms": 110.0,
     "max_ms": 538.0,
     "argmax_case": 9
    },
    "determinism_ok": true,
    "seconds": 7.3523098749999996,
    "n_cases": 16,
    "n_questions": 32,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-93583fda4fc64946",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "00bddd1",
     "date_utc": "2026-09-28T16:27:36Z"
    }
   },
   "paw": {
    "lane": "paw (hosted)",
    "model": "paw-ft-bs48-20260530",
    "posture": "hosted-anonymous",
    "program_id": "e5c9b00b49fb144810e4,1788474fc4e77eb38e3f",
    "spec_file": "scripts/paw_specs/code_fixtures.module.txt,scripts/paw_specs/code_fixtures.is_pub.txt",
    "spec_blake3": "c327469170449a2c0934e754ddc037222622be993bb9c76dff8112d5ddef6752,d1c6382cc8a44a44f96233d8e1afb318232c77a09a07997795f644e2295bdbb2",
    "compile_cache_hit": false,
    "compile_wall_s": 246.145184917,
    "n_cases": 16,
    "n_questions": 32,
    "n_answered": 31,
    "refusals": 1,
    "quote_stripped": 0,
    "accuracy": 0.625,
    "answered_accuracy": 0.6451612903225806,
    "refusal_rate": 0.03125,
    "score_mae_answered": null,
    "within_1_answered": null,
    "refusal_samples": [
     "harness::families"
    ],
    "server_latency_p50_ms": 77.0,
    "determinism_ok": true,
    "cases_digest": "fnv1a64-93583fda4fc64946",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "bb86069",
     "date_utc": "2026-09-29T00:02:21Z"
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A1",
    "hard": {
     "n": 32,
     "accuracy": 0.5625,
     "ece": 0.2524139366578311,
     "mean_confidence": 0.6191526229958981,
     "acc_at_50_coverage": 0.75
    },
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "latency_p50_ms": 0.002,
    "latency_p99_ms": 0.0218,
    "latency_tail_support": 1,
    "serves": "tier-fallback",
    "served_by": "Instinct (A1)",
    "fallback_note": "dead by law \u2014 the encoder reference reads 0.3575, 26.7 pt under the bar, and the class's measured head-lift ceiling (+14.7 pt) cannot close it (riir-train issue 600 T5)",
    "derived": true,
    "source_run": {
     "git_sha": "07119b4",
     "date_utc": "2026-09-30T00:59:53Z"
    },
    "cases_digest": "fnv1a64-93583fda4fc64946"
   },
   "bekko": {
    "lane": "bekko (reference)",
    "model": "hotchpotch/bekko-system-one-v0-400m",
    "hard": {
     "n": 32,
     "accuracy": 0.59375,
     "macro_f1": 0.3242845117845118,
     "ece": 0.18598971096798778,
     "brier": 0.5396329874521034,
     "nll": 1.3008256907880433,
     "aurc": 0.18061896030663246,
     "mean_confidence": 0.5993335810489953,
     "acc_at_50_coverage": 0.9375,
     "acc_at_80_coverage": 0.68
    },
    "readout_ece": 0.18598971096798778,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "abstain_causes": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 16,
    "n_questions": 32,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "jdi_chance": 0.375,
    "jdi_skill": 0.35,
    "cases_digest": "fnv1a64-93583fda4fc64946",
    "latency_quotable": false,
    "source_run": {
     "git_sha": "8ca8770",
     "date_utc": "2026-10-02T12:08:12Z"
    }
   }
  },
  {
   "name": "thai_wisesight",
   "n_questions": 400,
   "n_cases": 400,
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 400,
     "accuracy": 0.475,
     "macro_f1": 0.4011828455419324,
     "ece": 0.36124100334942344,
     "brier": 0.841119386323466,
     "nll": 1.7111618732338834,
     "aurc": 0.4940801666417539,
     "mean_confidence": 0.8349697407335043,
     "acc_at_50_coverage": 0.52,
     "acc_at_80_coverage": 0.484375
    },
    "readout_ece": 0.25726806574518457,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 107.0,
    "latency_p99_ms": 393.0,
    "latency_tail_support": 5,
    "latency_extremes": {
     "first_ms": 119.0,
     "max_ms": 946.0,
     "argmax_case": 123
    },
    "determinism_ok": true,
    "seconds": 46.735881167,
    "n_cases": 400,
    "n_questions": 400,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-fb7285428b60cfa1",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "3a7be30",
     "date_utc": "2026-09-28T15:36:40Z"
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 400,
       "accuracy": 0.4675,
       "macro_f1": 0.39621142694817335,
       "ece": 0.36819564446806907,
       "brier": 0.8414794855041283,
       "nll": 1.713083528154888,
       "aurc": 0.49315841829612755,
       "mean_confidence": 0.8356956444680691,
       "acc_at_50_coverage": 0.515,
       "acc_at_80_coverage": 0.484375
      },
      "readout_ece": 0.26103172260204727,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 57.0,
      "latency_p99_ms": 96.0,
      "latency_tail_support": 5,
      "latency_extremes": {
       "first_ms": 89.0,
       "max_ms": 166.0,
       "argmax_case": 123
      },
      "determinism_ok": true,
      "seconds": 25.007472,
      "n_cases": 400,
      "n_questions": 400,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-fb7285428b60cfa1",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T14:02:42Z"
      }
     }
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-ref (laya-multilingual encoder, reference logits \u2014 no head earned)",
    "hard": {
     "n": 400,
     "accuracy": 0.4075
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "serves": "\u2717 (record-only; no head earned and the suite is unsold in every instinct lane)",
    "gate": "acc 0.4075 (163/400 rows) \u2014 the laya-multilingual REFERENCE read (untrained logits, gold-only eval, class-relative gate REFUSED the head: holdout 0.4213 < ref holdout 0.4550). Below the openthai comparison lane's 0.475 \u2014 the multilingual encoder is not yet competitive here. Train-side protocol (case wire mirrors reflex build_thai_wisesight verbatim; 4000-row train pool, 800 holdout, fresh NLEH fit); encode wall 6.3 s / 400 rows (M3 metal, loaded box).",
    "device": "metal",
    "record_only": true,
    "latency_quotable": false,
    "source_run": {
     "git_sha": "e0440e3",
     "date_utc": "2026-10-01T19:20:00Z"
    }
   }
  },
  {
   "name": "thai_sib200",
   "n_questions": 204,
   "n_cases": 204,
   "openthai": {
    "lane": "openthai (reference)",
    "model": "openthai-systemone",
    "hard": {
     "n": 204,
     "accuracy": 0.8382352941176471,
     "macro_f1": 0.8178733804911904,
     "ece": 0.06707383619219648,
     "brier": 0.2748790805227198,
     "nll": 0.5618465140813915,
     "aurc": 0.07182042533936529,
     "mean_confidence": 0.848810897592236,
     "acc_at_50_coverage": 0.9411764705882353,
     "acc_at_80_coverage": 0.8773006134969326
    },
    "readout_ece": 0.09746096452409891,
    "raw_abstain": null,
    "calibrated_abstain": null,
    "readout_ece_raw": null,
    "readout_ece_calibrated": null,
    "floor_ece": null,
    "g1_pass": null,
    "g1_verdict": null,
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "latency_p50_ms": 112.0,
    "latency_p99_ms": 178.0,
    "latency_tail_support": 3,
    "latency_extremes": {
     "first_ms": 116.0,
     "max_ms": 227.0,
     "argmax_case": 102
    },
    "determinism_ok": true,
    "seconds": 27.023358125,
    "n_cases": 204,
    "n_questions": 204,
    "score_threshold": null,
    "distance_threshold": null,
    "threshold_recommendation": null,
    "corpus_cap": null,
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": null,
    "pair_head_ab": null,
    "cases_digest": "fnv1a64-b6e532604b031e8d",
    "latency_quotable": true,
    "source_run": {
     "git_sha": "3a7be30",
     "date_utc": "2026-09-28T15:36:40Z"
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "openthai": {
      "lane": "openthai (reference)",
      "model": "openthai-systemone",
      "hard": {
       "n": 204,
       "accuracy": 0.8382352941176471,
       "macro_f1": 0.8178733804911904,
       "ece": 0.06511045860893587,
       "brier": 0.27483704765782135,
       "nll": 0.5621628036142624,
       "aurc": 0.07176961221068476,
       "mean_confidence": 0.8493531900001507,
       "acc_at_50_coverage": 0.9411764705882353,
       "acc_at_80_coverage": 0.8773006134969326
      },
      "readout_ece": 0.09236668563918003,
      "raw_abstain": null,
      "calibrated_abstain": null,
      "readout_ece_raw": null,
      "readout_ece_calibrated": null,
      "floor_ece": null,
      "g1_pass": null,
      "g1_verdict": null,
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "latency_p50_ms": 60.0,
      "latency_p99_ms": 75.0,
      "latency_tail_support": 3,
      "latency_extremes": {
       "first_ms": 71.0,
       "max_ms": 80.0,
       "argmax_case": 11
      },
      "determinism_ok": true,
      "seconds": 13.9695856,
      "n_cases": 204,
      "n_questions": 204,
      "score_threshold": null,
      "distance_threshold": null,
      "threshold_recommendation": null,
      "corpus_cap": null,
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": null,
      "pair_head_ab": null,
      "cases_digest": "fnv1a64-b6e532604b031e8d",
      "latency_quotable": null,
      "source_run": {
       "git_sha": "bee92d3",
       "date_utc": "2026-09-28T14:02:42Z"
      }
     }
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "ENC-ref (laya-multilingual encoder, reference logits \u2014 no head earned)",
    "hard": {
     "n": 204,
     "accuracy": 0.7843137254901961
    },
    "consult_rate": 1.0,
    "latency_scope": "arm-only",
    "latency_rows": "questions",
    "serves": "\u2717 (record-only; no head earned and the suite is unsold in every instinct lane)",
    "gate": "acc 0.7843 (160/204 rows) \u2014 the laya-multilingual REFERENCE read (untrained logits, gold-only eval, class-relative gate REFUSED the head: holdout 0.6800 < ref holdout 0.7200). Below the openthai comparison lane's 0.838. Train-side protocol (case wire mirrors reflex build_thai_sib200 verbatim; 701-row train pool, 200 holdout, fresh NLEH fit); encode wall 3.6 s / 204 rows (M3 metal, loaded box).",
    "device": "metal",
    "record_only": true,
    "latency_quotable": false,
    "source_run": {
     "git_sha": "e0440e3",
     "date_utc": "2026-10-01T19:20:00Z"
    }
   }
  },
  {
   "name": "s1mb_choice",
   "n_questions": 4329,
   "n_cases": 4329,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 4329,
     "accuracy": 0.22961422961422961,
     "macro_f1": 0.05728477985464882,
     "ece": 0.04368157599910449,
     "brier": 0.7720725868820079,
     "nll": 1.843002186277205,
     "aurc": 0.6642965880999749,
     "mean_confidence": 0.24446859121964598,
     "acc_at_50_coverage": 0.3202402957486137,
     "acc_at_80_coverage": 0.2682645105399942
    },
    "readout_ece": 0.20603674965280053,
    "raw_abstain": {
     "abstain_rate": 0.7803187803187803,
     "selective_accuracy": 0.138801261829653,
     "selective_n": 951
    },
    "calibrated_abstain": {
     "abstain_rate": 0.7803187803187803,
     "selective_accuracy": 0.138801261829653,
     "selective_n": 951
    },
    "abstain_causes": {
     "score_gate": 2686,
     "distance_gate": 692,
     "grammar_invalid": 0
    },
    "readout_ece_raw": 0.20603674965280053,
    "readout_ece_calibrated": 0.20603674965280053,
    "floor_ece": 0.40613977927410666,
    "g1_pass": false,
    "g1_verdict": "fail",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 4329,
    "n_questions": 4329,
    "score_threshold": 0.010039693,
    "distance_threshold": 0.5756306,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.010039693,
      "status": "fitted",
      "targetAccuracy": 0.15,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     },
     "distance": {
      "threshold": 0.5756306,
      "status": "fitted",
      "targetAccuracy": 0.16428572,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 16,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": [
     {
      "gold": "not_mentioned",
      "pred": "contradiction",
      "count": 312,
      "share_of_errors": 0.09355322338830585
     },
     {
      "gold": "entailment",
      "pred": "contradiction",
      "count": 257,
      "share_of_errors": 0.07706146926536732
     },
     {
      "gold": "not_mentioned",
      "pred": "entailment",
      "count": 108,
      "share_of_errors": 0.032383808095952024
     },
     {
      "gold": "entailment",
      "pred": "neutral",
      "count": 80,
      "share_of_errors": 0.0239880059970015
     },
     {
      "gold": "contradiction",
      "pred": "entailment",
      "count": 61,
      "share_of_errors": 0.018290854572713643
     },
     {
      "gold": "contradiction",
      "pred": "neutral",
      "count": 53,
      "share_of_errors": 0.015892053973013492
     },
     {
      "gold": "option_1",
      "pred": "option_0",
      "count": 53,
      "share_of_errors": 0.015892053973013492
     },
     {
      "gold": "neutral",
      "pred": "entailment",
      "count": 46,
      "share_of_errors": 0.013793103448275862
     },
     {
      "gold": "option_2",
      "pred": "option_3",
      "count": 26,
      "share_of_errors": 0.007796101949025487
     },
     {
      "gold": "A",
      "pred": "B",
      "count": 25,
      "share_of_errors": 0.0074962518740629685
     },
     {
      "gold": "2",
      "pred": "3",
      "count": 24,
      "share_of_errors": 0.00719640179910045
     },
     {
      "gold": "option_0",
      "pred": "option_1",
      "count": 22,
      "share_of_errors": 0.006596701649175412
     }
    ],
    "pair_head_ab": null,
    "jdi_chance": 0.23331023331023332,
    "jdi_skill": 0.0,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.13536291792988778
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.09499485112726685
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.15231328934431076
      }
     ]
    },
    "corpus_fallbacks": [
     "getting_spare_card",
     "verify_my_identity",
     "virtual_card_not_working",
     "disposable_card_limits",
     "reverted_card_payment?",
     "pending_cash_withdrawal",
     "get_physical_card",
     "lost_or_stolen_card",
     "top_up_by_card_charge",
     "extra_charge_on_statement",
     "automatic_top_up",
     "get_disposable_virtual_card",
     "transfer_not_received_by_recipient",
     "request_refund",
     "pending_card_payment",
     "declined_transfer",
     "age_limit",
     "card_payment_fee_charged",
     "verify_source_of_funds",
     "declined_card_payment",
     "card_payment_wrong_exchange_rate",
     "wrong_amount_of_cash_received",
     "atm_support",
     "transaction_charged_twice",
     "wrong_exchange_rate_for_cash_withdrawal",
     "supported_cards_and_currencies",
     "balance_not_updated_after_bank_transfer",
     "card_swallowed",
     "failed_transfer",
     "cancel_transfer",
     "direct_debit_payment_not_recognised",
     "top_up_limits",
     "exchange_via_app",
     "passcode_forgotten",
     "top_up_by_cash_or_cheque",
     "top_up_by_bank_transfer_charge",
     "beneficiary_not_allowed",
     "compromised_card",
     "unable_to_verify_identity",
     "card_not_working",
     "exchange_rate",
     "card_payment_not_recognised",
     "transfer_into_account",
     "cash_withdrawal_charge",
     "card_arrival",
     "card_acceptance",
     "activate_my_card",
     "transfer_timing",
     "terminate_account",
     "topping_up_by_card",
     "cash_withdrawal_not_recognised",
     "getting_virtual_card",
     "country_support",
     "Refund_not_showing_up",
     "balance_not_updated_after_cheque_or_cash_deposit",
     "change_pin",
     "pending_transfer",
     "edit_personal_details",
     "verify_top_up",
     "pending_top_up",
     "why_verify_identity",
     "card_about_to_expire",
     "exchange_charge",
     "transfer_fee_charged",
     "top_up_failed",
     "fiat_currency_support",
     "card_delivery_estimate",
     "rewards_balance",
     "translate",
     "cancel",
     "maybe",
     "restaurant_reviews",
     "repeat",
     "todo_list",
     "flight_status",
     "last_maintenance",
     "pto_balance",
     "directions",
     "restaurant_reservation",
     "confirm_reservation",
     "order_status",
     "tell_joke",
     "oos",
     "calculator",
     "gas_type",
     "travel_suggestion",
     "share_location",
     "shopping_list",
     "change_language",
     "timezone",
     "yes",
     "book_flight",
     "payday",
     "no",
     "meeting_schedule",
     "do_you_have_pets",
     "order",
     "credit_limit_change",
     "recipe",
     "calendar_update",
     "improve_credit_score",
     "pin_change",
     "update_playlist",
     "cancel_reservation",
     "date",
     "measurement_conversion",
     "definition",
     "oil_change_how",
     "whisper_mode",
     "flip_coin",
     "find_phone",
     "todo_list_update",
     "redeem_rewards",
     "change_user_name",
     "change_ai_name",
     "min_payment",
     "spelling",
     "how_busy",
     "change_accent",
     "schedule_meeting",
     "food_last",
     "international_fees",
     "book_hotel",
     "where_are_you_from",
     "change_volume",
     "are_you_a_bot",
     "tire_change",
     "ingredients_list",
     "current_location",
     "goodbye",
     "insurance_change",
     "w2",
     "pto_used",
     "new_card",
     "traffic",
     "balance",
     "fun_fact",
     "ingredient_substitution",
     "transfer",
     "freeze_account",
     "schedule_maintenance",
     "thank_you",
     "what_is_your_name",
     "gas",
     "vaccines",
     "accept_reservations",
     "transactions",
     "cook_time",
     "distance",
     "routing",
     "report_lost_card",
     "next_song",
     "carry_on",
     "what_song",
     "travel_notification",
     "apr",
     "nutrition_info",
     "who_made_you",
     "next_holiday",
     "bill_balance",
     "reminder_update",
     "how_old_are_you",
     "mpg",
     "rollover_401k",
     "account_blocked",
     "plug_type",
     "card_declined",
     "smart_home",
     "meal_suggestion",
     "direct_deposit",
     "sync_device",
     "pay_bill",
     "uber",
     "pto_request_status",
     "taxes",
     "order_checks",
     "greeting",
     "income",
     "spending_history",
     "calories",
     "reset_settings",
     "expiration_date",
     "tire_pressure",
     "meaning_of_life",
     "make_call",
     "insurance",
     "restaurant_suggestion",
     "interest_rate",
     "credit_limit",
     "bill_due",
     "who_do_you_work_for",
     "report_fraud",
     "roll_dice",
     "change_speed",
     "damaged_card",
     "shopping_list_update",
     "what_are_your_hobbies",
     "lost_luggage",
     "oil_change_when",
     "text",
     "travel_alert",
     "car_rental",
     "international_visa",
     "application_status",
     "jump_start",
     "credit_score",
     "what_can_i_ask_you",
     "replacement_card_duration",
     "pto_request",
     "national_id",
     "company_name",
     "account_number",
     "password",
     "url",
     "-47",
     "-4",
     "14",
     "34",
     "24",
     "5760",
     "5756",
     "5750",
     "5765",
     "33",
     "-2",
     "8",
     "12",
     "29",
     "16",
     "52",
     "-5",
     "17",
     "72",
     "80",
     "78",
     "103",
     "98",
     "48",
     "200",
     "150",
     "180",
     "196",
     "64",
     "114",
     "59",
     "84",
     "38",
     "63",
     "43",
     "32",
     "21",
     "20",
     "-18",
     "27",
     "149",
     "197",
     "172",
     "147",
     "88",
     "93",
     "-1",
     "-3",
     "350",
     "397",
     "400",
     "375",
     "248",
     "254",
     "255",
     "251",
     "6580",
     "6598",
     "6600",
     "6593",
     "293",
     "305",
     "288",
     "13997",
     "14000",
     "13993",
     "14007",
     "-33",
     "475",
     "504",
     "501",
     "500",
     "11",
     "40",
     "90",
     "70",
     "89",
     "1198",
     "1194",
     "1202",
     "1173",
     "297",
     "310",
     "-15",
     "-10",
     "71",
     "121",
     "-21",
     "36",
     "122",
     "128",
     "125",
     "190",
     "207",
     "47",
     "39",
     "92",
     "130",
     "76",
     "195",
     "41",
     "46",
     "283",
     "273",
     "44",
     "18",
     "-8",
     "62",
     "5",
     "55",
     "play game",
     "iot coffee",
     "transport traffic",
     "calendar query",
     "qa factoid",
     "play audiobook",
     "iot cleaning",
     "alarm remove",
     "general negate",
     "recommendation events",
     "alarm set",
     "play podcasts",
     "transport query",
     "general commandstop",
     "iot wemo off",
     "music settings",
     "general explain",
     "lists query",
     "general joke",
     "iot hue lightup",
     "email query",
     "takeaway query",
     "play radio",
     "lists remove",
     "music query",
     "recommendation movies",
     "close_no_action",
     "delivery",
     "stop",
     "harmful",
     "qa_stock",
     "recommendation_movies",
     "audio_volume_up",
     "play_audiobook",
     "recommendation_locations",
     "cooking_recipe",
     "datetime_convert",
     "cooking_query",
     "qa_currency",
     "weather_query",
     "music_query",
     "iot_hue_lightdim",
     "iot_hue_lightup",
     "qa_definition",
     "audio_volume_down",
     "general_greet",
     "iot_hue_lightoff",
     "alarm_set",
     "general_quirky",
     "play_podcasts",
     "email_addcontact",
     "cooking",
     "IN:LIKE_MUSIC",
     "IN:GET_RECIPES",
     "IN:GET_TIMER",
     "IN:CREATE_PLAYLIST_MUSIC",
     "IN:PAUSE_MUSIC",
     "IN:SET_DEFAULT_PROVIDER_CALLING",
     "IN:SWITCH_CALL",
     "IN:SET_UNAVAILABLE",
     "IN:GET_EDUCATION_DEGREE",
     "IN:GET_MAJOR",
     "IN:GET_TRACK_INFO_MUSIC",
     "IN:GET_ATTENDEE_EVENT",
     "IN:UPDATE_METHOD_CALL",
     "IN:GET_AVAILABILITY",
     "IN:GET_EMPLOYMENT_TIME",
     "IN:GET_REMINDER_DATE_TIME",
     "IN:CREATE_REMINDER",
     "IN:SKIP_TRACK_MUSIC",
     "IN:GET_AGE",
     "IN:GET_LIFE_EVENT",
     "IN:STOP_SHUFFLE_MUSIC",
     "IN:GET_CALL",
     "IN:MERGE_CALL",
     "IN:DELETE_ALARM",
     "IN:UNLOOP_MUSIC",
     "IN:SET_DEFAULT_PROVIDER_MUSIC",
     "IN:RESUME_CALL",
     "IN:CREATE_CALL",
     "IN:GET_CALL_TIME",
     "IN:GET_CONTACT_METHOD",
     "IN:GET_UNDERGRAD",
     "IN:GET_GENDER",
     "IN:SHARE_EVENT",
     "IN:RESTART_TIMER",
     "IN:SET_RSVP_INTERESTED",
     "IN:QUESTION_NEWS",
     "IN:GET_DETAILS_NEWS",
     "IN:ADD_TIME_TIMER",
     "IN:GET_MESSAGE_CONTACT",
     "IN:ANSWER_CALL",
     "IN:SET_AVAILABLE",
     "IN:UPDATE_REMINDER_LOCATION",
     "IN:REMOVE_FROM_PLAYLIST_MUSIC",
     "IN:PLAY_MEDIA",
     "IN:HOLD_CALL",
     "IN:GET_STORIES_NEWS",
     "IN:GET_EVENT",
     "IN:FOLLOW_MUSIC",
     "IN:GET_AIRQUALITY",
     "IN:CANCEL_CALL",
     "IN:GET_REMINDER_AMOUNT",
     "IN:GET_WEATHER",
     "IN:UPDATE_REMINDER_DATE_TIME",
     "IN:CANCEL_MESSAGE",
     "IN:IS_TRUE_RECIPES",
     "IN:GET_GROUP",
     "IN:PAUSE_TIMER",
     "IN:DISPREFER",
     "IN:CREATE_TIMER",
     "option_9",
     "option_10",
     "option_11",
     "option_12",
     "option_13",
     "option_14",
     "option_15",
     "mixed",
     "SearchHotel",
     "MakePayment",
     "RequestPayment",
     "GetTimesForMovie",
     "BuyMovieTickets",
     "FindRestaurants",
     "SearchHouse",
     "SearchOnewayFlight",
     "LookupMusic",
     "AddAlarm",
     "GetWeather",
     "GetRide",
     "ShareLocation",
     "teacher",
     "supervisor",
     "clerk",
     "lawyer",
     "baker",
     "carpenter",
     "developer",
     "auditor"
    ],
    "corpus_digest": "fnv1a64-e5cf849bd33ea8e1",
    "cases_digest": "fnv1a64-46281107df052d0e",
    "source_run": {
     "git_sha": "d009604",
     "date_utc": "2026-10-03T08:35:02Z"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A0",
    "hard": {
     "n": 4329,
     "accuracy": 0.22961422961422961,
     "ece": 0.20603674965280053,
     "mean_confidence": 0.021036477420426566,
     "acc_at_50_coverage": 0.17236598890942698
    },
    "consult_rate": 0.0,
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "serves": "A0",
    "gate": "served by the reflex half \u2014 the best measured arm on this suite",
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 4329,
       "accuracy": 0.22961422961422961,
       "macro_f1": 0.05728477985464882,
       "ece": 0.043681576005988834,
       "brier": 0.7720725868744388,
       "nll": 1.8430021862719967,
       "aurc": 0.6642965880999749,
       "mean_confidence": 0.2444685912265303,
       "acc_at_50_coverage": 0.3202402957486137,
       "acc_at_80_coverage": 0.2682645105399942
      },
      "readout_ece": 0.2060367496941066,
      "raw_abstain": {
       "abstain_rate": 0.7803187803187803,
       "selective_accuracy": 0.138801261829653,
       "selective_n": 951
      },
      "calibrated_abstain": {
       "abstain_rate": 0.7803187803187803,
       "selective_accuracy": 0.138801261829653,
       "selective_n": 951
      },
      "abstain_causes": {
       "score_gate": 2686,
       "distance_gate": 692,
       "grammar_invalid": 0
      },
      "readout_ece_raw": 0.2060367496941066,
      "readout_ece_calibrated": 0.2060367496941066,
      "floor_ece": 0.40613977927410666,
      "g1_pass": false,
      "g1_verdict": "fail",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "determinism_ok": true,
      "determinism_n": 10,
      "n_cases": 4329,
      "n_questions": 4329,
      "score_threshold": 0.010039693,
      "distance_threshold": 0.5756306,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.010039693,
        "status": "fitted",
        "targetAccuracy": 0.15,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       },
       "distance": {
        "threshold": 0.5756306,
        "status": "fitted",
        "targetAccuracy": 0.16428572,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 16,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": [
       {
        "gold": "not_mentioned",
        "pred": "contradiction",
        "count": 312,
        "share_of_errors": 0.09355322338830585
       },
       {
        "gold": "entailment",
        "pred": "contradiction",
        "count": 257,
        "share_of_errors": 0.07706146926536732
       },
       {
        "gold": "not_mentioned",
        "pred": "entailment",
        "count": 108,
        "share_of_errors": 0.032383808095952024
       },
       {
        "gold": "entailment",
        "pred": "neutral",
        "count": 80,
        "share_of_errors": 0.0239880059970015
       },
       {
        "gold": "contradiction",
        "pred": "entailment",
        "count": 61,
        "share_of_errors": 0.018290854572713643
       },
       {
        "gold": "contradiction",
        "pred": "neutral",
        "count": 53,
        "share_of_errors": 0.015892053973013492
       },
       {
        "gold": "option_1",
        "pred": "option_0",
        "count": 53,
        "share_of_errors": 0.015892053973013492
       },
       {
        "gold": "neutral",
        "pred": "entailment",
        "count": 46,
        "share_of_errors": 0.013793103448275862
       },
       {
        "gold": "option_2",
        "pred": "option_3",
        "count": 26,
        "share_of_errors": 0.007796101949025487
       },
       {
        "gold": "A",
        "pred": "B",
        "count": 25,
        "share_of_errors": 0.0074962518740629685
       },
       {
        "gold": "2",
        "pred": "3",
        "count": 24,
        "share_of_errors": 0.00719640179910045
       },
       {
        "gold": "option_0",
        "pred": "option_1",
        "count": 22,
        "share_of_errors": 0.006596701649175412
       }
      ],
      "pair_head_ab": null,
      "jdi_chance": 0.23331023331023332,
      "jdi_skill": 0.0,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.13536291792988778
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.09499485112726685
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.15231328934431076
        }
       ]
      },
      "corpus_fallbacks": [
       "getting_spare_card",
       "verify_my_identity",
       "virtual_card_not_working",
       "disposable_card_limits",
       "reverted_card_payment?",
       "pending_cash_withdrawal",
       "get_physical_card",
       "lost_or_stolen_card",
       "top_up_by_card_charge",
       "extra_charge_on_statement",
       "automatic_top_up",
       "get_disposable_virtual_card",
       "transfer_not_received_by_recipient",
       "request_refund",
       "pending_card_payment",
       "declined_transfer",
       "age_limit",
       "card_payment_fee_charged",
       "verify_source_of_funds",
       "declined_card_payment",
       "card_payment_wrong_exchange_rate",
       "wrong_amount_of_cash_received",
       "atm_support",
       "transaction_charged_twice",
       "wrong_exchange_rate_for_cash_withdrawal",
       "supported_cards_and_currencies",
       "balance_not_updated_after_bank_transfer",
       "card_swallowed",
       "failed_transfer",
       "cancel_transfer",
       "direct_debit_payment_not_recognised",
       "top_up_limits",
       "exchange_via_app",
       "passcode_forgotten",
       "top_up_by_cash_or_cheque",
       "top_up_by_bank_transfer_charge",
       "beneficiary_not_allowed",
       "compromised_card",
       "unable_to_verify_identity",
       "card_not_working",
       "exchange_rate",
       "card_payment_not_recognised",
       "transfer_into_account",
       "cash_withdrawal_charge",
       "card_arrival",
       "card_acceptance",
       "activate_my_card",
       "transfer_timing",
       "terminate_account",
       "topping_up_by_card",
       "cash_withdrawal_not_recognised",
       "getting_virtual_card",
       "country_support",
       "Refund_not_showing_up",
       "balance_not_updated_after_cheque_or_cash_deposit",
       "change_pin",
       "pending_transfer",
       "edit_personal_details",
       "verify_top_up",
       "pending_top_up",
       "why_verify_identity",
       "card_about_to_expire",
       "exchange_charge",
       "transfer_fee_charged",
       "top_up_failed",
       "fiat_currency_support",
       "card_delivery_estimate",
       "rewards_balance",
       "translate",
       "cancel",
       "maybe",
       "restaurant_reviews",
       "repeat",
       "todo_list",
       "flight_status",
       "last_maintenance",
       "pto_balance",
       "directions",
       "restaurant_reservation",
       "confirm_reservation",
       "order_status",
       "tell_joke",
       "oos",
       "calculator",
       "gas_type",
       "travel_suggestion",
       "share_location",
       "shopping_list",
       "change_language",
       "timezone",
       "yes",
       "book_flight",
       "payday",
       "no",
       "meeting_schedule",
       "do_you_have_pets",
       "order",
       "credit_limit_change",
       "recipe",
       "calendar_update",
       "improve_credit_score",
       "pin_change",
       "update_playlist",
       "cancel_reservation",
       "date",
       "measurement_conversion",
       "definition",
       "oil_change_how",
       "whisper_mode",
       "flip_coin",
       "find_phone",
       "todo_list_update",
       "redeem_rewards",
       "change_user_name",
       "change_ai_name",
       "min_payment",
       "spelling",
       "how_busy",
       "change_accent",
       "schedule_meeting",
       "food_last",
       "international_fees",
       "book_hotel",
       "where_are_you_from",
       "change_volume",
       "are_you_a_bot",
       "tire_change",
       "ingredients_list",
       "current_location",
       "goodbye",
       "insurance_change",
       "w2",
       "pto_used",
       "new_card",
       "traffic",
       "balance",
       "fun_fact",
       "ingredient_substitution",
       "transfer",
       "freeze_account",
       "schedule_maintenance",
       "thank_you",
       "what_is_your_name",
       "gas",
       "vaccines",
       "accept_reservations",
       "transactions",
       "cook_time",
       "distance",
       "routing",
       "report_lost_card",
       "next_song",
       "carry_on",
       "what_song",
       "travel_notification",
       "apr",
       "nutrition_info",
       "who_made_you",
       "next_holiday",
       "bill_balance",
       "reminder_update",
       "how_old_are_you",
       "mpg",
       "rollover_401k",
       "account_blocked",
       "plug_type",
       "card_declined",
       "smart_home",
       "meal_suggestion",
       "direct_deposit",
       "sync_device",
       "pay_bill",
       "uber",
       "pto_request_status",
       "taxes",
       "order_checks",
       "greeting",
       "income",
       "spending_history",
       "calories",
       "reset_settings",
       "expiration_date",
       "tire_pressure",
       "meaning_of_life",
       "make_call",
       "insurance",
       "restaurant_suggestion",
       "interest_rate",
       "credit_limit",
       "bill_due",
       "who_do_you_work_for",
       "report_fraud",
       "roll_dice",
       "change_speed",
       "damaged_card",
       "shopping_list_update",
       "what_are_your_hobbies",
       "lost_luggage",
       "oil_change_when",
       "text",
       "travel_alert",
       "car_rental",
       "international_visa",
       "application_status",
       "jump_start",
       "credit_score",
       "what_can_i_ask_you",
       "replacement_card_duration",
       "pto_request",
       "national_id",
       "company_name",
       "account_number",
       "password",
       "url",
       "-47",
       "-4",
       "14",
       "34",
       "24",
       "5760",
       "5756",
       "5750",
       "5765",
       "33",
       "-2",
       "8",
       "12",
       "29",
       "16",
       "52",
       "-5",
       "17",
       "72",
       "80",
       "78",
       "103",
       "98",
       "48",
       "200",
       "150",
       "180",
       "196",
       "64",
       "114",
       "59",
       "84",
       "38",
       "63",
       "43",
       "32",
       "21",
       "20",
       "-18",
       "27",
       "149",
       "197",
       "172",
       "147",
       "88",
       "93",
       "-1",
       "-3",
       "350",
       "397",
       "400",
       "375",
       "248",
       "254",
       "255",
       "251",
       "6580",
       "6598",
       "6600",
       "6593",
       "293",
       "305",
       "288",
       "13997",
       "14000",
       "13993",
       "14007",
       "-33",
       "475",
       "504",
       "501",
       "500",
       "11",
       "40",
       "90",
       "70",
       "89",
       "1198",
       "1194",
       "1202",
       "1173",
       "297",
       "310",
       "-15",
       "-10",
       "71",
       "121",
       "-21",
       "36",
       "122",
       "128",
       "125",
       "190",
       "207",
       "47",
       "39",
       "92",
       "130",
       "76",
       "195",
       "41",
       "46",
       "283",
       "273",
       "44",
       "18",
       "-8",
       "62",
       "5",
       "55",
       "play game",
       "iot coffee",
       "transport traffic",
       "calendar query",
       "qa factoid",
       "play audiobook",
       "iot cleaning",
       "alarm remove",
       "general negate",
       "recommendation events",
       "alarm set",
       "play podcasts",
       "transport query",
       "general commandstop",
       "iot wemo off",
       "music settings",
       "general explain",
       "lists query",
       "general joke",
       "iot hue lightup",
       "email query",
       "takeaway query",
       "play radio",
       "lists remove",
       "music query",
       "recommendation movies",
       "close_no_action",
       "delivery",
       "stop",
       "harmful",
       "qa_stock",
       "recommendation_movies",
       "audio_volume_up",
       "play_audiobook",
       "recommendation_locations",
       "cooking_recipe",
       "datetime_convert",
       "cooking_query",
       "qa_currency",
       "weather_query",
       "music_query",
       "iot_hue_lightdim",
       "iot_hue_lightup",
       "qa_definition",
       "audio_volume_down",
       "general_greet",
       "iot_hue_lightoff",
       "alarm_set",
       "general_quirky",
       "play_podcasts",
       "email_addcontact",
       "cooking",
       "IN:LIKE_MUSIC",
       "IN:GET_RECIPES",
       "IN:GET_TIMER",
       "IN:CREATE_PLAYLIST_MUSIC",
       "IN:PAUSE_MUSIC",
       "IN:SET_DEFAULT_PROVIDER_CALLING",
       "IN:SWITCH_CALL",
       "IN:SET_UNAVAILABLE",
       "IN:GET_EDUCATION_DEGREE",
       "IN:GET_MAJOR",
       "IN:GET_TRACK_INFO_MUSIC",
       "IN:GET_ATTENDEE_EVENT",
       "IN:UPDATE_METHOD_CALL",
       "IN:GET_AVAILABILITY",
       "IN:GET_EMPLOYMENT_TIME",
       "IN:GET_REMINDER_DATE_TIME",
       "IN:CREATE_REMINDER",
       "IN:SKIP_TRACK_MUSIC",
       "IN:GET_AGE",
       "IN:GET_LIFE_EVENT",
       "IN:STOP_SHUFFLE_MUSIC",
       "IN:GET_CALL",
       "IN:MERGE_CALL",
       "IN:DELETE_ALARM",
       "IN:UNLOOP_MUSIC",
       "IN:SET_DEFAULT_PROVIDER_MUSIC",
       "IN:RESUME_CALL",
       "IN:CREATE_CALL",
       "IN:GET_CALL_TIME",
       "IN:GET_CONTACT_METHOD",
       "IN:GET_UNDERGRAD",
       "IN:GET_GENDER",
       "IN:SHARE_EVENT",
       "IN:RESTART_TIMER",
       "IN:SET_RSVP_INTERESTED",
       "IN:QUESTION_NEWS",
       "IN:GET_DETAILS_NEWS",
       "IN:ADD_TIME_TIMER",
       "IN:GET_MESSAGE_CONTACT",
       "IN:ANSWER_CALL",
       "IN:SET_AVAILABLE",
       "IN:UPDATE_REMINDER_LOCATION",
       "IN:REMOVE_FROM_PLAYLIST_MUSIC",
       "IN:PLAY_MEDIA",
       "IN:HOLD_CALL",
       "IN:GET_STORIES_NEWS",
       "IN:GET_EVENT",
       "IN:FOLLOW_MUSIC",
       "IN:GET_AIRQUALITY",
       "IN:CANCEL_CALL",
       "IN:GET_REMINDER_AMOUNT",
       "IN:GET_WEATHER",
       "IN:UPDATE_REMINDER_DATE_TIME",
       "IN:CANCEL_MESSAGE",
       "IN:IS_TRUE_RECIPES",
       "IN:GET_GROUP",
       "IN:PAUSE_TIMER",
       "IN:DISPREFER",
       "IN:CREATE_TIMER",
       "option_9",
       "option_10",
       "option_11",
       "option_12",
       "option_13",
       "option_14",
       "option_15",
       "mixed",
       "SearchHotel",
       "MakePayment",
       "RequestPayment",
       "GetTimesForMovie",
       "BuyMovieTickets",
       "FindRestaurants",
       "SearchHouse",
       "SearchOnewayFlight",
       "LookupMusic",
       "AddAlarm",
       "GetWeather",
       "GetRide",
       "ShareLocation",
       "teacher",
       "supervisor",
       "clerk",
       "lawyer",
       "baker",
       "carpenter",
       "developer",
       "auditor"
      ],
      "corpus_digest": "fnv1a64-e5cf849bd33ea8e1",
      "cases_digest": "fnv1a64-46281107df052d0e",
      "source_run": {
       "git_sha": "3cad008",
       "date_utc": "2026-10-02T20:23:08Z"
      }
     }
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A0",
    "hard": {
     "n": 4329,
     "accuracy": 0.22961422961422961,
     "ece": 0.20603674965280053,
     "mean_confidence": 0.021036477420426566,
     "acc_at_50_coverage": 0.17236598890942698
    },
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": null,
    "latency_p99_ms": null,
    "latency_tail_support": null,
    "serves": "tier-fallback",
    "served_by": "Instinct (A0)",
    "fallback_note": "no seated arm for this lane on this suite \u2014 the served answer is the tier shown (the full-coverage serving law, instinct 057d31a)",
    "derived": true,
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   }
  },
  {
   "name": "s1mb_noul",
   "n_questions": 6173,
   "n_cases": 6173,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 6173,
     "accuracy": 0.7054916572169123,
     "macro_f1": 0.42913836905658626,
     "ece": 0.20416973503486036,
     "brier": 0.5001310888121616,
     "nll": 0.6932784373256037,
     "aurc": 0.37506022202551853,
     "mean_confidence": 0.501321922182052,
     "acc_at_50_coverage": 0.6970187945560596,
     "acc_at_80_coverage": 0.6871202916160389
    },
    "readout_ece": 0.006089712481202825,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.4676818402721529,
     "selective_accuracy": 0.30401704199634816,
     "selective_n": 3286
    },
    "abstain_causes": {
     "score_gate": 0,
     "distance_gate": 2887,
     "grammar_invalid": 0
    },
    "readout_ece_raw": 0.10980077466141282,
    "readout_ece_calibrated": 0.006089712481202825,
    "floor_ece": 0.06142461191525924,
    "g1_pass": true,
    "g1_verdict": "pass",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": null,
    "within_1": null,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 6173,
    "n_questions": 6173,
    "score_threshold": 0.45501745,
    "distance_threshold": 0.6012192,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.45501745,
      "status": "fitted",
      "targetAccuracy": 0.455,
      "support": {
       "n": 200,
       "nPass": 200,
       "nAbstain": 0
      }
     },
     "distance": {
      "threshold": 0.6012192,
      "status": "fitted",
      "targetAccuracy": 0.47142857,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 64,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": [],
    "pair_head_ab": null,
    "jdi_chance": 0.7357848695933906,
    "jdi_skill": 0.0,
    "readout_report": {
     "best_on_cal": "max_prob",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.11997524559497832
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.0015637028217315718
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 0.11997524559497832
      }
     ]
    },
    "corpus_digest": "fnv1a64-0c8778c6d51742f4",
    "cases_digest": "fnv1a64-a887ca350a4a8b7c",
    "source_run": {
     "git_sha": "d009604",
     "date_utc": "2026-10-03T08:35:02Z"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A0",
    "hard": {
     "n": 6173,
     "accuracy": 0.7054916572169123,
     "ece": 0.10980077466141282,
     "mean_confidence": 3.2369676834370933e-05,
     "acc_at_50_coverage": 0.6970187945560596
    },
    "consult_rate": 0.0,
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "serves": "A0",
    "gate": "served by the reflex half (A0 is the argmax)",
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 6173,
       "accuracy": 0.7054916572169123,
       "macro_f1": 0.42913836905658626,
       "ece": 0.20416973503486036,
       "brier": 0.5001310888121616,
       "nll": 0.6932784373256037,
       "aurc": 0.37506022202551853,
       "mean_confidence": 0.501321922182052,
       "acc_at_50_coverage": 0.6970187945560596,
       "acc_at_80_coverage": 0.6871202916160389
      },
      "readout_ece": 0.006089712481202825,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.4676818402721529,
       "selective_accuracy": 0.30401704199634816,
       "selective_n": 3286
      },
      "abstain_causes": {
       "score_gate": 0,
       "distance_gate": 2887,
       "grammar_invalid": 0
      },
      "readout_ece_raw": 0.10980077466141282,
      "readout_ece_calibrated": 0.006089712481202825,
      "floor_ece": 0.06142461191525924,
      "g1_pass": true,
      "g1_verdict": "pass",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": null,
      "within_1": null,
      "determinism_ok": true,
      "determinism_n": 10,
      "n_cases": 6173,
      "n_questions": 6173,
      "score_threshold": 0.45501745,
      "distance_threshold": 0.6012192,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.45501745,
        "status": "fitted",
        "targetAccuracy": 0.455,
        "support": {
         "n": 200,
         "nPass": 200,
         "nAbstain": 0
        }
       },
       "distance": {
        "threshold": 0.6012192,
        "status": "fitted",
        "targetAccuracy": 0.47142857,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 64,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": [],
      "pair_head_ab": null,
      "jdi_chance": 0.7357848695933906,
      "jdi_skill": 0.0,
      "readout_report": {
       "best_on_cal": "max_prob",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.11997524559497832
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.0015637028217315718
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 0.11997524559497832
        }
       ]
      },
      "corpus_digest": "fnv1a64-0c8778c6d51742f4",
      "cases_digest": "fnv1a64-a887ca350a4a8b7c",
      "source_run": {
       "git_sha": "3cad008",
       "date_utc": "2026-10-02T20:23:08Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 6173,
        "accuracy": 0.6964198930827799,
        "macro_f1": 0.6277703084304177,
        "ece": 0.09739886602948292,
        "brier": 0.405517668958367,
        "nll": 0.6462810084974737,
        "aurc": 0.18015165205991684,
        "mean_confidence": 0.7911902478535553,
        "acc_at_50_coverage": 0.8204795852235904,
        "acc_at_80_coverage": 0.7353179424868368
       },
       "readout_ece": 0.09739886602948293,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "abstain_causes": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 18.0,
       "latency_p99_ms": 32.0,
       "latency_tail_support": 62,
       "latency_extremes": {
        "first_ms": 17.0,
        "max_ms": 43.0,
        "argmax_case": 3626
       },
       "determinism_ok": true,
       "determinism_n": 10,
       "seconds": 126.8324202,
       "n_cases": 6173,
       "n_questions": 6173,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "jdi_chance": 0.7357848695933906,
       "jdi_skill": 0.0,
       "cases_digest": "fnv1a64-a887ca350a4a8b7c",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "3cad008",
        "date_utc": "2026-10-02T20:23:08Z"
       }
      }
     }
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A0",
    "hard": {
     "n": 6173,
     "accuracy": 0.7054916572169123,
     "ece": 0.10980077466141282,
     "mean_confidence": 3.2369676834370933e-05,
     "acc_at_50_coverage": 0.6970187945560596
    },
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": null,
    "latency_p99_ms": null,
    "latency_tail_support": null,
    "serves": "tier-fallback",
    "served_by": "Instinct (A0)",
    "fallback_note": "no seated arm for this lane on this suite \u2014 the served answer is the tier shown (the full-coverage serving law, instinct 057d31a)",
    "derived": true,
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   }
  },
  {
   "name": "s1mb_score",
   "n_questions": 2571,
   "n_cases": 2571,
   "modelless": {
    "lane": "KatGPT",
    "model": "modelless",
    "hard": {
     "n": 2571,
     "accuracy": 0.49669389342668224,
     "macro_f1": 0.09700252081297644,
     "ece": 0.2918895636288539,
     "brier": 0.7940942525693938,
     "nll": 1.585251007273525,
     "aurc": 0.580947914748886,
     "mean_confidence": 0.20631818539701702,
     "acc_at_50_coverage": 0.45603112840466925,
     "acc_at_80_coverage": 0.48249027237354086
    },
    "readout_ece": 0.2665456929579174,
    "raw_abstain": {
     "abstain_rate": 1.0,
     "selective_accuracy": 0.0,
     "selective_n": 0
    },
    "calibrated_abstain": {
     "abstain_rate": 0.07701283547257877,
     "selective_accuracy": 0.5132743362831859,
     "selective_n": 2373
    },
    "abstain_causes": {
     "score_gate": 0,
     "distance_gate": 198,
     "grammar_invalid": 0
    },
    "readout_ece_raw": 0.00638683968796335,
    "readout_ece_calibrated": 0.2665456929579174,
    "floor_ece": 0.4304924231429053,
    "g1_pass": false,
    "g1_verdict": "fail",
    "by_question_type": null,
    "soft_acc": null,
    "brier_soft": null,
    "score_mae": 1.4454512420669667,
    "within_1": 0.29015947102294826,
    "determinism_ok": true,
    "determinism_n": 10,
    "n_cases": 2571,
    "n_questions": 2571,
    "score_threshold": 0.23013736,
    "distance_threshold": 0.41533497,
    "threshold_recommendation": {
     "posture": {
      "posture": "percentile",
      "rho": 0.3
     },
     "score": {
      "threshold": 0.23013736,
      "status": "fitted",
      "targetAccuracy": 0.23,
      "support": {
       "n": 200,
       "nPass": 200,
       "nAbstain": 0
      }
     },
     "distance": {
      "threshold": 0.41533497,
      "status": "fitted",
      "targetAccuracy": 0.27857143,
      "support": {
       "n": 200,
       "nPass": 140,
       "nAbstain": 60
      }
     }
    },
    "corpus_cap": {
     "effective": 16,
     "source": "registry",
     "selection": null
    },
    "head_selection": null,
    "nb_selection": null,
    "oc_selection": null,
    "ridge_selection": null,
    "genome_selection": null,
    "transductive": null,
    "confusion": [
     {
      "gold": "1.0",
      "pred": "0.0",
      "count": 509,
      "share_of_errors": 0.39335394126738793
     },
     {
      "gold": "3.0",
      "pred": "0.0",
      "count": 354,
      "share_of_errors": 0.2735703245749614
     },
     {
      "gold": "2.0",
      "pred": "0.0",
      "count": 241,
      "share_of_errors": 0.18624420401854713
     },
     {
      "gold": "4.0",
      "pred": "0.0",
      "count": 135,
      "share_of_errors": 0.10432766615146831
     },
     {
      "gold": "0.3333333333333333",
      "pred": "0.0",
      "count": 12,
      "share_of_errors": 0.00927357032457496
     },
     {
      "gold": "0.5",
      "pred": "0.0",
      "count": 12,
      "share_of_errors": 0.00927357032457496
     },
     {
      "gold": "0.75",
      "pred": "0.0",
      "count": 10,
      "share_of_errors": 0.0077279752704791345
     },
     {
      "gold": "0.25",
      "pred": "0.0",
      "count": 8,
      "share_of_errors": 0.0061823802163833074
     },
     {
      "gold": "0.6666666666666666",
      "pred": "0.0",
      "count": 8,
      "share_of_errors": 0.0061823802163833074
     },
     {
      "gold": "0.1111111111111111",
      "pred": "0.4444444444444444",
      "count": 1,
      "share_of_errors": 0.0007727975270479134
     },
     {
      "gold": "0.2",
      "pred": "0.0",
      "count": 1,
      "share_of_errors": 0.0007727975270479134
     },
     {
      "gold": "0.4",
      "pred": "0.0",
      "count": 1,
      "share_of_errors": 0.0007727975270479134
     }
    ],
    "pair_head_ab": null,
    "jdi_chance": 0.4955270322831583,
    "jdi_skill": 0.0023130300693909285,
    "readout_report": {
     "best_on_cal": "dispatch",
     "candidates": [
      {
       "mode": "dispatch",
       "cal_ece": 0.0006642995774745942
      },
      {
       "mode": "max_prob",
       "cal_ece": 0.09429798692464829
      },
      {
       "mode": "inv_entropy",
       "cal_ece": 2.0314157009124756e-05
      }
     ]
    },
    "corpus_fallbacks": [
     "5",
     "6",
     "7",
     "8",
     "9"
    ],
    "corpus_digest": "fnv1a64-f5d9f67a1fd69443",
    "cases_digest": "fnv1a64-81f40d7a9b57e9b6",
    "source_run": {
     "git_sha": "d009604",
     "date_utc": "2026-10-03T08:35:02Z"
    }
   },
   "hybrid": {
    "lane": "Instinct",
    "model": "A0",
    "hard": {
     "n": 2571,
     "accuracy": 0.49669389342668224,
     "ece": 0.00638683968796335,
     "mean_confidence": 0.0002253734586722004,
     "acc_at_50_coverage": 0.4217898832684825
    },
    "consult_rate": 0.0,
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "serves": "A0",
    "gate": "served by the reflex half \u2014 the best measured arm on this suite",
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   },
   "extra_host_lanes": {
    "4090-win": {
     "modelless": {
      "lane": "KatGPT",
      "model": "modelless",
      "hard": {
       "n": 2571,
       "accuracy": 0.49669389342668224,
       "macro_f1": 0.09700252081297644,
       "ece": 0.2918895636288539,
       "brier": 0.7940942525693938,
       "nll": 1.585251007273525,
       "aurc": 0.580947914748886,
       "mean_confidence": 0.20631818539701702,
       "acc_at_50_coverage": 0.45603112840466925,
       "acc_at_80_coverage": 0.48249027237354086
      },
      "readout_ece": 0.2665456929579174,
      "raw_abstain": {
       "abstain_rate": 1.0,
       "selective_accuracy": 0.0,
       "selective_n": 0
      },
      "calibrated_abstain": {
       "abstain_rate": 0.07701283547257877,
       "selective_accuracy": 0.5132743362831859,
       "selective_n": 2373
      },
      "abstain_causes": {
       "score_gate": 0,
       "distance_gate": 198,
       "grammar_invalid": 0
      },
      "readout_ece_raw": 0.00638683968796335,
      "readout_ece_calibrated": 0.2665456929579174,
      "floor_ece": 0.4304924231429053,
      "g1_pass": false,
      "g1_verdict": "fail",
      "by_question_type": null,
      "soft_acc": null,
      "brier_soft": null,
      "score_mae": 1.4454512420669667,
      "within_1": 0.29015947102294826,
      "determinism_ok": true,
      "determinism_n": 10,
      "n_cases": 2571,
      "n_questions": 2571,
      "score_threshold": 0.23013736,
      "distance_threshold": 0.41533497,
      "threshold_recommendation": {
       "posture": {
        "posture": "percentile",
        "rho": 0.3
       },
       "score": {
        "threshold": 0.23013736,
        "status": "fitted",
        "targetAccuracy": 0.23,
        "support": {
         "n": 200,
         "nPass": 200,
         "nAbstain": 0
        }
       },
       "distance": {
        "threshold": 0.41533497,
        "status": "fitted",
        "targetAccuracy": 0.27857143,
        "support": {
         "n": 200,
         "nPass": 140,
         "nAbstain": 60
        }
       }
      },
      "corpus_cap": {
       "effective": 16,
       "source": "registry",
       "selection": null
      },
      "head_selection": null,
      "nb_selection": null,
      "oc_selection": null,
      "ridge_selection": null,
      "genome_selection": null,
      "transductive": null,
      "confusion": [
       {
        "gold": "1.0",
        "pred": "0.0",
        "count": 509,
        "share_of_errors": 0.39335394126738793
       },
       {
        "gold": "3.0",
        "pred": "0.0",
        "count": 354,
        "share_of_errors": 0.2735703245749614
       },
       {
        "gold": "2.0",
        "pred": "0.0",
        "count": 241,
        "share_of_errors": 0.18624420401854713
       },
       {
        "gold": "4.0",
        "pred": "0.0",
        "count": 135,
        "share_of_errors": 0.10432766615146831
       },
       {
        "gold": "0.3333333333333333",
        "pred": "0.0",
        "count": 12,
        "share_of_errors": 0.00927357032457496
       },
       {
        "gold": "0.5",
        "pred": "0.0",
        "count": 12,
        "share_of_errors": 0.00927357032457496
       },
       {
        "gold": "0.75",
        "pred": "0.0",
        "count": 10,
        "share_of_errors": 0.0077279752704791345
       },
       {
        "gold": "0.25",
        "pred": "0.0",
        "count": 8,
        "share_of_errors": 0.0061823802163833074
       },
       {
        "gold": "0.6666666666666666",
        "pred": "0.0",
        "count": 8,
        "share_of_errors": 0.0061823802163833074
       },
       {
        "gold": "0.1111111111111111",
        "pred": "0.4444444444444444",
        "count": 1,
        "share_of_errors": 0.0007727975270479134
       },
       {
        "gold": "0.2",
        "pred": "0.0",
        "count": 1,
        "share_of_errors": 0.0007727975270479134
       },
       {
        "gold": "0.4",
        "pred": "0.0",
        "count": 1,
        "share_of_errors": 0.0007727975270479134
       }
      ],
      "pair_head_ab": null,
      "jdi_chance": 0.4955270322831583,
      "jdi_skill": 0.0023130300693909285,
      "readout_report": {
       "best_on_cal": "dispatch",
       "candidates": [
        {
         "mode": "dispatch",
         "cal_ece": 0.0006642995774745942
        },
        {
         "mode": "max_prob",
         "cal_ece": 0.09429798692464829
        },
        {
         "mode": "inv_entropy",
         "cal_ece": 2.0314157009124756e-05
        }
       ]
      },
      "corpus_fallbacks": [
       "5",
       "6",
       "7",
       "8",
       "9"
      ],
      "corpus_digest": "fnv1a64-f5d9f67a1fd69443",
      "cases_digest": "fnv1a64-81f40d7a9b57e9b6",
      "source_run": {
       "git_sha": "3cad008",
       "date_utc": "2026-10-02T20:23:08Z"
      }
     },
     "laya": {
      "english": {
       "lane": "laya (rust)",
       "model": "english",
       "hard": {
        "n": 2571,
        "accuracy": 0.2590431738623104,
        "macro_f1": 0.12722887517853823,
        "ece": 0.18714566316608316,
        "brier": 0.8355351283547244,
        "nll": 1.7237879218114467,
        "aurc": 0.6122756154382043,
        "mean_confidence": 0.44618883702839446,
        "acc_at_50_coverage": 0.3478599221789883,
        "acc_at_80_coverage": 0.28745136186770426
       },
       "readout_ece": 0.07684850252819911,
       "raw_abstain": null,
       "calibrated_abstain": null,
       "abstain_causes": null,
       "readout_ece_raw": null,
       "readout_ece_calibrated": null,
       "floor_ece": null,
       "g1_pass": null,
       "g1_verdict": null,
       "by_question_type": null,
       "soft_acc": null,
       "brier_soft": null,
       "score_mae": null,
       "within_1": null,
       "latency_p50_ms": 24.0,
       "latency_p99_ms": 31.0,
       "latency_tail_support": 26,
       "latency_extremes": {
        "first_ms": 23.0,
        "max_ms": 33.0,
        "argmax_case": 2024
       },
       "determinism_ok": true,
       "determinism_n": 10,
       "seconds": 65.2643958,
       "n_cases": 2571,
       "n_questions": 2571,
       "score_threshold": null,
       "distance_threshold": null,
       "threshold_recommendation": null,
       "corpus_cap": null,
       "head_selection": null,
       "nb_selection": null,
       "oc_selection": null,
       "ridge_selection": null,
       "genome_selection": null,
       "transductive": null,
       "confusion": null,
       "pair_head_ab": null,
       "jdi_chance": 0.4955270322831583,
       "jdi_skill": 0.0,
       "cases_digest": "fnv1a64-81f40d7a9b57e9b6",
       "latency_quotable": null,
       "source_run": {
        "git_sha": "3cad008",
        "date_utc": "2026-10-02T20:23:08Z"
       }
      }
     }
    }
   },
   "encoder": {
    "lane": "Rethink",
    "model": "A0",
    "hard": {
     "n": 2571,
     "accuracy": 0.49669389342668224,
     "ece": 0.00638683968796335,
     "mean_confidence": 0.0002253734586722004,
     "acc_at_50_coverage": 0.4217898832684825
    },
    "latency_scope": "seat+arm",
    "latency_rows": "questions",
    "latency_p50_ms": null,
    "latency_p99_ms": null,
    "latency_tail_support": null,
    "serves": "tier-fallback",
    "served_by": "Instinct (A0)",
    "fallback_note": "no seated arm for this lane on this suite \u2014 the served answer is the tier shown (the full-coverage serving law, instinct 057d31a)",
    "derived": true,
    "source_run": {
     "git_sha": "b6425bf",
     "date_utc": "2026-10-03T09:03:51Z"
    }
   }
  }
 ],
 "areas": {
  "version": 4,
  "edition": "2026-10",
  "chance_digest": "9b08e3970c7a38a566ddeba3f37ad8d7",
  "scale": "chance-corrected accuracy: cc = (acc - chance) / (1 - chance); 0 = random guessing, 1 = every question right; per-suite chance = the mean per-question random-pick probability of the harness's own option construction (a dataset fact, not a measurement); values below 0 are BELOW CHANCE and stay negative on purpose \u2014 a lane scoring under random guessing must read that way, and negatives pull area means and the index down by design",
  "suites": {
   "ag_news": {
    "area": "language",
    "chance": 0.25
   },
   "massive_intent_en": {
    "area": "language",
    "chance": 0.05
   },
   "banking77": {
    "area": "language",
    "chance": 0.012987
   },
   "sst5": {
    "area": "sentiment",
    "chance": 0.2
   },
   "emotion": {
    "area": "sentiment",
    "chance": 0.166667
   },
   "xnli_en": {
    "area": "reasoning",
    "chance": 0.333333
   },
   "prompt_injections": {
    "area": "reasoning",
    "chance": 0.5
   },
   "typed_decisions": {
    "area": "decisions",
    "chance": 0.3175
   },
   "code_fixtures": {
    "area": "decisions",
    "chance": 0.3125
   }
  },
  "areas": [
   {
    "id": "language",
    "label": "Language & intent",
    "suites": [
     "ag_news",
     "massive_intent_en",
     "banking77"
    ]
   },
   {
    "id": "sentiment",
    "label": "Sentiment",
    "suites": [
     "sst5",
     "emotion"
    ]
   },
   {
    "id": "reasoning",
    "label": "Reasoning & safety",
    "suites": [
     "xnli_en",
     "prompt_injections"
    ]
   },
   {
    "id": "decisions",
    "label": "Decisions & code",
    "suites": [
     "typed_decisions",
     "code_fixtures"
    ]
   }
  ],
  "lanes": {
   "modelless": {
    "display": "Reflex",
    "color_key": "katgpt",
    "kind": "modelless-in-process",
    "per_suite": {
     "ag_news": {
      "acc": 0.8825,
      "cc": 0.843333
     },
     "massive_intent_en": {
      "acc": 0.78,
      "cc": 0.768421
     },
     "banking77": {
      "acc": 0.842,
      "cc": 0.839921
     },
     "sst5": {
      "acc": 0.396667,
      "cc": 0.245833
     },
     "emotion": {
      "acc": 0.885,
      "cc": 0.862
     },
     "xnli_en": {
      "acc": 0.523333,
      "cc": 0.285
     },
     "prompt_injections": {
      "acc": 0.767241,
      "cc": 0.534483
     },
     "typed_decisions": {
      "acc": 0.5725,
      "cc": 0.373626
     },
     "code_fixtures": {
      "acc": 0.375,
      "cc": 0.090909
     }
    },
    "areas": {
     "language": 0.817225,
     "sentiment": 0.553917,
     "reasoning": 0.409741,
     "decisions": 0.232268
    },
    "index": 0.503288,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "hybrid": {
    "display": "Instinct",
    "color_key": "instinct",
    "kind": "trained-head",
    "per_suite": {
     "ag_news": {
      "acc": 0.8975,
      "cc": 0.863333
     },
     "massive_intent_en": {
      "acc": 0.84,
      "cc": 0.831579
     },
     "banking77": {
      "acc": 0.854,
      "cc": 0.852079
     },
     "sst5": {
      "acc": 0.421667,
      "cc": 0.277083
     },
     "emotion": {
      "acc": 0.885,
      "cc": 0.862
     },
     "xnli_en": {
      "acc": 0.523333,
      "cc": 0.285
     },
     "prompt_injections": {
      "acc": 0.853448,
      "cc": 0.706897
     },
     "typed_decisions": {
      "acc": 0.6475,
      "cc": 0.483516
     },
     "code_fixtures": {
      "acc": 0.5625,
      "cc": 0.363636
     }
    },
    "areas": {
     "language": 0.848997,
     "sentiment": 0.569542,
     "reasoning": 0.495949,
     "decisions": 0.423576
    },
    "index": 0.584516,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "encoder": {
    "display": "Rethink",
    "color_key": "instinct-encoder",
    "kind": "encoder",
    "per_suite": {
     "ag_news": {
      "acc": 0.9475,
      "cc": 0.93
     },
     "massive_intent_en": {
      "acc": 0.826667,
      "cc": 0.817544,
      "fallback": true,
      "served_by": "Instinct (H2(\u03b2=1,nmin=2,\u03c4=8))",
      "record_acc": 0.656667
     },
     "banking77": {
      "acc": 0.854,
      "cc": 0.852079,
      "fallback": true,
      "served_by": "Instinct (H2(\u03b2=2,nmin=8,\u03c4=8))",
      "record_acc": 0.442
     },
     "sst5": {
      "acc": 0.526667,
      "cc": 0.408333
     },
     "emotion": {
      "acc": 0.885,
      "cc": 0.862,
      "fallback": true,
      "served_by": "Instinct (A0)"
     },
     "xnli_en": {
      "acc": 0.86,
      "cc": 0.79
     },
     "prompt_injections": {
      "acc": 0.853448,
      "cc": 0.706897,
      "fallback": true,
      "served_by": "Instinct (A1)",
      "record_acc": 0.801724
     },
     "typed_decisions": {
      "acc": 0.755,
      "cc": 0.641026
     },
     "code_fixtures": {
      "acc": 0.5625,
      "cc": 0.363636,
      "fallback": true,
      "served_by": "Instinct (A1)"
     }
    },
    "areas": {
     "language": 0.866541,
     "sentiment": 0.635166,
     "reasoning": 0.748449,
     "decisions": 0.502331
    },
    "index": 0.688122,
    "coverage": {
     "suites": 4,
     "of": 9
    },
    "complete": false,
    "fallback_suites": [
     "banking77",
     "code_fixtures",
     "emotion",
     "massive_intent_en",
     "prompt_injections"
    ],
    "served_coverage": {
     "suites": 9,
     "of": 9
    }
   },
   "laya": {
    "display": "laya (rust)",
    "color_key": "rust",
    "kind": "encoder",
    "per_suite": {
     "ag_news": {
      "acc": 0.95,
      "cc": 0.933333,
      "ck": "english"
     },
     "massive_intent_en": {
      "acc": 0.693333,
      "cc": 0.677193,
      "ck": "english"
     },
     "banking77": {
      "acc": 0.422,
      "cc": 0.414395,
      "ck": "english"
     },
     "sst5": {
      "acc": 0.371667,
      "cc": 0.214583,
      "ck": "english"
     },
     "emotion": {
      "acc": 0.5925,
      "cc": 0.511,
      "ck": "english"
     },
     "xnli_en": {
      "acc": 0.86,
      "cc": 0.79,
      "ck": "english"
     },
     "prompt_injections": {
      "acc": 0.698276,
      "cc": 0.396552,
      "ck": "english"
     },
     "typed_decisions": {
      "acc": 0.7445,
      "cc": 0.625641,
      "ck": "typed"
     },
     "code_fixtures": {
      "acc": 0.40625,
      "cc": 0.136364,
      "ck": "english"
     }
    },
    "areas": {
     "language": 0.674974,
     "sentiment": 0.362791,
     "reasoning": 0.593276,
     "decisions": 0.381003
    },
    "index": 0.503011,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "python": {
    "display": "laya (python)",
    "color_key": "python",
    "kind": "encoder",
    "per_suite": {
     "ag_news": {
      "acc": 0.95,
      "cc": 0.933333,
      "ck": "english"
     },
     "massive_intent_en": {
      "acc": 0.693333,
      "cc": 0.677193,
      "ck": "english"
     },
     "banking77": {
      "acc": 0.422,
      "cc": 0.414395,
      "ck": "english"
     },
     "sst5": {
      "acc": 0.371667,
      "cc": 0.214583,
      "ck": "english"
     },
     "emotion": {
      "acc": 0.5925,
      "cc": 0.511,
      "ck": "english"
     },
     "xnli_en": {
      "acc": 0.86,
      "cc": 0.79,
      "ck": "english"
     },
     "prompt_injections": {
      "acc": 0.698276,
      "cc": 0.396552,
      "ck": "english"
     },
     "typed_decisions": {
      "acc": 0.7445,
      "cc": 0.625641,
      "ck": "typed"
     },
     "code_fixtures": {
      "acc": 0.40625,
      "cc": 0.136364,
      "ck": "english"
     }
    },
    "areas": {
     "language": 0.674974,
     "sentiment": 0.362791,
     "reasoning": 0.593276,
     "decisions": 0.381003
    },
    "index": 0.503011,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "bekko": {
    "display": "bekko",
    "color_key": "bekko",
    "kind": "python-subprocess",
    "per_suite": {
     "ag_news": {
      "acc": 0.915,
      "cc": 0.886667
     },
     "massive_intent_en": {
      "acc": 0.91,
      "cc": 0.905263
     },
     "banking77": {
      "acc": 0.792,
      "cc": 0.789263
     },
     "sst5": {
      "acc": 0.455,
      "cc": 0.31875
     },
     "emotion": {
      "acc": 0.5825,
      "cc": 0.499
     },
     "xnli_en": {
      "acc": 0.86,
      "cc": 0.79
     },
     "prompt_injections": {
      "acc": 0.508621,
      "cc": 0.017241
     },
     "typed_decisions": {
      "acc": 0.6235,
      "cc": 0.448352
     },
     "code_fixtures": {
      "acc": 0.59375,
      "cc": 0.409091
     }
    },
    "areas": {
     "language": 0.860398,
     "sentiment": 0.408875,
     "reasoning": 0.40362,
     "decisions": 0.428721
    },
    "index": 0.525404,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "openthai": {
    "display": "openthai",
    "color_key": "openthai",
    "kind": "http-oracle",
    "per_suite": {
     "ag_news": {
      "acc": 0.89,
      "cc": 0.853333
     },
     "massive_intent_en": {
      "acc": 0.92,
      "cc": 0.915789
     },
     "banking77": {
      "acc": 0.656,
      "cc": 0.651474
     },
     "sst5": {
      "acc": 0.43,
      "cc": 0.2875
     },
     "emotion": {
      "acc": 0.59,
      "cc": 0.508
     },
     "xnli_en": {
      "acc": 0.896667,
      "cc": 0.845
     },
     "prompt_injections": {
      "acc": 0.62931,
      "cc": 0.258621
     },
     "typed_decisions": {
      "acc": 0.5345,
      "cc": 0.317949
     },
     "code_fixtures": {
      "acc": 0.59375,
      "cc": 0.409091
     }
    },
    "areas": {
     "language": 0.806865,
     "sentiment": 0.39775,
     "reasoning": 0.55181,
     "decisions": 0.36352
    },
    "index": 0.529986,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "paw": {
    "display": "paw (hosted)",
    "color_key": "paw",
    "kind": "compiled-program",
    "per_suite": {
     "ag_news": {
      "acc": 0.79,
      "cc": 0.72
     },
     "massive_intent_en": {
      "acc": 0.51,
      "cc": 0.484211
     },
     "banking77": {
      "acc": 0.42,
      "cc": 0.412368
     },
     "sst5": {
      "acc": 0.395,
      "cc": 0.24375
     },
     "emotion": {
      "acc": 0.5,
      "cc": 0.4
     },
     "xnli_en": {
      "acc": 0.72,
      "cc": 0.58
     },
     "prompt_injections": {
      "acc": 0.637931,
      "cc": 0.275862
     },
     "typed_decisions": {
      "acc": 0.5925,
      "cc": 0.40293
     },
     "code_fixtures": {
      "acc": 0.625,
      "cc": 0.454545
     }
    },
    "areas": {
     "language": 0.53886,
     "sentiment": 0.321875,
     "reasoning": 0.427931,
     "decisions": 0.428737
    },
    "index": 0.429351,
    "coverage": {
     "suites": 9,
     "of": 9
    },
    "complete": true
   },
   "clm@4090-win": {
    "display": "clm",
    "color_key": "clm",
    "kind": "http-oracle",
    "per_suite": {
     "ag_news": {
      "acc": 0.4025,
      "cc": 0.203333
     },
     "massive_intent_en": {
      "acc": 0.266667,
      "cc": 0.22807
     },
     "banking77": {
      "acc": 0.01,
      "cc": -0.003026
     },
     "sst5": {
      "acc": 0.235,
      "cc": 0.04375
     },
     "emotion": {
      "acc": 0.295,
      "cc": 0.154
     },
     "xnli_en": {
      "acc": 0.616667,
      "cc": 0.425
     },
     "prompt_injections": {
      "acc": 0.534483,
      "cc": 0.068966
     },
     "typed_decisions": {
      "acc": 0.3465,
      "cc": 0.042491
     }
    },
    "areas": {
     "language": 0.142792,
     "sentiment": 0.098875,
     "reasoning": 0.246983,
     "decisions": 0.042491
    },
    "index": 0.132785,
    "coverage": {
     "suites": 8,
     "of": 9
    },
    "complete": false,
    "host": "4090-win"
   },
   "gliner@4090-win": {
    "display": "gliner",
    "color_key": "gliner",
    "kind": "python-subprocess",
    "per_suite": {
     "ag_news": {
      "acc": 0.7025,
      "cc": 0.603333
     },
     "massive_intent_en": {
      "acc": 0.823333,
      "cc": 0.814035
     },
     "banking77": {
      "acc": 0.706,
      "cc": 0.702132
     },
     "sst5": {
      "acc": 0.438333,
      "cc": 0.297917
     },
     "emotion": {
      "acc": 0.565,
      "cc": 0.478
     },
     "xnli_en": {
      "acc": 0.476667,
      "cc": 0.215
     },
     "prompt_injections": {
      "acc": 0.681034,
      "cc": 0.362069
     },
     "typed_decisions": {
      "acc": 0.528,
      "cc": 0.308425
     }
    },
    "areas": {
     "language": 0.7065,
     "sentiment": 0.387958,
     "reasoning": 0.288534,
     "decisions": 0.308425
    },
    "index": 0.422854,
    "coverage": {
     "suites": 8,
     "of": 9
    },
    "complete": false,
    "host": "4090-win"
   },
   "agentjev@4090-win": {
    "display": "agentjev",
    "color_key": "agentjev",
    "kind": "http-oracle",
    "per_suite": {
     "ag_news": {
      "acc": 0.8,
      "cc": 0.733333
     },
     "massive_intent_en": {
      "acc": 0.623333,
      "cc": 0.603509
     },
     "banking77": {
      "acc": 0.546,
      "cc": 0.540026
     },
     "sst5": {
      "acc": 0.438333,
      "cc": 0.297917
     },
     "emotion": {
      "acc": 0.4225,
      "cc": 0.307
     },
     "xnli_en": {
      "acc": 0.456667,
      "cc": 0.185
     },
     "prompt_injections": {
      "acc": 0.482759,
      "cc": -0.034483
     },
     "typed_decisions": {
      "acc": 0.7715,
      "cc": 0.665201
     }
    },
    "areas": {
     "language": 0.625623,
     "sentiment": 0.302458,
     "reasoning": 0.075259,
     "decisions": 0.665201
    },
    "index": 0.417135,
    "coverage": {
     "suites": 8,
     "of": 9
    },
    "complete": false,
    "host": "4090-win"
   },
   "paw_local@4090-win": {
    "display": "paw (local)",
    "color_key": "paw",
    "kind": "compiled-program",
    "per_suite": {
     "ag_news": {
      "acc": 0.8,
      "cc": 0.733333
     },
     "banking77": {
      "acc": 0.412,
      "cc": 0.404263
     },
     "sst5": {
      "acc": 0.416667,
      "cc": 0.270833
     },
     "emotion": {
      "acc": 0.4875,
      "cc": 0.385
     }
    },
    "areas": {
     "language": 0.568798,
     "sentiment": 0.327916
    },
    "index": 0.448357,
    "coverage": {
     "suites": 4,
     "of": 9
    },
    "complete": false,
    "host": "4090-win"
   }
  },
  "timing": {
   "modelless": {
    "clock": "in-process",
    "method": "Rust in-process per-decision read (the engine's own decision-time measurement; zero network, zero IPC)",
    "suites": 9,
    "p50_geomean_ms": 0.1626,
    "n_used": 9,
    "n_unquotable": 0,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "hybrid": {
    "clock": "in-process",
    "method": "Rust in-process over the same seat (specialist compose + modelless base, one process)",
    "suites": 9,
    "p50_geomean_ms": 0.0035,
    "n_used": 9,
    "n_unquotable": 0,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "encoder": {
    "clock": "in-process",
    "method": "Rust in-process GPU encode+head forward (Metal on the m3 hosts; the cell's `device` field names it)",
    "suites": 9,
    "p50_geomean_ms": 16.7464,
    "n_used": 2,
    "n_unquotable": 1,
    "n_unjudged": 6,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted); 5 fallback spoke(s) carry the answering tier's timing, unjudged here \u2014 shown in the suite tables, never plotted"
   },
   "laya": {
    "clock": "in-process",
    "method": "Rust in-process forward (device per host row: laya_device \u2014 metal/cuda/cpu)",
    "suites": 9,
    "p50_geomean_ms": 31.3825,
    "n_used": 9,
    "n_unquotable": 0,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "python": {
    "clock": "subprocess-jsonl",
    "method": "JSONL subprocess round-trip to the reference torch runtime \u2014 IPC included, by the lane's own protocol",
    "suites": 9,
    "p50_geomean_ms": 50.2341,
    "n_used": 9,
    "n_unquotable": 0,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "bekko": {
    "clock": "subprocess-jsonl",
    "method": "JSONL subprocess round-trip to their BekkoSentenceTransformer runtime",
    "suites": 9,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 9,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "openthai": {
    "clock": "http",
    "method": "HTTP round-trip to their OpenThai-SystemOne teacher over a loopback FastAPI subprocess",
    "suites": 9,
    "p50_geomean_ms": 274.7061,
    "n_used": 9,
    "n_unquotable": 0,
    "n_unjudged": 0,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "paw": {
    "clock": "http",
    "method": "HTTP round-trip to their hosted REST compile+answer service (accuracy cells only today \u2014 the published posture strips latency)",
    "suites": 9,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 5,
    "n_unjudged": 4,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "clm@4090-win": {
    "clock": "http",
    "method": "HTTP round-trip to their served reference (vLLM pooling on the 4090 window)",
    "suites": 8,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 0,
    "n_unjudged": 8,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "gliner@4090-win": {
    "clock": "subprocess-python",
    "method": "Python subprocess round-trip (their gliner2 package answers per batch; interpreter load excluded)",
    "suites": 8,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 0,
    "n_unjudged": 8,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "agentjev@4090-win": {
    "clock": "http",
    "method": "HTTP round-trip to their jev_service (step-600 tensors; bf16 wobble disclosed in the bench record)",
    "suites": 8,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 0,
    "n_unjudged": 8,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   },
   "paw_local@4090-win": {
    "clock": "local-runtime",
    "method": "Local llama.cpp runtime behind a Python subprocess (their programasweights paw.function; warm cache)",
    "suites": 4,
    "p50_geomean_ms": null,
    "n_used": 0,
    "n_unquotable": 0,
    "n_unjudged": 4,
    "note": "population = the suites behind the lane's index; p50 geometric mean over latency_quotable cells only (unfit timing is shown in the tables, never plotted)"
   }
  },
  "scope": "primary-host rows; a lane the primary host never ran rolls up from its serving host under a host-tagged lane key (clm@4090-win); a product lane's fallback spokes roll up from the answering tier (marked fallback, drawn as triangles \u2014 coverage stays the lane's own measured count)"
 },
 "s1mb": {
  "version": 1,
  "suites": [
   "s1mb_noul",
   "s1mb_score",
   "s1mb_choice"
  ],
  "lanes": [
   {
    "key": "modelless",
    "display": "Reflex",
    "model": "the modelless engine (count tables + sigmoid calibration)",
    "note": "the free floor \u2014 runs in the browser tab",
    "cells": {
     "s1mb_noul": {
      "n": 6173,
      "acc": 0.7055
     },
     "s1mb_score": {
      "n": 2571,
      "acc": 0.4967
     },
     "s1mb_choice": {
      "n": 4329,
      "acc": 0.2296
     }
    },
    "avg": 0.4773,
    "coverage": {
     "suites": 3,
     "of": 3
    }
   },
   {
    "key": "hybrid",
    "display": "Instinct",
    "model": "the served hybrid arm over the same seat",
    "note": "the served arm is the modelless half (A0) on every s1mb suite today \u2014 no specialist arm cleared its certification gate, so the lane's cells are the A0 reads (shown, labeled, per the show-losses law)",
    "cells": {
     "s1mb_noul": {
      "n": 6173,
      "acc": 0.7055,
      "serves": "A0"
     },
     "s1mb_score": {
      "n": 2571,
      "acc": 0.4967,
      "serves": "A0"
     },
     "s1mb_choice": {
      "n": 4329,
      "acc": 0.2296,
      "serves": "A0"
     }
    },
    "avg": 0.4773,
    "coverage": {
     "suites": 3,
     "of": 3
    }
   },
   {
    "key": "laya",
    "display": "laya (rust)",
    "model": "the zero-shot encoder lane (english checkpoint, round-1 breadth)",
    "note": "the encoder family's measured zero-shot reading on this benchmark \u2014 the s1mb suites route the english checkpoint, measured on the GPU serving host; the encoder head budget refuses the pick-an-answer suite's option list (it runs to hundreds of rendered options per question), so that cell is an honest gap, not a zero",
    "cells": {
     "s1mb_noul": {
      "n": 6173,
      "acc": 0.6964
     },
     "s1mb_score": {
      "n": 2571,
      "acc": 0.259
     }
    },
    "avg": 0.4777,
    "coverage": {
     "suites": 2,
     "of": 3
    },
    "gap_suites": [
     "s1mb_choice"
    ]
   },
   {
    "key": "encoder",
    "display": "Rethink",
    "model": "trained per-option encoder head",
    "cells": {
     "s1mb_noul": {
      "n": 6173,
      "acc": 0.7055,
      "serves": "tier-fallback",
      "fallback": true
     },
     "s1mb_score": {
      "n": 2571,
      "acc": 0.4967,
      "serves": "tier-fallback",
      "fallback": true
     },
     "s1mb_choice": {
      "n": 4329,
      "acc": 0.2296,
      "serves": "tier-fallback",
      "fallback": true
     }
    },
    "avg": null,
    "coverage": {
     "suites": 0,
     "of": 3
    },
    "note": "no trained head has passed its earn gate on these suites yet, so the lane serves the answering tier (shown above as the served-by reading) \u2014 the gap is the disclosure, never a padded number"
   }
  ],
  "disclosure": "forced-pick accuracy on our own 50/50 corpus/test split of the benchmark (every lane reads the same halves) \u2014 NOT comparable to the benchmark's own leaderboard, whose skill score is a different scale we do not publish. avg is the unweighted mean over the three suite accuracies. A missing cell is a disclosed gap, never a zero."
 }
}
