{
  "claim": "Every open-model-beats-Jev headline this site has covered was measured on the challenger's own data. Where the same system has since been run on a third-party suite, the ordering does not survive. Where it has not been run, nobody knows.",
  "method": "For each published head-to-head, the left number is the one its authors published on data they chose. The right number is the same system's hard-tier accuracy on JevBench v1.3.0, a third-party suite of 220 items (111 public, 109 held out) frozen 2026-09-19, which none of these systems trained on. The two columns are different tasks, so the levels are not comparable; the ordering between two systems in the same column is.",
  "source": "https://benchmarkheaven.com/api/jevbench/v1.2",
  "captured": "2026-09-22",
  "note": "JevBench is one benchmark, by one author, and half its hard tier is public — at least one entrant records using the public items as a development gate. Read it as a second opinion, not as ground truth. The point of the table is not that JevBench is right; it is that a number measured on the challenger's own generator and a number measured by somebody else are different kinds of number, and only one of them was ever published for most of these.",
  "columns": [
    { "key": "system", "label": "system" },
    { "key": "home", "label": "its own published number", "mono": true },
    { "key": "kind", "label": "what kind of number" },
    { "key": "hard", "label": "JevBench hard, 220 items", "align": "right", "mono": true },
    { "key": "survives", "label": "would the headline survive?" }
  ],
  "rows": [
    {
      "system": "Jev 1.13.0",
      "home": "0.727",
      "kind": "generalist, cold — had never seen the benchmark",
      "hard": "163/220 · 0.741",
      "survives": "control. The only row whose two numbers agree."
    },
    {
      "system": "Laya (ModernBERT-large 421M)",
      "home": "0.766",
      "kind": "specialist — fine-tuned on that benchmark's own 1,200-case train split",
      "hard": "75/220 · 0.341",
      "survives": "No. Ahead of Jev by 3.9 points at home, behind by 40.0 away."
    },
    {
      "system": "Bespoke Nimble 9B",
      "home": "0.9012",
      "kind": "specialist — LoRA on 2,676 rows from the generator that made the eval",
      "hard": "144/220 · 0.655",
      "survives": "Yes, directionally. Nimble already lost its own head-to-head (Jev 0.9321), and it stays second here."
    },
    {
      "system": "open-jev-deberta-v3-large",
      "home": "0.85 in-domain, 0.69 out",
      "kind": "specialist — and the only one that built a held-out-question split itself",
      "hard": "80/220 · 0.364",
      "survives": "Its own OOD number already said so. The card publishes both."
    },
    {
      "system": "cua-s1-forms (706K params)",
      "home": "0.997 vs Jev's 0.836",
      "kind": "specialist — 150,000 training rows from the generator that made the 196-case eval",
      "hard": "not entered",
      "survives": "Unknown, and the source release says no checkpoint claim is established by it."
    },
    {
      "system": "kev family (0.5B–8B)",
      "home": "Jev 0.857 vs kev-8b about 0.78",
      "kind": "already out-of-domain — 764 records from six public sources kev never trained on",
      "hard": "104/220 · 0.473 (kev 8B)",
      "survives": "Yes. It was an OOD comparison when it was published, and Jev won it."
    },
    {
      "system": "AgentJev-0.6B",
      "home": "79.25 vs 77.00, reported",
      "kind": "unsourced — I could not find the primary measurement",
      "hard": "not listed under that name",
      "survives": "Unknown. Flagging it rather than repeating it."
    }
  ]
}
