{
 "benchmark": "JevBench",
 "revision": "v1.4.2.2",
 "revision_log": [
  {
   "revision": "v1.4.2 display amendment",
   "date": "2026-09-26",
   "note": "Display-only amendment at the operator's request: the hosted Gemma system measured through the Autoloops API is now listed as \"Autoloops – Gemma 4 31B IT\" and its author field reads Autoloops. No measurement, score, axis, rank or eligibility changed. The artifact bytes before this amendment had SHA-256 ac14e206dde51ae28e40dc1ea2ff1fecc4a449b941d098e9ecb5618bd533e5be."
  },
  {
   "revision": "v1.3.0",
   "date": "2026-09-22",
   "note": "Scoring-only release; the task set and measurements are unchanged. Intelligence is now accuracy above each item's uniform-guessing baseline, aggregated with the existing tier weights. If chance-corrected Intelligence is below 50, the composite receives a growing (Intelligence / 50)^2 penalty. Calibration, Speed, Cost and ranking eligibility are unchanged."
  },
  {
   "revision": "v1.2.16",
   "date": "2026-09-21",
   "note": "Added the reranker class and five Apache-2.0 open rerankers. A neutral adapter, identical task instructions, public-only temperature/yes-no calibration grids, and a no-instruction public baseline were preregistered before the held-out run. Every row covers all 534 frozen decisions; no earlier row or task changed."
  },
  {
   "revision": "v1.2.15",
   "date": "2026-09-21",
   "note": "Added Open-Jev 2B and Open-Jev 9B by Zefan Cai. Both public checkpoints ran all 534 frozen decisions through the author's pinned LoRA-plus-decision-head server, serially on our RunPod H100 with prefix caching off. An exact normalized-text audit found no JevBench public state or instruction in the 79,116-row public training projection. No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.14",
   "date": "2026-09-21",
   "note": "Added Winnow-12B Q8 after its author requested evaluation. The pinned Q8_0 GGUF ran all 534 unchanged decisions through the author's pinned TypeSafe-compatible server, serially on our lium.io RTX 4090. The row records the applicable Apache-2.0 Gemma 4 terms, non-zero hosted-reference cost and the independently unverifiable private-training overlap claim. No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.13",
   "date": "2026-09-21",
   "note": "Added OpenJev (thinking, BF16) using OpenJev's native typed-API think=512 switch over DiffusionGemma. It ran all 534 unchanged decisions serially on the same H200. No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.12",
   "date": "2026-09-21",
   "note": "Added djev (thinking), an experimental full-generation run over the same open DiffusionGemma checkpoint used by djev-dev. Thinking was enabled with a frozen 8,192-token output cap on all 534 unchanged decisions. Current djev-dev itself hard-codes thinking off, one denoising step and read-only inference, so this row is not presented as a switch in its published typed API. No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.10",
   "date": "2026-09-21",
   "note": "Added smalljev semantic-v9 after its author requested evaluation. The public Apache-2.0 MiniCPM5-2B-Base LoRA and native decision heads ran all 534 frozen decisions serially on our lium.io A6000. Its mapping, endpoint condition, non-zero hosted-reference cost basis and public-benchmark-directed training disclosure were committed before the run (docs/v1.2-additions-smalljev.md). No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.9",
   "date": "2026-09-21",
   "note": "Added Certo v1 (AltSlate Labs) after its author requested evaluation. The public MIT ModernBERT-large checkpoint ran all 534 frozen decisions through the author's DecisionModel, serially on our RunPod RTX 3090. Its mapping, published 64-token state and 48-token option limits, endpoint condition and hosted-price cost basis were pushed before the run (docs/v1.2-additions-certo.md). No earlier measurement or task changed."
  },
  {
   "revision": "v1.2.8",
   "date": "2026-09-21",
   "note": "Added requested systems on the unchanged frozen 534-decision set, each through its author's own server and the existing TypeSafe adapter, one request at a time: decider-35b-a3b and reflex-27b (issues #4, #5), decider-2b (#2), reflex 4B (#3), OpenDecision, jev-local and LitJev on our RunPod GPUs; GLiNER2 large on our CPU; and decision-machine-1 (#8), a closed decision model behind milliseconds.ai's production API, shown in its own class. jqv (#6, #9) was re-run in full on our own GPU from its now-public serving code; that complete run replaces the v1.2.7 partial row. Bespoke Nimble 9B was re-run at Bespoke Labs' request after they raised its serving prompt limit from 2,048 to 8,192 tokens; the complete re-run replaces the v1.1.3 row (its old score is kept under superseded_rows). Mappings, endpoint conditions and cost bases were pushed before the runs (docs/v1.2-additions-run4.md, docs/v1.2-additions-run4b.md). No earlier measurement changed."
  },
  {
   "revision": "v1.2.7",
   "date": "2026-09-20",
   "note": "Added three systems: jqv (a stock Qwen3-32B read as a decision model, submitted with a public endpoint) and the GLiNER2.5 small and multi checkpoints. The GLiNER2.5 rows ran the full frozen 534-decision set on our CPU with the same mapping as the GLiNER2 row. jqv is a partial row: its endpoint is the submitter's own machine, and this revision stopped sending held-out items to an endpoint a submitter operates. The easy and standard/judge tiers had already been sent in full when that was decided; the 109 held-out hard items never were, so the row covers 425 of 534 decisions and carries no rank. Mappings, endpoint conditions and cost bases were committed before any row was aggregated and before the published GLiNER2.5 runs started (docs/v1.2-additions-run3.md); jqv's run had begun about ten minutes earlier, but it needs no mapping and is priced at its base model's public tariff. No earlier row changed."
  },
  {
   "revision": "v1.2.6",
   "date": "2026-09-20",
   "note": "Added openJev Verdict 1.4 and the identified SimpleJev public-demo configurations on the unchanged frozen 534-decision set. No earlier row changed."
  },
  {
   "revision": "v1.2.5",
   "date": "2026-09-20",
   "note": "Added kev 0.5B and the 0.6B, 4B and 8B research previews. Each ran the full frozen v1.2 set (534 decisions including held-out items) through kev's native TypeSafe-compatible endpoint on an RTX 3090. No other row changed."
  },
  {
   "revision": "v1.2.4",
   "date": "2026-09-20",
   "note": "classifier.dev (fast tier) leaves the ranking and becomes an honorable mention. It is not its own model: its own pages say \"The fast tier is Jev, TypeSafe's decision model\" (https://classifier.dev/benchmark), so ranking it against Jev ranks Jev's model against Jev's model at a different price. General rule from this revision on: a service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Its numbers, axes, cost basis, radars and per-task outcomes are unchanged; only its rank is gone. Every other row moves up one place; no score changed."
  },
  {
   "revision": "v1.2.3",
   "date": "2026-09-20",
   "note": "Cost correction. Every row's $ per 1,000 decisions is recomputed with each of the 534 decisions counted exactly once and priced exactly once. Three arithmetic mistakes were fixed: the 242-decision standard+judge run was averaged twice in the v1.1-tier price (556 rows instead of 314); rows priced from the gemini-3.1-flash-lite token counts used that run's standard+judge-only average (452 input tokens per decision) for all 314 v1.1 decisions instead of its average over all 314 (383); and requests whose answer came back unparseable were left unpriced although they were billed (9 DeepSeek V4.1 Flash decisions). The first two made the affected rows look 1.5-11 % more expensive than they are; the third made DeepSeek look 2.6 % cheaper. No tariff, no measurement, no item and no answer changed, and no rank changed. Details: results/v1.2/cost-correction-v1.2.3.json."
  },
  {
   "revision": "v1.2.2",
   "date": "2026-09-19",
   "note": "Added five systems requested by readers: Laya, jeff, GLiNER2, openJev Verdict and classifier.dev (fast tier). Full v1.2 set each (534 decisions incl. held-out), scored with the unchanged v1.2 rules. Local systems ran on our CPU (4 threads) with the usual self-hosted latency adjustment; classifier.dev is a production API. Mappings were fixed before the runs (docs/v1.2-additions.md). No other row changed."
  },
  {
   "revision": "v1.2.1",
   "date": "2026-09-19",
   "note": "Added djev (Maisa, diffusion-gemma): full v1.2 set (534 decisions incl. held-out) through its production API, scored with the unchanged v1.2 rules. Cost at djev's announced price ($0.035/M input tokens, output free), which is not yet charged (free preview). No other row changed."
  },
  {
   "revision": "v1.2",
   "date": "2026-09-19",
   "note": "Final JevBench Score: 4 axes, geometric mean."
  },
  {
   "revision": "v1.4.0",
   "date": "2026-09-23",
   "note": "Added 308 sealed decisions, the selected 20% scoring blend, a generalization penalty, harmonic mean and separate Speed/Cost Jev-class gates. Sealed items and answers are not published."
  },
  {
   "revision": "v1.4.1",
   "date": "2026-09-23",
   "note": "Added 6 systems measured on the frozen 534-decision public set but omitted from v1.4.0, with completed sealed measurements. The v1.4 scoring formulas and all v1.4.0 measurements are unchanged. Sealed items and answers are not published."
  },
  {
   "revision": "v1.4.2",
   "date": "2026-09-25",
   "note": "Added 11 newly measured systems and swanOne's completed sealed run, all on the full 842-decision protocol. The v1.4 scoring formulas and all v1.4.1 measurements are unchanged. Systems without a public bookable price carry a labelled estimate from their base model's public price. Sealed items and answers are not published."
  },
  {
   "revision": "v1.4.2.1",
   "date": "2026-09-27",
   "note": "Added Plumb-4B, measured on the full v1.4 protocol. The v1.4.2 scorer is unchanged. Its estimated Cost uses the bookable EmpirioLabs exact-base Qwen3.5-4B input price of $0.04/M; prior v1.4.2 rows are unchanged."
  },
  {
   "revision": "v1.4.2.2",
   "date": "2026-09-27",
   "note": "Added Imajev-4B, measured on the full v1.4 protocol with one rotation and calibration.json. The v1.4.2 scorer is unchanged; earlier rows are unchanged."
  },
  {
   "revision": "v1.4.2.2 cost-basis wording correction",
   "date": "2026-09-28",
   "note": "Text-only correction: the Imajev-4B cost basis now states that full-forward input tokens are counted once for the single pinned server pass, matching the one-rotation run receipt. Previous v1.4.2.2 result SHA-256: f0dfdd8f1601cadb16864061413e6e43c8b2dfa07b10ffd0716c67fc3c4b9952. Previous candidate-row SHA-256: c26650b9bedc52445d693d2d8d67f047c5a21e40c3a2abd294c5c13b6424bb61. No measurement, score, axis, rank, eligibility, or numeric value changed."
  }
 ],
 "protocol": "jevbench::v1.4",
 "status": "final",
 "generated_utc": "2026-09-27T18:49:51+00:00",
 "measured_in": "Frozen v1.2 public/held-out items plus 308 sealed v1.4 items; all item text and answers remain private",
 "revision_note": "v1.4.2.2 adds the verified Imajev-4B row to v1.4.2.1 using the exact live v1.4.2 scoring code. Earlier measurements and score fields are unchanged; only ranks and preset ranks move where the new row changes the ordering. A 28 Sep text-only amendment clarifies that full-forward input tokens are counted once for the single pinned server pass; no measurement, score, axis, rank, eligibility, or numeric value changed.",
 "score_name": "JevBench Score",
 "score_one_liner": "Intelligence, Calibration, Speed and Cost — equal-weight harmonic mean, with generalization and Jev-class gates.",
 "tiers": {
  "easy": 72,
  "judge": 146,
  "standard": 96,
  "hard": 220,
  "sealed": 308
 },
 "tier_weights": {
  "easy": 0.14,
  "standard": 0.28,
  "judge": 0.28,
  "hard": 0.3
 },
 "axis_weights": {
  "intelligence": 0.25,
  "calibration": 0.25,
  "speed": 0.25,
  "cost": 0.25
 },
 "presets": {
  "JevBench Score (25:25:25:25)": {
   "intelligence": 0.25,
   "calibration": 0.25,
   "speed": 0.25,
   "cost": 0.25
  },
  "Balanced 33:33:33 (no calibration)": {
   "intelligence": 0.3333333333333333,
   "calibration": 0.0,
   "speed": 0.3333333333333333,
   "cost": 0.3333333333333333
  },
  "Emphasis on Accuracy 60:20:20": {
   "intelligence": 0.6,
   "calibration": 0.0,
   "speed": 0.2,
   "cost": 0.2
  },
  "Emphasis on Speed 20:60:20": {
   "intelligence": 0.2,
   "calibration": 0.0,
   "speed": 0.6,
   "cost": 0.2
  },
  "Emphasis on Cost 20:20:60": {
   "intelligence": 0.2,
   "calibration": 0.0,
   "speed": 0.2,
   "cost": 0.6
  },
  "Intelligence only": {
   "intelligence": 1.0,
   "calibration": 0.0,
   "speed": 0.0,
   "cost": 0.0
  }
 },
 "main": "JevBench Score (25:25:25:25)",
 "speed_note": "Latency of self-hosted and demo endpoints is adjusted ×2 (+0.15 s on our own servers) to approximate production load — an assumption, not a measurement; raw measurements are in the table and the repo.",
 "cost_unit": {
  "unit": "$ per 1,000 decisions",
  "not_unit": "$ per 1,000 tokens",
  "one_liner": "Dollars per 1,000 decisions, not per 1,000 tokens: one decision is a whole question — state, rubric and options.",
  "worked_example": "One decision is a whole question, not a token. Jev 1.13.0 reads 950 input tokens per decision on average over the 534 v1.2 decisions. At its public tariff of $0.042 per MILLION input tokens (output tokens are free, https://docs.typesafe.ai/models), 1,000 decisions therefore cost 950 x 1,000 x $0.042 / 1,000,000 = $0.0399. That is what the Cost column shows: $0.0399 per 1,000 decisions, not per 1,000 tokens.",
  "short_note": "One decision ≈ 950 input tokens on average; at Jev's $0.042 per million input tokens that is $0.0399 per 1,000 decisions.",
  "mean_input_tokens_per_decision_jev": 950.3389513108614
 },
 "cost_correction": {
  "revision": "v1.2.3",
  "file": "results/v1.2/cost-correction-v1.2.3.json",
  "what_was_wrong": [
   "The v1.1 and v1.1.3 aggregations built their cost average from a row list that contained the 242-decision standard+judge run twice (once as the standard tier, once as the judge tier) and the 72 easy decisions once: 556 rows instead of 314. The standard and judge tiers were therefore over-weighted in the price, which made the affected rows look 1.5-3.3 % more expensive than they are.",
   "Rows without their own token counts were priced at the input tokens of the gemini-3.1-flash-lite run measured on the 242 standard+judge decisions only (452 per decision) and that figure was applied to all 314 v1.1 decisions, which excludes the shorter easy tier. Over all 314 decisions the same run averages 383.41 input tokens, which is the figure used from v1.2.3 on. This made the affected rows look 4-11 % more expensive.",
   "A metered row's price left out the requests whose answer came back unparseable. Those requests returned HTTP 200 with generated tokens and were billed, and JevBench already counts them as wrong answers, so from v1.2.3 they are priced too. Only DeepSeek V4.1 Flash had any (9 of its 314 v1.1 decisions); its price rises by 2.6 %.",
   "No tariff was wrong. The hard-tier costs, and classifier.dev's flat plan price, were already correct."
  ],
  "rule": "usd_per_1000_v11_tiers = 1000 x (mean input tokens per decision x $/M in + output tokens charged x $/M out) / 1e6, over all 314 v1.1 decisions (72 easy + 242 standard+judge), each decision counted exactly once and priced exactly once. A metered row uses the provider's own tariff and its own measured token counts, including the requests whose answer could not be parsed; an estimated row uses the reference tariff for its weights or size class and, when the run reports no usage, the input tokens of the gemini-3.1-flash-lite run on the same prompts over the same 314 decisions. usd_per_1000 = (v11 x 314 + hard x 220) / 534."
 },
 "cost_correction_table": {
  "classifier-dev-fast": {
   "old": 0.003333333333333333,
   "new": 0.003333333333333333,
   "pct": 0.0,
   "unchanged": true
  },
  "jev-1.13.0": {
   "old": 0.04061412193840432,
   "new": 0.03991423595505617,
   "pct": -1.7232576994021993,
   "unchanged": false
  },
  "semif-qwen3.5-4b": {
   "old": 0.022975040619190038,
   "new": 0.02244460674157303,
   "pct": -2.3087396727993554,
   "unchanged": false
  },
  "djev": {
   "old": 0.025951254681647943,
   "new": 0.025951254681647943,
   "pct": 0.0,
   "unchanged": true
  },
  "laya": {
   "old": 0.0028831273408239703,
   "new": 0.0028831273408239703,
   "pct": 0.0,
   "unchanged": true
  },
  "open-alternative-jev": {
   "old": 0.022171404494382024,
   "new": 0.022171404494382024,
   "pct": 0.0,
   "unchanged": true
  },
  "system-one-open": {
   "old": 0.015685659145076917,
   "new": 0.014880936329588014,
   "pct": -5.130309208213748,
   "unchanged": false
  },
  "openjev-razorback16": {
   "old": 0.0671907566081966,
   "new": 0.06560455056179774,
   "pct": -2.3607503866169424,
   "unchanged": false
  },
  "jeff": {
   "old": 0.006036722846441948,
   "new": 0.006036722846441948,
   "pct": 0.0,
   "unchanged": true
  },
  "openjev-sglang": {
   "old": 0.134615554522674,
   "new": 0.13127546816479402,
   "pct": -2.481203877013638,
   "unchanged": false
  },
  "openjev-verdict": {
   "old": 0.003872921348314607,
   "new": 0.00367125468164794,
   "pct": -5.20709429729129,
   "unchanged": false
  },
  "gpt-5.6-luna": {
   "old": 0.24728613154420276,
   "new": 0.24191123595505612,
   "pct": -2.173553185365702,
   "unchanged": false
  },
  "open-jev-deberta-v3-large": {
   "old": 0.007742829572538459,
   "new": 0.007340468164794008,
   "pct": -5.196568050154547,
   "unchanged": false
  },
  "nimble-9b": {
   "old": 0.10850016283992836,
   "new": 0.10494152501680593,
   "pct": -3.2798456057365897,
   "unchanged": false
  },
  "gemini-3.1-flash-lite": {
   "old": 0.26820170492819223,
   "new": 0.26378698501872655,
   "pct": -1.6460446851550288,
   "unchanged": false
  },
  "deepseek-flash": {
   "old": 0.5788175142273461,
   "new": 0.593682584269663,
   "pct": 2.5681790334489274,
   "unchanged": false
  },
  "system-one-sg": {
   "old": 0.09153429437798077,
   "new": 0.08944116853932584,
   "pct": -2.286712158408745,
   "unchanged": false
  },
  "gliner2": {
   "old": 0.003872921348314607,
   "new": 0.00367125468164794,
   "pct": -5.20709429729129,
   "unchanged": false
  },
  "qwen3.8-27b": {
   "old": 2.7110372815546095,
   "new": 2.669088141927965,
   "pct": -1.5473464681603164,
   "unchanged": false
  },
  "needle-3-tools": {
   "old": 0.016219537190082647,
   "new": 0.014372006369426754,
   "pct": -11.3907739721794,
   "unchanged": false
  },
  "needle-3": {
   "old": 0.02492563984585384,
   "new": 0.023839264044943822,
   "pct": -4.358467054921875,
   "unchanged": false
  }
 },
 "scoring": {
  "jevbench_score": "Equal-weight harmonic mean of Intelligence, Calibration, Speed and Cost (power mean p=-1). If Intelligence <50 multiply by (I/50)^2. For Speed and Cost separately, if below 50 multiply by (axis/50)^2.",
  "intelligence": "0.8 × v1.3 chance-corrected Intelligence on the frozen v1.2 items + 0.2 × 100 × max(0, (sealed accuracy − 0.293)/(1 − 0.293)); then multiply by 1 − max(0, public-minus-sealed accuracy gap in percentage points − 25)/100. Original tier weights: easy .14, standard .28, judge .28, hard .30.",
  "hard_tier": "220 new decisions (111 public, 109 held out): long multi-condition policy documents (2-6k tokens), priority trade-offs, deliberately ambiguous cases with a 'no clear answer' label, traps, multi-hop lookups, date/number reasoning, adversarial distractors, subtle answer-judging, overlapping routing, and probability items with an exact gold distribution. Half written by Claude Opus 5, half by GPT-5.6 Sol; each item reviewed blind and then against its gold by the other model; one discussion round; frozen and hashed before any benchmarked system saw an item. No item was selected on any system's answers.",
  "calibration": "v1.3 Calibration + (v1.4 candidate Calibration − v1.3 Calibration) × min(1, 0.2/0.35). Label-only systems contribute zero.",
  "speed": "Mean of score(p50) and score(p95) of the serial 242-decision standard+judge run; score(s) = 100 - 20 log10(s / 0.1 s), clipped to 0..100 (0.1 s = 100, 1 s = 80, 10 s = 60). Latency of self-hosted and demo endpoints is adjusted ×2 (+0.15 s on our own servers) to approximate production load — an assumption, not a measurement; raw measurements are in the table and the repo. Production APIs (Jev, djev, classifier.dev, OpenAI, Google, DeepSeek, Chutes) are not adjusted.",
  "cost": "US dollars per 1,000 DECISIONS — not per 1,000 tokens. One decision is one whole question: its state, its rubric and its options, which is hundreds to thousands of input tokens. Pooled over all 534 v1.2 decisions; score = 100 - 30 log10(usd / 0.001), clipped to 0..100 ($0.001 = 100, $0.01 = 70, $0.10 = 40, $1 = 10). Measured = public tariff x measured tokens. est. = hosted-provider list price of the same weights or size class x tokens (for a flat-rate service, its published plan price at full use). announced = the provider's published price, not yet charged (free preview), x measured tokens.",
  "ranked": "Ranked: a system's own model, with every tier attempted for >= 95 % of its decisions. Partial runs are shown below the ranking, marked, without a rank. A service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Ranking it would rank the same model twice, once at the model's own price and once at the service's. The row keeps every number, axis, cost basis and per-task outcome; it carries no rank number.",
  "presets": "Other views reweight the same four axes with a weighted harmonic mean and the same gates. They are not the official JevBench Score.",
  "sealed": "308 current sealed decisions; only aggregate results are published. API flags identify operator endpoints that received item text without answers."
 },
 "hard_dataset": {
  "frozen_utc": "2026-09-19T11:43:54+00:00",
  "n_items": 220,
  "n_public": 111,
  "n_heldout": 109,
  "families": {
   "adversarial": 12,
   "ambiguous": 14,
   "judge_hard": 33,
   "long_policy": 38,
   "multi_hop": 35,
   "probability": 20,
   "routing_hard": 10,
   "temporal_numeric": 30,
   "tradeoff": 12,
   "trap": 16
  },
  "types": {
   "choice": 129,
   "score": 14,
   "noul": 77
  },
  "authors": {
   "gpt-5.6-sol": 110,
   "claude-opus-5": 110
  },
  "sha256_public_file": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb",
  "sha256_heldout_file": "4e8adf72988766534c87c2b83808bde7f0f934f515f8eeb01cc74eb90b8778b1",
  "dataset_hash_all": "ec200ccd3db28153c93bfeaed55acb18b4909610ba403cf73482abbc4074ef6b",
  "review": {
   "opus-a": {
    "authored": 40,
    "accepted": 40,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/opus-a.jsonl"
   },
   "opus-b": {
    "authored": 40,
    "accepted": 40,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/opus-b.jsonl"
   },
   "sol-a": {
    "authored": 40,
    "accepted": 40,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/sol-a.v2.jsonl"
   },
   "sol-b": {
    "authored": 40,
    "accepted": 40,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/sol-b.v2.jsonl"
   },
   "opus-c": {
    "authored": 30,
    "accepted": 30,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/opus-c.jsonl"
   },
   "sol-c": {
    "authored": 30,
    "accepted": 30,
    "rejected": [],
    "no_verdict": [],
    "source_file": "authoring/sol-c.v3.jsonl"
   }
  },
  "rule": "Items authored by Claude Opus 5 were reviewed by GPT-5.6 Sol and vice versa (blind answer, then gold verdict); one discussion round; anything not accepted afterwards was dropped. No item was selected or dropped on the basis of any benchmarked system's answers. Frozen before any benchmarked system saw an item."
 },
 "footnotes": {
  "open-alternative-jev": "With the options in reverse order (A. no, B. yes) the same model scored 21 % instead of 72 % on yes/no answer-judging items — small models are very sensitive to option order. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "bge-reranker-v2-m3": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "certo": "The public Certo v1 checkpoint through the author's DecisionModel, serially on our rented GPU. The question instruction is prepended to the state because Certo exposes state + runtime options but no separate question field; the published 64-token state and 48-token option limits are unchanged. The model card says v1 does not yet transfer to arbitrary natural-language prose. Cost is an estimate from same-size hosted encoders times the checkpoint's retained input tokens, not free/100. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "classifier-dev-fast": "Its own benchmark page says the fast tier is Jev. Free for us; the price is its published Pro plan ($20/month for 200,000 fast classifications a day) at full use, $0.0033 per 1,000 decisions. | unranked / honorable mention | sealed item text (no golds) was sent to the operator endpoint (classifier.dev), as for every API measurement",
  "decider-2b": "The author's TypeSafe-compatible server and published weights (Qwen3.5-2B-Base with a trained one-pass decision readout), run serially on our GPU. Self-host latency gets the standard ×2 + 0.15 s adjustment. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "decider-35b-a3b": "The author's TypeSafe-compatible server and published FP8 weights, run serially on our H100 NVL. The exhaustive startup batch warmup was skipped; each required serial shape captured lazily before its measured request. Self-host latency receives the standard ×2 + 0.15 s adjustment. Cost uses the closest hosted 35B-A3B input tariff and is not the temporary rental charge. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "decision-machine-1": "A closed-weights decision model behind a production API that serves TypeSafe's wire format, so the unchanged typesafe adapter ran it. Run on a free test key (30 requests a minute, 2.2 s between requests); the provider states the inference infrastructure is the same as for paid keys. Cost is the public paid tariff, $0.04 per million input tokens (output free), times the input tokens the API reported. | already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (Milliseconds API), as for every API measurement",
  "djev-thinking": "Experimental full-generation path over the same DiffusionGemma checkpoint as djev-dev: thinking was enabled and the model could generate up to 8,192 tokens before returning its distribution. Current djev-dev itself hard-codes enable_thinking=false, diffusion_max_steps=1 and read_only=true, so this is not a switch in its published typed API. It is substantially slower/costlier, and 72/534 requests exhausted the output budget without a parseable distribution; those are failures. Cost uses measured tokens and a same-size hosted reference, not the H200 rental bill. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 220/308 sealed items answered validly (failures count as wrong)",
  "djev": "The measured endpoint was Maisa's hosted API in free preview; the cost uses its announced price ($0.035 per million input tokens, output free), and nothing was charged. The self-hostable djev-dev runtime is Apache-2.0 and applies a structured one-step inference method to Google's Apache-2.0 diffusiongemma-26B-A4B-it checkpoint; it adds no separately trained djev weights. Probabilities are djev's own (its docs call them experimental and uncalibrated). | already on the live v1.3.0 board | v1.4: hosted api.djev.dev was paused by its operator (\"Serving is paused by the administrator\"); sealed tier measured on the public djev runtime (Davipar/djev-dev 3ce907e, same weights, default mode) self-hosted on an H100; Speed/Cost kept from v1.3. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gliner2-large": "The large checkpoint of Fastino's earlier GLiNER2 family, same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gliner2.5-multi": "The multilingual GLiNER2.5 checkpoint (287M), same family and same documented mapping as the GLiNER2 row. JevBench items are English only, so its multilingual training is not exercised here. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gliner2.5-small": "The small GLiNER2.5 checkpoint (74M), same family and same documented mapping as the GLiNER2 row: the question goes in front of the text and the probabilities are the model's own single-label softmax over the labels, read out in full. A general schema classifier, not a Jev rebuild. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gliner2": "A general schema classifier, not a Jev rebuild. The question goes in front of the text; the probabilities are GLiNER2's own single-label softmax over the labels, read out in full (mapping fixed before the run). | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gte-reranker-modernbert-base": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jeff": "Self-hosted from its GitHub repo with server defaults, on our CPU (the author recommends a GPU, e.g. an L4), through the same TypeSafe-compatible API as Jev. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jev-local": "The author's local Jev-compatible server in its default full configuration: a frozen Qwen3.5-9B scores each option by its mean log-probability (one forward pass per option, no generation, no decision training). Run serially on our GPU. It re-reads the state once per option; if its reported token count covers one pass only, a per-token hosted price would be higher than this estimate. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jqv": "A stock Qwen3-32B with no decision training: the state is prefilled once, each question is an isolated branch and the answer is read from the option-letter logits, with one fitted temperature (3.02, 400 MMLU validation items). Re-run in v1.2.8 on our own GPU from the now-public serving code (Octalab-Inc/jqv 0189b67), so all 534 decisions including the held-out hard items were asked; this full run replaces the v1.2.7 partial row, which had been measured on the submitter's machine. Cost is the base model's public per-token tariff, not free. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "kev-0.5b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. This is the v0.1 release. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 307/308 sealed items answered validly (failures count as wrong)",
  "kev-0.6b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 307/308 sealed items answered validly (failures count as wrong)",
  "kev-4b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 306/308 sealed items answered validly (failures count as wrong)",
  "kev-8b": "Self-hosted from the author's repository at commit 20fa626 through its native TypeSafe-compatible `/v1/systemone` server, BF16 on an RTX 3090; measured serially from Sandy over the internet. The author labels this checkpoint a research preview. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "laya": "The English checkpoint (repo root), run on our CPU through its own `laya` package. Its budget is 512 tokens per question, so long hard-tier states are cut by the package itself. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "litjev": "The author's reproduction of Jev's decision layer on an off-the-shelf model, in its default configuration: Qwen3.8-27B, scores read from the output head, no training and no calibration file (its README says probabilities are not calibrated by default). Run serially on our GPU through an SSH tunnel, because its server binds to localhost; the request still crosses the internet and gets the ×2 + 0.15 s adjustment. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "mxbai-rerank-base-v2": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "nimble-9b": "Re-run in v1.2.8 at Bespoke Labs' request after they raised the serving prompt limit from 2,048 to 8,192 tokens (bespokelabsai/nimble PR #4). Same recipe as the v1.1.3 run — the published LoRA merged into Qwen3.5-9B with the author's PEFT safe-merge, served with SGLang and the author's Jev-compatible API — now from current nimble main; the adapter weights are unchanged. Hard-tier accuracy rose from 43.6 % to 65.5 %, yet the score fell: the long hard items that used to fail at once are now answered and priced (so Cost fell), and this pod was in Canada while the v1.1.3 run's was in Sweden, so part of the lower Speed is network distance from our server in Germany. This complete run replaces the earlier row; its old score is kept in the artifact under superseded_rows. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "open-jev-zefan-2b": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "open-jev-zefan-9b": "The author's pinned LoRA adapter, trained scalar decision head and calibration temperature, served by the author's Open-Jev server with prefix caching off, batch size 1 and 4,096-token limit. Serial requests were measured from Sandy over an SSH tunnel to the H100. Self-host latency receives the standing x2 + 0.15 s adjustment. Cost uses the exact Qwen3.5-9B hosted input tariff for 9B and the same conservative same-family proxy for the unlisted 2B; neither receives an automatic 100. Exact normalized comparison found no JevBench public task state or instruction in the 79,116-row public training projection. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "opendecision": "A zero-shot NLI classifier behind a TypeSafe-compatible server, not a trained decision model: it scores each option as an entailment hypothesis with ModernBERT-large-zeroshot-v2.0. Its choice path runs several NLI passes over the same state, which the reported token count does not include, so a per-token hosted price would be higher than the estimate here. Pre-registered for our CPU in v1.2.7, run on our GPU because the CPU was far too slow. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "openjev-thinking": "OpenJev's real typed-API thinking switch at think=512, using its own /v1/systemone server over BF16 DiffusionGemma. The thought is generated first, then native probability reads are taken after it. All 534 requests returned valid distributions. Cost counts the server's billed input and thought output tokens. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "openjev-verdict-1.4": "Same public weights as the earlier Verdict row, run through the author's fixed v1.4 engine. That engine auto-loads the calibrator for every option count, frames candidate labels as NLI sentences and uses a 512-token context budget. Run locally on our CPU, serially. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "openjev-verdict": "The openJev-verdict-2.0 Hugging Face repo ships no weights; its config is byte-identical to heman10x/rlcd-modernbert-151m, whose published weights we ran with the author's engine. The 'verdict2-base' checkpoint behind the README's numbers is not downloadable yet (Git LFS 404); we will run it once it is. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "qwen3-reranker-4b": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "reflex-27b": "The frozen public Qwen3.8-27B checkpoint through reflex at the requested pinned commit, with two option orders averaged and temperature 1. No adapter or fitted calibration file. Run serially on our H100 NVL. Self-host latency receives the standard ×2 + 0.15 s adjustment; cost uses the exact base model's public hosted input tariff. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "reflex-4b": "The author's reflex-serve: Qwen3.5-4B with the published LoRA and its per-primitive calibration file; the state is encoded once and each question read from the label logits. Run serially on our GPU; the author discloses that the 231 public items were used four times as a development gate. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "simplejev-qwen3.6-35b-a3b": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100. | already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (Featherless demo), as for every API measurement",
  "simplejev-qwen3.8-27b": "Author's no-login shared demo, model id recorded verbatim, one request at a time at or below its 2 RPS limit. SimpleJev reads answer-token logits and returns the complete distribution; it does not generate an answer. Speed uses the public-demo x2 load adjustment; cost uses a hosted size-class input price and is not free/100. | already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (Featherless demo), as for every API measurement",
  "smalljev": "The public semantic-v9 LoRA and native heads over MiniCPM5-2B-Base, through the mapping frozen before the run. It has a typed Python contract but no TypeSafe-compatible HTTP route. The released training recipe explicitly hill-climbed against JevBench's public shape and source families; this allowed public benchmark-directed development is disclosed. Cost is $0.04/M measured input tokens, not free/100. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "winnow-12b": "The submitted Q8_0 GGUF ran through the pinned author's TypeSafe-compatible /v1/systemone server with 8,192 context, four resident decision branches, Q8 KV, and full GPU offload. The private training corpus was not released. The author's checksum-based audit reports zero exact public-item overlap, but that claim cannot be independently reproduced; our scan found no exact public state or instruction text in the released artifacts. Cost uses the $0.05/M-input hosted Gemma 3 12B reference, not free/100. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "zerank-2": "Neutral documented reranker adapter; instruction and no-instruction public calibration were run, then frozen before one held-out pass. | already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gpt-6-luna": "OpenAI direct API baseline; reasoning effort default medium; strict JSON-schema probability response; temperature unset; max_completion_tokens=4096; price cost from returned usage at official standard list rates. | API measurement: sealed item text (no golds) was sent to OpenAI",
  "gpt-6-luna-low": "OpenAI direct API baseline; reasoning effort low; strict JSON-schema probability response; temperature unset; max_completion_tokens=4096; price cost from returned usage at official standard list rates. | API measurement: sealed item text (no golds) was sent to OpenAI",
  "jevk5-v02": "Author says no JevBench items or outputs were used for training, tuning, or selection; public results are reported. Scan found only one generic instruction shared by 8 public hard items; unreleased teacher/replay corpora were unavailable. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jev-1.13.0": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (TypeSafe API), as for every API measurement",
  "hopper": "Round 4 unpublished. Author discloses heavy public-benchmark-directed development (26 model/prompt configs, 20+ calibration-map variants observed against the public half); scan of 17 released files vs 231 public tasks found 0 state/instruction matches; training corpus not released so overlap not independently verifiable. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gpt-5.6-luna": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (OpenAI API), as for every API measurement",
  "ninfer-qwen3.8-flash-next": "Round 4 unpublished. Engine scan (1,758 files) vs 231 public tasks: 0 exact matches. Submitter discloses no training for Flash-Next but repeated consultation of public items and a public-hard temperature sweep. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jevone": "Round 4 unpublished. Scan vs 231 public tasks: 0 matches. Training corpus/provenance not disclosed — overlap unknown. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "ninfer-qwen3.8-27b-t1.5": "Round 4 unpublished. T=1.5 is an offline recomputation from raw logits of the same run, not a second execution. T=1.5 was chosen by sweeping public hard items (development-set calibrated, disclosed). | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "deepseek-flash": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (DeepSeek API), as for every API measurement; 298/308 sealed items answered validly (failures count as wrong)",
  "semif-qwen3.5-4b": "already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jobe-qwen3.5-4b": "Round 4 unpublished. No trained weights/LoRA/calibration fit; release explicitly rejects fitted temperature/order averaging. Scan of 37 files vs 231 public tasks: 0 matches. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "localjev-qwen3.5-4b": "Round 4 unpublished. Scan of 72 files: 0 matches. Author discloses choosing JSON layout after observing results on the 231 public items (benchmark-directed choice, disclosed). | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "ninfer-qwen3.8-27b": "Round 4 unpublished. Raw T=1.0 row. Author discloses repeated public-item consultation and public-hard tuning (applies to both NInfer 27B rows). | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "openjev-sglang": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (author's Modal demo), as for every API measurement",
  "metask-jev-4b": "Round 4 unpublished. Model card discloses 44.8k+16.1k+390 training rows incl. synthetic families intentionally mirroring JevBench hard families, plus repeated evaluation on all 231 public items (benchmark-directed development disclosed). Scan of 61 files: 0 exact matches. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "openjev-razorback16": "already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "gemini-3.1-flash-lite": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (Google API), as for every API measurement; 307/308 sealed items answered validly (failures count as wrong)",
  "qwen35-9b-jev-data-mix-v2": "The author disclosed development on the public JevBench set and public-result comparisons; released-data overlap scan found no matches, but 764 gap and 382 replay training rows are unreleased. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "jev-qwen3.5-9b-base-nvfp4": "Round 4 unpublished. Byte-identical to upstream March-2026 NVFP4 checkpoint, predates JevBench v1.2, no task-specific training added. Had 3 preliminary scope/policy scoring errors corrected during review (pooled-ECE, full-run latency, cost estimate); score above is final corrected value. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "opensourcejev-qwen35-4b-q4km": "DM submission, measure-only (JevBench publishing HOLD in force). Same author as the existing simplejev-qwen3.5-0.8b row (sabeel111/Featherless AI). Round-4 audit of an earlier commit could not be measured (no public GGUF, Windows-only DLL loader, unpinned llama.cpp build); this round the author published the exact unsloth Q4_K_M GGUF (hash/size independently verified) and we built llama.cpp CUDA from current upstream master on Linux ourselves -- its ABI matched the ctypes bindings exactly, so only a loader file-naming fix was needed (documented diff), no code/scoring/calibration change. Our public-231 subset exactly reproduced the author-reported table: easy 48/48, standard 67/72, hard 66/111, schema 231/231. Calibration (noul temperature) fit only on Google BoolQ, not JevBench. Repo docs name 3 public task IDs while describing benchmark-directed algorithm fixes on the public half (disclosed); 0 exact state/instruction text matches in a released-file scan. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "system-one-open": "already on the live v1.3.0 board | sealed item text (no golds) was sent to the operator endpoint (author's Modal demo), as for every API measurement",
  "raw-qwen3-4b-instruct-2507": "Round 4 unpublished. Neutral raw-logit control. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "system-one-sg": "already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "raw-qwen3-8b": "Round 4 unpublished. Neutral raw-logit control. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "raw-phi-4-mini": "Round 4 unpublished. Neutral raw-logit control, not JevBench-directed. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "open-jev-json-canvas-joshuasp": "Round 4 unpublished. Returns only a final label, not a probability distribution — calibration counts as 0 in the composite. Scan of 39 files: 0 matches. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "von-395m": "The author disclosed that its temperature calibration map used the 231 public JevBench items; monotonic scaling does not change accuracy. Estimated cost is USD 0.00551 per 1,000 decisions. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "raw-qwen3-1.7b": "Round 4 unpublished. Neutral raw-logit control. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "open-jev-deberta-v3-large": "already on the live v1.3.0 board | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 297/308 sealed items answered validly (failures count as wrong)",
  "raw-qwen3-0.6b": "Round 4 unpublished. Neutral raw-logit control. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "simplejev-qwen3.5-0.8b": "Round 4 unpublished. 143 pinned files scanned: 0 exact matches. No JevBench-specific fine-tuning. | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
  "mirror": "Round 4 unpublished. 171 intended HTTP 422 context rejections (over 512-token limit) counted once each as misses; only 363/534 valid distributions returned. Found via Gmail submission (Lewis). | re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 102/308 sealed items answered validly (failures count as wrong)",
  "qwen3.8-27b": "unranked / honorable mention | sealed item text (no golds) was sent to the operator endpoint (Chutes), as for every API measurement; 69/308 sealed items answered validly (failures count as wrong); partial: Chutes rate limit stopped the run after 81/308 items; unranked as in v1.3",
  "needle-3-tools": "unranked / honorable mention | v1.4: Not re-measured: same as needle-3.",
  "needle-3": "unranked / honorable mention | v1.4: Not re-measured: ~100-250 s per item on a rented CPU pod (19 s on Sandy); needs a dedicated CPU host.",
  "decision-2b": "flymy-ai/decision-2b-preview revision df57b75db927acc9ad91ec8115508c1e487086eb (checkpoint minicpm5_reduced_v16_4k_v59), base openbmb/MiniCPM5-2B revision 12a3808a956f869c767195e9266b59c4d21d92e2, the submitter's own FlyMyJevPackageAdapter and frozen calibrator, bf16, unmerged adapter, 4096-token packing, transformers 4.57.6 / peft 0.15.2 as pinned, torch 2.8.0 from the pod image, on our RunPod L40 in Czechia | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
  "decision-fast": "flymy-ai/decision-fast-preview revision 4225d41c66119fe28e95a2631bb0103decae6d56 (checkpoint qwen3_06b_headfirst_ep2a_v53), base Qwen/Qwen3-0.6B-Base revision da87bfb608c14b7cf20ba1ce41287e8de496c0cd, the submitter's own FlyMyJevPackageAdapter and frozen calibrator, bf16, unmerged adapter, 4096-token packing, transformers 4.57.6 / peft 0.15.2 as pinned, torch 2.8.0 from the pod image, on our RunPod L40 in Czechia | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
  "spark-s1-4b-v6": "abhishek085/spark-s1-4b-v6 revision 93d49ddbfb29212e3296635a75a3e80cf69da027, code github.com/abhishek085/open-spark-jev 30ac6d89b7fa36c644cf86aac68f35c1d276a919, the author's own MenuScorer.decide with his fitted calibration.json temperature, bf16, base Qwen/Qwen3.5-4B, transformers 5.17.0 / torch 2.8.0 from the pod image, flash-linear-attention 0.5.2 installed, causal_conv1d not installable here (no wheel builds against this toolchain), on our RunPod L40 in Czechia | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
  "jev-omni": "akhilaaa3/Jev-Omni revision c050d51354147985d13286cf4acf90f562f2c631, the author's own load_model.py (merged text decision model + 256-way head) and his own predict(), transformers 5.17.0 / torch 2.8.0 from the pod image; built on the CPU and moved to CUDA with every nn.Linear weight cast to bfloat16 first - the same cast his reference loader jev_omni.py applies - because our 46 GB GPU cannot hold his fp32 copy; on our RunPod L40 in Czechia | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
  "lev-350m": "weights franckverrot/lev-350m revision ab08ad8b8f346994d983152917e114224f6adac7, code github.com/franckverrot/lev c48a945dbf629998d7458dcc5c16f58df964db94, the author's own lev.serve /v1/systemone endpoint with its shipped calibration temperature, base LiquidAI/LFM2.5-350M, on our RunPod L40 in Czechia | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
  "mghafiri-qwen3.5-0.8b-decision-model": "JevLite SystemOne on CPU; bundled per-question calibration; no operator endpoint or network access. | Local CPU measurement on the existing 534-decision v1.3 set plus 308 sealed v1.4 decisions; no operator endpoint received sealed text.",
  "decider-4b-v2": "decider-ai 1.2.2 (PyPI wheel identical to tag v1.2.2 abadc94), weights Mapika/decider-4b rev 7ab294cbdf6be6ac17fc818c10cdead744393d92 (decider_config version 4b-v2, T=1.935), uvicorn decider.serve:app. Disclosed by the author: 8,000 of the v2 LoRA rows come from generators written from the published names of the ten sealed families (no item read). | Independent #1 gate (24 Sep): LEGIT. The author's private stage-2 training rows could not be audited for overlap with public items. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B dense size class, as decider-2b), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "cygnet": "blockbrain-ai/cygnet-recipe 81974de: frozen google/gemma-4-12B-it rev 707f0a3b8a3c7ad586ed01e27eafbad8a27dd0f7 on unmodified vLLM 0.30.0 (image vllm/vllm-openai:v0.30.0@sha256:8a69ffad…), the author's shim: options as letters in the benchmark's label order, logits masked to the option letters, one calibration temperature T=3.4 fitted on the author's own items. The shim sends its own system prompt and renders structured state with json indent=1. Over-context/over-26-option inputs are HTTP 422 (a wrong answer). | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter google/gemma-3-12b-it list price $0.05/M input (the nearest hosted 12B Gemma; gemma-4-12b-it is not listed; the Winnow-12B / Jev-Omni precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "malkuth-4b": "dhtocks/malkuth-4b rev 11dc416995dab324803cb6c533f1d5c69e19d630 (rank-16 LoRA + pointer head over Qwen/Qwen3.5-4B-Base rev 1001bb4d826a52d1f399e183466143f4da7b741b), served by kev.serve from jaredpalmer/kev 557598fced1dada75dfbf36ed144dce309ac6ceb (the author's evaluation revision). The released head carries temperature 1.0 (no fitted temperature), although the card says calibrated. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B one-pass size class, as kev-4b), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "malkuth-2b": "dhtocks/malkuth-2b rev 401304b989451876070d83c488271b4f927d03ab (rank-16 LoRA + pointer head over empero-ai/Qwen3.8-2B-Distill rev e37a2dc4acc68ad75a91e07e63168cb04cc06345, fitted T=1.1755), served by kev.serve from jaredpalmer/kev 557598fced1dada75dfbf36ed144dce309ac6ceb. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (no hosted ~2B listed; the 4B price errs high, the decider-2b precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "swanone": "Round 4 unpublished. Draft vocabulary derived via AGPL-3.0 generator (provenance recorded separately). Submitter consulted all 231 public tasks and swept temperature over 111 public hard tasks. Score corrected during review from pooled-534 ECE/latency to hard-tier/242-item block. | blockbrain-ai/swanone-recipe abcca2e789316472e58d4e824b61c04f419c3ba5: Mia-AiLab/Qwen3.8-Flash-Next-NVFP4 rev 925d7be6c14c6c9442ef83e8f05b5a3c39304f69 on vllm/vllm-openai:qwen38-flash-next@sha256:0aea3024… with MiaAI Lab's nine patched vLLM files (hash-verified) and the author's shim (option letters in the benchmark's order, the model's own renormalised letter mass, no temperature), H100.md serve command, here on an RTX PRO 6000 Blackwell. The shim sends its own system prompt; structured state is rendered with json indent=1. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter qwen/qwen3.8-flash list price $0.15/M input (same underlying Flash-Next weights; the NInfer Flash-Next precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "standardone-8b": "StandardThinking/StandardOne-8B rev 0f14d009a9400e55ea5a00a89b4d859882db704e (Ministral-3-8B-Instruct-2512 + LoRA, merged) on stock SGLang 0.5.20 behind the author's jev-adapter from server/ at the same revision, nominated configuration: --prompt-wording native --native-system-prompt none --default-temperature 1.35, context 8192. Disclosed by the author: benchmark-directed development; the native wording was selected by an ablation on the public hard tier; 181 of 359,497 training rows reuse one generic 58-character public-hard instruction. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter mistralai/ministral-8b-2512 list price $0.15/M input (= the author's proposed Mistral API price for the exact base), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "typecastlm": "typecastlm[server] 1.1.2 (wheel identical to git e95911e), checkpoint mihailgribov/typecastlm-qwen3.5-3.8b rev dcfecfdc28e44ef82631901628535b71574e27af: frozen Qwen/Qwen3.5-4B blocks 0-27 plus a computed 39-row head, one forward pass, four calibration temperatures by question type; transformers 5.3.0, flash-linear-attention 0.5.2. States over 32,768 tokens are folded in the middle, not refused. The server reads a noul question's meaning from the order of its two criteria (first = true), not from their keys. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (the exact base weights; one pass, no output), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
  "clm-8b": "Contrastive-LM/CLM commit cca045ffdb07b3ebcfe6938537cdeac5e14899c9, head Contrastive-LM/CLM-v0.1-8B rev 87655cb835bd76fd66c2da78e1e3709f7fa11a94 (clm-latest), Qwen3-8B rev b968826d9c46dd6066d109eabc6255188de91218 last-token pooling via vLLM. Authors' documented system_one path: state + instructions as state text, each option description as a candidate action, softmax over contrastive scores at temperature 1.0. | Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod; no operator endpoint; no golds were exposed. Latency is in-process Engine.answer time on the serial standard+judge items with the self-hosted adjustment. Cost is estimated at the Qwen3-Embedding-8B hosted list price ($0.01/M input, same-size 8B pooling encoder) over CLM's measured encoder tokens; it is not a GPU bill.",
  "instinct": "Requested in GitHub issue #69. A free evaluation/demo endpoint that speaks TypeSafe's /v1/systemone wire format, so the unchanged typesafe adapter ran it and no mapping of ours was involved; we registered the account and created the key ourselves in their console. The author states frozen Qwen3.8-27B base weights with no fine-tuning, read in one forward pass per question with no autoregressive decoding; the serving stack is not public, so nothing about it could be reviewed and the row rests on the API's own behaviour. Cost is an ESTIMATE: ZooWork publishes no bookable price, so the base model's public reference price is used (OpenRouter model-level qwen/qwen3.8-27b, $0.42/M input, zero output for a direct-logit readout); the author-announced tariff is not used. Classified as a free evaluation/demo endpoint (no SLA, no status page, no terms, pricing 'to be announced'), so the x2 latency adjustment applies. Re-scored when a bookable price is published. | API measurement: sealed item text (no golds) was sent to the operator endpoint; all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) through JevBench's adapter.",
  "jevact": "Requested in GitHub issue #66. The author published an endpoint and an evaluation client, not weights, and his API is not the TypeSafe wire format, so JevBench's own `jevact_api` adapter implements exactly the mapping table in his eval_code.zip: state to state, instructions to the question, the labels with their criteria text as the options in label order, noul as false/true, and results[0].options[i].probability read back as the probability of labels[i] by index. His server answers HTTP 400 with `all inference items exceeded max_context_tokens or contained reserved markers` for an input it cannot take, and his own client and test script record exactly that as one failed item and continue; the adapter therefore surfaces that one documented refusal as the harness's 422 refusal path, which scores it as a wrong answer and does not count toward the stop rule - the same treatment the swanOne and Cygnet packages get for their own 422. Any other 400 or an outage still stops the run. The endpoint reports no token accounting, so cost is the labelled 2B size-class estimate. The endpoint is one machine in China and the latency includes that distance; it is not a production API and gets the standard x2 self-host adjustment, without the +0.15 s that only our own servers carry. | API measurement: sealed item text (no golds) was sent to the operator endpoint; all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) through JevBench's adapter.",
  "verdict-small": "Requested in GitHub issue #73. Run through the author's own `verdict serve` on our CPU, which speaks TypeSafe's /v1/systemone wire format, so JevBench's unchanged typesafe adapter ran it and no mapping of ours was involved. A 118M multilingual bi-encoder: every option is scored against the rendered state by cosine similarity, at the model's own scale (temperature 1.0, no calibrator fitted on JevBench items, as the author states). Structured state is rendered as `key: value` lines by his own code. Code review before the run: the only network call is the Hugging Face download of his own checkpoint, no telemetry, no key, no rule written against public items. The `usage.input_tokens` his server reports is a word count and not a tokeniser count, so the cost is the labelled size-class estimate rather than a measured token price. Self-host latency gets the standard x2 + 0.15 s adjustment. The author's own public-set figures were easy 0.938, standard 0.486, hard 0.396 on an Apple M5 CPU. | Offline measurement of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) on our own CPU through the author's server; no operator endpoint.",
  "kushal-gemma4-31b-it-autoloops": "Gemma 4 31B IT served through the Autoloops systemone API. Cost = the API's own token usage x Autoloops' published rates ($0.20/M input, $0.45/M output; no output tokens were used). | API measurement: public and sealed item content reached the operator endpoint; no gold labels or answers were sent.",
  "plumb-4b": "crh225/plumb-4b @ 55de037801a8a9b9de3db5c0e16cef86210c2186: merged bf16 weights, LoRA r16 on the attention projections of JevK5 v0.2 (itself Qwen3.5-4B), served by the author's documented command, jevk5 v0.2.0's own jevk5-serve (github.com/allebee/jevk5, Apache-2.0), unmodified. One forward pass per decision: softmax over the declared options' answer-letter logits at the last position divided by the package's own T = 2.07 (jevk5_config.json), up to 16 options, 0 generated tokens; inputs over 16,384 tokens are refused rather than truncated. Disclosed by the author: no JevBench item was trained or tuned on and every checkpoint and temperature choice came from his own held-out sets, but the 231 public items were scored after each of five training rounds and that aggregate feedback shaped the recipe (hard mining, long documents, document length) — development against the public distribution, stated plainly. His own overlap audit dropped 9 training items sharing more than two 8-word spans with a public item. | Independent top-five gate (25 Sep 2026): LEGIT — cost basis (same as its JevK5 parent), public-versus-sealed gap against like-for-like rows, an 8-gram screen of its released training data (no row shares more than two 8-grams with a public item) and an exact composite recomputation. | Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium H100 80 GB (Hopper) pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate at the EmpirioLabs Qwen3.5-4B public pay-as-you-go list price ($0.04/M input, no output generated), applied to this system's measured tokens; it is not a GPU bill. 25 Sep 2026 pre-release correction: retired DeepInfra $0.03/M input reference replaced by bookable EmpirioLabs $0.04/M exact-base input reference; draft score 67.09 to 65.84.",
  "imajev_4b": "Imajev-4B: requested adapter c9e5f132465da85d31735ec502d5557982671a7d and server a0134749e0900189c129cd6bb5000969f3b64bb5; one rotation with calibration.json. The optimized image included flash-linear-attention/fla-core 0.5.2 and causal-conv1d 1.7.0; CUDA profiling confirmed the pinned kernels ran. Cost is an estimate using the public DeepInfra Qwen/Qwen3.5-4B base-model reference price ($0.03/M input, $0.15/M output), with no generated output tokens; it is not a GPU bill."
 },
 "superseded_rows": {
  "nimble-9b": {
   "reason": "Re-run in v1.2.8 after Bespoke Labs raised the serving prompt limit from 2,048 to 8,192 tokens (bespokelabsai/nimble PR #4); the v1.1.3 run had failed long items on that limit.",
   "old_score": 61.45878169849573,
   "old_tiers": {
    "easy": 1.0,
    "standard": 0.9479166666666666,
    "judge": 0.8904109589041096,
    "hard": 0.43636363636363634
   },
   "old_endpoint_condition": "our RunPod GPU (A40 48 GB (EU-SE-1)), reached over the internet"
  }
 },
 "excluded_runs": [
  {
   "key": "open-alternative-jev-reversed-order",
   "run_key": "open-alternative-jev",
   "why_not_ranked": "Our first adapter put the options in reverse order (A. no, B. yes); the author's yes_no() helper builds A. yes, B. no. An adapter mistake, not a model weakness, so the author-order run is the ranked row. Raw run files are kept.",
   "tiers": {
    "easy": 1.0,
    "standard": 0.8125,
    "judge": 0.5068493150684932,
    "hard": 0.55
   },
   "footnote": "With the options in reverse order (A. no, B. yes) the same model scored 21 % instead of 72 % on yes/no answer-judging items — small models are very sensitive to option order."
  }
 ],
 "honorable_mentions": {
  "heading": "Honorable mentions — services built on another entrant's model",
  "rule": "A service that runs another entrant's model is listed with all of its scores and axes, but is not ranked against the models. Ranking it would rank the same model twice, once at the model's own price and once at the service's. The row keeps every number, axis, cost basis and per-task outcome; it carries no rank number.",
  "systems": {
   "classifier-dev-fast": {
    "runs_on_key": "jev-1.13.0",
    "runs_on": "Jev (TypeSafe)",
    "short_reason": "runs on Jev (TypeSafe) — listed, not ranked",
    "why_not_ranked": "classifier.dev is not its own model. Its own pages say so: \"The fast tier is Jev, TypeSafe's decision model\" (https://classifier.dev/benchmark, read 2026-09-20), and the API answers with \"model\": \"jev-1.13.0\" — the same model version this benchmark measures directly as Jev 1.13.0. What it adds is a price and, on its smart tier, an orchestration layer: \"The smart tier is Jev plus a reasoning model re-asking only the answers Jev put under 0.7 confidence\" — escalation on low confidence (a model cascade), not best-of-N, not self-consistency and not a committee. Its published escalation model is gemini-3.8-flash. Ranking it against Jev would rank Jev's model against Jev's model, so from v1.2.4 it is an honorable mention instead of #1.",
    "tier_measured": "Only the fast tier was measured. The smart tier's escalation was never run, so nothing here scores it.",
    "price_note": "$0.0033 per 1,000 decisions is an estimate from the published flat-rate plan at full use: classifier.dev Pro is $20/month for 200,000 fast classifications a day (https://classifier.dev/pricing, read 2026-09-20), and one classification is one decision. Lower use costs more per decision — at a tenth of that allowance it is $0.033 per 1,000 — and the free tier (20,000 fast classifications a day), which is what our run used, costs nothing. Their pages do not say how the flat rate is funded, so we do not know their cost basis; the only figure they publish is what the model costs a caller: \"The model behind the fast tier costs about $0.005 per thousand classifications and needs a TypeSafe key\" (https://classifier.dev/pricing) — for their short single-sentence inputs, not for JevBench's whole questions.",
    "not_pass_through": "On our set the fast tier scored 97.3 % on the judge tier against Jev's 94.5 %, and 70.5 % against 74.1 % on the hard tier. classifier.dev's own explanation for differences of this kind is batching (\"The fast tier is Jev, packed a thousand to a request\"); on their own two test sets they measured the same difference as noise.",
    "credit": "A legitimate, well-documented product: free without an account, open source (https://github.com/mrmps/classifier-dev), by Michael Ryaboy (@michael_chomsky).",
    "sources": [
     "https://classifier.dev",
     "https://classifier.dev/benchmark",
     "https://classifier.dev/pricing",
     "https://classifier.dev/about"
    ],
    "sources_read": "2026-09-20"
   }
  }
 },
 "systems": [
  {
   "key": "imajev_4b",
   "display": "Imajev-4B",
   "author": "mohit67890",
   "repo": "https://github.com/mohit67890/imajev",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (author adapter and server metadata)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": "author server class probabilities",
   "endpoint_condition": "Evaluator-owned Lium GPU pod; network-disabled, read-only container; the author's reviewed server ran on loopback.",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.8904109589041096,
    "hard_public": 0.7207207207207207,
    "hard_heldout": 0.7522935779816514,
    "sealed": 0.37012987012987014
   },
   "speed": {
    "p50_s_raw": 0.03957595250176382,
    "p95_s_raw": 0.11505696040330804,
    "p50_s_adjusted": 0.22915190500352764,
    "p95_s_adjusted": 0.3801139208066161,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial standard+judge requests; 242 decisions",
    "hardware": "evaluator-owned Lium GPU; exact model recorded in the run receipt",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's reviewed server at commit a0134749e0900189c129cd6bb5000969f3b64bb5 on loopback inside the offline container, single secure Lium GPU; exact model recorded in the run receipt, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02200522471910112,
    "basis": "ESTIMATE: DeepInfra public model catalog for exact base Qwen/Qwen3.5-4B, retrieved 2026-09-27 00:52 UTC: USD 0.03/M input and USD 0.15/M output; see receipts/BASE-PRICE-REFERENCE.json; full-forward input tokens are counted once for the single pinned server pass; no generated output tokens; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null,
    "previous_price_basis": null
   },
   "calibration": {
    "score": 80.35017974166122,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "raw_results_sha256": "0137a96dc10ff92161f590bfd998f9d5924b075b435a45fed3dd5ba9d59c92b1",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc",
    "scope_counts": {
     "public-source": 377,
     "historical-private": 157,
     "current-sealed": 308
    },
    "source_row_sha256": "1b1755820752ba62e9910236f6dd6e18a0aa485fc0ed6990170e40cc4962f76f",
    "result_rows_handoff_sha256": "51584c82047bf4392d5b05be1334775d6bfb51b2074cca682a87b1e7cdcaebf5"
   },
   "new_in": "v1.4.2.2",
   "source_round": "v1.4.2.2 fast-lane addition (official Imajev-4B v1.4 protocol measurement; one rotation)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8614718614718615,
   "sealed_accuracy": 0.37012987012987014,
   "public_minus_sealed_gap_pp": 49.134199134199136,
   "sealed_aggregate": {
    "n": 308,
    "attempted": 308,
    "answered_valid": 308,
    "failed_or_invalid": 0,
    "by_family": {
     "ambiguous_abstain": {
      "correct": 15,
      "n": 37,
      "accuracy": 0.40540540540540543
     },
     "judge_hard": {
      "correct": 18,
      "n": 41,
      "accuracy": 0.43902439024390244
     },
     "long_policy": {
      "correct": 13,
      "n": 40,
      "accuracy": 0.325
     },
     "multi_hop": {
      "correct": 13,
      "n": 38,
      "accuracy": 0.34210526315789475
     },
     "paraphrase_robustness": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "probability": {
      "correct": 7,
      "n": 28,
      "accuracy": 0.25
     },
     "safety_judge": {
      "correct": 7,
      "n": 16,
      "accuracy": 0.4375
     },
     "temporal_numeric": {
      "correct": 17,
      "n": 56,
      "accuracy": 0.30357142857142855
     },
     "tradeoff": {
      "correct": 10,
      "n": 26,
      "accuracy": 0.38461538461538464
     },
     "trap_adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     }
    },
    "by_panel_stratum": {
     "0/3": {
      "correct": 20,
      "n": 109,
      "accuracy": 0.1834862385321101
     },
     "1/3": {
      "correct": 39,
      "n": 97,
      "accuracy": 0.4020618556701031
     },
     "2/3": {
      "correct": 55,
      "n": 102,
      "accuracy": 0.5392156862745098
     }
    },
    "ece": 0.17209076066270215,
    "mean_tvd_gold_probs": 0.25805817680417786,
    "calibration": 69.88801509352089,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.17532467532467533,
      "accuracy": 0.4074074074074074,
      "n": 54
     },
     "conf>=0.9": {
      "coverage": 0.02922077922077922,
      "accuracy": 0.2222222222222222,
      "n": 9
     }
    }
   },
   "v130_comparison_rank": 1,
   "v130_comparison_score": 78.81325984024393,
   "axes": {
    "intelligence": 52.22351834450795,
    "calibration": 80.35017974166122,
    "speed": 90.59962753052734,
    "cost": 59.72422576086029
   },
   "jevbench_score": 67.36821557095253,
   "presets": {
    "JevBench Score (25:25:25:25)": 67.36821557095253,
    "Balanced 33:33:33 (no calibration)": 63.92546021840456,
    "Emphasis on Accuracy 60:20:20": 58.667143470951444,
    "Emphasis on Speed 20:60:20": 72.4587231323314,
    "Emphasis on Cost 20:20:60": 62.17598006001362,
    "Intelligence only": 52.22351834450795
   },
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium single secure Lium GPU; exact model recorded in the run receipt pod, through JevBench's unchanged typed-decision adapter against the author's reviewed server at commit a0134749e0900189c129cd6bb5000969f3b64bb5 on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled base-model reference estimate: DeepInfra public model catalog for exact base Qwen/Qwen3.5-4B, retrieved 2026-09-27 00:52 UTC: USD 0.03/M input and USD 0.15/M output; see receipts/BASE-PRICE-REFERENCE.json, applied to the exact full-forward input-token count captured across the author's one rotation; inference returns class probabilities without generated output tokens. The run image had flash-linear-attention 0.5.2 and causal-conv1d 1.7.0 installed and used; this is an optimized-kernel timing. It is not a GPU bill.",
   "pricing_change_note": null,
   "rank": 1,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 1,
    "Balanced 33:33:33 (no calibration)": 1,
    "Emphasis on Accuracy 60:20:20": 1,
    "Emphasis on Speed 20:60:20": 2,
    "Emphasis on Cost 20:20:60": 1,
    "Intelligence only": 3
   }
  },
  {
   "key": "plumb-4b",
   "display": "Plumb-4B (crh225, JevK5 v0.2 + LoRA)",
   "author": "crh225",
   "repo": "https://github.com/crh225/plumb",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (weights and code; NOTICE credits JevK5, Qwen and SemIf); base Qwen3.5-4B Apache-2.0",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (H100 80 GB (evaluator-owned Lium pod)), offline read-only container, author's model served on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.952054794520548,
    "hard": 0.7772727272727272
   },
   "speed": {
    "p50_s_raw": 0.016472740564495325,
    "p95_s_raw": 0.04735108860768378,
    "p50_s_adjusted": 0.18294548112899064,
    "p95_s_adjusted": 0.24470217721536755,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "H100 80 GB (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, H100 80 GB (Hopper), all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0298085393258427,
    "basis": "ESTIMATE: EmpirioLabs qwen3-5-4b public pay-as-you-go list price $0.04/M input, $0.07/M output; exact Qwen3.5-4B base model, 25 Sep 2026 cutoff; no output generated by this system. https://empiriolabs.ai/models/qwen3-5-4b",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null,
    "previous_price_basis": {
     "basis": "ESTIMATE: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (the exact base weights; one pass, no output tokens), read 2026-09-25; the server's own usage.input_tokens; nothing generated; estimated, not charged",
     "usd_per_1000": 0.02235640449438202
    }
   },
   "calibration": {
    "score": 75.48544394158392,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "raw_results_sha256": "293fbc168a47a44f86705c1ed281e395763468ff0b06b3ea6f54a7248aba2286",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc",
    "scope_counts": {
     "public-source": 377,
     "historical-private": 157,
     "current-sealed": 308
    },
    "aggregate_source_sha256": "aade49ff5c86936b14ed287d6a503d3c1ba2884eb2b9f745cc8f0e4b73296165",
    "row_sha256": "0fe248c8e2c1f267951c8d8c0fd3a480baea1cbf725caacd2911c087facda709"
   },
   "new_in": "v1.4.2.1",
   "source_round": "v1.4.2.1 fast-lane addition (official Plumb-4B v1.4 protocol measurement)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8961038961038961,
   "sealed_accuracy": 0.37987012987012986,
   "public_minus_sealed_gap_pp": 51.623376623376615,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.5946,
     "judge_hard": 0.2927,
     "long_policy": 0.3,
     "multi_hop": 0.4211,
     "paraphrase_robustness": 0.4286,
     "probability": 0.2857,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.375,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.3299,
     "2/3": 0.6373
    },
    "ece": 0.28215403489574986,
    "mean_tvd_gold_probs": 0.23924419273418004,
    "calibration": 59.82238687371601,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4481,
      "accuracy": 0.4203
     },
     "conf>=0.9": {
      "coverage": 0.2045,
      "accuracy": 0.4762
     }
    }
   },
   "v130_comparison_rank": 1,
   "v130_comparison_score": 79.73357327525596,
   "scoring_note": "Cost is a labelled estimate at the EmpirioLabs Qwen3.5-4B public pay-as-you-go list price ($0.04/M input, no output generated), applied to this system's measured tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 52.97892668438094,
    "calibration": 75.48544394158392,
    "speed": 93.49040479933188,
    "cost": 55.76977914062651
   },
   "jevbench_score": 65.84344821613185,
   "presets": {
    "JevBench Score (25:25:25:25)": 65.84344821613185,
    "Balanced 33:33:33 (no calibration)": 63.15447429227392,
    "Emphasis on Accuracy 60:20:20": 58.64866547345907,
    "Emphasis on Speed 20:60:20": 72.57405723994296,
    "Emphasis on Cost 20:20:60": 59.9777202388783,
    "Intelligence only": 52.97892668438094
   },
   "pricing_change_note": "25 Sep 2026 pre-release correction: retired DeepInfra $0.03/M input reference replaced by bookable EmpirioLabs $0.04/M exact-base input reference; draft score 67.09 to 65.84.",
   "rank": 2,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 2,
    "Balanced 33:33:33 (no calibration)": 2,
    "Emphasis on Accuracy 60:20:20": 2,
    "Emphasis on Speed 20:60:20": 1,
    "Emphasis on Cost 20:20:60": 3,
    "Intelligence only": 2
   }
  },
  {
   "key": "decider-4b-v2",
   "display": "decider-4b v2 (Mapika)",
   "author": "Mapika",
   "repo": "https://github.com/Mapika/decider",
   "class": "system-one-open",
   "licence": "Apache-2.0 (package and weights)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.8767123287671232,
    "hard": 0.6727272727272727
   },
   "speed": {
    "p50_s_raw": 0.01749592460691929,
    "p95_s_raw": 0.06258766227401794,
    "p50_s_adjusted": 0.18499184921383857,
    "p95_s_adjusted": 0.27517532454803584,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02006786516853933,
    "basis": "ESTIMATE: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B dense size class, as decider-2b), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 74.99659054834055,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "244a63753e86153a1faa53ec10a328702632bb94f2139ccb154bb1583e8e0ce4",
    "row_sha256": "78524df885397f6390200ee0db087ece0093ceeeee780003ab461beb833b54e7",
    "raw_results_sha256": "d389e2f4f830bd8766aaf09117769ad9b048046433b3c51636841a59313455d7",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8354978354978355,
   "sealed_accuracy": 0.3474025974025974,
   "public_minus_sealed_gap_pp": 48.80952380952381,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4324,
     "judge_hard": 0.3171,
     "long_policy": 0.3,
     "multi_hop": 0.3421,
     "paraphrase_robustness": 0.5,
     "probability": 0.3571,
     "safety_judge": 0.375,
     "temporal_numeric": 0.3036,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.3299,
     "2/3": 0.5392
    },
    "ece": 0.2542512987012987,
    "mean_tvd_gold_probs": 0.2849924242424242,
    "calibration": 60.32524891774892,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3052,
      "accuracy": 0.3298
     },
     "conf>=0.9": {
      "coverage": 0.0909,
      "accuracy": 0.1786
     }
    }
   },
   "v130_comparison_rank": 1,
   "v130_comparison_score": 77.91174129320584,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B dense size class, as decider-2b), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 49.35289661364111,
    "calibration": 74.99659054834055,
    "speed": 92.93237918931897,
    "cost": 60.92496476683517
   },
   "jevbench_score": 64.12889492043318,
   "presets": {
    "JevBench Score (25:25:25:25)": 64.12889492043318,
    "Balanced 33:33:33 (no calibration)": 61.616212524307535,
    "Emphasis on Accuracy 60:20:20": 55.381647309356744,
    "Emphasis on Speed 20:60:20": 70.6438537711419,
    "Emphasis on Cost 20:20:60": 60.69269292028249,
    "Intelligence only": 48.083706020529895
   },
   "rank": 3,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 3,
    "Balanced 33:33:33 (no calibration)": 3,
    "Emphasis on Accuracy 60:20:20": 4,
    "Emphasis on Speed 20:60:20": 3,
    "Emphasis on Cost 20:20:60": 2,
    "Intelligence only": 7
   }
  },
  {
   "key": "jev-1.13.0",
   "display": "Jev 1.13.0 (TypeSafe AI)",
   "class": "jev",
   "open": "no",
   "author": "TypeSafe AI",
   "repo": "https://docs.typesafe.ai",
   "licence": "proprietary API",
   "underlying": "closed",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "production API (api.typesafe.ai)",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.9452054794520548,
    "hard": 0.740909090909091
   },
   "axes": {
    "intelligence": 53.05904597275748,
    "calibration": 76.3389831504074,
    "speed": 83.26811926100174,
    "cost": 51.96616538951724
   },
   "jevbench_score": 63.29205745601932,
   "speed": {
    "p50_s_raw": 0.6524335257709026,
    "p95_s_raw": 0.7221905551850795,
    "p50_s_adjusted": 0.6524335257709026,
    "p95_s_adjusted": 0.7221905551850795,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.6717139892280102,
    "hard_tier_p95_s": 0.8295219600200652
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.03991423595505617,
    "basis": "public tariff x measured tokens (https://docs.typesafe.ai/models (output tokens not billed)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
    "usd_per_1000_v11_tiers": 0.024698140127388538,
    "usd_per_1000_hard": 0.06163175454545452,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 76.3389831504074,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.06061111111111118,
    "probability_fidelity": 77.42800000000001,
    "brier_hard": 0.339603665766944,
    "brier_standard_judge_v11": 0.05558677685950414,
    "note": null
   },
   "hard": {
    "run": "runs/jev-1.13.0--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.740909090909091,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857142857142857
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 30,
      "n": 35,
      "accuracy": 0.8571428571428571
     },
     "probability": {
      "correct": 16,
      "n": 20,
      "accuracy": 0.8
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.339603665766944,
    "ece": 0.06061111111111118,
    "probability_fidelity": 77.42800000000001,
    "calibration_score": 82.6528888888889,
    "onehot": {
     "ece": 0.25909090909090904,
     "probability_fidelity": 59.264999999999986,
     "calibration_score": 53.72340909090909
    },
    "latency_p50_s": 0.6717139892280102,
    "latency_p95_s": 0.8295219600200652,
    "mean_input_tokens": 1467.4227272727273,
    "mean_output_tokens": 45.086363636363636,
    "charged_usd": 0.013558985999999993
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 63.29205745601932,
    "Balanced 33:33:33 (no calibration)": 59.88069849063511,
    "Emphasis on Accuracy 60:20:20": 56.951843086859334,
    "Emphasis on Speed 20:60:20": 67.45962090738993,
    "Emphasis on Cost 20:20:60": 56.44220212537236,
    "Intelligence only": 53.05904597275748
   },
   "rank": 4,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 4,
    "Balanced 33:33:33 (no calibration)": 4,
    "Emphasis on Accuracy 60:20:20": 3,
    "Emphasis on Speed 20:60:20": 6,
    "Emphasis on Cost 20:20:60": 5,
    "Intelligence only": 1
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8658008658008658,
   "sealed_accuracy": 0.36688311688311687,
   "public_minus_sealed_gap_pp": 49.891774891774894,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.3415,
     "long_policy": 0.275,
     "multi_hop": 0.4474,
     "paraphrase_robustness": 0.6429,
     "probability": 0.5,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2857,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1009,
     "1/3": 0.3402,
     "2/3": 0.6765
    },
    "ece": 0.2203584546766365,
    "mean_tvd_gold_probs": 0.285059657177839,
    "calibration": 63.7111716734444,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2922,
      "accuracy": 0.4
     },
     "conf>=0.9": {
      "coverage": 0.0844,
      "accuracy": 0.4231
     }
    }
   },
   "v130_comparison_rank": 3,
   "v130_comparison_score": 74.40448849535433,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (TypeSafe API), as for every API measurement"
  },
  {
   "key": "jevk5-v02",
   "display": "JevK5 v0.2.0",
   "author": "allebee",
   "repo": "https://github.com/allebee/jevk5",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (code/adapter); Apache-2.0 (Qwen3.5-4B base)",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "endpoint_kind": "unknown",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.9452054794520548,
    "hard": 0.7
   },
   "speed": {
    "p50_s_raw": null,
    "p95_s_raw": null,
    "p50_s_adjusted": null,
    "p95_s_adjusted": null,
    "adjustment": "See original v1.3 measurement",
    "run": null,
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.022356404494382018,
    "basis": "Reconstructed from the frozen v1.3 Cost axis; same speed/cost measurement, not a new price observation",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 74.53428390286419,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9167
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857
     },
     "judge_hard": {
      "correct": 27,
      "n": 33,
      "accuracy": 0.8182
     },
     "long_policy": {
      "correct": 19,
      "n": 38,
      "accuracy": 0.5
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.3667
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    }
   },
   "source_round": "round 5 new",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8528138528138528,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 52.16450216450217,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4324,
     "judge_hard": 0.3659,
     "long_policy": 0.225,
     "multi_hop": 0.3947,
     "paraphrase_robustness": 0.1429,
     "probability": 0.3929,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1376,
     "1/3": 0.268,
     "2/3": 0.598
    },
    "ece": 0.22177826419666213,
    "mean_tvd_gold_probs": 0.2744738869604739,
    "calibration": 64.09847923231008,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2695,
      "accuracy": 0.4096
     },
     "conf>=0.9": {
      "coverage": 0.0682,
      "accuracy": 0.381
     }
    }
   },
   "v130_comparison_rank": 1,
   "v130_comparison_score": 77.29659511062113,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "axes": {
    "intelligence": 48.8938430961117,
    "calibration": 74.53428390286419,
    "speed": 91.09028776527087,
    "cost": 59.517941238875515
   },
   "jevbench_score": 62.04446598579722,
   "presets": {
    "JevBench Score (25:25:25:25)": 62.04446598579722,
    "Balanced 33:33:33 (no calibration)": 59.477415714241246,
    "Emphasis on Accuracy 60:20:20": 53.63884118449528,
    "Emphasis on Speed 20:60:20": 68.11966006176617,
    "Emphasis on Cost 20:20:60": 58.424671736663036,
    "Intelligence only": 46.75440288414063
   },
   "rank": 5,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 5,
    "Balanced 33:33:33 (no calibration)": 5,
    "Emphasis on Accuracy 60:20:20": 6,
    "Emphasis on Speed 20:60:20": 4,
    "Emphasis on Cost 20:20:60": 4,
    "Intelligence only": 8
   }
  },
  {
   "key": "cygnet",
   "display": "Cygnet (blockbrain, frozen Gemma-4-12B-it)",
   "author": "blockbrain",
   "repo": "https://github.com/blockbrain-ai/cygnet-recipe",
   "class": "system-one-open",
   "licence": "shim MIT; weights Apache-2.0 with Google's Gemma Prohibited Use Policy",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.952054794520548,
    "hard": 0.7545454545454545
   },
   "speed": {
    "p50_s_raw": 0.035219401121139526,
    "p95_s_raw": 0.11935583325102912,
    "p50_s_adjusted": 0.22043880224227905,
    "p95_s_adjusted": 0.38871166650205824,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.03736151685393259,
    "basis": "ESTIMATE: OpenRouter google/gemma-3-12b-it list price $0.05/M input (the nearest hosted 12B Gemma; gemma-4-12b-it is not listed; the Winnow-12B / Jev-Omni precedent), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 74.92580266255106,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "136858429923947953444101d7f2e93ca9c6a0753a2d66f3f76bb8dbb48fad3c",
    "row_sha256": "fde5ccf3e37aaaece5aef68538563f78e9d5d0b90b23da1ac52e7c395fbcb697",
    "raw_results_sha256": "c3e7a4acbb69d787842e4835f4646993d4eecadce66d88b8eb93152994aece89",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8787878787878788,
   "sealed_accuracy": 0.33766233766233766,
   "public_minus_sealed_gap_pp": 54.112554112554115,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3784,
     "judge_hard": 0.5366,
     "long_policy": 0.225,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.6429,
     "probability": 0.3571,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1376,
     "1/3": 0.299,
     "2/3": 0.5882
    },
    "ece": 0.25798077360743493,
    "mean_tvd_gold_probs": 0.2462126169437964,
    "calibration": 61.89129179206668,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3506,
      "accuracy": 0.4259
     },
     "conf>=0.9": {
      "coverage": 0.0974,
      "accuracy": 0.3667
     }
    }
   },
   "v130_comparison_rank": 2,
   "v130_comparison_score": 76.04533426828046,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter google/gemma-3-12b-it list price $0.05/M input (the nearest hosted 12B Gemma; gemma-4-12b-it is not listed; the Winnow-12B / Jev-Omni precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 49.51055476185943,
    "calibration": 74.92580266255106,
    "speed": 90.67084381968537,
    "cost": 52.827265000089206
   },
   "jevbench_score": 61.76221691847706,
   "presets": {
    "JevBench Score (25:25:25:25)": 61.76221691847706,
    "Balanced 33:33:33 (no calibration)": 58.64782258317542,
    "Emphasis on Accuracy 60:20:20": 54.14135859650023,
    "Emphasis on Speed 20:60:20": 67.88970443556094,
    "Emphasis on Cost 20:20:60": 55.70145603635848,
    "Intelligence only": 48.545990784103694
   },
   "rank": 6,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 6,
    "Balanced 33:33:33 (no calibration)": 6,
    "Emphasis on Accuracy 60:20:20": 5,
    "Emphasis on Speed 20:60:20": 5,
    "Emphasis on Cost 20:20:60": 6,
    "Intelligence only": 6
   }
  },
  {
   "key": "hopper",
   "display": "Hopper",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "HopitAI",
   "repo": "https://huggingface.co/HopitAI/hopper",
   "licence": "Component-specific terms recorded in RESULT.md; submitted adapter release and Qwen base retain their respective terms",
   "underlying": "Qwen3.5-4B plus HopitAI/hopper LoRA",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX A6000 48 GB), local loopback HTTP; serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.8356164383561644,
    "hard": 0.65
   },
   "axes": {
    "intelligence": 48.001323586436904,
    "calibration": 79.05762339459613,
    "speed": 86.80781551893452,
    "cost": 58.728026855585874
   },
   "jevbench_score": 59.433442874809664,
   "speed": {
    "p50_s_raw": 0.12881201645359397,
    "p95_s_raw": 0.18081656964495776,
    "p50_s_adjusted": 0.40762403290718796,
    "p95_s_adjusted": 0.5116331392899155,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our GPU (lium.io RTX A6000 48 GB), local loopback HTTP; serial",
    "hard_tier_p50_s": 0.13407477270811796,
    "hard_tier_p95_s": 0.3750792991835623
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02375376404494382,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.02375376404494382,
    "usd_per_1000_hard": 0.02375376404494382,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 79.05762339459613,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.06006635020486461,
    "probability_fidelity": 78.07297634980476,
    "brier_hard": 0.4290948535258254,
    "brier_standard_judge_v11": 0.1791607379566707,
    "note": null
   },
   "hard": {
    "run": "hopper/runs/results.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.65,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857142857142857
     },
     "judge_hard": {
      "correct": 23,
      "n": 33,
      "accuracy": 0.696969696969697
     },
     "long_policy": {
      "correct": 20,
      "n": 38,
      "accuracy": 0.5263157894736842
     },
     "multi_hop": {
      "correct": 24,
      "n": 35,
      "accuracy": 0.6857142857142857
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4290948535258254,
    "ece": 0.06006635020486461,
    "probability_fidelity": 78.07297634980476,
    "calibration_score": 83.02985315441592,
    "onehot": {
     "ece": 0.35,
     "probability_fidelity": 53.949000000000005,
     "calibration_score": 41.974500000000006
    },
    "latency_p50_s": 0.13407477270811796,
    "latency_p95_s": 0.3750792991835623,
    "mean_input_tokens": 1295.9272727272728,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 59.433442874809664,
    "Balanced 33:33:33 (no calibration)": 55.99324931573137,
    "Emphasis on Accuracy 60:20:20": 50.614780141412744,
    "Emphasis on Speed 20:60:20": 63.632776546044724,
    "Emphasis on Cost 20:20:60": 55.231405918209404,
    "Intelligence only": 44.24045955269003
   },
   "rank": 7,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 7,
    "Balanced 33:33:33 (no calibration)": 7,
    "Emphasis on Accuracy 60:20:20": 7,
    "Emphasis on Speed 20:60:20": 7,
    "Emphasis on Cost 20:20:60": 7,
    "Intelligence only": 10
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8225108225108225,
   "sealed_accuracy": 0.3409090909090909,
   "public_minus_sealed_gap_pp": 48.160173160173166,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.3902,
     "long_policy": 0.275,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3571,
     "safety_judge": 0.5,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.2018,
     "1/3": 0.3814,
     "2/3": 0.451
    },
    "ece": 0.15951254697758535,
    "mean_tvd_gold_probs": 0.2587116285456973,
    "calibration": 71.1131638749566,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1786,
      "accuracy": 0.2727
     },
     "conf>=0.9": {
      "coverage": 0.0325,
      "accuracy": 0.2
     }
    }
   },
   "v130_comparison_rank": 2,
   "v130_comparison_score": 75.40886356781982,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "winnow-12b",
   "display": "Winnow-12B Q8",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Eldan Ring",
   "repo": "https://huggingface.co/EldanRing/Winnow-12B",
   "licence": "Apache-2.0, including the applicable Gemma 4 base/derivative licence terms",
   "underlying": "google/gemma-4-12B-it LoRA fine-tune, merged and exported as Q8_0 GGUF",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX 4090 24 GB), reached over the internet from Germany; serial, one request at a time",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.910958904109589,
    "hard": 0.7090909090909091
   },
   "axes": {
    "intelligence": 48.303781520788064,
    "calibration": 64.81178680536223,
    "speed": 82.32624042236164,
    "cost": 52.92322256021786
   },
   "jevbench_score": 55.57545047877118,
   "speed": {
    "p50_s_raw": 0.22503555566072464,
    "p95_s_raw": 0.4126893173903227,
    "p50_s_adjusted": 0.6000711113214493,
    "p95_s_adjusted": 0.9753786347806455,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.2846274711191654,
    "hard_tier_p95_s": 0.8760518249124287
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.037087359550561805,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter google/gemma-3-12b-it hosted reference list price $0.05/M in, $0.0/M out (the nearest publicly hosted 12B Gemma sibling; Winnow reads answer logits in one forward pass and generates no answer tokens) x 393 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.01962611464968153,
    "usd_per_1000_hard": 0.06200931818181819,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 64.81178680536223,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.11992766665864553,
    "probability_fidelity": 67.92304895443986,
    "brier_hard": 0.4051575575819251,
    "brier_standard_judge_v11": 0.09509412904044282,
    "note": null
   },
   "hard": {
    "run": "runs/winnow-12b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7090909090909091,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 29,
      "n": 33,
      "accuracy": 0.8787878787878788
     },
     "long_policy": {
      "correct": 25,
      "n": 38,
      "accuracy": 0.6578947368421053
     },
     "multi_hop": {
      "correct": 24,
      "n": 35,
      "accuracy": 0.6857142857142857
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4051575575819251,
    "ece": 0.11992766665864553,
    "probability_fidelity": 67.92304895443986,
    "calibration_score": 71.96875781135537,
    "onehot": {
     "ece": 0.2909090909090909,
     "probability_fidelity": 47.842000000000006,
     "calibration_score": 44.83009090909091
    },
    "latency_p50_s": 0.2846274711191654,
    "latency_p95_s": 0.8760518249124287,
    "mean_input_tokens": 1240.1863636363637,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 55.57545047877118,
    "Balanced 33:33:33 (no calibration)": 54.1103209454223,
    "Emphasis on Accuracy 60:20:20": 50.097253489428546,
    "Emphasis on Speed 20:60:20": 61.37077334034029,
    "Emphasis on Cost 20:20:60": 52.11940222480541,
    "Intelligence only": 45.08202187528132
   },
   "rank": 8,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 8,
    "Balanced 33:33:33 (no calibration)": 8,
    "Emphasis on Accuracy 60:20:20": 8,
    "Emphasis on Speed 20:60:20": 9,
    "Emphasis on Cost 20:20:60": 10,
    "Intelligence only": 9
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8571428571428571,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 52.5974025974026,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3514,
     "judge_hard": 0.4634,
     "long_policy": 0.225,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.6429,
     "probability": 0.3214,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1376,
     "1/3": 0.268,
     "2/3": 0.598
    },
    "ece": 0.30773902331043035,
    "mean_tvd_gold_probs": 0.37456505751162006,
    "calibration": 50.49784479337596,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4838,
      "accuracy": 0.4094
     },
     "conf>=0.9": {
      "coverage": 0.2597,
      "accuracy": 0.4625
     }
    }
   },
   "v130_comparison_rank": 10,
   "v130_comparison_score": 71.21883011371422,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "reflex-4b",
   "display": "reflex 4B (kshetrajna12)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "kshetrajna12",
   "repo": "https://github.com/kshetrajna12/reflex",
   "licence": "MIT (code, adapter); Apache-2.0 (base)",
   "underlying": "Qwen/Qwen3.5-4B + kshetrajna12/reflex-qwen3.5-4b-lora",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9479166666666666,
    "judge": 0.9726027397260274,
    "hard": 0.6318181818181818
   },
   "axes": {
    "intelligence": 47.457309838375316,
    "calibration": 70.41256219336218,
    "speed": 67.96813879081057,
    "cost": 59.675719111465405
   },
   "jevbench_score": 53.99041264165809,
   "speed": {
    "p50_s_raw": 1.7996779419481754,
    "p95_s_raw": 2.0541166696697473,
    "p50_s_adjusted": 3.7493558838963508,
    "p95_s_adjusted": 4.258233339339495,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 1.8708765655755997,
    "hard_tier_p95_s": 2.159583227336406
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.022087303370786515,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (the exact base weights; one pass, no generated output) x 377 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.01132184713375796,
    "usd_per_1000_hard": 0.037452545454545454,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 70.41256219336218,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.10498626363636375,
    "probability_fidelity": 71.42537999999999,
    "brier_hard": 0.511049058192209,
    "brier_standard_judge_v11": 0.08995296889612805,
    "note": null
   },
   "hard": {
    "run": "runs/reflex-4b--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6318181818181818,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 20,
      "n": 38,
      "accuracy": 0.5263157894736842
     },
     "multi_hop": {
      "correct": 22,
      "n": 35,
      "accuracy": 0.6285714285714286
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.511049058192209,
    "ece": 0.10498626363636375,
    "probability_fidelity": 71.42537999999999,
    "calibration_score": 75.21406363636362,
    "onehot": {
     "ece": 0.36818181818181817,
     "probability_fidelity": 49.4625,
     "calibration_score": 37.91306818181818
    },
    "latency_p50_s": 1.8708765655755997,
    "latency_p95_s": 2.159583227336406,
    "mean_input_tokens": 1248.418181818182,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 53.99041264165809,
    "Balanced 33:33:33 (no calibration)": 51.43803305157935,
    "Emphasis on Accuracy 60:20:20": 47.57253953390997,
    "Emphasis on Speed 20:60:20": 54.953643340339966,
    "Emphasis on Cost 20:20:60": 52.342544429041496,
    "Intelligence only": 42.75327023592517
   },
   "rank": 9,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 9,
    "Balanced 33:33:33 (no calibration)": 11,
    "Emphasis on Accuracy 60:20:20": 10,
    "Emphasis on Speed 20:60:20": 12,
    "Emphasis on Cost 20:20:60": 9,
    "Intelligence only": 11
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7922077922077922,
   "sealed_accuracy": 0.2824675324675325,
   "public_minus_sealed_gap_pp": 50.974025974025984,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.3415,
     "long_policy": 0.2,
     "multi_hop": 0.3421,
     "paraphrase_robustness": 0.2143,
     "probability": 0.2857,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.0826,
     "1/3": 0.299,
     "2/3": 0.4804
    },
    "ece": 0.25715864935064936,
    "mean_tvd_gold_probs": 0.2694915151515152,
    "calibration": 60.80955930735931,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2662,
      "accuracy": 0.2683
     },
     "conf>=0.9": {
      "coverage": 0.1104,
      "accuracy": 0.2647
     }
    }
   },
   "v130_comparison_rank": 16,
   "v130_comparison_score": 70.31658532999862,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "djev",
   "display": "djev (Maisa, diffusion-gemma)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Maisa (David Villalón)",
   "repo": "https://github.com/Davipar/djev-dev",
   "licence": "Apache-2.0 code; Google DiffusionGemma Apache-2.0 weights; no djev-specific weights",
   "underlying": "inference method on google/diffusiongemma-26B-A4B-it (one structured denoising read), not a separately trained model",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "production API (api.djev.dev, free preview)",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9315068493150684,
    "hard": 0.6954545454545454
   },
   "axes": {
    "intelligence": 46.99529771527977,
    "calibration": 55.355844103478105,
    "speed": 91.35677310946093,
    "cost": 57.57524920597589
   },
   "jevbench_score": 52.228492110723586,
   "speed": {
    "p50_s_raw": 0.2370578795671463,
    "p95_s_raw": 0.30865143015980717,
    "p50_s_adjusted": 0.2370578795671463,
    "p95_s_adjusted": 0.30865143015980717,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.24707749113440514,
    "hard_tier_p95_s": 0.3622760258615016
   },
   "cost": {
    "kind": "announced",
    "usd_per_1000": 0.025951254681647943,
    "basis": "ANNOUNCED PRICE (free preview): djev's docs state $0.035 per million input tokens, output tokens free (https://api.djev.dev/docs, 'Usage & credits'; prepaid billing not yet switched on, 19 Sep 2026, so nothing was charged) x measured input tokens (741 per decision on average over all 534 decisions)",
    "usd_per_1000_v11_tiers": 0.013884410828025478,
    "usd_per_1000_hard": 0.04317393181818183,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 55.355844103478105,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.17503581317955752,
    "probability_fidelity": 65.83617501611904,
    "brier_hard": 0.4680494813450917,
    "brier_standard_judge_v11": 0.08268269187152161,
    "note": null
   },
   "hard": {
    "run": "runs/djev--hard (job djev-jevbench-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6954545454545454,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 10,
      "n": 14,
      "accuracy": 0.7142857142857143
     },
     "judge_hard": {
      "correct": 29,
      "n": 33,
      "accuracy": 0.8787878787878788
     },
     "long_policy": {
      "correct": 18,
      "n": 38,
      "accuracy": 0.47368421052631576
     },
     "multi_hop": {
      "correct": 29,
      "n": 35,
      "accuracy": 0.8285714285714286
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4680494813450917,
    "ece": 0.17503581317955752,
    "probability_fidelity": 65.83617501611904,
    "calibration_score": 65.41450619010376,
    "onehot": {
     "ece": 0.30454545454545456,
     "probability_fidelity": 52.93449999999999,
     "calibration_score": 46.01270454545454
    },
    "latency_p50_s": 0.24707749113440514,
    "latency_p95_s": 0.3622760258615016,
    "mean_input_tokens": 1233.540909090909,
    "mean_output_tokens": 8.0,
    "charged_usd": 0.009498265000000002
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 52.228492110723586,
    "Balanced 33:33:33 (no calibration)": 53.439971404052315,
    "Emphasis on Accuracy 60:20:20": 47.93353820823202,
    "Emphasis on Speed 20:60:20": 61.790302268464984,
    "Emphasis on Cost 20:20:60": 52.3786022999605,
    "Intelligence only": 41.516736430709585
   },
   "rank": 10,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 10,
    "Balanced 33:33:33 (no calibration)": 9,
    "Emphasis on Accuracy 60:20:20": 9,
    "Emphasis on Speed 20:60:20": 8,
    "Emphasis on Cost 20:20:60": 8,
    "Intelligence only": 12
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8398268398268398,
   "sealed_accuracy": 0.2987012987012987,
   "public_minus_sealed_gap_pp": 54.112554112554115,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.4878,
     "long_policy": 0.225,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.4286,
     "probability": 0.25,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.1651,
     "1/3": 0.2577,
     "2/3": 0.4804
    },
    "ece": 0.41862236643597567,
    "mean_tvd_gold_probs": 0.45798486852351306,
    "calibration": 35.23851993022678,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5162,
      "accuracy": 0.3459
     },
     "conf>=0.9": {
      "coverage": 0.3247,
      "accuracy": 0.32
     }
    }
   },
   "v130_comparison_rank": 6,
   "v130_comparison_score": 73.02927314568292,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jev-omni",
   "display": "Jev-Omni (akhilaaa3, Gemma-4-12B merged)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "akhilaaa3",
   "repo": "https://huggingface.co/akhilaaa3/Jev-Omni",
   "licence": "Apache-2.0, following Gemma 4; dataset rights stated separately by the author",
   "underlying": "google/gemma-4-12B-it fine-tuned and merged, with a trained 256-way decision head",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (L40 48 GB, Czechia), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.9246575342465754,
    "hard": 0.75
   },
   "axes": {
    "intelligence": 46.7523835806221,
    "calibration": 64.08815671004801,
    "speed": 81.54478729126231,
    "cost": 53.01727443048121
   },
   "jevbench_score": 51.34132749148857,
   "speed": {
    "p50_s_raw": 0.21708130463957787,
    "p95_s_raw": 0.5247324109077453,
    "p50_s_adjusted": 0.5841626092791558,
    "p95_s_adjusted": 1.1994648218154904,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.2663530632853508,
    "hard_tier_p95_s": 1.114325867965817
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.03682059925093633,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Gemma 3 12B input rate list price $0.05/M in, $0.0/M out (a 12B one-pass model with no generated output; the same reference the author uses in his own model card for this model) x 384 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.01918184713375796,
    "usd_per_1000_hard": 0.061995909090909095
   },
   "calibration": {
    "score": 64.08815671004801,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.10370494980703701,
    "probability_fidelity": 66.72734014629386,
    "brier_hard": 0.3437970517377786,
    "brier_standard_judge_v11": 0.08021950364110036
   },
   "hard": {
    "run": "runs/jev-omni--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.75,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 13,
      "n": 14,
      "accuracy": 0.9285714285714286
     },
     "judge_hard": {
      "correct": 29,
      "n": 33,
      "accuracy": 0.8787878787878788
     },
     "long_policy": {
      "correct": 22,
      "n": 38,
      "accuracy": 0.5789473684210527
     },
     "multi_hop": {
      "correct": 29,
      "n": 35,
      "accuracy": 0.8285714285714286
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3437970517377786,
    "ece": 0.10370494980703701,
    "probability_fidelity": 66.72734014629386,
    "calibration_score": 72.99317509244322,
    "onehot": {
     "ece": 0.25,
     "probability_fidelity": 50.017,
     "calibration_score": 50.0085
    },
    "latency_p50_s": 0.2663530632853508,
    "latency_p95_s": 1.114325867965817,
    "mean_input_tokens": 1239.918181818182,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "footnote": "Requested by Florian on X on 23 Sep with the words 'let's include it in v1.4', so this row is measured now and belongs to that round. A merged google/gemma-4-12B-it with a trained 256-way decision head; the author's own load_model.py and predict() run unchanged, one forward pass, nothing generated. Two staging differences, forced by our 46 GB GPU and fixed before the run: the model is built on the CPU and moved to CUDA afterwards, and every nn.Linear weight is cast to bfloat16 first - the exact cast his own reference loader applies, and consistent with his 'inference uses BF16 autocast'. JevBench is text-only, so the image, audio and video paths of this model are never exercised. Self-host latency gets the standard x2 + 0.15 s adjustment.",
   "note": "akhilaaa3/Jev-Omni revision c050d51354147985d13286cf4acf90f562f2c631, the author's own load_model.py (merged text decision model + 256-way head) and his own predict(), transformers 5.17.0 / torch 2.8.0 from the pod image; built on the CPU and moved to CUDA with every nn.Linear weight cast to bfloat16 first - the same cast his reference loader jev_omni.py applies - because our 46 GB GPU cannot hold his fp32 copy; on our RunPod L40 in Czechia",
   "source": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "scorer": "jevbench composite_v13.py, unchanged v1.3.0 copy",
   "scorer_sha256": "66562dffb9d6f70bcafe5a0d7a823e0f48195aa6cdfb12dfc7d32f064b433d33",
   "dataset": "frozen JevBench v1.2, exact 534-task set (72 easy, 242 standard+judge, 220 hard)",
   "presets": {
    "JevBench Score (25:25:25:25)": 51.34132749148857,
    "Balanced 33:33:33 (no calibration)": 49.9472416475538,
    "Emphasis on Accuracy 60:20:20": 45.875128726093514,
    "Emphasis on Speed 20:60:20": 56.74368286231025,
    "Emphasis on Cost 20:20:60": 48.44499828504404,
    "Intelligence only": 40.876270426043206
   },
   "source_round": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8874458874458875,
   "sealed_accuracy": 0.32142857142857145,
   "public_minus_sealed_gap_pp": 56.60173160173161,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3514,
     "judge_hard": 0.439,
     "long_policy": 0.175,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3571,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1284,
     "1/3": 0.299,
     "2/3": 0.549
    },
    "ece": 0.342834684997797,
    "mean_tvd_gold_probs": 0.3887682310992541,
    "calibration": 46.278119945257586,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4838,
      "accuracy": 0.3758
     },
     "conf>=0.9": {
      "coverage": 0.1786,
      "accuracy": 0.4909
     }
    }
   },
   "v130_comparison_rank": 9,
   "v130_comparison_score": 71.84658713390714,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
   "not_ranked_because": null,
   "rank": 11,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 11,
    "Balanced 33:33:33 (no calibration)": 12,
    "Emphasis on Accuracy 60:20:20": 12,
    "Emphasis on Speed 20:60:20": 11,
    "Emphasis on Cost 20:20:60": 12,
    "Intelligence only": 13
   }
  },
  {
   "key": "metask-jev-4b",
   "display": "metask-jev-4b",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Wayfind (metask-ai)",
   "repo": "https://github.com/metask-ai/metask-jev",
   "licence": "Apache-2.0",
   "underlying": "Qwen3.5-4B merged r16 LoRA, candidate-logit readout",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX 5090 32 GB); serial, in-process candidate-logit inference",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.8972602739726028,
    "hard": 0.5909090909090909
   },
   "axes": {
    "intelligence": 44.692072447140184,
    "calibration": 66.89723664642602,
    "speed": 89.10064969336227,
    "cost": 54.54015986693759
   },
   "jevbench_score": 47.78280701270882,
   "speed": {
    "p50_s_raw": 0.06571843556594104,
    "p95_s_raw": 0.1435365290963091,
    "p50_s_adjusted": 0.2814368711318821,
    "p95_s_adjusted": 0.4370730581926182,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.08556740445783362,
    "hard_tier_p95_s": 0.27962883535074057
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.032758801498127314,
    "basis": "ESTIMATE: hosted 4B reference rate USD 0.04/M input, USD 0/M output over the exact measured prompt-token counts of all 534 attempts; no generated answer tokens",
    "usd_per_1000_v11_tiers": 0.01964968152866242,
    "usd_per_1000_hard": 0.051469090909090895,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 66.89723664642602,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.1477715282049797,
    "probability_fidelity": 76.89411211538592,
    "brier_hard": 0.547023251993916,
    "brier_standard_judge_v11": 0.11398239572572733,
    "note": null
   },
   "hard": {
    "run": "runs/metask-jev-4b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 218,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.990909090909091,
    "accuracy": 0.5909090909090909,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 25,
      "n": 33,
      "accuracy": 0.7575757575757576
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 24,
      "n": 35,
      "accuracy": 0.6857142857142857
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.547023251993916,
    "ece": 0.1477715282049797,
    "probability_fidelity": 76.89411211538592,
    "calibration_score": 73.66990323719499,
    "onehot": {
     "ece": 0.4036697247706422,
     "probability_fidelity": 52.81,
     "calibration_score": 36.03802752293578
    },
    "latency_p50_s": 0.08556740445783362,
    "latency_p95_s": 0.27962883535074057,
    "mean_input_tokens": 1298.532110091743,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.011323199999999997
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 47.78280701270882,
    "Balanced 33:33:33 (no calibration)": 46.15225188618938,
    "Emphasis on Accuracy 60:20:20": 41.31756238832786,
    "Emphasis on Speed 20:60:20": 53.70731516057139,
    "Emphasis on Cost 20:20:60": 45.08561251809298,
    "Intelligence only": 35.70684461395281
   },
   "rank": 12,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 12,
    "Balanced 33:33:33 (no calibration)": 15,
    "Emphasis on Accuracy 60:20:20": 14,
    "Emphasis on Speed 20:60:20": 13,
    "Emphasis on Cost 20:20:60": 17,
    "Intelligence only": 18
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7965367965367965,
   "sealed_accuracy": 0.275974025974026,
   "public_minus_sealed_gap_pp": 52.056277056277054,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.3902,
     "long_policy": 0.275,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.1429,
     "probability": 0.2857,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1193,
     "1/3": 0.299,
     "2/3": 0.4216
    },
    "ece": 0.2983131309817638,
    "mean_tvd_gold_probs": 0.33633566873871057,
    "calibration": 53.35190346488809,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3604,
      "accuracy": 0.2703
     },
     "conf>=0.9": {
      "coverage": 0.0487,
      "accuracy": 0.2667
     }
    }
   },
   "v130_comparison_rank": 7,
   "v130_comparison_score": 72.36190467891471,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "semif-qwen3.5-4b",
   "display": "SemIf, formerly OpenJev (Qwen3.5-4B, TheoLeeCJ)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Theodore Lee (TheoLeeCJ)",
   "repo": "https://github.com/TheoLeeCJ/openjev",
   "licence": "MIT (code); Qwen3.5 weights Apache-2.0",
   "underlying": "Qwen/Qwen3.5-4B (frozen, BF16)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.952054794520548,
    "hard": 0.5954545454545455
   },
   "axes": {
    "intelligence": 44.436255677242116,
    "calibration": 66.76099782801955,
    "speed": 83.70422262345133,
    "cost": 59.46663998783557
   },
   "jevbench_score": 47.69091838825422,
   "speed": {
    "p50_s_raw": 0.19796114787459373,
    "p95_s_raw": 0.3153164997696876,
    "p50_s_adjusted": 0.5459222957491875,
    "p95_s_adjusted": 0.7806329995393753,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
    "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing",
    "hard_tier_p50_s": 0.22315140068531036,
    "hard_tier_p95_s": 0.6500909611582756
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02244460674157303,
    "basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (same weights (not on OpenRouter), as open-alternative-jev in v1.1.2) x 396 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1244 in / 0 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.012019777070063693,
    "usd_per_1000_hard": 0.03732368181818181,
    "self_host_sensitivity": {
     "usd_per_1000": 0.0347231924417574,
     "score": 61.4845088183062,
     "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)",
     "usd_per_h": 0.72,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.2083391546505444,
     "decisions_per_hour": 20735.42060418796
    }
   },
   "calibration": {
    "score": 66.76099782801955,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.12080109007622307,
    "probability_fidelity": 69.36301275995041,
    "brier_hard": 0.5419578429506192,
    "brier_standard_judge_v11": 0.06934070252416917,
    "note": null
   },
   "hard": {
    "run": "runs-gpu/semif-qwen3.5-4b--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5954545454545455,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714285714285714
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 16,
      "n": 38,
      "accuracy": 0.42105263157894735
     },
     "multi_hop": {
      "correct": 22,
      "n": 35,
      "accuracy": 0.6285714285714286
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5419578429506192,
    "ece": 0.12080109007622307,
    "probability_fidelity": 69.36301275995041,
    "calibration_score": 72.60139737235289,
    "onehot": {
     "ece": 0.40454545454545454,
     "probability_fidelity": 42.757,
     "calibration_score": 30.923954545454546
    },
    "latency_p50_s": 0.22315140068531036,
    "latency_p95_s": 0.6500909611582756,
    "mean_input_tokens": 1244.1227272727272,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 47.69091838825422,
    "Balanced 33:33:33 (no calibration)": 46.21864280395295,
    "Emphasis on Accuracy 60:20:20": 41.019417978610235,
    "Emphasis on Speed 20:60:20": 52.54284843987061,
    "Emphasis on Cost 20:20:60": 46.51576256989155,
    "Intelligence only": 35.09719124451025
   },
   "rank": 13,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 13,
    "Balanced 33:33:33 (no calibration)": 14,
    "Emphasis on Accuracy 60:20:20": 15,
    "Emphasis on Speed 20:60:20": 15,
    "Emphasis on Cost 20:20:60": 15,
    "Intelligence only": 19
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8095238095238095,
   "sealed_accuracy": 0.262987012987013,
   "public_minus_sealed_gap_pp": 54.65367965367965,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.4146,
     "long_policy": 0.15,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2143,
     "safety_judge": 0.375,
     "temporal_numeric": 0.25,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1101,
     "1/3": 0.3299,
     "2/3": 0.3627
    },
    "ece": 0.2930190833249826,
    "mean_tvd_gold_probs": 0.31235785856297704,
    "calibration": 55.08019873935288,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2727,
      "accuracy": 0.2857
     },
     "conf>=0.9": {
      "coverage": 0.1071,
      "accuracy": 0.1515
     }
    }
   },
   "v130_comparison_rank": 5,
   "v130_comparison_score": 73.08747495585776,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jobe-qwen3.5-4b",
   "display": "Jobe Qwen3.5-4B (frozen)",
   "class": "native-logit",
   "open": "yes",
   "author": "MantisShrimpdev",
   "repo": "https://github.com/MantisShrimpdev/jobe",
   "licence": "MIT code; Apache-2.0 weights",
   "underlying": "Qwen/Qwen3.5-4B frozen BF16 backbone",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX A6000 48 GB), in-process, serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9657534246575342,
    "hard": 0.5863636363636363
   },
   "axes": {
    "intelligence": 44.09907806068809,
    "calibration": 66.09860630363153,
    "speed": 85.5887262805767,
    "cost": 59.517941238875515
   },
   "jevbench_score": 46.93829466139942,
   "speed": {
    "p50_s_raw": 0.13457110384479165,
    "p95_s_raw": 0.25440939371474086,
    "p50_s_adjusted": 0.4191422076895833,
    "p95_s_adjusted": 0.6588187874294817,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.13874096516519785,
    "hard_tier_p95_s": 0.5290757374372332
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.022356404494382018,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3.5-4B hosted reference list price $0.03/M in, $0.0/M out (same underlying weights; one forward pass, no generated tokens) x 396 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.011869777070063692,
    "usd_per_1000_hard": 0.03732368181818181,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 66.09860630363153,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.1273374148796787,
    "probability_fidelity": 69.32947671915909,
    "brier_hard": 0.5389483114063452,
    "brier_standard_judge_v11": 0.06904116420024838,
    "note": null
   },
   "hard": {
    "run": "runs/jobe-qwen3.5-4b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5863636363636363,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714285714285714
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 22,
      "n": 35,
      "accuracy": 0.6285714285714286
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 5,
      "n": 30,
      "accuracy": 0.16666666666666666
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5389483114063452,
    "ece": 0.1273374148796787,
    "probability_fidelity": 69.32947671915909,
    "calibration_score": 71.93099687161168,
    "onehot": {
     "ece": 0.4136363636363637,
     "probability_fidelity": 42.757,
     "calibration_score": 30.01486363636363
    },
    "latency_p50_s": 0.13874096516519785,
    "latency_p95_s": 0.5290757374372332,
    "mean_input_tokens": 1244.1227272727272,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.008211209999999998
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 46.93829466139942,
    "Balanced 33:33:33 (no calibration)": 45.61374503032864,
    "Emphasis on Accuracy 60:20:20": 40.299381432488985,
    "Emphasis on Speed 20:60:20": 52.187018046217815,
    "Emphasis on Cost 20:20:60": 45.88520157659598,
    "Intelligence only": 34.30429684882838
   },
   "rank": 14,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 14,
    "Balanced 33:33:33 (no calibration)": 17,
    "Emphasis on Accuracy 60:20:20": 17,
    "Emphasis on Speed 20:60:20": 16,
    "Emphasis on Cost 20:20:60": 16,
    "Intelligence only": 24
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8095238095238095,
   "sealed_accuracy": 0.2564935064935065,
   "public_minus_sealed_gap_pp": 55.3030303030303,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.3659,
     "long_policy": 0.175,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2143,
     "safety_judge": 0.375,
     "temporal_numeric": 0.25,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1101,
     "1/3": 0.3196,
     "2/3": 0.3529
    },
    "ece": 0.2992288903669122,
    "mean_tvd_gold_probs": 0.312865715912751,
    "calibration": 54.43382516767123,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.276,
      "accuracy": 0.2824
     },
     "conf>=0.9": {
      "coverage": 0.1104,
      "accuracy": 0.1471
     }
    }
   },
   "v130_comparison_rank": 4,
   "v130_comparison_score": 73.37135548467025,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "localjev-qwen3.5-4b",
   "display": "local-jev Qwen3.5-4B",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Amith Chandrappa (amithgc)",
   "repo": "https://github.com/amithgc/local-jev",
   "licence": "MIT code; Apache-2.0 Qwen weights",
   "underlying": "Qwen/Qwen3.5-4B text model, zero-shot next-token letter probabilities",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), reached over the internet from Germany; serial, one request at a time",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.958904109589041,
    "hard": 0.6045454545454545
   },
   "axes": {
    "intelligence": 44.40785125416772,
    "calibration": 73.29981818181818,
    "speed": 74.96494013517304,
    "cost": 55.797342929598116
   },
   "jevbench_score": 46.79864791106764,
   "speed": {
    "p50_s_raw": 0.7075801230967045,
    "p95_s_raw": 0.9433971662074327,
    "p50_s_adjusted": 1.5651602461934089,
    "p95_s_adjusted": 2.0367943324148654,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.7749514579772949,
    "hard_tier_p95_s": 1.2899098664522168
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02974554307116105,
    "basis": "ESTIMATE: hosted-provider price, hosted 4B reference rate list price $0.04/M in, $0.0/M out (one forward pass over measured input tokens and no generated answer tokens) x 397 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.015883694267515923,
    "usd_per_1000_hard": 0.049530181818181813,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 73.29981818181818,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.08116227272727276,
    "probability_fidelity": 69.841,
    "brier_hard": 0.5113870788181817,
    "brier_standard_judge_v11": 0.11271555078512399,
    "note": null
   },
   "hard": {
    "run": "runs/localjev-qwen3.5-4b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6045454545454545,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714285714285714
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 18,
      "n": 38,
      "accuracy": 0.47368421052631576
     },
     "multi_hop": {
      "correct": 23,
      "n": 35,
      "accuracy": 0.6571428571428571
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 4,
      "n": 30,
      "accuracy": 0.13333333333333333
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5113870788181817,
    "ece": 0.08116227272727276,
    "probability_fidelity": 69.841,
    "calibration_score": 76.80427272727272,
    "onehot": {
     "ece": 0.3954545454545455,
     "probability_fidelity": 42.757,
     "calibration_score": 31.83304545454545
    },
    "latency_p50_s": 0.7749514579772949,
    "latency_p95_s": 1.2899098664522168,
    "mean_input_tokens": 1238.2545454545455,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.010896640000000008
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 46.79864791106764,
    "Balanced 33:33:33 (no calibration)": 44.002675001273296,
    "Emphasis on Accuracy 60:20:20": 39.9132424136161,
    "Emphasis on Speed 20:60:20": 49.02002679635665,
    "Emphasis on Cost 20:20:60": 44.007293045360136,
    "Intelligence only": 35.02993006258887
   },
   "rank": 15,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 15,
    "Balanced 33:33:33 (no calibration)": 19,
    "Emphasis on Accuracy 60:20:20": 19,
    "Emphasis on Speed 20:60:20": 20,
    "Emphasis on Cost 20:20:60": 19,
    "Intelligence only": 20
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8051948051948052,
   "sealed_accuracy": 0.2597402597402597,
   "public_minus_sealed_gap_pp": 54.545454545454554,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.4146,
     "long_policy": 0.1,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2857,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1284,
     "1/3": 0.2887,
     "2/3": 0.3725
    },
    "ece": 0.21122499999999994,
    "mean_tvd_gold_probs": 0.2517318181818182,
    "calibration": 66.2909090909091,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1396,
      "accuracy": 0.186
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 12,
   "v130_comparison_score": 70.92991890404122,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "system-one-open",
   "display": "system-one-open (Gemma 4 E2B LoRA on an L4)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "mithalouni",
   "repo": "https://github.com/mithalouni/system-one-open",
   "licence": "MIT (repository LICENSE; Gemma weights keep Google’s terms)",
   "underlying": "google/gemma-4-E2B-it + LoRA",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "author's public demo endpoint (Modal, L4) — not a production service",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9375,
    "judge": 0.8767123287671232,
    "hard": 0.4909090909090909
   },
   "axes": {
    "intelligence": 44.19587425502762,
    "calibration": 54.867197380012975,
    "speed": 76.96012500732209,
    "cost": 64.82109224519658
   },
   "jevbench_score": 45.11471760046746,
   "speed": {
    "p50_s_raw": 0.6517308317124844,
    "p95_s_raw": 0.7724301926791667,
    "p50_s_adjusted": 1.3034616634249687,
    "p95_s_adjusted": 1.5448603853583334,
    "adjustment": "x2 (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.6776973158121109,
    "hard_tier_p95_s": 1.1177304897457356
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.014880936329588014,
    "basis": "ESTIMATE: hosted-provider price, deepinfra google/gemma-4-E4B-it list price $0.02/M in, $0.1/M out (Gemma 4 E2B is not listed; the nearest larger sibling, Gemma 4 E4B, is listed only on DeepInfra) x 383 input and 2 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra google/gemma-4-E4B-it $0.02/M in, $0.1/M out x 1235 in / 2 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.00786815286624204,
    "usd_per_1000_hard": 0.024890090909090907,
    "self_host_sensitivity": {
     "usd_per_1000": 0.06593618349183544,
     "score": 54.521904855816615,
     "machine": "1x L4 24 GB (Gemma 4 E2B + LoRA)",
     "usd_per_h": 0.43,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.6624286341505328,
     "decisions_per_hour": 6521.45722163681
    }
   },
   "calibration": {
    "score": 54.867197380012975,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.25707820493262257,
    "probability_fidelity": 64.81503340934218,
    "brier_hard": 0.7469110891631627,
    "brier_standard_judge_v11": 0.13816078684373112,
    "note": null
   },
   "hard": {
    "run": "runs/system-one-open--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4909090909090909,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 25,
      "n": 33,
      "accuracy": 0.7575757575757576
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 15,
      "n": 35,
      "accuracy": 0.42857142857142855
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 3,
      "n": 12,
      "accuracy": 0.25
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7469110891631627,
    "ece": 0.25707820493262257,
    "probability_fidelity": 64.81503340934218,
    "calibration_score": 56.699696211408835,
    "onehot": {
     "ece": 0.509090909090909,
     "probability_fidelity": 44.011500000000005,
     "calibration_score": 22.005750000000003
    },
    "latency_p50_s": 0.6776973158121109,
    "latency_p95_s": 1.1177304897457356,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 45.11471760046746,
    "Balanced 33:33:33 (no calibration)": 45.91677424352335,
    "Emphasis on Accuracy 60:20:20": 40.5662721327779,
    "Emphasis on Speed 20:60:20": 50.71147194597174,
    "Emphasis on Cost 20:20:60": 47.6981452173596,
    "Intelligence only": 34.53068383831726
   },
   "rank": 16,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 16,
    "Balanced 33:33:33 (no calibration)": 16,
    "Emphasis on Accuracy 60:20:20": 16,
    "Emphasis on Speed 20:60:20": 18,
    "Emphasis on Cost 20:20:60": 13,
    "Intelligence only": 21
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.7316017316017316,
   "sealed_accuracy": 0.275974025974026,
   "public_minus_sealed_gap_pp": 45.56277056277056,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.2439,
     "long_policy": 0.275,
     "multi_hop": 0.1316,
     "paraphrase_robustness": 0.3571,
     "probability": 0.25,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.4615,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1927,
     "1/3": 0.2784,
     "2/3": 0.3627
    },
    "ece": 0.3141180161234028,
    "mean_tvd_gold_probs": 0.3477199734087694,
    "calibration": 51.20219971722125,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3312,
      "accuracy": 0.2843
     },
     "conf>=0.9": {
      "coverage": 0.1299,
      "accuracy": 0.2
     }
    }
   },
   "v130_comparison_rank": 26,
   "v130_comparison_score": 66.59745356122706,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (author's Modal demo), as for every API measurement"
  },
  {
   "key": "spark-s1-4b-v6",
   "display": "spark-s1-4b-v6 (Open Spark Jev, abhishek085)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Abhishek Rai (abhishek085)",
   "repo": "https://github.com/abhishek085/open-spark-jev",
   "licence": "Apache-2.0 (code and weights); base Qwen/Qwen3.5-4B Apache-2.0",
   "underlying": "Qwen/Qwen3.5-4B with a LoRA, read at the option-letter logits with a fitted temperature (no extra head)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (L40 48 GB, Czechia), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9178082191780822,
    "hard": 0.6
   },
   "axes": {
    "intelligence": 45.069442435695464,
    "calibration": 47.576491816220006,
    "speed": 80.96905763279898,
    "cost": 57.88923950418274
   },
   "jevbench_score": 44.623631936050636,
   "speed": {
    "p50_s_raw": 0.31425347179174423,
    "p95_s_raw": 0.4388090513646602,
    "p50_s_adjusted": 0.7785069435834885,
    "p95_s_adjusted": 1.0276181027293203,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.31987401843070984,
    "hard_tier_p95_s": 0.8502757232636212
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.025333314606741573,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B size-class reference list price $0.03/M in, $0.0/M out (a 2-4B one-pass model with no generated output; the board's 4B open-weights reference, as for reflex 4B and decider-2b) x 505 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.015161560509554141,
    "usd_per_1000_hard": 0.03985118181818181
   },
   "calibration": {
    "score": 47.576491816220006,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.26160901895804284,
    "probability_fidelity": 60.53319573049987,
    "brier_hard": 0.6177864353252894,
    "brier_standard_judge_v11": 0.08994350389919674
   },
   "hard": {
    "run": "runs/spark-s1-4b-v6--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 5,
      "n": 30,
      "accuracy": 0.16666666666666666
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6177864353252894,
    "ece": 0.26160901895804284,
    "probability_fidelity": 60.53319573049987,
    "calibration_score": 54.10569596944565,
    "onehot": {
     "ece": 0.4,
     "probability_fidelity": 51.14950000000002,
     "calibration_score": 35.57475000000001
    },
    "latency_p50_s": 0.31987401843070984,
    "latency_p95_s": 0.8502757232636212,
    "mean_input_tokens": 1328.3727272727272,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "footnote": "Requested twice by Florian on X (22 and 23 Sep). A Qwen3.5-4B with a LoRA, read at the option-letter logits and divided by a temperature fitted on the author's own held-out split - no extra head, nothing generated. Run through his own MenuScorer.decide, one state and one question per request; JevBench's noul/choice/score map one-to-one onto his Noul/Choice/Score, with the per-option descriptions in the prompt because his schema carries option strings only. Speed caveat in his own words: HF Transformers in-process serving is slow for this hybrid backbone without causal_conv1d and flash-linear-attention, and vLLM is his recommended path (his figure: 74.9 ms p50 on a DGX Spark). We measured the Transformers path his quickstart documents, with flash-linear-attention 0.5.2 installed and causal_conv1d unavailable (no wheel builds here), so the Speed axis is the slower of his two paths. He also reports JevBench public-tier numbers in his model card and used the public items during development; his 'public-proxy Intelligence 83.1' is his own metric over the 231 public items, not an official score. Self-host latency gets the standard x2 + 0.15 s adjustment.",
   "note": "abhishek085/spark-s1-4b-v6 revision 93d49ddbfb29212e3296635a75a3e80cf69da027, code github.com/abhishek085/open-spark-jev 30ac6d89b7fa36c644cf86aac68f35c1d276a919, the author's own MenuScorer.decide with his fitted calibration.json temperature, bf16, base Qwen/Qwen3.5-4B, transformers 5.17.0 / torch 2.8.0 from the pod image, flash-linear-attention 0.5.2 installed, causal_conv1d not installable here (no wheel builds against this toolchain), on our RunPod L40 in Czechia",
   "source": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "scorer": "jevbench composite_v13.py, unchanged v1.3.0 copy",
   "scorer_sha256": "66562dffb9d6f70bcafe5a0d7a823e0f48195aa6cdfb12dfc7d32f064b433d33",
   "dataset": "frozen JevBench v1.2, exact 534-task set (72 easy, 242 standard+judge, 220 hard)",
   "presets": {
    "JevBench Score (25:25:25:25)": 44.623631936050636,
    "Balanced 33:33:33 (no calibration)": 47.0445153493564,
    "Emphasis on Accuracy 60:20:20": 42.234782025292894,
    "Emphasis on Speed 20:60:20": 53.0952984313406,
    "Emphasis on Cost 20:20:60": 47.040754588107404,
    "Intelligence only": 36.619005654288806
   },
   "source_round": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7922077922077922,
   "sealed_accuracy": 0.2662337662337662,
   "public_minus_sealed_gap_pp": 52.5974025974026,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.3415,
     "long_policy": 0.175,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.4286,
     "probability": 0.3929,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.0917,
     "1/3": 0.3711,
     "2/3": 0.3529
    },
    "ece": 0.4458747923066029,
    "mean_tvd_gold_probs": 0.4178887451914198,
    "calibration": 34.51808350976873,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5455,
      "accuracy": 0.2202
     },
     "conf>=0.9": {
      "coverage": 0.3182,
      "accuracy": 0.2245
     }
    }
   },
   "v130_comparison_rank": 25,
   "v130_comparison_score": 66.64989846517598,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
   "not_ranked_because": null,
   "rank": 17,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 17,
    "Balanced 33:33:33 (no calibration)": 13,
    "Emphasis on Accuracy 60:20:20": 13,
    "Emphasis on Speed 20:60:20": 14,
    "Emphasis on Cost 20:20:60": 14,
    "Intelligence only": 16
   }
  },
  {
   "key": "malkuth-4b",
   "display": "Malkuth-4B (newfull5, Kev post-train)",
   "author": "newfull5 (dhtocks)",
   "repo": "https://github.com/newfull5/malkuth",
   "class": "jev-rebuild",
   "licence": "CC-BY-NC-4.0, research use only (XNLI and RACE in the training mix)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.910958904109589,
    "hard": 0.5272727272727272
   },
   "speed": {
    "p50_s_raw": 0.08862553583458066,
    "p95_s_raw": 0.14933330831117927,
    "p50_s_adjusted": 0.32725107166916134,
    "p95_s_adjusted": 0.4486666166223585,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.019154662921348316,
    "basis": "ESTIMATE: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B one-pass size class, as kev-4b), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 61.4864246031746,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "d52c6ff2d9cc9fb6c40104a868643a602f44efe027b41ebbd27ce73fb4fb265f",
    "row_sha256": "981a6763827626f3e9559d41351c75843eb11b24dc33a9513b1fd95530706b77",
    "raw_results_sha256": "4327bc1cdb685622a0dc7cf4652c8440c0f1d90cc36bd48429efab354820ff75",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7489177489177489,
   "sealed_accuracy": 0.23376623376623376,
   "public_minus_sealed_gap_pp": 51.515151515151516,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.2927,
     "long_policy": 0.175,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.4286,
     "probability": 0.2857,
     "safety_judge": 0.375,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.0826,
     "1/3": 0.2474,
     "2/3": 0.3824
    },
    "ece": 0.31640551948051954,
    "mean_tvd_gold_probs": 0.2681075757575757,
    "calibration": 54.95406926406926,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3084,
      "accuracy": 0.2316
     },
     "conf>=0.9": {
      "coverage": 0.1558,
      "accuracy": 0.2292
     }
    }
   },
   "v130_comparison_rank": 8,
   "v130_comparison_score": 71.29176478631354,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (4B one-pass size class, as kev-4b), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 43.149009407368084,
    "calibration": 61.4864246031746,
    "speed": 88.33195165837034,
    "cost": 61.531764586365185
   },
   "jevbench_score": 44.453846652872855,
   "presets": {
    "JevBench Score (25:25:25:25)": 44.453846652872855,
    "Balanced 33:33:33 (no calibration)": 44.02529080017372,
    "Emphasis on Accuracy 60:20:20": 38.34916927299591,
    "Emphasis on Speed 20:60:20": 50.73811995049493,
    "Emphasis on Cost 20:20:60": 44.72788190473142,
    "Intelligence only": 32.13456911275833
   },
   "rank": 18,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 18,
    "Balanced 33:33:33 (no calibration)": 18,
    "Emphasis on Accuracy 60:20:20": 22,
    "Emphasis on Speed 20:60:20": 17,
    "Emphasis on Cost 20:20:60": 18,
    "Intelligence only": 26
   }
  },
  {
   "key": "jqv",
   "display": "jqv (Qwen3-32B zero-shot)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "hjmurmur (Octalab)",
   "repo": "https://github.com/Octalab-Inc/jqv",
   "licence": "Apache-2.0 (Qwen3-32B weights); serving code public",
   "underlying": "Qwen/Qwen3-32B, bf16, read as a direct-logit classifier (no fine-tuning)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.9246575342465754,
    "hard": 0.6454545454545455
   },
   "axes": {
    "intelligence": 46.402753266614425,
    "calibration": 71.64600373856102,
    "speed": 74.62291018872223,
    "cost": 47.458062754792564
   },
   "jevbench_score": 44.35209587211248,
   "speed": {
    "p50_s_raw": 0.7473905384540558,
    "p95_s_raw": 0.9735059145838022,
    "p50_s_adjusted": 1.6447810769081115,
    "p95_s_adjusted": 2.0970118291676045,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.8068596571683884,
    "hard_tier_p95_s": 1.5256738737225528
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.05641543071161049,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-32b list price $0.08/M in, $0.0/M out (the exact base model this system reads logits from; nothing is generated) x 359 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.02870751592356688,
    "usd_per_1000_hard": 0.09596218181818182,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 71.64600373856102,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.08756655067713424,
    "probability_fidelity": 75.52526065485925,
    "brier_hard": 0.47242119377157676,
    "brier_standard_judge_v11": 0.13845653894523616,
    "note": null
   },
   "hard": {
    "run": "runs/jqv--hard-r4 (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6454545454545455,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6428571428571429
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 19,
      "n": 38,
      "accuracy": 0.5
     },
     "multi_hop": {
      "correct": 24,
      "n": 35,
      "accuracy": 0.6857142857142857
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.47242119377157676,
    "ece": 0.08756655067713424,
    "probability_fidelity": 75.52526065485925,
    "calibration_score": 79.0059752597162,
    "onehot": {
     "ece": 0.3545454545454545,
     "probability_fidelity": 51.3725,
     "calibration_score": 40.23170454545455
    },
    "latency_p50_s": 0.8068596571683884,
    "latency_p95_s": 1.5256738737225528,
    "mean_input_tokens": 1199.5272727272727,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 44.35209587211248,
    "Balanced 33:33:33 (no calibration)": 41.55153868274715,
    "Emphasis on Accuracy 60:20:20": 39.140090700977325,
    "Emphasis on Speed 20:60:20": 46.84273963027925,
    "Emphasis on Cost 20:20:60": 39.522230132931526,
    "Intelligence only": 36.005698839078825
   },
   "rank": 19,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 19,
    "Balanced 33:33:33 (no calibration)": 22,
    "Emphasis on Accuracy 60:20:20": 20,
    "Emphasis on Speed 20:60:20": 22,
    "Emphasis on Cost 20:20:60": 24,
    "Intelligence only": 17
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8008658008658008,
   "sealed_accuracy": 0.2824675324675325,
   "public_minus_sealed_gap_pp": 51.83982683982684,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.3415,
     "long_policy": 0.125,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.3571,
     "probability": 0.1786,
     "safety_judge": 0.5625,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1927,
     "1/3": 0.2784,
     "2/3": 0.3824
    },
    "ece": 0.2640164549910016,
    "mean_tvd_gold_probs": 0.33344587609298304,
    "calibration": 56.92606069625069,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2597,
      "accuracy": 0.2625
     },
     "conf>=0.9": {
      "coverage": 0.0325,
      "accuracy": 0.4
     }
    }
   },
   "v130_comparison_rank": 19,
   "v130_comparison_score": 68.62862396499413,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "qwen3-reranker-4b",
   "display": "Qwen3-Reranker-4B",
   "class": "reranker",
   "open": "yes",
   "author": "Qwen",
   "repo": "https://huggingface.co/Qwen/Qwen3-Reranker-4B",
   "licence": "Apache-2.0",
   "underlying": "4B instruction-aware generative reranker",
   "has_distribution": true,
   "probability_source": [
    "public_calibrated_reranker_softmax"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7916666666666666,
    "judge": 0.8767123287671232,
    "hard": 0.5
   },
   "axes": {
    "intelligence": 44.621963891405095,
    "calibration": 65.21858819057935,
    "speed": 78.72320187600475,
    "cost": 49.152310354989595
   },
   "jevbench_score": 43.489652534518434,
   "speed": {
    "p50_s_raw": 0.1302479188889265,
    "p95_s_raw": 1.5593349148519338,
    "p50_s_adjusted": 0.41049583777785303,
    "p95_s_adjusted": 3.2686698297038674,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.2625806527212262,
    "hard_tier_p95_s": 1.8920937991235407
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.04953623422454652,
    "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
    "usd_per_1000_v11_tiers": 0.03307339728309628,
    "usd_per_1000_hard": 0.07303319240461638,
    "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21."
   },
   "calibration": {
    "score": 65.21858819057935,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.19149945567314672,
    "probability_fidelity": 72.37803790114894,
    "brier_hard": 0.6573086779280183,
    "brier_standard_judge_v11": 0.23321289738391712,
    "note": null
   },
   "hard": {
    "run": "runs/qwen3-reranker-4b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 16,
      "n": 38,
      "accuracy": 0.42105263157894735
     },
     "multi_hop": {
      "correct": 18,
      "n": 35,
      "accuracy": 0.5142857142857142
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 10,
      "n": 16,
      "accuracy": 0.625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6573086779280183,
    "ece": 0.19149945567314672,
    "probability_fidelity": 72.37803790114894,
    "calibration_score": 67.03907338325979,
    "onehot": {
     "ece": 0.5,
     "probability_fidelity": 49.763999999999996,
     "calibration_score": 24.881999999999998
    },
    "latency_p50_s": 0.2625806527212262,
    "latency_p95_s": 1.8920937991235407,
    "mean_input_tokens": 4350.459090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 43.489652534518434,
    "Balanced 33:33:33 (no calibration)": 41.63524397286264,
    "Emphasis on Accuracy 60:20:20": 38.37644039240353,
    "Emphasis on Speed 20:60:20": 47.590695493833856,
    "Emphasis on Cost 20:20:60": 40.02533635515595,
    "Intelligence only": 34.34423889657737
   },
   "rank": 20,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 20,
    "Balanced 33:33:33 (no calibration)": 21,
    "Emphasis on Accuracy 60:20:20": 21,
    "Emphasis on Speed 20:60:20": 21,
    "Emphasis on Cost 20:20:60": 23,
    "Intelligence only": 23
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6796536796536796,
   "sealed_accuracy": 0.2987012987012987,
   "public_minus_sealed_gap_pp": 38.095238095238095,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.2927,
     "long_policy": 0.275,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2857,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.25,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.2268,
     "2/3": 0.4608
    },
    "ece": 0.2524012949436469,
    "mean_tvd_gold_probs": 0.2636450540083371,
    "calibration": 61.57761780521846,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2662,
      "accuracy": 0.378
     },
     "conf>=0.9": {
      "coverage": 0.1558,
      "accuracy": 0.3125
     }
    }
   },
   "v130_comparison_rank": 35,
   "v130_comparison_score": 63.82721228923758,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "decider-35b-a3b",
   "display": "decider-35b-a3b (Mapika)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Mapika",
   "repo": "https://huggingface.co/Mapika/decider-35b-a3b",
   "licence": "Apache-2.0",
   "underlying": "Qwen3.5-35B-A3B-Base with a trained decision readout, 34.7B parameters / 3B active",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.910958904109589,
    "hard": 0.6545454545454545
   },
   "axes": {
    "intelligence": 47.16904647304943,
    "calibration": 65.279582972583,
    "speed": 80.78478445085703,
    "cost": 45.30548488186928
   },
   "jevbench_score": 41.18326776449841,
   "speed": {
    "p50_s_raw": 0.29191894084215164,
    "p95_s_raw": 0.49371074326336223,
    "p50_s_adjusted": 0.7338378816843033,
    "p95_s_adjusted": 1.1374214865267245,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.3118191212415695,
    "hard_tier_p95_s": 0.6214927081018686
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0665503745318352,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the closest public hosted 35B-A3B direct-logit model; no output is generated) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.031202547770700636,
    "usd_per_1000_hard": 0.11700136363636363,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 65.279582972583,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.18867318181818157,
    "probability_fidelity": 80.742,
    "brier_hard": 0.4866369675454547,
    "brier_standard_judge_v11": 0.1055856861570248,
    "note": null
   },
   "hard": {
    "run": "runs/decider-35b-a3b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6545454545454545,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6428571428571429
     },
     "judge_hard": {
      "correct": 25,
      "n": 33,
      "accuracy": 0.7575757575757576
     },
     "long_policy": {
      "correct": 20,
      "n": 38,
      "accuracy": 0.5263157894736842
     },
     "multi_hop": {
      "correct": 26,
      "n": 35,
      "accuracy": 0.7428571428571429
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4866369675454547,
    "ece": 0.18867318181818157,
    "probability_fidelity": 80.742,
    "calibration_score": 71.50368181818185,
    "onehot": {
     "ece": 0.34545454545454546,
     "probability_fidelity": 52.62050000000001,
     "calibration_score": 41.76479545454546
    },
    "latency_p50_s": 0.3118191212415695,
    "latency_p95_s": 0.6214927081018686,
    "mean_input_tokens": 1170.0136363636364,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 41.18326776449841,
    "Balanced 33:33:33 (no calibration)": 39.3896136924476,
    "Emphasis on Accuracy 60:20:20": 37.2605590897094,
    "Emphasis on Speed 20:60:20": 45.43642480568188,
    "Emphasis on Cost 20:20:60": 36.60937791568228,
    "Intelligence only": 34.46615520388542
   },
   "rank": 21,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 21,
    "Balanced 33:33:33 (no calibration)": 26,
    "Emphasis on Accuracy 60:20:20": 24,
    "Emphasis on Speed 20:60:20": 25,
    "Emphasis on Cost 20:20:60": 27,
    "Intelligence only": 22
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8311688311688312,
   "sealed_accuracy": 0.31493506493506496,
   "public_minus_sealed_gap_pp": 51.62337662337663,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.2927,
     "long_policy": 0.25,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.4286,
     "probability": 0.2143,
     "safety_judge": 0.5,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.2202,
     "1/3": 0.2165,
     "2/3": 0.5098
    },
    "ece": 0.3186694805194805,
    "mean_tvd_gold_probs": 0.3060333333333333,
    "calibration": 52.83138528138528,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4058,
      "accuracy": 0.36
     },
     "conf>=0.9": {
      "coverage": 0.2305,
      "accuracy": 0.3239
     }
    }
   },
   "v130_comparison_rank": 23,
   "v130_comparison_score": 67.55405352545425,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "raw-qwen3-4b-instruct-2507",
   "display": "Raw Qwen3 4B Instruct 2507 direct logits",
   "class": "raw-logit-control",
   "open": "yes",
   "author": "Alibaba Qwen / neutral reproduction",
   "repo": "https://huggingface.co/Qwen/Qwen3-4B-Instruct-2507",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-4B-Instruct-2507 BF16",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.8645833333333334,
    "judge": 0.9383561643835616,
    "hard": 0.5181818181818182
   },
   "axes": {
    "intelligence": 46.39383596928202,
    "calibration": 29.102680770831846,
    "speed": 87.55255760839427,
    "cost": 59.683674963546565
   },
   "jevbench_score": 40.95282149140936,
   "speed": {
    "p50_s_raw": 0.08244980592280626,
    "p95_s_raw": 0.20396011807024478,
    "p50_s_adjusted": 0.31489961184561255,
    "p95_s_adjusted": 0.5579202361404896,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
    "hard_tier_p50_s": 0.10531842289492488,
    "hard_tier_p95_s": 0.4118952717166394
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.022073820224719102,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.022073820224719102,
    "usd_per_1000_hard": 0.022073820224719102,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 29.102680770831846,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.4516820019423525,
    "probability_fidelity": 54.55492230083715,
    "brier_hard": 0.9299748456250653,
    "brier_standard_judge_v11": 0.16925729523587418,
    "note": null
   },
   "hard": {
    "run": "raw-controls/runs/qwen3-4b-2507/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5181818181818182,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 22,
      "n": 33,
      "accuracy": 0.6666666666666666
     },
     "long_policy": {
      "correct": 14,
      "n": 38,
      "accuracy": 0.3684210526315789
     },
     "multi_hop": {
      "correct": 19,
      "n": 35,
      "accuracy": 0.5428571428571428
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.9299748456250653,
    "ece": 0.4516820019423525,
    "probability_fidelity": 54.55492230083715,
    "calibration_score": 32.10926095618333,
    "onehot": {
     "ece": 0.4818181818181818,
     "probability_fidelity": 50.44350000000001,
     "calibration_score": 27.039931818181824
    },
    "latency_p50_s": 0.10531842289492488,
    "latency_p95_s": 0.4118952717166394,
    "mean_input_tokens": 1232.2227272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 40.95282149140936,
    "Balanced 33:33:33 (no calibration)": 51.93641986091741,
    "Emphasis on Accuracy 60:20:20": 46.36744994359026,
    "Emphasis on Speed 20:60:20": 59.31508923982133,
    "Emphasis on Cost 20:20:60": 51.71442594607999,
    "Intelligence only": 39.94301462159372
   },
   "rank": 22,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 22,
    "Balanced 33:33:33 (no calibration)": 10,
    "Emphasis on Accuracy 60:20:20": 11,
    "Emphasis on Speed 20:60:20": 10,
    "Emphasis on Cost 20:20:60": 11,
    "Intelligence only": 14
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.696969696969697,
   "sealed_accuracy": 0.2727272727272727,
   "public_minus_sealed_gap_pp": 42.42424242424243,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.2927,
     "long_policy": 0.175,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.3571,
     "probability": 0.3929,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2294,
     "1/3": 0.268,
     "2/3": 0.3235
    },
    "ece": 0.6285346412298987,
    "mean_tvd_gold_probs": 0.5382095919974224,
    "calibration": 23.089520400128883,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8539,
      "accuracy": 0.2662
     },
     "conf>=0.9": {
      "coverage": 0.6916,
      "accuracy": 0.2535
     }
    }
   },
   "v130_comparison_rank": 49,
   "v130_comparison_score": 58.58927286491753,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "opensourcejev-qwen35-4b-q4km",
   "display": "OpenSourceJev (Qwen3.5-4B Q4_K_M, native llama.cpp)",
   "author": "sabeel111",
   "repo": "https://github.com/sabeel111/OpenSourceJev",
   "class": "jev-rebuild",
   "licence": "MIT (repository code); Apache-2.0 (Qwen/Qwen3.5-4B base and unsloth/Qwen3.5-4B-GGUF Q4_K_M conversion)",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "endpoint_kind": "unknown",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "hard": 0.5636363636363636,
    "judge": 0.8561643835616438,
    "standard": 0.9270833333333334
   },
   "speed": {
    "p50_s_raw": null,
    "p95_s_raw": null,
    "p50_s_adjusted": null,
    "p95_s_adjusted": null,
    "adjustment": "See original v1.3 measurement",
    "run": null,
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.015903033707865177,
    "basis": "Reconstructed from the frozen v1.3 Cost axis; same speed/cost measurement, not a new price observation",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 60.32926888167386,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6429
     },
     "judge_hard": {
      "correct": 23,
      "n": 33,
      "accuracy": 0.697
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.3947
     },
     "multi_hop": {
      "correct": 20,
      "n": 35,
      "accuracy": 0.5714
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.2667
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    }
   },
   "source_round": "round 5 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7835497835497836,
   "sealed_accuracy": 0.262987012987013,
   "public_minus_sealed_gap_pp": 52.056277056277054,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.3902,
     "long_policy": 0.1,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.5,
     "probability": 0.3214,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.3093,
     "2/3": 0.2745
    },
    "ece": 0.3257234805194805,
    "mean_tvd_gold_probs": 0.38307442424242427,
    "calibration": 48.27393073593073,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.289,
      "accuracy": 0.2584
     },
     "conf>=0.9": {
      "coverage": 0.0649,
      "accuracy": 0.35
     }
    }
   },
   "v130_comparison_rank": 14,
   "v130_comparison_score": 70.6590827344797,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "axes": {
    "intelligence": 41.778425732165964,
    "calibration": 60.32926888167386,
    "speed": 82.04136987594218,
    "cost": 63.955600615838556
   },
   "jevbench_score": 40.86697866022174,
   "presets": {
    "JevBench Score (25:25:25:25)": 40.86697866022174,
    "Balanced 33:33:33 (no calibration)": 40.46559491192694,
    "Emphasis on Accuracy 60:20:20": 35.03759959383248,
    "Emphasis on Speed 20:60:20": 45.84895676929366,
    "Emphasis on Cost 20:20:60": 42.042351732331795,
    "Intelligence only": 29.168641634430376
   },
   "rank": 23,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 23,
    "Balanced 33:33:33 (no calibration)": 23,
    "Emphasis on Accuracy 60:20:20": 27,
    "Emphasis on Speed 20:60:20": 24,
    "Emphasis on Cost 20:20:60": 20,
    "Intelligence only": 36
   }
  },
  {
   "key": "zerank-2",
   "display": "ZeroEntropy zerank-2",
   "class": "reranker",
   "open": "yes",
   "author": "ZeroEntropy",
   "repo": "https://huggingface.co/zeroentropy/zerank-2-reranker",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-Reranker-derived 4B cross-encoder",
   "has_distribution": true,
   "probability_source": [
    "public_calibrated_reranker_softmax"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7916666666666666,
    "judge": 0.8835616438356164,
    "hard": 0.4727272727272727
   },
   "axes": {
    "intelligence": 42.06697845466217,
    "calibration": 75.76484738035566,
    "speed": 78.96019184080492,
    "cost": 49.75473753033334
   },
   "jevbench_score": 40.20589573277385,
   "speed": {
    "p50_s_raw": 0.12663885252550244,
    "p95_s_raw": 1.5002395501825958,
    "p50_s_adjusted": 0.4032777050510049,
    "p95_s_adjusted": 3.1504791003651915,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.23386710975319147,
    "hard_tier_p95_s": 1.8308645181823509
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.04729792435014604,
    "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
    "usd_per_1000_v11_tiers": 0.03125511838232944,
    "usd_per_1000_hard": 0.07019538377693882,
    "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21."
   },
   "calibration": {
    "score": 75.76484738035566,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.09786559872743003,
    "probability_fidelity": 72.55111248672686,
    "brier_hard": 0.631716160803407,
    "brier_standard_judge_v11": 0.40128849491331786,
    "note": null
   },
   "hard": {
    "run": "runs/zerank-2--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4727272727272727,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 12,
      "n": 38,
      "accuracy": 0.3157894736842105
     },
     "multi_hop": {
      "correct": 19,
      "n": 35,
      "accuracy": 0.5428571428571428
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 12,
      "n": 16,
      "accuracy": 0.75
     }
    },
    "has_distribution": true,
    "brier_mean": 0.631716160803407,
    "ece": 0.09786559872743003,
    "probability_fidelity": 72.55111248672686,
    "calibration_score": 76.48899637062043,
    "onehot": {
     "ece": 0.5272727272727273,
     "probability_fidelity": 38.704499999999996,
     "calibration_score": 19.352249999999998
    },
    "latency_p50_s": 0.23386710975319147,
    "latency_p95_s": 1.8308645181823509,
    "mean_input_tokens": 4350.459090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 40.20589573277385,
    "Balanced 33:33:33 (no calibration)": 37.194334838869295,
    "Emphasis on Accuracy 60:20:20": 33.673045014020886,
    "Emphasis on Speed 20:60:20": 42.81031013414403,
    "Emphasis on Cost 20:20:60": 36.23025428669733,
    "Intelligence only": 29.485793451110546
   },
   "rank": 24,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 24,
    "Balanced 33:33:33 (no calibration)": 29,
    "Emphasis on Accuracy 60:20:20": 31,
    "Emphasis on Speed 20:60:20": 30,
    "Emphasis on Cost 20:20:60": 28,
    "Intelligence only": 34
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7012987012987013,
   "sealed_accuracy": 0.2857142857142857,
   "public_minus_sealed_gap_pp": 41.55844155844156,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.2927,
     "long_policy": 0.125,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5,
     "probability": 0.3929,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.2202,
     "1/3": 0.268,
     "2/3": 0.3725
    },
    "ece": 0.14686658277918277,
    "mean_tvd_gold_probs": 0.21993584644511172,
    "calibration": 74.31654939982613,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1234,
      "accuracy": 0.3158
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 30,
   "v130_comparison_score": 65.96713617823673,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "decision-machine-1",
   "display": "decision-machine-1 (milliseconds.ai)",
   "class": "decision-api",
   "open": "no",
   "author": "milliseconds.ai (Baptiste Laget)",
   "repo": "https://www.milliseconds.ai",
   "licence": "proprietary API, closed weights",
   "underlying": "decision-machine-1 (weights not published)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "production API (milliseconds.ai, served from its nearest region), measured from Germany",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7604166666666666,
    "judge": 0.8972602739726028,
    "hard": 0.4681818181818182
   },
   "axes": {
    "intelligence": 41.275558286402884,
    "calibration": 68.32820400405224,
    "speed": 92.92504013916033,
    "cost": 53.668063581045736
   },
   "jevbench_score": 39.93541656481836,
   "speed": {
    "p50_s_raw": 0.17223640158772469,
    "p95_s_raw": 0.29605407454073424,
    "p50_s_adjusted": 0.17223640158772469,
    "p95_s_adjusted": 0.29605407454073424,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.18790292367339134,
    "hard_tier_p95_s": 0.3872291069477796
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.035026591760299625,
    "basis": "public tariff x measured tokens: $0.04 per million input tokens, output free (https://docs.milliseconds.ai/reference/pricing, read 2026-09-21) x 496 input tokens per easy/standard/judge decision as reported by the API; the run used the free test key, the price is the paid one",
    "usd_per_1000_v11_tiers": 0.019859745222929936,
    "usd_per_1000_hard": 0.05667381818181818,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 68.32820400405224,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.15839024302206123,
    "probability_fidelity": 72.56853703703703,
    "brier_hard": 0.6458498833513563,
    "brier_standard_judge_v11": 0.2808058520000086,
    "note": null
   },
   "hard": {
    "run": "runs/decision-machine-1--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4681818181818182,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 21,
      "n": 33,
      "accuracy": 0.6363636363636364
     },
     "long_policy": {
      "correct": 13,
      "n": 38,
      "accuracy": 0.34210526315789475
     },
     "multi_hop": {
      "correct": 16,
      "n": 35,
      "accuracy": 0.45714285714285713
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 12,
      "n": 16,
      "accuracy": 0.75
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6458498833513563,
    "ece": 0.15839024302206123,
    "probability_fidelity": 72.56853703703703,
    "calibration_score": 70.4452442163124,
    "onehot": {
     "ece": 0.5318181818181817,
     "probability_fidelity": 45.02249999999999,
     "calibration_score": 22.511249999999993
    },
    "latency_p50_s": 0.18790292367339134,
    "latency_p95_s": 0.3872291069477796,
    "mean_input_tokens": 1416.8454545454545,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 39.93541656481836,
    "Balanced 33:33:33 (no calibration)": 38.126375698505385,
    "Emphasis on Accuracy 60:20:20": 33.38024125474112,
    "Emphasis on Speed 20:60:20": 45.343850177082906,
    "Emphasis on Cost 20:20:60": 37.48949947255078,
    "Intelligence only": 28.128000417414217
   },
   "rank": 25,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 25,
    "Balanced 33:33:33 (no calibration)": 27,
    "Emphasis on Accuracy 60:20:20": 32,
    "Emphasis on Speed 20:60:20": 26,
    "Emphasis on Cost 20:20:60": 26,
    "Intelligence only": 39
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.6753246753246753,
   "sealed_accuracy": 0.2564935064935065,
   "public_minus_sealed_gap_pp": 41.883116883116884,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.4146,
     "long_policy": 0.15,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.5,
     "probability": 0.25,
     "safety_judge": 0.1875,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2018,
     "1/3": 0.2784,
     "2/3": 0.2941
    },
    "ece": 0.23509657008642643,
    "mean_tvd_gold_probs": 0.24792438823650945,
    "calibration": 64.0941235795319,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2273,
      "accuracy": 0.3286
     },
     "conf>=0.9": {
      "coverage": 0.0162,
      "accuracy": 0.0
     }
    }
   },
   "v130_comparison_rank": 20,
   "v130_comparison_score": 68.33662413233554,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (Milliseconds API), as for every API measurement"
  },
  {
   "key": "malkuth-2b",
   "display": "Malkuth-2B (newfull5, Kev post-train)",
   "author": "newfull5 (dhtocks)",
   "repo": "https://github.com/newfull5/malkuth",
   "class": "jev-rebuild",
   "licence": "CC-BY-NC-4.0, research use only (XNLI and RACE in the training mix)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.8854166666666666,
    "judge": 0.863013698630137,
    "hard": 0.44545454545454544
   },
   "speed": {
    "p50_s_raw": 0.04497810127213597,
    "p95_s_raw": 0.07615387919358908,
    "p50_s_adjusted": 0.23995620254427194,
    "p95_s_adjusted": 0.30230775838717816,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.019154662921348316,
    "basis": "ESTIMATE: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (no hosted ~2B listed; the 4B price errs high, the decider-2b precedent), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 53.47207215007215,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "c5c8e642539f596fece95ec68020f714e2addba6956007ffcd28336c2a11b855",
    "row_sha256": "11469ca612340d0557749a92ee0d989a48ea300a2c46e1cfef679de9a4eac99f",
    "raw_results_sha256": "6e468641a599b2ca1c4bdf080d50c20f6e3f6abb24d4cad337eb8d0c65e522a8",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.696969696969697,
   "sealed_accuracy": 0.2435064935064935,
   "public_minus_sealed_gap_pp": 45.346320346320354,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1081,
     "judge_hard": 0.3415,
     "long_policy": 0.275,
     "multi_hop": 0.1316,
     "paraphrase_robustness": 0.4286,
     "probability": 0.25,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.2784,
     "2/3": 0.3137
    },
    "ece": 0.3487084415584415,
    "mean_tvd_gold_probs": 0.3226924242424243,
    "calibration": 48.99453463203463,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3474,
      "accuracy": 0.2897
     },
     "conf>=0.9": {
      "coverage": 0.1526,
      "accuracy": 0.234
     }
    }
   },
   "v130_comparison_rank": 19,
   "v130_comparison_score": 67.1284937381763,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (no hosted ~2B listed; the 4B price errs high, the decider-2b precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 41.30149581986909,
    "calibration": 53.47207215007215,
    "speed": 91.39418726661523,
    "cost": 61.531764586365185
   },
   "jevbench_score": 38.930553392688644,
   "presets": {
    "JevBench Score (25:25:25:25)": 38.930553392688644,
    "Balanced 33:33:33 (no calibration)": 39.820116873699945,
    "Emphasis on Accuracy 60:20:20": 34.174375131470484,
    "Emphasis on Speed 20:60:20": 46.55044949806488,
    "Emphasis on Cost 20:20:60": 40.65859779113492,
    "Intelligence only": 28.18106059688171
   },
   "rank": 26,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 26,
    "Balanced 33:33:33 (no calibration)": 25,
    "Emphasis on Accuracy 60:20:20": 28,
    "Emphasis on Speed 20:60:20": 23,
    "Emphasis on Cost 20:20:60": 22,
    "Intelligence only": 38
   }
  },
  {
   "key": "raw-phi-4-mini",
   "display": "Raw Phi-4 mini direct logits",
   "class": "raw-logit-control",
   "open": "yes",
   "author": "Microsoft / neutral reproduction",
   "repo": "https://huggingface.co/microsoft/Phi-4-mini-instruct",
   "licence": "MIT",
   "underlying": "Phi-4-mini-instruct BF16",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7395833333333334,
    "judge": 0.7808219178082192,
    "hard": 0.5227272727272727
   },
   "axes": {
    "intelligence": 41.791137926269585,
    "calibration": 58.81535803341269,
    "speed": 88.83321682255763,
    "cost": 49.570821980679646
   },
   "jevbench_score": 37.95731820837918,
   "speed": {
    "p50_s_raw": 0.06377661414444447,
    "p95_s_raw": 0.16066877208650113,
    "p50_s_adjusted": 0.27755322828888895,
    "p95_s_adjusted": 0.4713375441730022,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
    "hard_tier_p50_s": 0.0792750152759254,
    "hard_tier_p95_s": 0.31224897452630096
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.047970318352059935,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.047970318352059935,
    "usd_per_1000_hard": 0.047970318352059935,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 58.81535803341269,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2417753590606609,
    "probability_fidelity": 72.76407537993293,
    "brier_hard": 0.6986812465074166,
    "brier_standard_judge_v11": 0.3532999027223932,
    "note": null
   },
   "hard": {
    "run": "raw-controls/runs/phi4-mini/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5227272727272727,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 19,
      "n": 33,
      "accuracy": 0.5757575757575758
     },
     "long_policy": {
      "correct": 16,
      "n": 38,
      "accuracy": 0.42105263157894735
     },
     "multi_hop": {
      "correct": 20,
      "n": 35,
      "accuracy": 0.5714285714285714
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 5,
      "n": 30,
      "accuracy": 0.16666666666666666
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6986812465074166,
    "ece": 0.2417753590606609,
    "probability_fidelity": 72.76407537993293,
    "calibration_score": 62.204501783900376,
    "onehot": {
     "ece": 0.4772727272727273,
     "probability_fidelity": 51.07900000000001,
     "calibration_score": 27.812227272727274
    },
    "latency_p50_s": 0.0792750152759254,
    "latency_p95_s": 0.31224897452630096,
    "mean_input_tokens": 1140.8863636363637,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 37.95731820837918,
    "Balanced 33:33:33 (no calibration)": 37.21138069677622,
    "Emphasis on Accuracy 60:20:20": 33.26324366220059,
    "Emphasis on Speed 20:60:20": 44.088434788026966,
    "Emphasis on Cost 20:20:60": 35.87366987490255,
    "Intelligence only": 28.696227946101086
   },
   "rank": 27,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 27,
    "Balanced 33:33:33 (no calibration)": 28,
    "Emphasis on Accuracy 60:20:20": 33,
    "Emphasis on Speed 20:60:20": 28,
    "Emphasis on Cost 20:20:60": 29,
    "Intelligence only": 37
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.658008658008658,
   "sealed_accuracy": 0.2922077922077922,
   "public_minus_sealed_gap_pp": 36.580086580086586,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.2927,
     "long_policy": 0.25,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.3571,
     "probability": 0.3571,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2857,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.2018,
     "1/3": 0.2887,
     "2/3": 0.3922
    },
    "ece": 0.32396223009125014,
    "mean_tvd_gold_probs": 0.31133412916875286,
    "calibration": 52.03707053243734,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3799,
      "accuracy": 0.2906
     },
     "conf>=0.9": {
      "coverage": 0.1688,
      "accuracy": 0.25
     }
    }
   },
   "v130_comparison_rank": 36,
   "v130_comparison_score": 63.42593271007894,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jev-qwen3.5-9b-base-nvfp4",
   "display": "JEV Qwen3.5-9B Base NVFP4",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "WilfLin",
   "repo": "https://huggingface.co/WIlfLin/JEV-Qwen3.5-9B-Base-NVFP4",
   "licence": "Apache-2.0 entrant and upstream checkpoint",
   "underlying": "Qwen3.5-9B Base NVFP4 with compact BF16 decision head",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX 5090 32 GB), local loopback HTTP, native FP4; serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9166666666666666,
    "judge": 0.8972602739726028,
    "hard": 0.5863636363636363
   },
   "axes": {
    "intelligence": 46.83968447419993,
    "calibration": 67.67253092868599,
    "speed": 93.29099763267277,
    "cost": 43.327939808722704
   },
   "jevbench_score": 37.697189917260424,
   "speed": {
    "p50_s_raw": 0.019635691001894884,
    "p95_s_raw": 0.048818428552112894,
    "p50_s_adjusted": 0.18927138200378976,
    "p95_s_adjusted": 0.24763685710422578,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our GPU (lium.io RTX 5090 32 GB), local loopback HTTP, native FP4; serial",
    "hard_tier_p50_s": 0.03173043649803731,
    "hard_tier_p95_s": 0.11175313639869267
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.07745842696629213,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.07745842696629213,
    "usd_per_1000_hard": 0.07745842696629213,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 67.67253092868599,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.09833429740233855,
    "probability_fidelity": 70.58078219464421,
    "brier_hard": 0.529798054594463,
    "brier_standard_judge_v11": 0.17053084553058298,
    "note": null
   },
   "hard": {
    "run": "['jev9b/evidence/run-full/easy.jsonl', 'jev9b/evidence/run-full/standard_judge.jsonl', 'jev9b/evidence/run-full/hard.jsonl'] (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5863636363636363,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 10,
      "n": 14,
      "accuracy": 0.7142857142857143
     },
     "judge_hard": {
      "correct": 21,
      "n": 33,
      "accuracy": 0.6363636363636364
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 23,
      "n": 35,
      "accuracy": 0.6571428571428571
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.529798054594463,
    "ece": 0.09833429740233855,
    "probability_fidelity": 70.58078219464421,
    "calibration_score": 75.45696135708826,
    "onehot": {
     "ece": 0.4136363636363637,
     "probability_fidelity": 42.774000000000015,
     "calibration_score": 30.02336363636364
    },
    "latency_p50_s": 0.03173043649803731,
    "latency_p95_s": 0.11175313639869267,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 37.697189917260424,
    "Balanced 33:33:33 (no calibration)": 35.848634990400484,
    "Emphasis on Accuracy 60:20:20": 33.67484430617602,
    "Emphasis on Speed 20:60:20": 43.02301803587772,
    "Emphasis on Cost 20:20:60": 32.524486396267946,
    "Intelligence only": 30.867250325671467
   },
   "rank": 28,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 28,
    "Balanced 33:33:33 (no calibration)": 31,
    "Emphasis on Accuracy 60:20:20": 30,
    "Emphasis on Speed 20:60:20": 29,
    "Emphasis on Cost 20:20:60": 37,
    "Intelligence only": 30
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7532467532467533,
   "sealed_accuracy": 0.29545454545454547,
   "public_minus_sealed_gap_pp": 45.77922077922078,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.4634,
     "long_policy": 0.175,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.5,
     "probability": 0.1429,
     "safety_judge": 0.5,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.2294,
     "1/3": 0.3505,
     "2/3": 0.3137
    },
    "ece": 0.2909642591491922,
    "mean_tvd_gold_probs": 0.37599808026398673,
    "calibration": 52.10367007188144,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.237,
      "accuracy": 0.2603
     },
     "conf>=0.9": {
      "coverage": 0.0455,
      "accuracy": 0.2143
     }
    }
   },
   "v130_comparison_rank": 18,
   "v130_comparison_score": 68.88432501076402,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "openjev-razorback16",
   "display": "OpenJev (DiffusionGemma 26B-A4B NVFP4, razorback16)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "razorback16 / Codiv",
   "repo": "https://github.com/razorback16/openjev",
   "licence": "Apache-2.0 (repo and weights)",
   "underlying": "nvidia/diffusiongemma-26B-A4B-it-NVFP4",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.910958904109589,
    "hard": 0.6545454545454545
   },
   "axes": {
    "intelligence": 45.43523560052183,
    "calibration": 54.993853529501855,
    "speed": 83.1778984786352,
    "cost": 45.49198106171689
   },
   "jevbench_score": 36.85070900932601,
   "speed": {
    "p50_s_raw": 0.24127069488167763,
    "p95_s_raw": 0.3052692499011755,
    "p50_s_adjusted": 0.6325413897633553,
    "p95_s_adjusted": 0.760538499802351,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
    "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request); model loaded before timing",
    "hard_tier_p50_s": 0.2737487629055977,
    "hard_tier_p95_s": 0.5951106011867523
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.06560455056179774,
    "basis": "ESTIMATE: hosted-provider price, openrouter google/gemma-4-26b-a4b-it list price $0.09/M in, $0.3/M out (DiffusionGemma 26B-A4B is not listed; the same-size Gemma 4 26B-A4B MoE sibling is (size class moe_26B-A4B)) x 380 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter google/gemma-4-26b-a4b-it $0.09/M in, $0.3/M out x 1222 in / 0 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.034483662420382165,
    "usd_per_1000_hard": 0.11002254545454546,
    "self_host_sensitivity": {
     "usd_per_1000": 0.042441862057452644,
     "score": 59.305139262330144,
     "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)",
     "usd_per_h": 0.72,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.25465117234471585,
     "decisions_per_hour": 16964.38292517306
    }
   },
   "calibration": {
    "score": 54.993853529501855,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.17789083573394485,
    "probability_fidelity": 65.09839938295943,
    "brier_hard": 0.48442036776440417,
    "brier_standard_judge_v11": 0.12110731767657959,
    "note": null
   },
   "hard": {
    "run": "runs-gpu/openjev-razorback16--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6545454545454545,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 10,
      "n": 14,
      "accuracy": 0.7142857142857143
     },
     "judge_hard": {
      "correct": 29,
      "n": 33,
      "accuracy": 0.8787878787878788
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.48442036776440417,
    "ece": 0.17789083573394485,
    "probability_fidelity": 65.09839938295943,
    "calibration_score": 64.76011611808524,
    "onehot": {
     "ece": 0.34545454545454546,
     "probability_fidelity": 48.83850000000001,
     "calibration_score": 39.87379545454546
    },
    "latency_p50_s": 0.2737487629055977,
    "latency_p95_s": 0.5951106011867523,
    "mean_input_tokens": 1222.4727272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 36.85070900932601,
    "Balanced 33:33:33 (no calibration)": 36.610231362914796,
    "Emphasis on Accuracy 60:20:20": 34.16683066443873,
    "Emphasis on Speed 20:60:20": 42.69113501144666,
    "Emphasis on Cost 20:20:60": 34.185595124683815,
    "Intelligence only": 31.05761022179735
   },
   "rank": 29,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 29,
    "Balanced 33:33:33 (no calibration)": 30,
    "Emphasis on Accuracy 60:20:20": 29,
    "Emphasis on Speed 20:60:20": 31,
    "Emphasis on Cost 20:20:60": 32,
    "Intelligence only": 28
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8181818181818182,
   "sealed_accuracy": 0.2857142857142857,
   "public_minus_sealed_gap_pp": 53.24675324675325,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.4878,
     "long_policy": 0.225,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.4286,
     "probability": 0.2143,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.2165,
     "2/3": 0.4608
    },
    "ece": 0.43439819122402074,
    "mean_tvd_gold_probs": 0.42197705050525675,
    "calibration": 35.461328352335094,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5292,
      "accuracy": 0.319
     },
     "conf>=0.9": {
      "coverage": 0.3052,
      "accuracy": 0.266
     }
    }
   },
   "v130_comparison_rank": 27,
   "v130_comparison_score": 66.36329072785742,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "kev-4b",
   "display": "kev 4B (research preview)",
   "class": "jev-rebuild",
   "open": true,
   "author": "Jared Palmer",
   "repo": "https://github.com/jaredpalmer/kev",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-4B-Base + LoRA + learned pointer head; jaredpalmer/kev-4b",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9166666666666666,
    "judge": 0.8561643835616438,
    "hard": 0.42272727272727273
   },
   "axes": {
    "intelligence": 42.07543736300355,
    "calibration": 39.64120851500514,
    "speed": 75.73845187472214,
    "cost": 61.77327904537152
   },
   "jevbench_score": 36.136502368258114,
   "speed": {
    "p50_s_raw": 0.5502474755048752,
    "p95_s_raw": 0.9917014226317405,
    "p50_s_adjusted": 1.2504949510097503,
    "p95_s_adjusted": 2.1334028452634812,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.718296229839325,
    "hard_tier_p95_s": 1.7924389142543065
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.018802865168539323,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3.5-4B size-class reference list price $0.03/M in, $0.0/M out (a 4B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.008382324840764331,
    "usd_per_1000_hard": 0.033675818181818175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 39.64120851500514,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.40195757303003016,
    "probability_fidelity": 64.33339383938394,
    "brier_hard": 0.8860972000262979,
    "brier_standard_judge_v11": 0.17123441707403728,
    "note": null
   },
   "hard": {
    "run": "runs/kev-4b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.42272727272727273,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 9,
      "n": 38,
      "accuracy": 0.23684210526315788
     },
     "multi_hop": {
      "correct": 16,
      "n": 35,
      "accuracy": 0.45714285714285713
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 3,
      "n": 30,
      "accuracy": 0.1
     },
     "tradeoff": {
      "correct": 3,
      "n": 12,
      "accuracy": 0.25
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8860972000262979,
    "ece": 0.40195757303003016,
    "probability_fidelity": 64.33339383938394,
    "calibration_score": 41.97093961668896,
    "onehot": {
     "ece": 0.5772727272727273,
     "probability_fidelity": 45.35549999999999,
     "calibration_score": 22.677749999999996
    },
    "latency_p50_s": 0.718296229839325,
    "latency_p95_s": 1.7924389142543065,
    "mean_input_tokens": 1122.5272727272727,
    "mean_output_tokens": 61.42727272727273,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 36.136502368258114,
    "Balanced 33:33:33 (no calibration)": 39.96378758251536,
    "Emphasis on Accuracy 60:20:20": 35.163493019081514,
    "Emphasis on Speed 20:60:20": 44.50049674732379,
    "Emphasis on Cost 20:20:60": 41.394643286122175,
    "Intelligence only": 29.79517279783051
   },
   "rank": 30,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 30,
    "Balanced 33:33:33 (no calibration)": 24,
    "Emphasis on Accuracy 60:20:20": 26,
    "Emphasis on Speed 20:60:20": 27,
    "Emphasis on Cost 20:20:60": 21,
    "Intelligence only": 33
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6623376623376623,
   "sealed_accuracy": 0.22402597402597402,
   "public_minus_sealed_gap_pp": 43.83116883116883,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 306,
    "by_family": {
     "ambiguous_abstain": 0.1081,
     "judge_hard": 0.3171,
     "long_policy": 0.2,
     "multi_hop": 0.1316,
     "paraphrase_robustness": 0.2143,
     "probability": 0.2857,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.1429,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1651,
     "1/3": 0.2165,
     "2/3": 0.2941
    },
    "ece": 0.4649653895628323,
    "mean_tvd_gold_probs": 0.3704342946415854,
    "calibration": 34.9817463116375,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5032,
      "accuracy": 0.2581
     },
     "conf>=0.9": {
      "coverage": 0.3052,
      "accuracy": 0.234
     }
    }
   },
   "v130_comparison_rank": 48,
   "v130_comparison_score": 59.724673285425375,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 306/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "decision-2b",
   "display": "Decision 2B (FlyMy.AI, v59)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "FlyMy.AI (@denti)",
   "repo": "https://huggingface.co/flymy-ai/decision-2b-preview",
   "licence": "Apache-2.0 notices on the included code and the pinned base; the weights are an evaluation preview under EVALUATION-PERMISSION.md, not a cleared commercial release",
   "underlying": "openbmb/MiniCPM5-2B with a trained LoRA adapter and pointer head (26.2M trainable parameters)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (L40 48 GB, Czechia), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "tiers": {
    "easy": 1.0,
    "standard": 0.875,
    "judge": 0.6986301369863014,
    "hard": 0.5818181818181818
   },
   "axes": {
    "intelligence": 38.76224329634284,
    "calibration": 74.12548511090162,
    "speed": 84.28398951642146,
    "cost": 62.50937716584096
   },
   "jevbench_score": 35.800088231124995,
   "speed": {
    "p50_s_raw": 0.18901889398694038,
    "p95_s_raw": 0.2781067743897438,
    "p50_s_adjusted": 0.5280377879738808,
    "p95_s_adjusted": 0.7062135487794876,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.2644132226705551,
    "hard_tier_p95_s": 0.36454500518739197
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.017769999999999998,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B size-class reference list price $0.03/M in, $0.0/M out (a 2-4B one-pass model with no generated output; the board's 4B open-weights reference, as for reflex 4B and decider-2b) x 269 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.008063885350318472,
    "usd_per_1000_hard": 0.03162327272727272
   },
   "calibration": {
    "score": 74.12548511090162,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.07969462515005729,
    "probability_fidelity": 75.53950396053546,
    "brier_hard": 0.5206095137462398,
    "brier_standard_judge_v11": 0.34821268301133307
   },
   "hard": {
    "run": "runs/decision-2b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5818181818181818,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 10,
      "n": 14,
      "accuracy": 0.7142857142857143
     },
     "judge_hard": {
      "correct": 22,
      "n": 33,
      "accuracy": 0.6666666666666666
     },
     "long_policy": {
      "correct": 17,
      "n": 38,
      "accuracy": 0.4473684210526316
     },
     "multi_hop": {
      "correct": 22,
      "n": 35,
      "accuracy": 0.6285714285714286
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 8,
      "n": 10,
      "accuracy": 0.8
     },
     "temporal_numeric": {
      "correct": 4,
      "n": 30,
      "accuracy": 0.13333333333333333
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5206095137462398,
    "ece": 0.07969462515005729,
    "probability_fidelity": 75.53950396053546,
    "calibration_score": 79.800289465262,
    "onehot": {
     "ece": 0.4181818181818182,
     "probability_fidelity": 47.655499999999996,
     "calibration_score": 32.00956818181818
    },
    "latency_p50_s": 0.2644132226705551,
    "latency_p95_s": 0.36454500518739197,
    "mean_input_tokens": 1054.1090909090908,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "footnote": "A MiniCPM5-2B with a trained LoRA and pointer head (26.2M trainable parameters), run through the submitter's own packaged adapter and his frozen calibration temperature - one forward pass, nothing generated, no permutation or retry. The package verifies its own file hashes and its pinned public base before it loads. His disclosure, carried here unchanged: public JevBench results informed earlier experiment design and candidate selection, so this is not a benchmark-blind result; both training mixtures contain 96 SNI task289/Gigaword-derived records whose upstream entitlement is not established, source attribution for tasks 1186/1283/1284 is incomplete, and the 2B mixture holds 543 synthetic records labelled claude-fable-5-1 whose provider contract is not established. These are evaluation previews under an explicit evaluation permission, not a cleared commercial release. Self-host latency gets the standard x2 + 0.15 s adjustment.",
   "note": "flymy-ai/decision-2b-preview revision df57b75db927acc9ad91ec8115508c1e487086eb (checkpoint minicpm5_reduced_v16_4k_v59), base openbmb/MiniCPM5-2B revision 12a3808a956f869c767195e9266b59c4d21d92e2, the submitter's own FlyMyJevPackageAdapter and frozen calibrator, bf16, unmerged adapter, 4096-token packing, transformers 4.57.6 / peft 0.15.2 as pinned, torch 2.8.0 from the pod image, on our RunPod L40 in Czechia",
   "source": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "scorer": "jevbench composite_v13.py, unchanged v1.3.0 copy",
   "scorer_sha256": "66562dffb9d6f70bcafe5a0d7a823e0f48195aa6cdfb12dfc7d32f064b433d33",
   "dataset": "frozen JevBench v1.2, exact 534-task set (72 easy, 242 standard+judge, 220 hard)",
   "presets": {
    "JevBench Score (25:25:25:25)": 35.800088231124995,
    "Balanced 33:33:33 (no calibration)": 33.600360248579655,
    "Emphasis on Accuracy 60:20:20": 28.549347831252316,
    "Emphasis on Speed 20:60:20": 38.82967920896114,
    "Emphasis on Cost 20:20:60": 35.082560473432395,
    "Intelligence only": 23.296286610603048
   },
   "source_round": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7532467532467533,
   "sealed_accuracy": 0.2597402597402597,
   "public_minus_sealed_gap_pp": 49.350649350649356,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.2927,
     "long_policy": 0.175,
     "multi_hop": 0.1579,
     "paraphrase_robustness": 0.4286,
     "probability": 0.3571,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.25,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.1651,
     "1/3": 0.3299,
     "2/3": 0.2941
    },
    "ece": 0.2361596232750288,
    "mean_tvd_gold_probs": 0.27216322540632465,
    "calibration": 62.775876402180884,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1364,
      "accuracy": 0.1667
     },
     "conf>=0.9": {
      "coverage": 0.0097,
      "accuracy": 0.0
     }
    }
   },
   "v130_comparison_rank": 8,
   "v130_comparison_score": 72.03641785804933,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
   "not_ranked_because": null,
   "rank": 31,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 31,
    "Balanced 33:33:33 (no calibration)": 33,
    "Emphasis on Accuracy 60:20:20": 41,
    "Emphasis on Speed 20:60:20": 33,
    "Emphasis on Cost 20:20:60": 30,
    "Intelligence only": 46
   }
  },
  {
   "key": "qwen35-9b-jev-data-mix-v2",
   "display": "Qwen3.5-9B Jev-like data-mix v2",
   "author": "jsaurabh",
   "repo": "https://huggingface.co/jsaurabh/qwen3.5-9b-jev-data-mix-v2",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (adapter code and weights; Qwen3.5-9B base is Apache-2.0)",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "endpoint_kind": "unknown",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9479166666666666,
    "judge": 0.952054794520548,
    "hard": 0.6045454545454545
   },
   "speed": {
    "p50_s_raw": null,
    "p95_s_raw": null,
    "p50_s_adjusted": null,
    "p95_s_adjusted": null,
    "adjustment": "See original v1.3 measurement",
    "run": null,
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.08344007490636705,
    "basis": "Reconstructed from the frozen v1.3 Cost axis; same speed/cost measurement, not a new price observation",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 61.34859947411265,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714
     },
     "judge_hard": {
      "correct": 25,
      "n": 33,
      "accuracy": 0.7576
     },
     "long_policy": {
      "correct": 19,
      "n": 38,
      "accuracy": 0.5
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 3,
      "n": 30,
      "accuracy": 0.1
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    }
   },
   "source_round": "round 5 new",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7835497835497836,
   "sealed_accuracy": 0.2922077922077922,
   "public_minus_sealed_gap_pp": 49.134199134199136,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.3415,
     "long_policy": 0.225,
     "multi_hop": 0.1579,
     "paraphrase_robustness": 0.5714,
     "probability": 0.2857,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.4231,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.1284,
     "1/3": 0.2887,
     "2/3": 0.4706
    },
    "ece": 0.3331439466058434,
    "mean_tvd_gold_probs": 0.36924397736185965,
    "calibration": 48.223406471322676,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3734,
      "accuracy": 0.313
     },
     "conf>=0.9": {
      "coverage": 0.1429,
      "accuracy": 0.2273
     }
    }
   },
   "v130_comparison_rank": 32,
   "v130_comparison_score": 65.50770324701708,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "axes": {
    "intelligence": 47.3950810857469,
    "calibration": 61.34859947411265,
    "speed": 81.97617376178209,
    "cost": 42.35875944111046
   },
   "jevbench_score": 35.235946407850655,
   "presets": {
    "JevBench Score (25:25:25:25)": 35.235946407850655,
    "Balanced 33:33:33 (no calibration)": 33.99679186137923,
    "Emphasis on Accuracy 60:20:20": 32.53499523673245,
    "Emphasis on Speed 20:60:20": 39.658462257223675,
    "Emphasis on Cost 20:20:60": 30.96724362384556,
    "Intelligence only": 30.56372331106409
   },
   "rank": 32,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 32,
    "Balanced 33:33:33 (no calibration)": 32,
    "Emphasis on Accuracy 60:20:20": 34,
    "Emphasis on Speed 20:60:20": 32,
    "Emphasis on Cost 20:20:60": 38,
    "Intelligence only": 31
   }
  },
  {
   "key": "gpt-6-luna-low",
   "display": "GPT-6 Luna (low reasoning effort)",
   "author": "OpenAI",
   "repo": null,
   "class": "llm-baseline",
   "licence": "proprietary API",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "API measurement: sealed item text (no golds) was sent to OpenAI",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.9726027397260274,
    "hard": 0.9727272727272728
   },
   "speed": {
    "p50_s_raw": 1.4418320655822754,
    "p95_s_raw": 2.9878296986222264,
    "p50_s_adjusted": 1.4418320655822754,
    "p95_s_adjusted": 2.9878296986222264,
    "adjustment": "none (production API)",
    "run": "serial v1.2 standard+judge subset; used for Speed axis",
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.12699831460674169,
    "basis": "Measured API usage × list price",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 91.9779132545049,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 14,
      "n": 14,
      "accuracy": 1.0
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9091
     },
     "long_policy": {
      "correct": 38,
      "n": 38,
      "accuracy": 1.0
     },
     "multi_hop": {
      "correct": 35,
      "n": 35,
      "accuracy": 1.0
     },
     "probability": {
      "correct": 20,
      "n": 20,
      "accuracy": 1.0
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 28,
      "n": 30,
      "accuracy": 0.9333
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9167
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    }
   },
   "source_round": "GPT-6 API measurement 2026-09-23",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.9913419913419913,
   "sealed_accuracy": 0.9285714285714286,
   "public_minus_sealed_gap_pp": 6.2770562770562695,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.973,
     "judge_hard": 0.9268,
     "long_policy": 0.875,
     "multi_hop": 0.9474,
     "paraphrase_robustness": 1.0,
     "probability": 1.0,
     "safety_judge": 0.9375,
     "temporal_numeric": 0.8929,
     "tradeoff": 0.9231,
     "trap_adversarial": 0.8333
    },
    "by_panel_stratum": {
     "0/3": 0.8899,
     "1/3": 0.9278,
     "2/3": 0.9706
    },
    "ece": 0.11788425329158833,
    "mean_tvd_gold_probs": 0.027823548852892555,
    "calibration": 86.82039722819654,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8961,
      "accuracy": 0.9239
     },
     "conf>=0.9": {
      "coverage": 0.8929,
      "accuracy": 0.9236
     }
    }
   },
   "v130_comparison_rank": 13,
   "v130_comparison_score": 70.70041643027498,
   "scoring_note": "API measurement: sealed item text (no golds) was sent to OpenAI",
   "axes": {
    "intelligence": 95.78441561268662,
    "calibration": 91.9779132545049,
    "speed": 73.65729480447798,
    "cost": 36.88606127569502
   },
   "jevbench_score": 35.11224166817657,
   "presets": {
    "JevBench Score (25:25:25:25)": 35.11224166817657,
    "Balanced 33:33:33 (no calibration)": 31.934153327127547,
    "Emphasis on Accuracy 60:20:20": 37.79013373720287,
    "Emphasis on Speed 20:60:20": 34.762013763902566,
    "Emphasis on Cost 20:20:60": 25.830221069877904,
    "Intelligence only": 52.12900217803402
   },
   "rank": 33,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 33,
    "Balanced 33:33:33 (no calibration)": 36,
    "Emphasis on Accuracy 60:20:20": 23,
    "Emphasis on Speed 20:60:20": 41,
    "Emphasis on Cost 20:20:60": 47,
    "Intelligence only": 4
   }
  },
  {
   "key": "simplejev-qwen3.8-27b",
   "display": "SimpleJev Qwen3.8-27B",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Featherless AI",
   "repo": "https://github.com/featherless-ai/simple-jev",
   "licence": "Apache-2.0 (Qwen weights); repository licence not stated",
   "underlying": "Qwen3.8-27B through SimpleJev's direct-logit classifier",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "author's public demo endpoint (Featherless Classifier Demo) — not a production service",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.9315068493150684,
    "hard": 0.75
   },
   "axes": {
    "intelligence": 51.58275939184895,
    "calibration": 74.49514921690603,
    "speed": 71.17954293639775,
    "cost": 39.48943618731995
   },
   "jevbench_score": 34.56619633355993,
   "speed": {
    "p50_s_raw": 1.0138170085847378,
    "p95_s_raw": 1.8794299442321063,
    "p50_s_adjusted": 2.0276340171694756,
    "p95_s_adjusted": 3.7588598884642126,
    "adjustment": "x2 (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 1.5036342144012451,
    "hard_tier_p95_s": 2.3205738529562945
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.10399651685393257,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Gemma 4 26B-A4B size-class reference list price $0.09/M in, $0.0/M out (a public 27B dense model served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.07279222929936303,
    "usd_per_1000_hard": 0.14853354545454545,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 74.49514921690603,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.06112991220300847,
    "probability_fidelity": 74.51582146352575,
    "brier_hard": 0.2989611751696843,
    "brier_standard_judge_v11": 0.08157045921166492,
    "note": null
   },
   "hard": {
    "run": "runs/simplejev-qwen3.8-27b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.75,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 28,
      "n": 33,
      "accuracy": 0.8484848484848485
     },
     "long_policy": {
      "correct": 26,
      "n": 38,
      "accuracy": 0.6842105263157895
     },
     "multi_hop": {
      "correct": 29,
      "n": 35,
      "accuracy": 0.8285714285714286
     },
     "probability": {
      "correct": 14,
      "n": 20,
      "accuracy": 0.7
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.2989611751696843,
    "ece": 0.06112991220300847,
    "probability_fidelity": 74.51582146352575,
    "calibration_score": 81.14491951146204,
    "onehot": {
     "ece": 0.25,
     "probability_fidelity": 55.716,
     "calibration_score": 52.858000000000004
    },
    "latency_p50_s": 1.5036342144012451,
    "latency_p95_s": 2.3205738529562945,
    "mean_input_tokens": 1650.3727272727272,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 34.56619633355993,
    "Balanced 33:33:33 (no calibration)": 31.847268130857827,
    "Emphasis on Accuracy 60:20:20": 31.977786869375375,
    "Emphasis on Speed 20:60:20": 35.90786466232381,
    "Emphasis on Cost 20:20:60": 28.507211848988284,
    "Intelligence only": 32.17558326378193
   },
   "rank": 34,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 34,
    "Balanced 33:33:33 (no calibration)": 37,
    "Emphasis on Accuracy 60:20:20": 35,
    "Emphasis on Speed 20:60:20": 39,
    "Emphasis on Cost 20:20:60": 43,
    "Intelligence only": 25
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8658008658008658,
   "sealed_accuracy": 0.35714285714285715,
   "public_minus_sealed_gap_pp": 50.86580086580086,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4595,
     "judge_hard": 0.5122,
     "long_policy": 0.25,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3929,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.3918,
     "2/3": 0.4804
    },
    "ece": 0.2217556156276108,
    "mean_tvd_gold_probs": 0.3325765961888979,
    "calibration": 61.19560862779402,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.263,
      "accuracy": 0.4691
     },
     "conf>=0.9": {
      "coverage": 0.0422,
      "accuracy": 0.4615
     }
    }
   },
   "v130_comparison_rank": 28,
   "v130_comparison_score": 66.29860868418298,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (Featherless demo), as for every API measurement"
  },
  {
   "key": "ninfer-qwen3.8-flash-next",
   "display": "NInfer Qwen3.8-Flash-Next mixed",
   "class": "native-logit",
   "open": "yes",
   "author": "Igor L. / NInfer contributors",
   "repo": "https://github.com/igorls/ninfer",
   "licence": "Qwen Community License 1.0 (model); Apache-2.0 (engine)",
   "underlying": "Qwen/Qwen3.8-Flash-Next, mixed NVFP4/FP8/INT4 NInfer artifact",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX PRO 6000 Blackwell 96 GB; NInfer and harness co-located; serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.952054794520548,
    "hard": 0.7727272727272727
   },
   "axes": {
    "intelligence": 49.52312902943435,
    "calibration": 78.5836892196356,
    "speed": 88.20497686265139,
    "cost": 38.91358504738628
   },
   "jevbench_score": 33.977533774383616,
   "speed": {
    "p50_s_raw": 0.07892056694254279,
    "p95_s_raw": 0.17055323952808976,
    "p50_s_adjusted": 0.3078411338850856,
    "p95_s_adjusted": 0.49110647905617955,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.10779002541676164,
    "hard_tier_p95_s": 0.41232560444623206
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.10869606741573033,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3.8-flash hosted list reference list price $0.15/M in, $0.0/M out (same underlying Flash-Next weights; native one-pass option-logit readout) x 366 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.054927229299363056,
    "usd_per_1000_hard": 0.18543886363636364,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 78.5836892196356,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.053760269138428435,
    "probability_fidelity": 79.67080253564572,
    "brier_hard": 0.27546730210553816,
    "brier_standard_judge_v11": 0.03739829817270268,
    "note": null
   },
   "hard": {
    "run": "runs/ninfer-qwen3.8-flash-next--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7727272727272727,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 13,
      "n": 14,
      "accuracy": 0.9285714285714286
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9090909090909091
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 30,
      "n": 35,
      "accuracy": 0.8571428571428571
     },
     "probability": {
      "correct": 15,
      "n": 20,
      "accuracy": 0.75
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.27546730210553816,
    "ece": 0.053760269138428435,
    "probability_fidelity": 79.67080253564572,
    "calibration_score": 84.45937435398001,
    "onehot": {
     "ece": 0.2272727272727273,
     "probability_fidelity": 57.965999999999994,
     "calibration_score": 56.25572727272727
    },
    "latency_p50_s": 0.10779002541676164,
    "latency_p95_s": 0.41232560444623206,
    "mean_input_tokens": 1236.259090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.04079654999999999
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 33.977533774383616,
    "Balanced 33:33:33 (no calibration)": 31.149636090077085,
    "Emphasis on Accuracy 60:20:20": 30.436953963902603,
    "Emphasis on Speed 20:60:20": 37.183450244801044,
    "Emphasis on Cost 20:20:60": 27.351678565781576,
    "Intelligence only": 29.427048203523267
   },
   "rank": 35,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 35,
    "Balanced 33:33:33 (no calibration)": 40,
    "Emphasis on Accuracy 60:20:20": 37,
    "Emphasis on Speed 20:60:20": 36,
    "Emphasis on Cost 20:20:60": 46,
    "Intelligence only": 35
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8961038961038961,
   "sealed_accuracy": 0.3409090909090909,
   "public_minus_sealed_gap_pp": 55.519480519480524,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.5122,
     "long_policy": 0.275,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.5,
     "probability": 0.3214,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.268,
     "2/3": 0.5784
    },
    "ece": 0.19232581023201106,
    "mean_tvd_gold_probs": 0.2787020005170424,
    "calibration": 66.83231895094677,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2338,
      "accuracy": 0.4722
     },
     "conf>=0.9": {
      "coverage": 0.0422,
      "accuracy": 0.3846
     }
    }
   },
   "v130_comparison_rank": 11,
   "v130_comparison_score": 70.94796786702575,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "swanone",
   "display": "swanOne (blockbrain, Qwen3.8-Flash-Next NVFP4)",
   "author": "blockbrain",
   "repo": "https://github.com/blockbrain-ai/swanone-recipe",
   "class": "system-one-open",
   "licence": "patches Apache-2.0, shim MIT; weights under the Qwen licence (LICENSE-NOTICE.md)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.9315068493150684,
    "hard": 0.7545454545454545
   },
   "speed": {
    "p50_s_raw": 0.2812396320514381,
    "p95_s_raw": 0.3177051310893148,
    "p50_s_adjusted": 0.7124792641028762,
    "p95_s_adjusted": 0.7854102621786296,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.11099550561797752,
    "basis": "ESTIMATE: OpenRouter qwen/qwen3.8-flash list price $0.15/M input (same underlying Flash-Next weights; the NInfer Flash-Next precedent), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 71.02567022903239,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "e2fa276cd8c1e9c64943f64d33ba41862a099578e55004e76f195d50fd357c15",
    "row_sha256": "e3f21199eb9a25669394c5b9f51fae9e58426f0d5a58d5743b0778cec3621436",
    "raw_results_sha256": "14c33bf1c4de9fa4495fe2bbfbd8a751472e1489f67f4ac9f2e003e149bdf010",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8874458874458875,
   "sealed_accuracy": 0.38636363636363635,
   "public_minus_sealed_gap_pp": 50.10822510822511,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.5405,
     "judge_hard": 0.3415,
     "long_policy": 0.325,
     "multi_hop": 0.5263,
     "paraphrase_robustness": 0.2857,
     "probability": 0.25,
     "safety_judge": 0.5,
     "temporal_numeric": 0.25,
     "tradeoff": 0.4231,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.3299,
     "2/3": 0.6961
    },
    "ece": 0.21794939292376322,
    "mean_tvd_gold_probs": 0.3902122440325648,
    "calibration": 58.69444850599543,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3734,
      "accuracy": 0.4696
     },
     "conf>=0.9": {
      "coverage": 0.1071,
      "accuracy": 0.4848
     }
    }
   },
   "v130_comparison_rank": 19,
   "v130_comparison_score": 67.52914018037747,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter qwen/qwen3.8-flash list price $0.15/M input (same underlying Flash-Next weights; the NInfer Flash-Next precedent), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 52.596343642944674,
    "calibration": 71.02567022903239,
    "speed": 82.52131199313874,
    "cost": 38.64083818365132
   },
   "jevbench_score": 33.60524727984119,
   "presets": {
    "JevBench Score (25:25:25:25)": 33.60524727984119,
    "Balanced 33:33:33 (no calibration)": 31.4283815940468,
    "Emphasis on Accuracy 60:20:20": 31.422204222703737,
    "Emphasis on Speed 20:60:20": 36.75524513056021,
    "Emphasis on Cost 20:20:60": 27.454808927228996,
    "Intelligence only": 31.412942717546485
   },
   "rank": 36,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 36,
    "Balanced 33:33:33 (no calibration)": 39,
    "Emphasis on Accuracy 60:20:20": 36,
    "Emphasis on Speed 20:60:20": 37,
    "Emphasis on Cost 20:20:60": 45,
    "Intelligence only": 27
   }
  },
  {
   "key": "gpt-6-luna",
   "display": "GPT-6 Luna (default medium reasoning effort)",
   "author": "OpenAI",
   "repo": null,
   "class": "llm-baseline",
   "licence": "proprietary API",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "API measurement: sealed item text (no golds) was sent to OpenAI",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 1.0,
    "judge": 0.9726027397260274,
    "hard": 0.9863636363636363
   },
   "speed": {
    "p50_s_raw": 1.477266326546669,
    "p95_s_raw": 3.731053405255079,
    "p50_s_adjusted": 1.477266326546669,
    "p95_s_adjusted": 3.731053405255079,
    "adjustment": "none (production API)",
    "run": "serial v1.2 standard+judge subset; used for Speed axis",
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.13544588014981276,
    "basis": "Measured API usage × list price",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 93.45967507606923,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 14,
      "n": 14,
      "accuracy": 1.0
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9091
     },
     "long_policy": {
      "correct": 38,
      "n": 38,
      "accuracy": 1.0
     },
     "multi_hop": {
      "correct": 35,
      "n": 35,
      "accuracy": 1.0
     },
     "probability": {
      "correct": 20,
      "n": 20,
      "accuracy": 1.0
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 30,
      "n": 30,
      "accuracy": 1.0
     },
     "tradeoff": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    }
   },
   "source_round": "GPT-6 API measurement 2026-09-23",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.9956709956709957,
   "sealed_accuracy": 0.9545454545454546,
   "public_minus_sealed_gap_pp": 4.112554112554112,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.9459,
     "judge_hard": 0.9024,
     "long_policy": 0.95,
     "multi_hop": 0.9737,
     "paraphrase_robustness": 1.0,
     "probability": 1.0,
     "safety_judge": 0.9375,
     "temporal_numeric": 0.9464,
     "tradeoff": 0.9615,
     "trap_adversarial": 1.0
    },
    "by_panel_stratum": {
     "0/3": 0.9358,
     "1/3": 0.9278,
     "2/3": 1.0
    },
    "ece": 0.09774373485292245,
    "mean_tvd_gold_probs": 0.010314550895454543,
    "calibration": 89.70989896993504,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8961,
      "accuracy": 0.9493
     },
     "conf>=0.9": {
      "coverage": 0.8929,
      "accuracy": 0.9491
     }
    }
   },
   "v130_comparison_rank": 15,
   "v130_comparison_score": 70.36922576521634,
   "scoring_note": "API measurement: sealed item text (no golds) was sent to OpenAI",
   "axes": {
    "intelligence": 97.35385336870618,
    "calibration": 93.45967507606923,
    "speed": 72.58709736103097,
    "cost": 36.04702601026741
   },
   "jevbench_score": 33.26981698087274,
   "presets": {
    "JevBench Score (25:25:25:25)": 33.26981698087274,
    "Balanced 33:33:33 (no calibration)": 30.10752497739428,
    "Emphasis on Accuracy 60:20:20": 35.92769142892222,
    "Emphasis on Speed 20:60:20": 32.75368934568851,
    "Emphasis on Cost 20:20:60": 24.22582268702631,
    "Intelligence only": 50.60017480671213
   },
   "rank": 37,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 37,
    "Balanced 33:33:33 (no calibration)": 43,
    "Emphasis on Accuracy 60:20:20": 25,
    "Emphasis on Speed 20:60:20": 44,
    "Emphasis on Cost 20:20:60": 53,
    "Intelligence only": 5
   }
  },
  {
   "key": "open-alternative-jev",
   "display": "open-alternative-jev (Qwen3.5-4B, IkerMoel)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "IkerMoel",
   "repo": "https://github.com/ikermoel/open-alternative-jev",
   "licence": "Apache-2.0 (code and weights)",
   "underlying": "Qwen/Qwen3.5-4B (frozen, BF16)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.84375,
    "judge": 0.7465753424657534,
    "hard": 0.5681818181818182
   },
   "axes": {
    "intelligence": 38.59574775248901,
    "calibration": 58.69063349321708,
    "speed": 83.47780372641819,
    "cost": 59.62620384142349
   },
   "jevbench_score": 33.24214483148707,
   "speed": {
    "p50_s_raw": 0.20686038956046104,
    "p95_s_raw": 0.32322231084108355,
    "p50_s_adjusted": 0.5637207791209221,
    "p95_s_adjusted": 0.7964446216821671,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
    "measured_where": "as open-alternative-jev",
    "hard_tier_p50_s": 0.2411614954471588,
    "hard_tier_p95_s": 0.6258101891726254
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.022171404494382024,
    "basis": "ESTIMATE: hosted-provider price, deepinfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.15/M out (as open-alternative-jev) x 383 input and 1 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) | ESTIMATE: deepinfra Qwen/Qwen3.5-4B $0.03/M in, $0.15/M out x 1235 in / 1 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.011652229299363059,
    "usd_per_1000_hard": 0.03718513636363636,
    "self_host_sensitivity": {
     "usd_per_1000": 0.036831988551922504,
     "score": 60.84437082593016,
     "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)",
     "usd_per_h": 0.72,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.22099193131153502,
     "decisions_per_hour": 19548.22501600768
    }
   },
   "calibration": {
    "score": 58.69063349321708,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.19036693270964333,
    "probability_fidelity": 64.40341152711024,
    "brier_hard": 0.6106512084932694,
    "brier_standard_judge_v11": 0.2681967810959945,
    "note": null
   },
   "hard": {
    "run": "runs-gpu/open-alternative-jev-yesfirst--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5681818181818182,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6428571428571429
     },
     "judge_hard": {
      "correct": 23,
      "n": 33,
      "accuracy": 0.696969696969697
     },
     "long_policy": {
      "correct": 17,
      "n": 38,
      "accuracy": 0.4473684210526316
     },
     "multi_hop": {
      "correct": 19,
      "n": 35,
      "accuracy": 0.5428571428571428
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6106512084932694,
    "ece": 0.19036693270964333,
    "probability_fidelity": 64.40341152711024,
    "calibration_score": 63.16501249259078,
    "onehot": {
     "ece": 0.43181818181818177,
     "probability_fidelity": 42.308499999999995,
     "calibration_score": 27.97243181818182
    },
    "latency_p50_s": 0.2411614954471588,
    "latency_p95_s": 0.6258101891726254,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 33.24214483148707,
    "Balanced 33:33:33 (no calibration)": 32.703238293812,
    "Emphasis on Accuracy 60:20:20": 27.97978223668259,
    "Emphasis on Speed 20:60:20": 37.895226816908355,
    "Emphasis on Cost 20:20:60": 33.777627572542954,
    "Intelligence only": 22.9973804230676
   },
   "run_key": "open-alternative-jev-yesfirst",
   "rank": 38,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 38,
    "Balanced 33:33:33 (no calibration)": 35,
    "Emphasis on Accuracy 60:20:20": 42,
    "Emphasis on Speed 20:60:20": 35,
    "Emphasis on Cost 20:20:60": 35,
    "Intelligence only": 47
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7402597402597403,
   "sealed_accuracy": 0.2435064935064935,
   "public_minus_sealed_gap_pp": 49.67532467532468,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.3171,
     "long_policy": 0.15,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.5714,
     "probability": 0.1786,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.1538,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1927,
     "1/3": 0.2371,
     "2/3": 0.3039
    },
    "ece": 0.35199515504627493,
    "mean_tvd_gold_probs": 0.30117218001805685,
    "calibration": 49.74187549446966,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3149,
      "accuracy": 0.2577
     },
     "conf>=0.9": {
      "coverage": 0.1104,
      "accuracy": 0.1176
     }
    }
   },
   "v130_comparison_rank": 24,
   "v130_comparison_score": 66.9883439146374,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jev-local",
   "display": "jev-local (Qwen3.5-9B)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "us (GitHub)",
   "repo": "https://github.com/us/jev-local",
   "licence": "no licence stated in the repository (public code); Apache-2.0 base weights",
   "underlying": "Qwen/Qwen3.5-9B, frozen, per-option mean log-probability",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.84375,
    "judge": 0.8904109589041096,
    "hard": 0.5909090909090909
   },
   "axes": {
    "intelligence": 45.15000569598809,
    "calibration": 64.17815548340548,
    "speed": 69.18842978642493,
    "cost": 43.327939808722704
   },
   "jevbench_score": 32.542400749826804,
   "speed": {
    "p50_s_raw": 1.0450106970965862,
    "p95_s_raw": 2.6157593585550782,
    "p50_s_adjusted": 2.2400213941931724,
    "p95_s_adjusted": 5.381518717110157,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 1.3785462081432343,
    "hard_tier_p95_s": 3.9017020471394064
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.07745842696629213,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3.5-9b list price $0.1/M in, $0.0/M out (the exact base weights; scored by log-probabilities, nothing is generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.0452,
    "usd_per_1000_hard": 0.1235,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 64.17815548340548,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.14593909090909096,
    "probability_fidelity": 66.67524999999999,
    "brier_hard": 0.5767452409090912,
    "brier_standard_judge_v11": 0.1765305730578512,
    "note": null
   },
   "hard": {
    "run": "runs/jev-local--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5909090909090909,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714285714285714
     },
     "judge_hard": {
      "correct": 23,
      "n": 33,
      "accuracy": 0.696969696969697
     },
     "long_policy": {
      "correct": 17,
      "n": 38,
      "accuracy": 0.4473684210526316
     },
     "multi_hop": {
      "correct": 20,
      "n": 35,
      "accuracy": 0.5714285714285714
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5767452409090912,
    "ece": 0.14593909090909096,
    "probability_fidelity": 66.67524999999999,
    "calibration_score": 68.7437159090909,
    "onehot": {
     "ece": 0.40909090909090906,
     "probability_fidelity": 39.6865,
     "calibration_score": 28.934159090909095
    },
    "latency_p50_s": 1.3785462081432343,
    "latency_p95_s": 3.9017020471394064,
    "mean_input_tokens": 636.8863636363636,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 32.542400749826804,
    "Balanced 33:33:33 (no calibration)": 30.77892288007467,
    "Emphasis on Accuracy 60:20:20": 29.44415974758906,
    "Emphasis on Speed 20:60:20": 34.559432386987844,
    "Emphasis on Cost 20:20:60": 28.925940905517265,
    "Intelligence only": 27.645820867824607
   },
   "rank": 39,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 39,
    "Balanced 33:33:33 (no calibration)": 41,
    "Emphasis on Accuracy 60:20:20": 38,
    "Emphasis on Speed 20:60:20": 42,
    "Emphasis on Cost 20:20:60": 41,
    "Intelligence only": 40
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7489177489177489,
   "sealed_accuracy": 0.29545454545454547,
   "public_minus_sealed_gap_pp": 45.34632034632034,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.3659,
     "long_policy": 0.25,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.3571,
     "probability": 0.4643,
     "safety_judge": 0.25,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1743,
     "1/3": 0.2887,
     "2/3": 0.4314
    },
    "ece": 0.259294805194805,
    "mean_tvd_gold_probs": 0.38046969696969685,
    "calibration": 55.04703463203465,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3019,
      "accuracy": 0.3548
     },
     "conf>=0.9": {
      "coverage": 0.1526,
      "accuracy": 0.2553
     }
    }
   },
   "v130_comparison_rank": 42,
   "v130_comparison_score": 61.79680786600313,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "decision-fast",
   "display": "Decision Fast (FlyMy.AI, v53a)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "FlyMy.AI (@denti)",
   "repo": "https://huggingface.co/flymy-ai/decision-fast-preview",
   "licence": "Apache-2.0 notices on the included code and the pinned base; the weights are an evaluation preview under EVALUATION-PERMISSION.md, not a cleared commercial release",
   "underlying": "Qwen/Qwen3-0.6B-Base with a trained LoRA adapter and pointer head (10.6M trainable parameters)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (L40 48 GB, Czechia), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.78125,
    "judge": 0.7465753424657534,
    "hard": 0.38636363636363635
   },
   "axes": {
    "intelligence": 37.074471307080934,
    "calibration": 65.30757379702415,
    "speed": 81.59778170772206,
    "cost": 76.07683824704311
   },
   "jevbench_score": 32.492203456270644,
   "speed": {
    "p50_s_raw": 0.24040156602859497,
    "p95_s_raw": 0.47365329600870587,
    "p50_s_adjusted": 0.63080313205719,
    "p95_s_adjusted": 1.0973065920174117,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.24019427970051765,
    "hard_tier_p95_s": 0.30370312519371506
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0062724719101123596,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output, the same reference the kev 0.5B/0.6B rows use) x 280 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0027984713375796178,
    "usd_per_1000_hard": 0.011230818181818182
   },
   "calibration": {
    "score": 65.30757379702415,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.17070899271518503,
    "probability_fidelity": 64.15601134661442,
    "brier_hard": 0.7227289657304032,
    "brier_standard_judge_v11": 0.37929172170710185
   },
   "hard": {
    "run": "runs/decision-fast--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.38636363636363635,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 5,
      "n": 20,
      "accuracy": 0.25
     },
     "routing_hard": {
      "correct": 7,
      "n": 10,
      "accuracy": 0.7
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7227289657304032,
    "ece": 0.17070899271518503,
    "probability_fidelity": 64.15601134661442,
    "calibration_score": 65.00710640178872,
    "onehot": {
     "ece": 0.6136363636363636,
     "probability_fidelity": 35.4665,
     "calibration_score": 17.73325
    },
    "latency_p50_s": 0.24019427970051765,
    "latency_p95_s": 0.30370312519371506,
    "mean_input_tokens": 1123.081818181818,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "footnote": "The 0.6B sibling of Decision 2B: Qwen3-0.6B-Base with a trained LoRA and pointer head (10.6M trainable parameters), same packaged adapter, same frozen-calibration pointer readout, nothing generated. Its historical readout was a 256-way route; this package evaluates every item through the pointer head, which the submitter states changed no argmax on the items he had measured before. The same provenance disclosure as Decision 2B applies, minus the synthetic claude-fable-5-1 records. Self-host latency gets the standard x2 + 0.15 s adjustment.",
   "note": "flymy-ai/decision-fast-preview revision 4225d41c66119fe28e95a2631bb0103decae6d56 (checkpoint qwen3_06b_headfirst_ep2a_v53), base Qwen/Qwen3-0.6B-Base revision da87bfb608c14b7cf20ba1ce41287e8de496c0cd, the submitter's own FlyMyJevPackageAdapter and frozen calibrator, bf16, unmerged adapter, 4096-token packing, transformers 4.57.6 / peft 0.15.2 as pinned, torch 2.8.0 from the pod image, on our RunPod L40 in Czechia",
   "source": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "scorer": "jevbench composite_v13.py, unchanged v1.3.0 copy",
   "scorer_sha256": "66562dffb9d6f70bcafe5a0d7a823e0f48195aa6cdfb12dfc7d32f064b433d33",
   "dataset": "frozen JevBench v1.2, exact 534-task set (72 easy, 242 standard+judge, 220 hard)",
   "presets": {
    "JevBench Score (25:25:25:25)": 32.492203456270644,
    "Balanced 33:33:33 (no calibration)": 31.493956619710993,
    "Emphasis on Accuracy 60:20:20": 25.85668884226624,
    "Emphasis on Speed 20:60:20": 35.756033348527204,
    "Emphasis on Cost 20:20:60": 34.94749987457829,
    "Intelligence only": 20.38378786979466
   },
   "source_round": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6320346320346321,
   "sealed_accuracy": 0.2564935064935065,
   "public_minus_sealed_gap_pp": 37.55411255411256,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.2683,
     "long_policy": 0.1,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.5714,
     "probability": 0.25,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.2844,
     "1/3": 0.2474,
     "2/3": 0.2353
    },
    "ece": 0.19621019688435656,
    "mean_tvd_gold_probs": 0.28940943448138623,
    "calibration": 65.90850858749502,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1136,
      "accuracy": 0.4857
     },
     "conf>=0.9": {
      "coverage": 0.013,
      "accuracy": 0.75
     }
    }
   },
   "v130_comparison_rank": 21,
   "v130_comparison_score": 68.00397642460794,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
   "not_ranked_because": null,
   "rank": 40,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 40,
    "Balanced 33:33:33 (no calibration)": 38,
    "Emphasis on Accuracy 60:20:20": 45,
    "Emphasis on Speed 20:60:20": 40,
    "Emphasis on Cost 20:20:60": 31,
    "Intelligence only": 54
   }
  },
  {
   "key": "decider-2b",
   "display": "decider-2b (Mapika)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Mapika",
   "repo": "https://huggingface.co/Mapika/decider-2b",
   "licence": "Apache-2.0",
   "underlying": "Qwen3.5-2B-Base with a trained decision readout, 1.9B",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.8541666666666666,
    "judge": 0.773972602739726,
    "hard": 0.4727272727272727
   },
   "axes": {
    "intelligence": 38.54932735543702,
    "calibration": 43.48691774891777,
    "speed": 83.17774245885059,
    "cost": 60.99184724027941
   },
   "jevbench_score": 30.7375448157181,
   "speed": {
    "p50_s_raw": 0.26079703494906425,
    "p95_s_raw": 0.2831697516143322,
    "p50_s_adjusted": 0.6715940698981285,
    "p95_s_adjusted": 0.7163395032286645,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.2675044983625412,
    "hard_tier_p95_s": 0.45449532456696035
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.01996511235955056,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen/Qwen3.5-4B list price $0.03/M in, $0.0/M out (no hosted ~2B Qwen3.5 is listed, so the 4B price is used and errs high; one pass, no output) x 312 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.00936076433121019,
    "usd_per_1000_hard": 0.035100409090909085,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 43.48691774891777,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.3221518181818179,
    "probability_fidelity": 57.608000000000004,
    "brier_hard": 0.8062077985909092,
    "brier_standard_judge_v11": 0.2826622928099173,
    "note": null
   },
   "hard": {
    "run": "runs/decider-2b--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4727272727272727,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 21,
      "n": 33,
      "accuracy": 0.6363636363636364
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 17,
      "n": 35,
      "accuracy": 0.4857142857142857
     },
     "probability": {
      "correct": 6,
      "n": 20,
      "accuracy": 0.3
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 5,
      "n": 30,
      "accuracy": 0.16666666666666666
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8062077985909092,
    "ece": 0.3221518181818179,
    "probability_fidelity": 57.608000000000004,
    "calibration_score": 46.58881818181821,
    "onehot": {
     "ece": 0.5272727272727273,
     "probability_fidelity": 39.31099999999999,
     "calibration_score": 19.655499999999996
    },
    "latency_p50_s": 0.2675044983625412,
    "latency_p95_s": 0.45449532456696035,
    "mean_input_tokens": 1170.0136363636364,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 30.7375448157181,
    "Balanced 33:33:33 (no calibration)": 32.805331344237096,
    "Emphasis on Accuracy 60:20:20": 27.97523227121188,
    "Emphasis on Speed 20:60:20": 37.907634763228145,
    "Emphasis on Cost 20:20:60": 34.10323064925486,
    "Intelligence only": 22.914501028410267
   },
   "rank": 41,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 41,
    "Balanced 33:33:33 (no calibration)": 34,
    "Emphasis on Accuracy 60:20:20": 43,
    "Emphasis on Speed 20:60:20": 34,
    "Emphasis on Cost 20:20:60": 34,
    "Intelligence only": 48
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.70995670995671,
   "sealed_accuracy": 0.24675324675324675,
   "public_minus_sealed_gap_pp": 46.32034632034633,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.2927,
     "long_policy": 0.225,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2143,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.156,
     "1/3": 0.2474,
     "2/3": 0.3431
    },
    "ece": 0.4128347402597403,
    "mean_tvd_gold_probs": 0.4286681818181817,
    "calibration": 37.28311688311688,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5,
      "accuracy": 0.2532
     },
     "conf>=0.9": {
      "coverage": 0.2792,
      "accuracy": 0.2326
     }
    }
   },
   "v130_comparison_rank": 43,
   "v130_comparison_score": 61.6816875223098,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jeff",
   "display": "jeff (Logan Markewich, GLiFormer 400M)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Logan Markewich",
   "repo": "https://github.com/logan-markewich/jeff",
   "licence": "MIT (code); GLiFormer weights per their model card",
   "underlying": "GLiFormer large (knowledgator/gliformer-large-v1, ~400M) behind a TypeSafe-compatible /v1/systemone server",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7604166666666666,
    "judge": 0.6164383561643836,
    "hard": 0.37727272727272726
   },
   "axes": {
    "intelligence": 36.768981386976094,
    "calibration": 67.87788564213564,
    "speed": 63.49232389662579,
    "cost": 76.57596288088385
   },
   "jevbench_score": 30.579484596127113,
   "speed": {
    "p50_s_raw": 0.9379300177097321,
    "p95_s_raw": 10.969045254960655,
    "p50_s_adjusted": 2.025860035419464,
    "p95_s_adjusted": 22.088090509921308,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 2.2362988367676735,
    "hard_tier_p95_s": 55.099180381745
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.006036722846441948,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 272 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0027223885350318475,
    "usd_per_1000_hard": 0.010767181818181818,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 67.87788564213564,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.18531090909090905,
    "probability_fidelity": 66.18625,
    "brier_hard": 0.7454531705909093,
    "brier_standard_judge_v11": 0.5159533382231405,
    "note": null
   },
   "hard": {
    "run": "runs/jeff--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.37727272727272726,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 0,
      "n": 14,
      "accuracy": 0.0
     },
     "judge_hard": {
      "correct": 17,
      "n": 33,
      "accuracy": 0.5151515151515151
     },
     "long_policy": {
      "correct": 5,
      "n": 38,
      "accuracy": 0.13157894736842105
     },
     "multi_hop": {
      "correct": 17,
      "n": 35,
      "accuracy": 0.4857142857142857
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 4,
      "n": 10,
      "accuracy": 0.4
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 11,
      "n": 16,
      "accuracy": 0.6875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7454531705909093,
    "ece": 0.18531090909090905,
    "probability_fidelity": 66.18625,
    "calibration_score": 64.5620340909091,
    "onehot": {
     "ece": 0.6227272727272728,
     "probability_fidelity": 37.17100000000001,
     "calibration_score": 18.585500000000003
    },
    "latency_p50_s": 2.2362988367676735,
    "latency_p95_s": 55.099180381745,
    "mean_input_tokens": 1076.7181818181818,
    "mean_output_tokens": 6.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 30.579484596127113,
    "Balanced 33:33:33 (no calibration)": 28.967579522179133,
    "Emphasis on Accuracy 60:20:20": 24.49213124095356,
    "Emphasis on Speed 20:60:20": 30.899929407973797,
    "Emphasis on Cost 20:20:60": 32.9249740108452,
    "Intelligence only": 19.884047300996603
   },
   "rank": 42,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 42,
    "Balanced 33:33:33 (no calibration)": 45,
    "Emphasis on Accuracy 60:20:20": 51,
    "Emphasis on Speed 20:60:20": 50,
    "Emphasis on Cost 20:20:60": 36,
    "Intelligence only": 55
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6277056277056277,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 29.65367965367965,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2703,
     "judge_hard": 0.5366,
     "long_policy": 0.275,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3214,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2857,
     "tradeoff": 0.1538,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.2752,
     "1/3": 0.3814,
     "2/3": 0.3431
    },
    "ece": 0.12291623376623377,
    "mean_tvd_gold_probs": 0.2639757575757576,
    "calibration": 74.50958874458874,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1851,
      "accuracy": 0.5789
     },
     "conf>=0.9": {
      "coverage": 0.0065,
      "accuracy": 1.0
     }
    }
   },
   "v130_comparison_rank": 54,
   "v130_comparison_score": 54.38199454750733,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "laya",
   "display": "Laya (Convai Innovations, ModernBERT-large 421M)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Convai Innovations",
   "repo": "https://huggingface.co/convaiinnovations/laya",
   "licence": "Apache-2.0",
   "underlying": "ModernBERT-large encoder + option-marker decision head, 421M, RLCD-trained",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9444444444444444,
    "standard": 0.7291666666666666,
    "judge": 0.6917808219178082,
    "hard": 0.3409090909090909
   },
   "axes": {
    "intelligence": 36.132988467190394,
    "calibration": 63.67551046176046,
    "speed": 71.05966297458822,
    "cost": 86.20408526325653
   },
   "jevbench_score": 30.251281619829957,
   "speed": {
    "p50_s_raw": 0.787067785859108,
    "p95_s_raw": 2.197125389799475,
    "p50_s_adjusted": 1.7241355717182159,
    "p95_s_adjusted": 4.544250779598951,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 1.9288412630558014,
    "hard_tier_p95_s": 2.2888929322361946
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0028831273408239703,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 205 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.002048248407643312,
    "usd_per_1000_hard": 0.004074727272727273,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 63.67551046176046,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.20550909090909092,
    "probability_fidelity": 66.03125,
    "brier_hard": 0.7660523635454545,
    "brier_standard_judge_v11": 0.41431437239669405,
    "note": null
   },
   "hard": {
    "run": "runs/laya--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.3409090909090909,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 12,
      "n": 33,
      "accuracy": 0.36363636363636365
     },
     "long_policy": {
      "correct": 12,
      "n": 38,
      "accuracy": 0.3157894736842105
     },
     "multi_hop": {
      "correct": 12,
      "n": 35,
      "accuracy": 0.34285714285714286
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 3,
      "n": 10,
      "accuracy": 0.3
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 3,
      "n": 16,
      "accuracy": 0.1875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7660523635454545,
    "ece": 0.20550909090909092,
    "probability_fidelity": 66.03125,
    "calibration_score": 62.46471590909091,
    "onehot": {
     "ece": 0.6590909090909092,
     "probability_fidelity": 40.04649999999999,
     "calibration_score": 20.023249999999994
    },
    "latency_p50_s": 1.9288412630558014,
    "latency_p95_s": 2.2888929322361946,
    "mean_input_tokens": 407.4727272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 30.251281619829957,
    "Balanced 33:33:33 (no calibration)": 29.36743247536606,
    "Emphasis on Accuracy 60:20:20": 24.022017847348167,
    "Emphasis on Speed 20:60:20": 32.041461709468976,
    "Emphasis on Cost 20:20:60": 34.11113829878133,
    "Intelligence only": 18.869988637264445
   },
   "rank": 43,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 43,
    "Balanced 33:33:33 (no calibration)": 44,
    "Emphasis on Accuracy 60:20:20": 52,
    "Emphasis on Speed 20:60:20": 46,
    "Emphasis on Cost 20:20:60": 33,
    "Intelligence only": 56
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5844155844155844,
   "sealed_accuracy": 0.30844155844155846,
   "public_minus_sealed_gap_pp": 27.597402597402592,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.3902,
     "long_policy": 0.275,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.5,
     "probability": 0.2143,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.2477,
     "1/3": 0.3402,
     "2/3": 0.3431
    },
    "ece": 0.17190551948051955,
    "mean_tvd_gold_probs": 0.3342469696969697,
    "calibration": 66.09709956709956,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1818,
      "accuracy": 0.375
     },
     "conf>=0.9": {
      "coverage": 0.0195,
      "accuracy": 0.1667
     }
    }
   },
   "v130_comparison_rank": 55,
   "v130_comparison_score": 54.35373809070198,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "kushal-gemma4-31b-it-autoloops",
   "display": "Autoloops – Gemma 4 31B IT",
   "author": "Autoloops",
   "repo": null,
   "class": "jev",
   "licence": "Gemma terms of use (Google) for the base weights; hosted Autoloops service",
   "open": "no",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "Autoloops production API, reached from a Hetzner server in Germany",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.958904109589041,
    "hard": 0.8272727272727273
   },
   "speed": {
    "p50_s_raw": 0.6035970412194729,
    "p95_s_raw": 0.6620777942240238,
    "p50_s_adjusted": 0.6035970412194729,
    "p95_s_adjusted": 0.6620777942240238,
    "adjustment": "none (production API)",
    "run": "242 public standard+judge requests, sequential API calls",
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.136347191011236,
    "basis": "Measured API usage × list price",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 79.1909136580842,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "candidate_sha256": "d92411c795410382d6f57d87acc8528afb1149982b3132d4ad7b642e0013300f"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (Autoloops operator API, official full-protocol measurement, 24 Sep 2026)",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.9283819628647215,
   "sealed_accuracy": 0.4577922077922078,
   "public_minus_sealed_gap_pp": 47.05897550725137,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4595,
     "judge_hard": 0.7317,
     "long_policy": 0.325,
     "multi_hop": 0.5263,
     "paraphrase_robustness": 0.5714,
     "probability": 0.4643,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2857,
     "tradeoff": 0.4231,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": null,
    "ece": 0.13753618589150965,
    "mean_tvd_gold_probs": 0.23223795589980412,
    "calibration": 74.63448361585884,
    "label_only": false,
    "risk_coverage": null
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": null,
   "scoring_note": "API measurement: public and sealed item content reached the operator endpoint; no gold labels or answers were sent.",
   "axes": {
    "intelligence": 59.83714395843067,
    "calibration": 79.1909136580842,
    "speed": 83.98343875671114,
    "cost": 35.96061414550688
   },
   "jevbench_score": 29.962547814634828,
   "presets": {
    "JevBench Score (25:25:25:25)": 29.962547814634828,
    "Balanced 33:33:33 (no calibration)": 27.50083499400292,
    "Emphasis on Accuracy 60:20:20": 28.784545635578965,
    "Emphasis on Speed 20:60:20": 32.2318213029474,
    "Emphasis on Cost 20:20:60": 23.083230812891586,
    "Intelligence only": 30.951738529988752
   },
   "rank": 44,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 44,
    "Balanced 33:33:33 (no calibration)": 47,
    "Emphasis on Accuracy 60:20:20": 40,
    "Emphasis on Speed 20:60:20": 45,
    "Emphasis on Cost 20:20:60": 55,
    "Intelligence only": 29
   }
  },
  {
   "key": "standardone-8b",
   "display": "Standard One 8B (Standard Thinking)",
   "author": "Standard Thinking (myeongho12)",
   "repo": "https://huggingface.co/StandardThinking/StandardOne-8B",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (jev-adapter server, LoRA and merged weights; base Ministral 3 8B Apache-2.0)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9479166666666666,
    "judge": 0.9657534246575342,
    "hard": 0.5681818181818182
   },
   "speed": {
    "p50_s_raw": 0.023836581502109766,
    "p95_s_raw": 0.08334439168684185,
    "p50_s_adjusted": 0.19767316300421953,
    "p95_s_adjusted": 0.31668878337368367,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.10444269662921347,
    "basis": "ESTIMATE: OpenRouter mistralai/ministral-8b-2512 list price $0.15/M input (= the author's proposed Mistral API price for the exact base), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 65.85255199514542,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "7bf28891ae6843655d91fe145c380dc41affa70b18069787f2572dab1ed242cb",
    "row_sha256": "f7b874c485ab44cfdf8df4fd80e2275714bd6ecff8f99e85bc3b12fadd7c0110",
    "raw_results_sha256": "6cd6ee61f2c4342612cae5b57de420d28ccc346f8ade2f7d04a90f5134a91878",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7662337662337663,
   "sealed_accuracy": 0.2662337662337662,
   "public_minus_sealed_gap_pp": 50.0,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.3171,
     "long_policy": 0.2,
     "multi_hop": 0.1579,
     "paraphrase_robustness": 0.3571,
     "probability": 0.5,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.1538,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1743,
     "1/3": 0.2165,
     "2/3": 0.4118
    },
    "ece": 0.315018714327478,
    "mean_tvd_gold_probs": 0.2417063393340918,
    "calibration": 56.41281160054761,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3344,
      "accuracy": 0.2524
     },
     "conf>=0.9": {
      "coverage": 0.1169,
      "accuracy": 0.1389
     }
    }
   },
   "v130_comparison_rank": 20,
   "v130_comparison_score": 66.63753487554017,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: OpenRouter mistralai/ministral-8b-2512 list price $0.15/M input (= the author's proposed Mistral API price for the exact base), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 46.19294302571464,
    "calibration": 65.85255199514542,
    "speed": 92.03419606845978,
    "cost": 39.4336577067554
   },
   "jevbench_score": 29.066687887780347,
   "presets": {
    "JevBench Score (25:25:25:25)": 29.066687887780347,
    "Balanced 33:33:33 (no calibration)": 27.520185929320764,
    "Emphasis on Accuracy 60:20:20": 26.23768532556003,
    "Emphasis on Speed 20:60:20": 33.34576833428673,
    "Emphasis on Cost 20:20:60": 24.444522296145394,
    "Intelligence only": 24.52341826983675
   },
   "rank": 45,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 45,
    "Balanced 33:33:33 (no calibration)": 46,
    "Emphasis on Accuracy 60:20:20": 44,
    "Emphasis on Speed 20:60:20": 43,
    "Emphasis on Cost 20:20:60": 51,
    "Intelligence only": 45
   }
  },
  {
   "key": "lev-350m",
   "display": "lev-350m (Franck Verrot, LFM2.5-350M)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Franck Verrot (franckverrot)",
   "repo": "https://github.com/franckverrot/lev",
   "licence": "Apache-2.0 (code); weights under LiquidAI's LFM1.0 licence, following the LFM2.5-350M base",
   "underlying": "LiquidAI/LFM2.5-350M with a LoRA and a 6.5M-parameter pointer head (a kev clone)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (L40 48 GB, Czechia), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.71875,
    "judge": 0.6917808219178082,
    "hard": 0.36818181818181817
   },
   "axes": {
    "intelligence": 34.75234376694234,
    "calibration": 70.61387806637806,
    "speed": 85.31334412225277,
    "cost": 76.09548365186328
   },
   "jevbench_score": 28.501129766483672,
   "speed": {
    "p50_s_raw": 0.17216148599982262,
    "p95_s_raw": 0.22259443029761308,
    "p50_s_adjusted": 0.49432297199964526,
    "p95_s_adjusted": 0.5951888605952261,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.1572176218032837,
    "hard_tier_p95_s": 0.23697478026151647
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.006263501872659176,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output, the same reference the kev 0.5B/0.6B rows use) x 280 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0028040764331210195,
    "usd_per_1000_hard": 0.011201045454545455
   },
   "calibration": {
    "score": 70.61387806637806,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.12270454545454548,
    "probability_fidelity": 66.5325,
    "brier_hard": 0.6945723181818187,
    "brier_standard_judge_v11": 0.45654067768595025
   },
   "hard": {
    "run": "runs/lev-350m--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.36818181818181817,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 17,
      "n": 33,
      "accuracy": 0.5151515151515151
     },
     "long_policy": {
      "correct": 8,
      "n": 38,
      "accuracy": 0.21052631578947367
     },
     "multi_hop": {
      "correct": 10,
      "n": 35,
      "accuracy": 0.2857142857142857
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 6,
      "n": 10,
      "accuracy": 0.6
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6945723181818187,
    "ece": 0.12270454545454548,
    "probability_fidelity": 66.5325,
    "calibration_score": 70.99579545454546,
    "onehot": {
     "ece": 0.6318181818181818,
     "probability_fidelity": 37.15249999999999,
     "calibration_score": 18.576249999999995
    },
    "latency_p50_s": 0.1572176218032837,
    "latency_p95_s": 0.23697478026151647,
    "mean_input_tokens": 1120.1045454545454,
    "mean_output_tokens": 62.872727272727275,
    "charged_usd": 0.0
   },
   "footnote": "Requested by Florian on X on 22 Sep (2102305995007406540), replying to @franckverrot. A weekend clone of Jared Palmer's kev with the Qwen backbone swapped for LiquidAI's LFM2.5-350M: a LoRA and a 6.5M-parameter pointer head over frozen 350M weights, one prefill pass, nothing generated. Served by the author's own lev.serve on his TypeSafe-compatible /v1/systemone endpoint with the calibration temperature shipped beside the weights, so JevBench's own typesafe adapter is used unchanged - no mapping of ours. Its probabilities arrive rounded to three decimals by his API, which limits how fine a calibration measurement can be. The weights follow LiquidAI's LFM1.0 licence rather than Apache-2.0; the code is Apache-2.0. Self-host latency gets the standard x2 + 0.15 s adjustment.",
   "note": "weights franckverrot/lev-350m revision ab08ad8b8f346994d983152917e114224f6adac7, code github.com/franckverrot/lev c48a945dbf629998d7458dcc5c16f58df964db94, the author's own lev.serve /v1/systemone endpoint with its shipped calibration temperature, base LiquidAI/LFM2.5-350M, on our RunPod L40 in Czechia",
   "source": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "scorer": "jevbench composite_v13.py, unchanged v1.3.0 copy",
   "scorer_sha256": "66562dffb9d6f70bcafe5a0d7a823e0f48195aa6cdfb12dfc7d32f064b433d33",
   "dataset": "frozen JevBench v1.2, exact 534-task set (72 easy, 242 standard+judge, 220 hard)",
   "presets": {
    "JevBench Score (25:25:25:25)": 28.501129766483672,
    "Balanced 33:33:33 (no calibration)": 27.019514128234633,
    "Emphasis on Accuracy 60:20:20": 21.72402548595142,
    "Emphasis on Speed 20:60:20": 31.33656333518096,
    "Emphasis on Cost 20:20:60": 30.22309867362729,
    "Intelligence only": 16.788515273155372
   },
   "source_round": "round 8 new (add-requests run 8, 23 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5844155844155844,
   "sealed_accuracy": 0.25,
   "public_minus_sealed_gap_pp": 33.44155844155844,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.3171,
     "long_policy": 0.2,
     "multi_hop": 0.1316,
     "paraphrase_robustness": 0.4286,
     "probability": 0.2143,
     "safety_judge": 0.25,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.4231,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.2062,
     "2/3": 0.3333
    },
    "ece": 0.16306168831168832,
    "mean_tvd_gold_probs": 0.2768757575757576,
    "calibration": 69.85004329004329,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0779,
      "accuracy": 0.2083
     },
     "conf>=0.9": {
      "coverage": 0.0065,
      "accuracy": 0.0
     }
    }
   },
   "v130_comparison_rank": 44,
   "v130_comparison_score": 61.57472520374522,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container; no operator endpoint received sealed text.",
   "not_ranked_because": null,
   "rank": 46,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 46,
    "Balanced 33:33:33 (no calibration)": 49,
    "Emphasis on Accuracy 60:20:20": 55,
    "Emphasis on Speed 20:60:20": 48,
    "Emphasis on Cost 20:20:60": 39,
    "Intelligence only": 60
   }
  },
  {
   "key": "openjev-sglang",
   "display": "openjev-sglang (Qwen3.6-35B-A3B on SGLang)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "ekzhang",
   "repo": "https://github.com/ekzhang/openjev-sglang",
   "licence": "no licence file in the repository as of 2026-09-19; Qwen3.6 weights keep their own terms",
   "underlying": "Qwen3.6-35B-A3B",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "author's public demo endpoint (Modal) — not a production service",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.952054794520548,
    "hard": 0.7136363636363636
   },
   "axes": {
    "intelligence": 49.41096142357457,
    "calibration": 69.24191796746032,
    "speed": 77.059799513554,
    "cost": 36.45449272670061
   },
   "jevbench_score": 27.653667300053762,
   "speed": {
    "p50_s_raw": 0.6776718497276306,
    "p95_s_raw": 0.7260066717863083,
    "p50_s_adjusted": 1.3553436994552612,
    "p95_s_adjusted": 1.4520133435726166,
    "adjustment": "x2 (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.6935733407735825,
    "hard_tier_p95_s": 0.8754492454230784
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.13127546816479402,
    "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.6-35b-a3b list price $0.1/M in, $0.9/M out (same base weights) x 610 input and 2 output tokens per decision [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3.6-35b-a3b $0.1/M in, $0.9/M out x 2272 in / 2 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.0628296178343949,
    "usd_per_1000_hard": 0.22896636363636363,
    "self_host_sensitivity": {
     "usd_per_1000": 0.13621106532407892,
     "score": 46.64469025955533,
     "machine": "1x L40S 48 GB (Qwen3.6-35B-A3B, SGLang)",
     "usd_per_h": 0.86,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.6842230258139779,
     "decisions_per_hour": 6313.7308114715415
    }
   },
   "calibration": {
    "score": 69.24191796746032,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.09423841992624496,
    "probability_fidelity": 73.67122146328776,
    "brier_hard": 0.4005126115129668,
    "brier_standard_judge_v11": 0.08533204520463213,
    "note": null
   },
   "hard": {
    "run": "runs/openjev-sglang--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7136363636363636,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857142857142857
     },
     "judge_hard": {
      "correct": 27,
      "n": 33,
      "accuracy": 0.8181818181818182
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 11,
      "n": 20,
      "accuracy": 0.55
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 12,
      "n": 30,
      "accuracy": 0.4
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4005126115129668,
    "ece": 0.09423841992624496,
    "probability_fidelity": 73.67122146328776,
    "calibration_score": 77.41176873901938,
    "onehot": {
     "ece": 0.2863636363636364,
     "probability_fidelity": 49.77550000000001,
     "calibration_score": 46.25138636363637
    },
    "latency_p50_s": 0.6935733407735825,
    "latency_p95_s": 0.8754492454230784,
    "mean_input_tokens": 2271.663636363636,
    "mean_output_tokens": 2.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 27.653667300053762,
    "Balanced 33:33:33 (no calibration)": 25.679225980847427,
    "Emphasis on Accuracy 60:20:20": 25.667637481870678,
    "Emphasis on Speed 20:60:20": 29.97211304240087,
    "Emphasis on Cost 20:20:60": 22.470880934700617,
    "Intelligence only": 25.650274331318737
   },
   "rank": 47,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 47,
    "Balanced 33:33:33 (no calibration)": 51,
    "Emphasis on Accuracy 60:20:20": 46,
    "Emphasis on Speed 20:60:20": 51,
    "Emphasis on Cost 20:20:60": 56,
    "Intelligence only": 42
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8528138528138528,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 52.16450216450217,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.3902,
     "long_policy": 0.35,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.2857,
     "probability": 0.3214,
     "safety_judge": 0.5625,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.2784,
     "2/3": 0.5392
    },
    "ece": 0.2807123157213715,
    "mean_tvd_gold_probs": 0.3805310400704129,
    "calibration": 52.9022164243422,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3312,
      "accuracy": 0.3039
     },
     "conf>=0.9": {
      "coverage": 0.1201,
      "accuracy": 0.2162
     }
    }
   },
   "v130_comparison_rank": 33,
   "v130_comparison_score": 65.26826354532687,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (author's Modal demo), as for every API measurement"
  },
  {
   "key": "von-395m",
   "display": "Von (wfzyx, Option-Marker 395M)",
   "author": "wfzyx (Victor Hugo)",
   "repo": "https://github.com/wfzyx/von",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 code; Apache-2.0 weights (answerdotai/ModernBERT-large base)",
   "open": null,
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "endpoint_kind": "unknown",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9305555555555556,
    "standard": 0.6875,
    "judge": 0.684931506849315,
    "hard": 0.37272727272727274
   },
   "speed": {
    "p50_s_raw": null,
    "p95_s_raw": null,
    "p50_s_adjusted": null,
    "p95_s_adjusted": null,
    "adjustment": "See original v1.3 measurement",
    "run": null,
    "hardware": null,
    "measured_where": null,
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.005513183520599252,
    "basis": "Reconstructed from the frozen v1.3 Cost axis; same speed/cost measurement, not a new price observation",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 75.70851839826841,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5455
     },
     "long_policy": {
      "correct": 8,
      "n": 38,
      "accuracy": 0.2105
     },
     "multi_hop": {
      "correct": 8,
      "n": 35,
      "accuracy": 0.2286
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 6,
      "n": 10,
      "accuracy": 0.6
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4167
     },
     "trap": {
      "correct": 10,
      "n": 16,
      "accuracy": 0.625
     }
    }
   },
   "source_round": "round 5 new",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5714285714285714,
   "sealed_accuracy": 0.2792207792207792,
   "public_minus_sealed_gap_pp": 29.22077922077922,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.4634,
     "long_policy": 0.2,
     "multi_hop": 0.3947,
     "paraphrase_robustness": 0.0714,
     "probability": 0.1786,
     "safety_judge": 0.375,
     "temporal_numeric": 0.0893,
     "tradeoff": 0.5,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2202,
     "1/3": 0.268,
     "2/3": 0.3529
    },
    "ece": 0.10662467532467533,
    "mean_tvd_gold_probs": 0.22925909090909097,
    "calibration": 77.87457792207792,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0617,
      "accuracy": 0.4737
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 56,
   "v130_comparison_score": 53.08478248718057,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest",
   "axes": {
    "intelligence": 34.490709713287984,
    "calibration": 75.70851839826841,
    "speed": 70.45969264486726,
    "cost": 77.75792651988814
   },
   "jevbench_score": 27.483645331198645,
   "presets": {
    "JevBench Score (25:25:25:25)": 27.483645331198645,
    "Balanced 33:33:33 (no calibration)": 25.470586015162723,
    "Emphasis on Accuracy 60:20:20": 20.86431647532491,
    "Emphasis on Speed 20:60:20": 28.179346192301338,
    "Emphasis on Cost 20:20:60": 29.097498359912535,
    "Intelligence only": 16.412184256378776
   },
   "rank": 48,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 48,
    "Balanced 33:33:33 (no calibration)": 52,
    "Emphasis on Accuracy 60:20:20": 56,
    "Emphasis on Speed 20:60:20": 56,
    "Emphasis on Cost 20:20:60": 40,
    "Intelligence only": 62
   }
  },
  {
   "key": "ninfer-qwen3.8-27b-t1.5",
   "display": "NInfer Qwen3.8-27B NVFP4 (T=1.5)",
   "class": "native-logit",
   "open": "yes",
   "author": "Igor L. / NInfer contributors",
   "repo": "https://github.com/igorls/ninfer",
   "licence": "Apache-2.0 (engine and submitted 27B artifact/base)",
   "underlying": "Qwen/Qwen3.8-27B, NInfer NVFP4/FP8 artifact",
   "has_distribution": true,
   "probability_source": [
    "native-option-letter-softmax-calibrated-t1.5"
   ],
   "endpoint_condition": "our lium.io RTX 5090 32 GB, reached over the public internet from Sandy; serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9657534246575342,
    "hard": 0.7045454545454546
   },
   "axes": {
    "intelligence": 51.470093582226795,
    "calibration": 75.95962347375385,
    "speed": 80.14121364631829,
    "cost": 35.16542294913728
   },
   "jevbench_score": 26.916169238404322,
   "speed": {
    "p50_s_raw": 0.368209820240736,
    "p95_s_raw": 0.4710209038108587,
    "p50_s_adjusted": 0.886419640481472,
    "p95_s_adjusted": 1.0920418076217173,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.405510064214468,
    "hard_tier_p95_s": 0.8191776685416697
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.14492808988764044,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B hosted list reference list price $0.2/M in, $0.0/M out (same underlying weights; native one-pass option-logit readout) x 366 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.07323630573248406,
    "usd_per_1000_hard": 0.24725181818181818,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 75.95962347375385,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.05665602032632619,
    "probability_fidelity": 72.64096353077629,
    "brier_hard": 0.35349645913273137,
    "brier_standard_judge_v11": 0.06436642368664976,
    "note": null
   },
   "hard": {
    "run": "runs/ninfer-qwen3.8-27b-t1.5--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7045454545454546,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 21,
      "n": 38,
      "accuracy": 0.5526315789473685
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714285714285715
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.35349645913273137,
    "ece": 0.05665602032632619,
    "probability_fidelity": 72.64096353077629,
    "calibration_score": 80.65487973275552,
    "onehot": {
     "ece": 0.2954545454545454,
     "probability_fidelity": 54.20399999999999,
     "calibration_score": 47.55654545454546
    },
    "latency_p50_s": 0.405510064214468,
    "latency_p95_s": 0.8191776685416697,
    "mean_input_tokens": 1236.259090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.054395400000000045
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 26.916169238404322,
    "Balanced 33:33:33 (no calibration)": 24.591249162988152,
    "Emphasis on Accuracy 60:20:20": 24.9312721363086,
    "Emphasis on Speed 20:60:20": 28.994401869980386,
    "Emphasis on Cost 20:20:60": 21.099302385416387,
    "Intelligence only": 25.459310612668556
   },
   "rank": 49,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 49,
    "Balanced 33:33:33 (no calibration)": 55,
    "Emphasis on Accuracy 60:20:20": 49,
    "Emphasis on Speed 20:60:20": 54,
    "Emphasis on Cost 20:20:60": 62,
    "Intelligence only": 44
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8311688311688312,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 50.0,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.439,
     "long_policy": 0.2,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2857,
     "safety_judge": 0.625,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.3196,
     "2/3": 0.5392
    },
    "ece": 0.20248643451982903,
    "mean_tvd_gold_probs": 0.26364491184533145,
    "calibration": 66.56911095575052,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1916,
      "accuracy": 0.3729
     },
     "conf>=0.9": {
      "coverage": 0.026,
      "accuracy": 0.5
     }
    }
   },
   "v130_comparison_rank": 29,
   "v130_comparison_score": 66.18806782927027,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "ninfer-qwen3.8-27b",
   "display": "NInfer Qwen3.8-27B NVFP4",
   "class": "native-logit",
   "open": "yes",
   "author": "Igor L. / NInfer contributors",
   "repo": "https://github.com/igorls/ninfer",
   "licence": "Apache-2.0 (engine and submitted 27B artifact/base)",
   "underlying": "Qwen/Qwen3.8-27B, NInfer NVFP4/FP8 artifact",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX 5090 32 GB, reached over the public internet from Sandy; serial",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9657534246575342,
    "hard": 0.7045454545454546
   },
   "axes": {
    "intelligence": 51.470093582226795,
    "calibration": 67.1613598982882,
    "speed": 80.14121364631829,
    "cost": 35.16542294913728
   },
   "jevbench_score": 26.29915101653767,
   "speed": {
    "p50_s_raw": 0.368209820240736,
    "p95_s_raw": 0.4710209038108587,
    "p50_s_adjusted": 0.886419640481472,
    "p95_s_adjusted": 1.0920418076217173,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.405510064214468,
    "hard_tier_p95_s": 0.8191776685416697
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.14492808988764044,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B hosted list reference list price $0.2/M in, $0.0/M out (same underlying weights; native one-pass option-logit readout) x 366 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.07323630573248406,
    "usd_per_1000_hard": 0.24725181818181818,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 67.1613598982882,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.11149757299657584,
    "probability_fidelity": 69.53135462799072,
    "brier_hard": 0.37073865170440246,
    "brier_standard_judge_v11": 0.05306458153291101,
    "note": null
   },
   "hard": {
    "run": "runs/ninfer-qwen3.8-27b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7045454545454546,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 21,
      "n": 38,
      "accuracy": 0.5526315789473685
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714285714285715
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.37073865170440246,
    "ece": 0.11149757299657584,
    "probability_fidelity": 69.53135462799072,
    "calibration_score": 73.61592001433777,
    "onehot": {
     "ece": 0.2954545454545454,
     "probability_fidelity": 54.20399999999999,
     "calibration_score": 47.55654545454546
    },
    "latency_p50_s": 0.405510064214468,
    "latency_p95_s": 0.8191776685416697,
    "mean_input_tokens": 1236.259090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.054395400000000045
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 26.29915101653767,
    "Balanced 33:33:33 (no calibration)": 24.591249162988152,
    "Emphasis on Accuracy 60:20:20": 24.9312721363086,
    "Emphasis on Speed 20:60:20": 28.994401869980386,
    "Emphasis on Cost 20:20:60": 21.099302385416387,
    "Intelligence only": 25.459310612668556
   },
   "rank": 50,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 50,
    "Balanced 33:33:33 (no calibration)": 54,
    "Emphasis on Accuracy 60:20:20": 48,
    "Emphasis on Speed 20:60:20": 53,
    "Emphasis on Cost 20:20:60": 61,
    "Intelligence only": 43
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8311688311688312,
   "sealed_accuracy": 0.33116883116883117,
   "public_minus_sealed_gap_pp": 50.0,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.439,
     "long_policy": 0.2,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.3571,
     "probability": 0.2857,
     "safety_judge": 0.625,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.3196,
     "2/3": 0.5392
    },
    "ece": 0.29928035033403366,
    "mean_tvd_gold_probs": 0.31639450600815167,
    "calibration": 54.252239666189055,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3604,
      "accuracy": 0.3514
     },
     "conf>=0.9": {
      "coverage": 0.1299,
      "accuracy": 0.375
     }
    }
   },
   "v130_comparison_rank": 34,
   "v130_comparison_score": 64.69414495970757,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "kev-8b",
   "display": "kev 8B (research preview)",
   "class": "jev-rebuild",
   "open": true,
   "author": "Jared Palmer",
   "repo": "https://github.com/jaredpalmer/kev",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-8B-Base + LoRA + learned pointer head; jaredpalmer/kev-8b",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9270833333333334,
    "judge": 0.9041095890410958,
    "hard": 0.4727272727272727
   },
   "axes": {
    "intelligence": 41.80839400991245,
    "calibration": 40.165779287885506,
    "speed": 74.85933162711135,
    "cost": 44.04134083457654
   },
   "jevbench_score": 25.56370172722739,
   "speed": {
    "p50_s_raw": 0.5904169715940952,
    "p95_s_raw": 1.1521932914853092,
    "p50_s_adjusted": 1.3308339431881904,
    "p95_s_adjusted": 2.4543865829706184,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.7628460973501205,
    "hard_tier_p95_s": 2.0858752064406856
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.07333117415730338,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter qwen/qwen3-8b list price list price $0.117/M in, $0.0/M out (the same-size Qwen3-8B weights; kev generates no output tokens) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.03269106687898089,
    "usd_per_1000_hard": 0.13133569090909092,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 40.165779287885506,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.36420177926883585,
    "probability_fidelity": 61.20875487548755,
    "brier_hard": 0.8373365943511044,
    "brier_standard_judge_v11": 0.13167773335725774,
    "note": null
   },
   "hard": {
    "run": "runs/kev-8b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4727272727272727,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 18,
      "n": 35,
      "accuracy": 0.5142857142857142
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 0,
      "n": 30,
      "accuracy": 0.0
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8373365943511044,
    "ece": 0.36420177926883585,
    "probability_fidelity": 61.20875487548755,
    "calibration_score": 44.18419951086019,
    "onehot": {
     "ece": 0.5272727272727273,
     "probability_fidelity": 47.43100000000001,
     "calibration_score": 23.715500000000006
    },
    "latency_p50_s": 0.7628460973501205,
    "latency_p95_s": 2.0858752064406856,
    "mean_input_tokens": 1122.5272727272727,
    "mean_output_tokens": 61.21363636363636,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 25.56370172722739,
    "Balanced 33:33:33 (no calibration)": 27.130719773145287,
    "Emphasis on Accuracy 60:20:20": 25.155773653293025,
    "Emphasis on Speed 20:60:20": 31.283850576368625,
    "Emphasis on Cost 20:20:60": 25.73467246713809,
    "Intelligence only": 22.67939701244948
   },
   "rank": 51,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 51,
    "Balanced 33:33:33 (no calibration)": 48,
    "Emphasis on Accuracy 60:20:20": 47,
    "Emphasis on Speed 20:60:20": 49,
    "Emphasis on Cost 20:20:60": 48,
    "Intelligence only": 50
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7142857142857143,
   "sealed_accuracy": 0.21753246753246752,
   "public_minus_sealed_gap_pp": 49.67532467532468,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1081,
     "judge_hard": 0.2683,
     "long_policy": 0.15,
     "multi_hop": 0.1316,
     "paraphrase_robustness": 0.1429,
     "probability": 0.2857,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.125,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.2062,
     "2/3": 0.3039
    },
    "ece": 0.47359168059663115,
    "mean_tvd_gold_probs": 0.41023786196801504,
    "calibration": 32.12893884193613,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5065,
      "accuracy": 0.2308
     },
     "conf>=0.9": {
      "coverage": 0.3117,
      "accuracy": 0.2292
     }
    }
   },
   "v130_comparison_rank": 51,
   "v130_comparison_score": 56.383551097366656,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jevone",
   "display": "JevOne",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Juspay",
   "repo": "https://huggingface.co/juspay/jev-one",
   "licence": "Component-specific JevOne/Qwen/SGLang terms recorded in RESULT.md",
   "underlying": "Qwen3.6-35B-A3B BF16 with JevOne bidirectional option-logit mapping",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io RTX PRO 6000 Blackwell 96 GB), local loopback HTTP, TP1; serial; independently validated one-card condition",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.96875,
    "judge": 0.9246575342465754,
    "hard": 0.75
   },
   "axes": {
    "intelligence": 47.587794717975754,
    "calibration": 77.59848665223666,
    "speed": 88.46835142022954,
    "cost": 35.86750288568889
   },
   "jevbench_score": 25.512210128129595,
   "speed": {
    "p50_s_raw": 0.08668531733565032,
    "p95_s_raw": 0.14500587752554567,
    "p50_s_adjusted": 0.32337063467130067,
    "p95_s_adjusted": 0.44001175505109136,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our GPU (lium.io RTX PRO 6000 Blackwell 96 GB), local loopback HTTP, TP1; serial; independently validated one-card condition",
    "hard_tier_p50_s": 0.10946263652294874,
    "hard_tier_p95_s": 0.2206121566006913
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.1373250936329588,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.1373250936329588,
    "usd_per_1000_hard": 0.1373250936329588,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 77.59848665223666,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.029812272727272604,
    "probability_fidelity": 77.68325,
    "brier_hard": 0.35137750199999995,
    "brier_standard_judge_v11": 0.14412887595041324,
    "note": null
   },
   "hard": {
    "run": "jevone/runs/results.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.75,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6428571428571429
     },
     "judge_hard": {
      "correct": 28,
      "n": 33,
      "accuracy": 0.8484848484848485
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714285714285715
     },
     "probability": {
      "correct": 16,
      "n": 20,
      "accuracy": 0.8
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 15,
      "n": 30,
      "accuracy": 0.5
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.35137750199999995,
    "ece": 0.029812272727272604,
    "probability_fidelity": 77.68325,
    "calibration_score": 85.86039772727274,
    "onehot": {
     "ece": 0.25,
     "probability_fidelity": 59.564499999999995,
     "calibration_score": 54.78225
    },
    "latency_p50_s": 0.10946263652294874,
    "latency_p95_s": 0.2206121566006913,
    "mean_input_tokens": 1193.9272727272728,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 25.512210128129595,
    "Balanced 33:33:33 (no calibration)": 23.230313587862376,
    "Emphasis on Accuracy 60:20:20": 22.799497677980995,
    "Emphasis on Speed 20:60:20": 28.14679119170592,
    "Emphasis on Cost 20:20:60": 20.099305909734298,
    "Intelligence only": 22.182424137285917
   },
   "rank": 52,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 52,
    "Balanced 33:33:33 (no calibration)": 59,
    "Emphasis on Accuracy 60:20:20": 54,
    "Emphasis on Speed 20:60:20": 57,
    "Emphasis on Cost 20:20:60": 64,
    "Intelligence only": 52
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8961038961038961,
   "sealed_accuracy": 0.33766233766233766,
   "public_minus_sealed_gap_pp": 55.84415584415584,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.3659,
     "long_policy": 0.175,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.2857,
     "probability": 0.3214,
     "safety_judge": 0.375,
     "temporal_numeric": 0.3929,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.156,
     "1/3": 0.3196,
     "2/3": 0.549
    },
    "ece": 0.22276396103896104,
    "mean_tvd_gold_probs": 0.33297878787878793,
    "calibration": 61.074664502164495,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2403,
      "accuracy": 0.3649
     },
     "conf>=0.9": {
      "coverage": 0.0195,
      "accuracy": 0.1667
     }
    }
   },
   "v130_comparison_rank": 17,
   "v130_comparison_score": 69.25524530948347,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "typecastlm",
   "display": "typecastlm (Mikhail Gribov, Qwen3.5-4B computed head)",
   "author": "Mikhail Gribov",
   "repo": "https://github.com/mihail-gribov/typecastlm",
   "class": "system-one-open",
   "licence": "Apache-2.0 (package and weights)",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.9375,
    "judge": 0.410958904109589,
    "hard": 0.5818181818181818
   },
   "speed": {
    "p50_s_raw": 0.027890041936188936,
    "p95_s_raw": 0.05798874022439122,
    "p50_s_adjusted": 0.20578008387237787,
    "p95_s_adjusted": 0.26597748044878244,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge requests through the typesafe adapter to the author's server on loopback inside the offline container, RTX PRO 6000 Blackwell 96 GB, all 842 items serially; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.02126808988764045,
    "basis": "ESTIMATE: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (the exact base weights; one pass, no output), read 2026-09-24; the server's own usage.input_tokens; nothing generated; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 62.08651094359049,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "8949544e51c4d509fcea5714cd93db96f40b3c3217d65c22c733ad910e430985",
    "row_sha256": "7d1f2a3cb04bf31f6e89cf26847e91611bbdced2d06ae03d18d063bc742bd727",
    "raw_results_sha256": "5cf64733ba8267b1edccbe241d57a645fe082164b41d225011c995bf05865885",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026, jevbench-add-requests run 12)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7792207792207793,
   "sealed_accuracy": 0.2987012987012987,
   "public_minus_sealed_gap_pp": 48.05194805194806,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.3902,
     "long_policy": 0.125,
     "multi_hop": 0.3421,
     "paraphrase_robustness": 0.4286,
     "probability": 0.3214,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.1538,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1651,
     "1/3": 0.3608,
     "2/3": 0.3824
    },
    "ece": 0.3025436269869194,
    "mean_tvd_gold_probs": 0.3384578951841035,
    "calibration": 52.82274254210289,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3182,
      "accuracy": 0.2857
     },
     "conf>=0.9": {
      "coverage": 0.0519,
      "accuracy": 0.375
     }
    }
   },
   "v130_comparison_rank": 19,
   "v130_comparison_score": 67.23322895313632,
   "scoring_note": "Offline self-hosted inference of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 Blackwell 96 GB pod, through JevBench's unchanged typesafe adapter against the author's own server on loopback; no operator endpoint; no golds were exposed. Latency is the serial request wall time on the standard+judge items with the self-hosted adjustment. Cost is a labelled estimate: DeepInfra Qwen/Qwen3.5-4B list price $0.03/M input (the exact base weights; one pass, no output), read 2026-09-24, over the server's own usage.input_tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 33.95536030388151,
    "calibration": 62.08651094359049,
    "speed": 92.61751792384507,
    "cost": 60.168145385439914
   },
   "jevbench_score": 25.279442237462565,
   "presets": {
    "JevBench Score (25:25:25:25)": 25.279442237462565,
    "Balanced 33:33:33 (no calibration)": 24.32948520645221,
    "Emphasis on Accuracy 60:20:20": 19.918487711222703,
    "Emphasis on Speed 20:60:20": 29.389243370391767,
    "Emphasis on Cost 20:20:60": 25.59082691549351,
    "Intelligence only": 15.659757080223736
   },
   "rank": 53,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 53,
    "Balanced 33:33:33 (no calibration)": 56,
    "Emphasis on Accuracy 60:20:20": 59,
    "Emphasis on Speed 20:60:20": 52,
    "Emphasis on Cost 20:20:60": 49,
    "Intelligence only": 65
   }
  },
  {
   "key": "simplejev-qwen3.6-35b-a3b",
   "display": "SimpleJev Qwen3.6-35B-A3B",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Featherless AI",
   "repo": "https://github.com/featherless-ai/simple-jev",
   "licence": "Apache-2.0 (Qwen weights); repository licence not stated",
   "underlying": "Qwen3.6-35B-A3B through SimpleJev's direct-logit classifier",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "author's public demo endpoint (Featherless Classifier Demo) — not a production service",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9375,
    "judge": 0.9315068493150684,
    "hard": 0.6636363636363637
   },
   "axes": {
    "intelligence": 45.71661934152712,
    "calibration": 59.795389546888345,
    "speed": 74.98811602029707,
    "cost": 38.11671147049971
   },
   "jevbench_score": 24.86157915547749,
   "speed": {
    "p50_s_raw": 0.8519573211669922,
    "p95_s_raw": 0.9304875515401363,
    "p50_s_adjusted": 1.7039146423339844,
    "p95_s_adjusted": 1.8609751030802726,
    "adjustment": "x2 (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.8764849826693535,
    "hard_tier_p95_s": 0.9821765016764402
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.11555168539325843,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.6-35B-A3B list price list price $0.1/M in, $0.0/M out (the same base weights served as a direct-logit classifier; no output is generated) x 809 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.08088025477707006,
    "usd_per_1000_hard": 0.16503727272727273,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 59.795389546888345,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.16640343503383082,
    "probability_fidelity": 67.39746615885267,
    "brier_hard": 0.4791052568647208,
    "brier_standard_judge_v11": 0.0940923781582149,
    "note": null
   },
   "hard": {
    "run": "runs/simplejev-qwen3.6-35b-a3b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6636363636363637,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 8,
      "n": 14,
      "accuracy": 0.5714285714285714
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 26,
      "n": 35,
      "accuracy": 0.7428571428571429
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.4791052568647208,
    "ece": 0.16640343503383082,
    "probability_fidelity": 67.39746615885267,
    "calibration_score": 67.05838957604325,
    "onehot": {
     "ece": 0.3363636363636363,
     "probability_fidelity": 47.1915,
     "calibration_score": 39.95938636363637
    },
    "latency_p50_s": 0.8764849826693535,
    "latency_p95_s": 0.9821765016764402,
    "mean_input_tokens": 1650.3727272727272,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 24.86157915547749,
    "Balanced 33:33:33 (no calibration)": 23.721218482031585,
    "Emphasis on Accuracy 60:20:20": 23.093250487062967,
    "Emphasis on Speed 20:60:20": 27.568749381733497,
    "Emphasis on Cost 20:20:60": 21.324962180030806,
    "Intelligence only": 22.211257909060556
   },
   "rank": 54,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 54,
    "Balanced 33:33:33 (no calibration)": 57,
    "Emphasis on Accuracy 60:20:20": 53,
    "Emphasis on Speed 20:60:20": 59,
    "Emphasis on Cost 20:20:60": 60,
    "Intelligence only": 51
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8138528138528138,
   "sealed_accuracy": 0.2824675324675325,
   "public_minus_sealed_gap_pp": 53.13852813852813,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3514,
     "judge_hard": 0.2439,
     "long_policy": 0.25,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.4286,
     "probability": 0.25,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.1376,
     "1/3": 0.268,
     "2/3": 0.451
    },
    "ece": 0.3683251373334364,
    "mean_tvd_gold_probs": 0.3579619355615567,
    "calibration": 45.269389488578526,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4058,
      "accuracy": 0.232
     },
     "conf>=0.9": {
      "coverage": 0.1656,
      "accuracy": 0.1373
     }
    }
   },
   "v130_comparison_rank": 40,
   "v130_comparison_score": 62.48305419264297,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (Featherless demo), as for every API measurement"
  },
  {
   "key": "kev-0.6b",
   "display": "kev 0.6B (research preview)",
   "class": "jev-rebuild",
   "open": true,
   "author": "Jared Palmer",
   "repo": "https://github.com/jaredpalmer/kev",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-0.6B-Base + LoRA + learned pointer head; jaredpalmer/kev-0.6b",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.8125,
    "judge": 0.6643835616438356,
    "hard": 0.4
   },
   "axes": {
    "intelligence": 34.20432736698408,
    "calibration": 49.957886527801605,
    "speed": 75.55649753874303,
    "cost": 76.08691668696139
   },
   "jevbench_score": 24.750427849352384,
   "speed": {
    "p50_s_raw": 0.5904003903269768,
    "p95_s_raw": 0.9702187780290842,
    "p50_s_adjusted": 1.3308007806539535,
    "p95_s_adjusted": 2.0904375560581685,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.6060735657811165,
    "hard_tier_p95_s": 1.4071057934314009
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0062676217228464426,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0027941082802547773,
    "usd_per_1000_hard": 0.011225272727272728,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 49.957886527801605,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.26937623307785324,
    "probability_fidelity": 56.05957240724073,
    "brier_hard": 0.8234993914874135,
    "brier_standard_judge_v11": 0.3857487253984507,
    "note": null
   },
   "hard": {
    "run": "runs/kev-0.6b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 11,
      "n": 35,
      "accuracy": 0.3142857142857143
     },
     "probability": {
      "correct": 6,
      "n": 20,
      "accuracy": 0.3
     },
     "routing_hard": {
      "correct": 5,
      "n": 10,
      "accuracy": 0.5
     },
     "temporal_numeric": {
      "correct": 10,
      "n": 30,
      "accuracy": 0.3333333333333333
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8234993914874135,
    "ece": 0.26937623307785324,
    "probability_fidelity": 56.05957240724073,
    "calibration_score": 51.092162895835045,
    "onehot": {
     "ece": 0.6,
     "probability_fidelity": 34.61249999999999,
     "calibration_score": 17.306249999999995
    },
    "latency_p50_s": 0.6060735657811165,
    "latency_p95_s": 1.4071057934314009,
    "mean_input_tokens": 1122.5272727272727,
    "mean_output_tokens": 62.20909090909091,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 24.750427849352384,
    "Balanced 33:33:33 (no calibration)": 25.244033869325918,
    "Emphasis on Accuracy 60:20:20": 20.50968612710913,
    "Emphasis on Speed 20:60:20": 28.505713072913565,
    "Emphasis on Cost 20:20:60": 28.56993990748217,
    "Intelligence only": 16.006749722374877
   },
   "rank": 55,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 55,
    "Balanced 33:33:33 (no calibration)": 53,
    "Emphasis on Accuracy 60:20:20": 57,
    "Emphasis on Speed 20:60:20": 55,
    "Emphasis on Cost 20:20:60": 42,
    "Intelligence only": 64
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6666666666666666,
   "sealed_accuracy": 0.24025974025974026,
   "public_minus_sealed_gap_pp": 42.640692640692635,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 307,
    "by_family": {
     "ambiguous_abstain": 0.1351,
     "judge_hard": 0.3659,
     "long_policy": 0.225,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.6429,
     "probability": 0.0,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2477,
     "1/3": 0.1856,
     "2/3": 0.2843
    },
    "ece": 0.31029333226482253,
    "mean_tvd_gold_probs": 0.4256266596356605,
    "calibration": 47.689333791734725,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.289,
      "accuracy": 0.3596
     },
     "conf>=0.9": {
      "coverage": 0.1299,
      "accuracy": 0.35
     }
    }
   },
   "v130_comparison_rank": 39,
   "v130_comparison_score": 62.489006255662105,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 307/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "raw-qwen3-8b",
   "display": "Raw Qwen3 8B direct logits",
   "class": "raw-logit-control",
   "open": "yes",
   "author": "Alibaba Qwen / neutral reproduction",
   "repo": "https://huggingface.co/Qwen/Qwen3-8B",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-8B BF16",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.8645833333333334,
    "judge": 0.9657534246575342,
    "hard": 0.4636363636363636
   },
   "axes": {
    "intelligence": 45.65959354886205,
    "calibration": 24.10350736927811,
    "speed": 86.33902733346537,
    "cost": 41.88109987449869
   },
   "jevbench_score": 23.676144280945373,
   "speed": {
    "p50_s_raw": 0.08492440730333328,
    "p95_s_raw": 0.288180502736941,
    "p50_s_adjusted": 0.3198488146066666,
    "p95_s_adjusted": 0.726361005473882,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
    "hard_tier_p50_s": 0.15026657423004508,
    "hard_tier_p95_s": 0.6204535091295839
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0865558988764045,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.0865558988764045,
    "usd_per_1000_hard": 0.0865558988764045,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 24.10350736927811,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.47954385778952974,
    "probability_fidelity": 47.80992141349076,
    "brier_hard": 0.9930717989711458,
    "brier_standard_judge_v11": 0.1352067143558659,
    "note": null
   },
   "hard": {
    "run": "raw-controls/runs/qwen3-8/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4636363636363636,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 23,
      "n": 33,
      "accuracy": 0.696969696969697
     },
     "long_policy": {
      "correct": 9,
      "n": 38,
      "accuracy": 0.23684210526315788
     },
     "multi_hop": {
      "correct": 17,
      "n": 35,
      "accuracy": 0.4857142857142857
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 3,
      "n": 30,
      "accuracy": 0.1
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.9930717989711458,
    "ece": 0.47954385778952974,
    "probability_fidelity": 47.80992141349076,
    "calibration_score": 25.95057492779241,
    "onehot": {
     "ece": 0.5363636363636364,
     "probability_fidelity": 40.8955,
     "calibration_score": 20.44775
    },
    "latency_p50_s": 0.15026657423004508,
    "latency_p95_s": 0.6204535091295839,
    "mean_input_tokens": 1236.2227272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 23.676144280945373,
    "Balanced 33:33:33 (no calibration)": 30.600465022266736,
    "Emphasis on Accuracy 60:20:20": 28.918025682951072,
    "Emphasis on Speed 20:60:20": 36.32947700761297,
    "Emphasis on Cost 20:20:60": 27.830841016407017,
    "Intelligence only": 26.714820672376973
   },
   "rank": 56,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 56,
    "Balanced 33:33:33 (no calibration)": 42,
    "Emphasis on Accuracy 60:20:20": 39,
    "Emphasis on Speed 20:60:20": 38,
    "Emphasis on Cost 20:20:60": 44,
    "Intelligence only": 41
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.683982683982684,
   "sealed_accuracy": 0.262987012987013,
   "public_minus_sealed_gap_pp": 42.0995670995671,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.3659,
     "long_policy": 0.15,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.5714,
     "probability": 0.2143,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.2784,
     "2/3": 0.3039
    },
    "ece": 0.6451195930737063,
    "mean_tvd_gold_probs": 0.5918125549550098,
    "calibration": 20.409372252249508,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8669,
      "accuracy": 0.2697
     },
     "conf>=0.9": {
      "coverage": 0.7532,
      "accuracy": 0.2672
     }
    }
   },
   "v130_comparison_rank": 58,
   "v130_comparison_score": 50.415533525795176,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "system-one-sg",
   "display": "system-one (Qwen3-8B, Sean Goedecke)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Sean Goedecke",
   "repo": "https://github.com/sgoedecke/system-one",
   "licence": "no licence file in the repository as of 19 Sep; Qwen3 weights Apache-2.0",
   "underlying": "Qwen/Qwen3-8B (frozen, BF16), the model of the author's demos",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (RTX PRO 4500 Blackwell 32 GB (EU-RO-1)), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.90625,
    "judge": 0.9178082191780822,
    "hard": 0.5
   },
   "axes": {
    "intelligence": 43.58090741395149,
    "calibration": 32.828064074368044,
    "speed": 84.3621367799104,
    "cost": 41.45387606228526
   },
   "jevbench_score": 23.369044226391765,
   "speed": {
    "p50_s_raw": 0.1661309413611889,
    "p95_s_raw": 0.30472867079079147,
    "p50_s_adjusted": 0.4822618827223778,
    "p95_s_adjusted": 0.759457341581583,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RunPod RTX PRO 4500 Blackwell 32 GB (EU-RO-1)",
    "measured_where": "from a Hetzner server in Germany over the internet to the pod's public TCP port (plain HTTP, one connection per request) through a thin transport around the author's library; model loaded before timing",
    "hard_tier_p50_s": 0.20696432143449783,
    "hard_tier_p95_s": 0.65277008600533
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.08944116853932584,
    "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3-8b list price $0.117/M in, $0.455/M out (same weights, listed on OpenRouter) x 412 input and 1 output tokens per decision (input tokens measured) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3-8b $0.117/M in, $0.455/M out x 1258 in / 1 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.04869626114649681,
    "usd_per_1000_hard": 0.14759526363636366,
    "self_host_sensitivity": {
     "usd_per_1000": 0.030492280369226854,
     "score": 62.895252391243304,
     "machine": "1x RTX PRO 4500 Blackwell 32 GB (EU-RO-1) (on-demand, RunPod secure)",
     "usd_per_h": 0.72,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 0.18295368221536112,
     "decisions_per_hour": 23612.533771879913
    }
   },
   "calibration": {
    "score": 32.828064074368044,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.42428537094007235,
    "probability_fidelity": 58.38569410230225,
    "brier_hard": 0.9241372110675431,
    "brier_standard_judge_v11": 0.1602412905726563,
    "note": null
   },
   "hard": {
    "run": "runs-gpu/system-one-sg--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.5,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 5,
      "n": 14,
      "accuracy": 0.35714285714285715
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 12,
      "n": 38,
      "accuracy": 0.3157894736842105
     },
     "multi_hop": {
      "correct": 17,
      "n": 35,
      "accuracy": 0.4857142857142857
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.9241372110675431,
    "ece": 0.42428537094007235,
    "probability_fidelity": 58.38569410230225,
    "calibration_score": 36.764309957143894,
    "onehot": {
     "ece": 0.5,
     "probability_fidelity": 45.96800000000001,
     "calibration_score": 22.984000000000005
    },
    "latency_p50_s": 0.20696432143449783,
    "latency_p95_s": 0.65277008600533,
    "mean_input_tokens": 1257.6090909090908,
    "mean_output_tokens": 1.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 23.369044226391765,
    "Balanced 33:33:33 (no calibration)": 26.587747581123555,
    "Emphasis on Accuracy 60:20:20": 24.91105456622965,
    "Emphasis on Speed 20:60:20": 31.599124759816654,
    "Emphasis on Cost 20:20:60": 24.363704596451157,
    "Intelligence only": 22.758261208173618
   },
   "rank": 57,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 57,
    "Balanced 33:33:33 (no calibration)": 50,
    "Emphasis on Accuracy 60:20:20": 50,
    "Emphasis on Speed 20:60:20": 47,
    "Emphasis on Cost 20:20:60": 52,
    "Intelligence only": 49
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7186147186147186,
   "sealed_accuracy": 0.2435064935064935,
   "public_minus_sealed_gap_pp": 47.51082251082251,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.2927,
     "long_policy": 0.125,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.2857,
     "probability": 0.3214,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.2062,
     "2/3": 0.3431
    },
    "ece": 0.6526329163774862,
    "mean_tvd_gold_probs": 0.5008885538236733,
    "calibration": 24.955572308816336,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8571,
      "accuracy": 0.2462
     },
     "conf>=0.9": {
      "coverage": 0.6916,
      "accuracy": 0.2441
     }
    }
   },
   "v130_comparison_rank": 53,
   "v130_comparison_score": 54.830990505255514,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "opendecision",
   "display": "OpenDecision (ModernBERT-large zero-shot)",
   "class": "classifier",
   "open": "yes",
   "author": "Deepan Wadhwa",
   "repo": "https://github.com/deepanwadhwa/OpenDecision",
   "licence": "Apache-2.0",
   "underlying": "MoritzLaurer/ModernBERT-large-zeroshot-v2.0 (~400M) used as a zero-shot NLI decision engine",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.875,
    "standard": 0.625,
    "judge": 0.7123287671232876,
    "hard": 0.33181818181818185
   },
   "axes": {
    "intelligence": 31.79943294249068,
    "calibration": 57.06329219214104,
    "speed": 79.89955851182286,
    "cost": 75.34414706615897
   },
   "jevbench_score": 21.641705888024706,
   "speed": {
    "p50_s_raw": 0.3378134034574032,
    "p95_s_raw": 0.5447697393596171,
    "p50_s_adjusted": 0.8256268069148064,
    "p95_s_adjusted": 1.2395394787192342,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.35639120638370514,
    "hard_tier_p95_s": 0.49497928358614446
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.006635318352059925,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0032921656050955415,
    "usd_per_1000_hard": 0.011406909090909093,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 57.06329219214104,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2598902093754573,
    "probability_fidelity": 64.16270788765885,
    "brier_hard": 0.8135612674812179,
    "brier_standard_judge_v11": 0.4328156823860972,
    "note": null
   },
   "hard": {
    "run": "runs/opendecision--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.33181818181818185,
    "by_family": {
     "adversarial": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 16,
      "n": 33,
      "accuracy": 0.48484848484848486
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 5,
      "n": 35,
      "accuracy": 0.14285714285714285
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 1,
      "n": 10,
      "accuracy": 0.1
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 3,
      "n": 12,
      "accuracy": 0.25
     },
     "trap": {
      "correct": 10,
      "n": 16,
      "accuracy": 0.625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8135612674812179,
    "ece": 0.2598902093754573,
    "probability_fidelity": 64.16270788765885,
    "calibration_score": 56.092333006283695,
    "onehot": {
     "ece": 0.6681818181818182,
     "probability_fidelity": 41.01049999999999,
     "calibration_score": 20.505249999999997
    },
    "latency_p50_s": 0.35639120638370514,
    "latency_p95_s": 0.49497928358614446,
    "mean_input_tokens": 1140.6909090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 21.641705888024706,
    "Balanced 33:33:33 (no calibration)": 21.201004588708706,
    "Emphasis on Accuracy 60:20:20": 16.835237948336356,
    "Emphasis on Speed 20:60:20": 24.5835619187879,
    "Emphasis on Cost 20:20:60": 24.13947539642414,
    "Intelligence only": 12.862284694787563
   },
   "rank": 58,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 58,
    "Balanced 33:33:33 (no calibration)": 60,
    "Emphasis on Accuracy 60:20:20": 63,
    "Emphasis on Speed 20:60:20": 60,
    "Emphasis on Cost 20:20:60": 54,
    "Intelligence only": 67
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5324675324675324,
   "sealed_accuracy": 0.2564935064935065,
   "public_minus_sealed_gap_pp": 27.597402597402592,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.4634,
     "long_policy": 0.15,
     "multi_hop": 0.0789,
     "paraphrase_robustness": 0.2143,
     "probability": 0.25,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1927,
     "1/3": 0.2371,
     "2/3": 0.3431
    },
    "ece": 0.24082116910268542,
    "mean_tvd_gold_probs": 0.3382534505175144,
    "calibration": 59.00521056385574,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2305,
      "accuracy": 0.2958
     },
     "conf>=0.9": {
      "coverage": 0.1104,
      "accuracy": 0.2353
     }
    }
   },
   "v130_comparison_rank": 59,
   "v130_comparison_score": 40.58744897671299,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "litjev",
   "display": "LitJev (Qwen3.8-27B)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Zhengxu Yu",
   "repo": "https://github.com/zhengxuyu/litjev",
   "licence": "Apache-2.0 (code); Apache-2.0 base weights",
   "underlying": "Qwen/Qwen3.8-27B, frozen, read at the output head",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.8835616438356164,
    "hard": 0.7318181818181818
   },
   "axes": {
    "intelligence": 46.25717739910522,
    "calibration": 76.62350985013104,
    "speed": 66.72450940291681,
    "cost": 33.63149944045428
   },
   "jevbench_score": 19.51031342209139,
   "speed": {
    "p50_s_raw": 2.025244139134884,
    "p95_s_raw": 2.4555754292756315,
    "p50_s_adjusted": 4.200488278269768,
    "p95_s_adjusted": 5.0611508585512635,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 2.2890688702464104,
    "hard_tier_p95_s": 2.9710183084011077
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.16303594007490638,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B (as the reflex-27b row) list price $0.214/M in, $0.0/M out (the exact base weights; nothing is generated) x 418 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.08940292993630572,
    "usd_per_1000_hard": 0.2681303272727273,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 76.62350985013104,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.047705096932661215,
    "probability_fidelity": 76.57508897365247,
    "brier_hard": 0.3296182961504934,
    "brier_standard_judge_v11": 0.11729020229951063,
    "note": null
   },
   "hard": {
    "run": "runs/litjev--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7318181818181818,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 24,
      "n": 33,
      "accuracy": 0.7272727272727273
     },
     "long_policy": {
      "correct": 26,
      "n": 38,
      "accuracy": 0.6842105263157895
     },
     "multi_hop": {
      "correct": 29,
      "n": 35,
      "accuracy": 0.8285714285714286
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3296182961504934,
    "ece": 0.047705096932661215,
    "probability_fidelity": 76.57508897365247,
    "calibration_score": 83.51703479356011,
    "onehot": {
     "ece": 0.2681818181818182,
     "probability_fidelity": 53.50899999999999,
     "calibration_score": 49.93631818181818
    },
    "latency_p50_s": 2.2890688702464104,
    "latency_p95_s": 2.9710183084011077,
    "mean_input_tokens": 1252.9454545454546,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 19.51031342209139,
    "Balanced 33:33:33 (no calibration)": 17.51141001937946,
    "Emphasis on Accuracy 60:20:20": 17.66956875746804,
    "Emphasis on Speed 20:60:20": 20.102705183040463,
    "Emphasis on Cost 20:20:60": 15.389860500685673,
    "Intelligence only": 17.912237121958725
   },
   "rank": 59,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 59,
    "Balanced 33:33:33 (no calibration)": 66,
    "Emphasis on Accuracy 60:20:20": 62,
    "Emphasis on Speed 20:60:20": 66,
    "Emphasis on Cost 20:20:60": 69,
    "Intelligence only": 58
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8614718614718615,
   "sealed_accuracy": 0.30844155844155846,
   "public_minus_sealed_gap_pp": 55.3030303030303,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.3171,
     "long_policy": 0.25,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.2857,
     "probability": 0.2143,
     "safety_judge": 0.5,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1376,
     "1/3": 0.2371,
     "2/3": 0.5588
    },
    "ece": 0.22889566781015255,
    "mean_tvd_gold_probs": 0.2854794651142363,
    "calibration": 62.83645996327293,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1851,
      "accuracy": 0.3509
     },
     "conf>=0.9": {
      "coverage": 0.0195,
      "accuracy": 0.1667
     }
    }
   },
   "v130_comparison_rank": 38,
   "v130_comparison_score": 62.69075853252832,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "openjev-verdict-1.4",
   "display": "openJev Verdict 1.4",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Hemant (heman10x)",
   "repo": "https://huggingface.co/heman10x/rlcd-modernbert-151m",
   "licence": "Apache-2.0",
   "underlying": "GLiClass ModernBERT-base fine-tuned decision model, 151M; v1.4 fixed inference engine",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8611111111111112,
    "standard": 0.6770833333333334,
    "judge": 0.5616438356164384,
    "hard": 0.37727272727272726
   },
   "axes": {
    "intelligence": 29.409876443502302,
    "calibration": 72.03302398440775,
    "speed": 78.08595973190066,
    "cost": 82.35883967864214
   },
   "jevbench_score": 19.001051642863935,
   "speed": {
    "p50_s_raw": 0.31351133808493614,
    "p95_s_raw": 0.9248626325279473,
    "p50_s_adjusted": 0.7770226761698723,
    "p95_s_adjusted": 1.9997252650558945,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 0.6292409487068653,
    "hard_tier_p95_s": 0.8346716947853564
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.003872921348314607,
    "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.0022600000000000003,
    "usd_per_1000_hard": 0.006175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 72.03302398440775,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.11565818173180892,
    "probability_fidelity": 71.40255301097424,
    "brier_hard": 0.6941169946619116,
    "brier_standard_judge_v11": 0.5357783992553375,
    "note": null
   },
   "hard": {
    "run": "runs/openjev-verdict-1.4--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.37727272727272726,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 6,
      "n": 35,
      "accuracy": 0.17142857142857143
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 3,
      "n": 10,
      "accuracy": 0.3
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6941169946619116,
    "ece": 0.11565818173180892,
    "probability_fidelity": 71.40255301097424,
    "calibration_score": 74.13545833230623,
    "onehot": {
     "ece": 0.6227272727272728,
     "probability_fidelity": 41.260499999999986,
     "calibration_score": 20.630249999999993
    },
    "latency_p50_s": 0.6292409487068653,
    "latency_p95_s": 0.8346716947853564,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 19.001051642863935,
    "Balanced 33:33:33 (no calibration)": 17.60676781983142,
    "Emphasis on Accuracy 60:20:20": 13.62595043620767,
    "Emphasis on Speed 20:60:20": 20.456631316251524,
    "Emphasis on Cost 20:20:60": 20.783217661813037,
    "Intelligence only": 10.175121204989262
   },
   "rank": 60,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 60,
    "Balanced 33:33:33 (no calibration)": 65,
    "Emphasis on Accuracy 60:20:20": 71,
    "Emphasis on Speed 20:60:20": 65,
    "Emphasis on Cost 20:20:60": 63,
    "Intelligence only": 74
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5757575757575758,
   "sealed_accuracy": 0.2792207792207792,
   "public_minus_sealed_gap_pp": 29.65367965367966,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.3415,
     "long_policy": 0.375,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.2857,
     "probability": 0.2143,
     "safety_judge": 0.1875,
     "temporal_numeric": 0.125,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.1856,
     "2/3": 0.4412
    },
    "ece": 0.18085664849055377,
    "mean_tvd_gold_probs": 0.28172359724667667,
    "calibration": 67.82815528861079,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0552,
      "accuracy": 0.2353
     },
     "conf>=0.9": {
      "coverage": 0.0032,
      "accuracy": 0.0
     }
    }
   },
   "v130_comparison_rank": 60,
   "v130_comparison_score": 38.93683519894202,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "kev-0.5b",
   "display": "kev 0.5B",
   "class": "jev-rebuild",
   "open": true,
   "author": "Jared Palmer",
   "repo": "https://github.com/jaredpalmer/kev",
   "licence": "Apache-2.0",
   "underlying": "Qwen2.5-0.5B + LoRA + learned pointer head; jaredpalmer/kev-0.5b",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud CA), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9583333333333334,
    "standard": 0.5208333333333334,
    "judge": 0.7123287671232876,
    "hard": 0.3090909090909091
   },
   "axes": {
    "intelligence": 30.536372423403222,
    "calibration": 49.68828397919909,
    "speed": 76.95661397000733,
    "cost": 76.08687775906971
   },
   "jevbench_score": 18.88295792074435,
   "speed": {
    "p50_s_raw": 0.43036095052957535,
    "p95_s_raw": 0.9219581566751004,
    "p50_s_adjusted": 1.0107219010591506,
    "p95_s_adjusted": 1.9939163133502007,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.5884033516049385,
    "hard_tier_p95_s": 1.2630689594894642
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0062676404494382025,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3-Embedding-0.6B size-class reference list price $0.01/M in, $0.0/M out (a <=0.6B one-pass model with no generated output) x 279 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0027941401273885347,
    "usd_per_1000_hard": 0.011225272727272728,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 49.68828397919909,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.3138481166298448,
    "probability_fidelity": 57.50547134713471,
    "brier_hard": 0.8464522868993845,
    "brier_standard_judge_v11": 0.46104057750389293,
    "note": null
   },
   "hard": {
    "run": "runs/kev-0.5b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.3090909090909091,
    "by_family": {
     "adversarial": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 7,
      "n": 38,
      "accuracy": 0.18421052631578946
     },
     "multi_hop": {
      "correct": 11,
      "n": 35,
      "accuracy": 0.3142857142857143
     },
     "probability": {
      "correct": 4,
      "n": 20,
      "accuracy": 0.2
     },
     "routing_hard": {
      "correct": 0,
      "n": 10,
      "accuracy": 0.0
     },
     "temporal_numeric": {
      "correct": 4,
      "n": 30,
      "accuracy": 0.13333333333333333
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8464522868993845,
    "ece": 0.3138481166298448,
    "probability_fidelity": 57.50547134713471,
    "calibration_score": 47.367924010582875,
    "onehot": {
     "ece": 0.6909090909090909,
     "probability_fidelity": 30.452500000000004,
     "calibration_score": 15.226250000000002
    },
    "latency_p50_s": 0.5884033516049385,
    "latency_p95_s": 1.2630689594894642,
    "mean_input_tokens": 1122.5272727272727,
    "mean_output_tokens": 62.163636363636364,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 18.88295792074435,
    "Balanced 33:33:33 (no calibration)": 19.002519035980097,
    "Emphasis on Accuracy 60:20:20": 14.99380571820519,
    "Emphasis on Speed 20:60:20": 21.973119189886095,
    "Emphasis on Cost 20:20:60": 21.89647787888633,
    "Intelligence only": 11.389700975579167
   },
   "rank": 61,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 61,
    "Balanced 33:33:33 (no calibration)": 61,
    "Emphasis on Accuracy 60:20:20": 68,
    "Emphasis on Speed 20:60:20": 61,
    "Emphasis on Cost 20:20:60": 57,
    "Intelligence only": 70
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.4935064935064935,
   "sealed_accuracy": 0.2727272727272727,
   "public_minus_sealed_gap_pp": 22.07792207792208,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 307,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.4878,
     "long_policy": 0.175,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.5,
     "probability": 0.1071,
     "safety_judge": 0.375,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.2477,
     "1/3": 0.2474,
     "2/3": 0.3235
    },
    "ece": 0.2630974270065443,
    "mean_tvd_gold_probs": 0.3872250676582809,
    "calibration": 54.32900391643152,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.263,
      "accuracy": 0.4938
     },
     "conf>=0.9": {
      "coverage": 0.026,
      "accuracy": 0.5
     }
    }
   },
   "v130_comparison_rank": 63,
   "v130_comparison_score": 33.24350332804402,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 307/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "nimble-9b",
   "display": "Bespoke Nimble 9B (Bespoke Labs)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Bespoke Labs",
   "repo": "https://github.com/bespokelabsai/nimble",
   "licence": "Apache-2.0 (weights); repository without a licence file as of 19 Sep",
   "underlying": "bespokelabs/Bespoke-Nimble-9B (LoRA, adapter unchanged since 93ec5d6), merged into Qwen/Qwen3.5-9B@c202236 with the author's PEFT safe-merge",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (A40 48 GB, Canada), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9479166666666666,
    "judge": 0.8904109589041096,
    "hard": 0.6545454545454545
   },
   "axes": {
    "intelligence": 46.27509427808829,
    "calibration": 56.44746075341965,
    "speed": 78.68440871747626,
    "cost": 33.41009520749539
   },
   "jevbench_score": 18.66373818414425,
   "speed": {
    "p50_s_raw": 0.3889440931379795,
    "p95_s_raw": 0.6545137587934732,
    "p50_s_adjusted": 0.927888186275959,
    "p95_s_adjusted": 1.4590275175869463,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.444174911826849,
    "hard_tier_p95_s": 0.977684601396322
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.16583014981273408,
    "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.5-9b list price $0.1/M in, $0.15/M out (a LoRA merge of Qwen3.5-9B; the base weights are listed on OpenRouter (size class dense_9B), as in the v1.1.3 row) x 970 input and 1 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.09719840764331211,
    "usd_per_1000_hard": 0.26378636363636365,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 56.44746075341965,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.19055518585790857,
    "probability_fidelity": 68.71798264250582,
    "brier_hard": 0.5272551742334669,
    "brier_standard_judge_v11": 0.1554427753099633,
    "note": null
   },
   "hard": {
    "run": "runs/nimble-9b--hard-r4 (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6545454545454545,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 25,
      "n": 33,
      "accuracy": 0.7575757575757576
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5272551742334669,
    "ece": 0.19055518585790857,
    "probability_fidelity": 68.71798264250582,
    "calibration_score": 65.30347273546205,
    "onehot": {
     "ece": 0.34545454545454546,
     "probability_fidelity": 51.72950000000001,
     "calibration_score": 41.319295454545454
    },
    "latency_p50_s": 0.444174911826849,
    "latency_p95_s": 0.977684601396322,
    "mean_input_tokens": 2636.3636363636365,
    "mean_output_tokens": 2.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 18.66373818414425,
    "Balanced 33:33:33 (no calibration)": 17.857406085425893,
    "Emphasis on Accuracy 60:20:20": 17.793187521996664,
    "Emphasis on Speed 20:60:20": 21.32567222569046,
    "Emphasis on Cost 20:20:60": 15.40727611794657,
    "Intelligence only": 17.697721062548045
   },
   "rank": 62,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 62,
    "Balanced 33:33:33 (no calibration)": 64,
    "Emphasis on Accuracy 60:20:20": 61,
    "Emphasis on Speed 20:60:20": 63,
    "Emphasis on Cost 20:20:60": 68,
    "Intelligence only": 59
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7965367965367965,
   "sealed_accuracy": 0.288961038961039,
   "public_minus_sealed_gap_pp": 50.75757575757576,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2703,
     "judge_hard": 0.2927,
     "long_policy": 0.225,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.5,
     "probability": 0.3929,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1651,
     "1/3": 0.2371,
     "2/3": 0.4706
    },
    "ece": 0.4030108266010678,
    "mean_tvd_gold_probs": 0.4192696110111676,
    "calibration": 38.73543678933484,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4903,
      "accuracy": 0.2781
     },
     "conf>=0.9": {
      "coverage": 0.2435,
      "accuracy": 0.32
     }
    }
   },
   "v130_comparison_rank": 45,
   "v130_comparison_score": 60.47515137925396,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "gpt-5.6-luna",
   "display": "GPT-5.6 Luna (low reasoning effort)",
   "class": "llm-baseline",
   "open": "no",
   "author": "OpenAI",
   "repo": null,
   "licence": "proprietary API",
   "underlying": "closed",
   "has_distribution": true,
   "probability_source": [
    "verbalized"
   ],
   "endpoint_condition": "production API (OpenAI), reasoning effort low",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9791666666666666,
    "judge": 0.9657534246575342,
    "hard": 0.9454545454545454
   },
   "axes": {
    "intelligence": 93.13761299887643,
    "calibration": 87.37467197114547,
    "speed": 77.5464346646158,
    "cost": 28.490318789992514
   },
   "jevbench_score": 18.506333238992966,
   "speed": {
    "p50_s_raw": 0.9680032916367054,
    "p95_s_raw": 1.8175220962613816,
    "p50_s_adjusted": 0.9680032916367054,
    "p95_s_adjusted": 1.8175220962613816,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 1.2202091813087463,
    "hard_tier_p95_s": 3.4094011016190024
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.24191123595505612,
    "basis": "public tariff x measured tokens (https://platform.openai.com/docs/pricing (standard tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
    "usd_per_1000_v11_tiers": 0.15501464968152867,
    "usd_per_1000_hard": 0.3659363636363635,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 87.37467197114547,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.07068160957409142,
    "probability_fidelity": 93.72047507875,
    "brier_hard": 0.11785378708136751,
    "brier_standard_judge_v11": 0.05621211608720952,
    "note": null
   },
   "hard": {
    "run": "runs/gpt-5.6-luna--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.9454545454545454,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 13,
      "n": 14,
      "accuracy": 0.9285714285714286
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9090909090909091
     },
     "long_policy": {
      "correct": 36,
      "n": 38,
      "accuracy": 0.9473684210526315
     },
     "multi_hop": {
      "correct": 33,
      "n": 35,
      "accuracy": 0.9428571428571428
     },
     "probability": {
      "correct": 18,
      "n": 20,
      "accuracy": 0.9
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 28,
      "n": 30,
      "accuracy": 0.9333333333333333
     },
     "tradeoff": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.11785378708136751,
    "ece": 0.07068160957409142,
    "probability_fidelity": 93.72047507875,
    "calibration_score": 89.79207658196586,
    "onehot": {
     "ece": 0.054545454545454564,
     "probability_fidelity": 62.242,
     "calibration_score": 75.66645454545454
    },
    "latency_p50_s": 1.2202091813087463,
    "latency_p95_s": 3.4094011016190024,
    "mean_input_tokens": 1185.5545454545454,
    "mean_output_tokens": 107.35454545454546,
    "charged_usd": 0.08050599999999997
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 18.506333238992966,
    "Balanced 33:33:33 (no calibration)": 16.584466226695312,
    "Emphasis on Accuracy 60:20:20": 20.240452772554722,
    "Emphasis on Speed 20:60:20": 19.206578762546428,
    "Emphasis on Cost 20:20:60": 12.59118185766796,
    "Intelligence only": 30.239855541859214
   },
   "rank": 63,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 63,
    "Balanced 33:33:33 (no calibration)": 67,
    "Emphasis on Accuracy 60:20:20": 58,
    "Emphasis on Speed 20:60:20": 68,
    "Emphasis on Cost 20:20:60": 73,
    "Intelligence only": 32
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.974025974025974,
   "sealed_accuracy": 0.8896103896103896,
   "public_minus_sealed_gap_pp": 8.441558441558438,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.8649,
     "judge_hard": 0.878,
     "long_policy": 0.85,
     "multi_hop": 0.8947,
     "paraphrase_robustness": 0.9286,
     "probability": 1.0,
     "safety_judge": 0.9375,
     "temporal_numeric": 0.8393,
     "tradeoff": 0.9231,
     "trap_adversarial": 0.9167
    },
    "by_panel_stratum": {
     "0/3": 0.8991,
     "1/3": 0.866,
     "2/3": 0.902
    },
    "ece": 0.1549736113861939,
    "mean_tvd_gold_probs": 0.03925552223751804,
    "calibration": 82.5398627495047,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8864,
      "accuracy": 0.8828
     },
     "conf>=0.9": {
      "coverage": 0.8669,
      "accuracy": 0.8839
     }
    }
   },
   "v130_comparison_rank": 31,
   "v130_comparison_score": 65.94418818052425,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (OpenAI API), as for every API measurement"
  },
  {
   "key": "openjev-verdict",
   "display": "openJev Verdict (heman10x, ModernBERT-base 151M)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Hemant (heman10x)",
   "repo": "https://github.com/Heman10x-NGU/openJev-verdict-2.0",
   "licence": "Apache-2.0",
   "underlying": "GLiClass ModernBERT-base (knowledgator/gliclass-modern-base-v2.0) fine-tuned, 151M",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8611111111111112,
    "standard": 0.65625,
    "judge": 0.6095890410958904,
    "hard": 0.38181818181818183
   },
   "axes": {
    "intelligence": 30.017651584940374,
    "calibration": 46.99073252218098,
    "speed": 76.68207818819491,
    "cost": 83.05556459946699
   },
   "jevbench_score": 18.094581304233536,
   "speed": {
    "p50_s_raw": 0.27805980294942856,
    "p95_s_raw": 1.4451411496847864,
    "p50_s_adjusted": 0.7061196058988571,
    "p95_s_adjusted": 3.0402822993695726,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 0.750414352864027,
    "hard_tier_p95_s": 1.7721023652702563
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.00367125468164794,
    "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]",
    "usd_per_1000_v11_tiers": 0.0019170382165605096,
    "usd_per_1000_hard": 0.006175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 46.99073252218098,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.30116600532276583,
    "probability_fidelity": 62.826009020276516,
    "brier_hard": 0.8456557054244584,
    "brier_standard_judge_v11": 0.4858771320072627,
    "note": null
   },
   "hard": {
    "run": "runs/openjev-verdict--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.38181818181818183,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 5,
      "n": 35,
      "accuracy": 0.14285714285714285
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 3,
      "n": 10,
      "accuracy": 0.3
     },
     "temporal_numeric": {
      "correct": 12,
      "n": 30,
      "accuracy": 0.4
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 8,
      "n": 16,
      "accuracy": 0.5
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8456557054244584,
    "ece": 0.30116600532276583,
    "probability_fidelity": 62.826009020276516,
    "calibration_score": 51.29640397786167,
    "onehot": {
     "ece": 0.6181818181818182,
     "probability_fidelity": 43.343999999999994,
     "calibration_score": 21.671999999999997
    },
    "latency_p50_s": 0.750414352864027,
    "latency_p95_s": 1.7721023652702563,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 18.094581304233536,
    "Balanced 33:33:33 (no calibration)": 18.516593021077217,
    "Emphasis on Accuracy 60:20:20": 14.414392212669837,
    "Emphasis on Speed 20:60:20": 21.332799104647492,
    "Emphasis on Cost 20:20:60": 21.85048942262095,
    "Intelligence only": 10.819074930759779
   },
   "rank": 64,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 64,
    "Balanced 33:33:33 (no calibration)": 63,
    "Emphasis on Accuracy 60:20:20": 70,
    "Emphasis on Speed 20:60:20": 62,
    "Emphasis on Cost 20:20:60": 58,
    "Intelligence only": 72
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5541125541125541,
   "sealed_accuracy": 0.24675324675324675,
   "public_minus_sealed_gap_pp": 30.73593073593074,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.3415,
     "long_policy": 0.275,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.1429,
     "probability": 0.2857,
     "safety_judge": 0.125,
     "temporal_numeric": 0.1607,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.211,
     "1/3": 0.1649,
     "2/3": 0.3627
    },
    "ece": 0.3988990974201011,
    "mean_tvd_gold_probs": 0.43461401294340607,
    "calibration": 38.37938961081959,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4253,
      "accuracy": 0.3282
     },
     "conf>=0.9": {
      "coverage": 0.1851,
      "accuracy": 0.3509
     }
    }
   },
   "v130_comparison_rank": 61,
   "v130_comparison_score": 38.059545049974176,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "raw-qwen3-1.7b",
   "display": "Raw Qwen3 1.7B direct logits",
   "class": "raw-logit-control",
   "open": "yes",
   "author": "Alibaba Qwen / neutral reproduction",
   "repo": "https://huggingface.co/Qwen/Qwen3-1.7B",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-1.7B BF16",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8194444444444444,
    "standard": 0.5520833333333334,
    "judge": 0.7534246575342466,
    "hard": 0.43636363636363634
   },
   "axes": {
    "intelligence": 33.22626585162718,
    "calibration": 24.285299854089764,
    "speed": 89.71855887469545,
    "cost": 64.8957758569641
   },
   "jevbench_score": 18.055721418631492,
   "speed": {
    "p50_s_raw": 0.06664170743897557,
    "p95_s_raw": 0.11331849200651051,
    "p50_s_adjusted": 0.28328341487795117,
    "p95_s_adjusted": 0.37663698401302104,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
    "hard_tier_p50_s": 0.07238452648743987,
    "hard_tier_p95_s": 0.2022612498141825
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.014795880149812733,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.014795880149812733,
    "usd_per_1000_hard": 0.014795880149812733,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 24.285299854089764,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.5165099099472441,
    "probability_fidelity": 53.52808773650845,
    "brier_hard": 1.0836421065443438,
    "brier_standard_judge_v11": 0.6254593558189092,
    "note": null
   },
   "hard": {
    "run": "raw-controls/runs/qwen3-17/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.43636363636363634,
    "by_family": {
     "adversarial": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 18,
      "n": 35,
      "accuracy": 0.5142857142857142
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 8,
      "n": 10,
      "accuracy": 0.8
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 8,
      "n": 16,
      "accuracy": 0.5
     }
    },
    "has_distribution": true,
    "brier_mean": 1.0836421065443438,
    "ece": 0.5165099099472441,
    "probability_fidelity": 53.52808773650845,
    "calibration_score": 26.764043868254227,
    "onehot": {
     "ece": 0.5636363636363637,
     "probability_fidelity": 47.017999999999994,
     "calibration_score": 23.508999999999997
    },
    "latency_p50_s": 0.07238452648743987,
    "latency_p95_s": 0.2022612498141825,
    "mean_input_tokens": 1236.2227272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 18.055721418631492,
    "Balanced 33:33:33 (no calibration)": 23.38456916110271,
    "Emphasis on Accuracy 60:20:20": 18.896518956257463,
    "Emphasis on Speed 20:60:20": 27.968844912459065,
    "Emphasis on Cost 20:20:60": 25.24241626266092,
    "Intelligence only": 14.672516219420602
   },
   "rank": 65,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 65,
    "Balanced 33:33:33 (no calibration)": 58,
    "Emphasis on Accuracy 60:20:20": 60,
    "Emphasis on Speed 20:60:20": 58,
    "Emphasis on Cost 20:20:60": 50,
    "Intelligence only": 66
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5411255411255411,
   "sealed_accuracy": 0.2597402597402597,
   "public_minus_sealed_gap_pp": 28.13852813852814,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.2927,
     "long_policy": 0.175,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.2143,
     "probability": 0.3214,
     "safety_judge": 0.1875,
     "temporal_numeric": 0.125,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1743,
     "1/3": 0.268,
     "2/3": 0.3431
    },
    "ece": 0.66250416149145,
    "mean_tvd_gold_probs": 0.6134437634847831,
    "calibration": 19.327811825760843,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8831,
      "accuracy": 0.2794
     },
     "conf>=0.9": {
      "coverage": 0.7727,
      "accuracy": 0.2857
     }
    }
   },
   "v130_comparison_rank": 62,
   "v130_comparison_score": 37.3907084877778,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "reflex-27b",
   "display": "reflex-27b (Qwen3.8-27B)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "kshetrajna12",
   "repo": "https://github.com/kshetrajna12/reflex",
   "licence": "MIT code; Apache-2.0 Qwen weights",
   "underlying": "Qwen3.8-27B, frozen, direct-logit readout averaged across two option orders",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 NVL 96 GB), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9583333333333334,
    "judge": 0.958904109589041,
    "hard": 0.759090909090909
   },
   "axes": {
    "intelligence": 46.38779077055885,
    "calibration": 77.23797885642135,
    "speed": 67.46599369477367,
    "cost": 32.26119927292868
   },
   "jevbench_score": 17.844519244120388,
   "speed": {
    "p50_s_raw": 1.8879061974585056,
    "p95_s_raw": 2.2076592873781915,
    "p50_s_adjusted": 3.925812394917011,
    "p95_s_adjusted": 4.565318574756383,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 2.1106490306556225,
    "hard_tier_p95_s": 2.5633741833269594
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.18111733707865169,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.8-27B list price list price $0.214/M in, $0.0/M out (the exact public base weights used as a direct-logit classifier; no output is generated) x 481 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.10295512738853504,
    "usd_per_1000_hard": 0.29267612727272724,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 77.23797885642135,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.046629459090909105,
    "probability_fidelity": 81.6821575,
    "brier_hard": 0.3173506562125362,
    "brier_standard_judge_v11": 0.0736030978986818,
    "note": null
   },
   "hard": {
    "run": "runs/reflex-27b--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.759090909090909,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 13,
      "n": 14,
      "accuracy": 0.9285714285714286
     },
     "judge_hard": {
      "correct": 27,
      "n": 33,
      "accuracy": 0.8181818181818182
     },
     "long_policy": {
      "correct": 26,
      "n": 38,
      "accuracy": 0.6842105263157895
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 13,
      "n": 20,
      "accuracy": 0.65
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 12,
      "n": 30,
      "accuracy": 0.4
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3173506562125362,
    "ece": 0.046629459090909105,
    "probability_fidelity": 81.6821575,
    "calibration_score": 86.17813284090909,
    "onehot": {
     "ece": 0.24090909090909096,
     "probability_fidelity": 53.07349999999998,
     "calibration_score": 52.44584090909089
    },
    "latency_p50_s": 2.1106490306556225,
    "latency_p95_s": 2.5633741833269594,
    "mean_input_tokens": 1367.6454545454546,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 17.844519244120388,
    "Balanced 33:33:33 (no calibration)": 15.955121345286392,
    "Emphasis on Accuracy 60:20:20": 16.215474849652626,
    "Emphasis on Speed 20:60:20": 18.466798097500842,
    "Emphasis on Cost 20:20:60": 13.849134974111864,
    "Intelligence only": 16.622336393075177
   },
   "rank": 66,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 66,
    "Balanced 33:33:33 (no calibration)": 69,
    "Emphasis on Accuracy 60:20:20": 64,
    "Emphasis on Speed 20:60:20": 69,
    "Emphasis on Cost 20:20:60": 70,
    "Intelligence only": 61
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8701298701298701,
   "sealed_accuracy": 0.29545454545454547,
   "public_minus_sealed_gap_pp": 57.467532467532465,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3514,
     "judge_hard": 0.3659,
     "long_policy": 0.15,
     "multi_hop": 0.3947,
     "paraphrase_robustness": 0.3571,
     "probability": 0.1429,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5
    },
    "by_panel_stratum": {
     "0/3": 0.1101,
     "1/3": 0.2887,
     "2/3": 0.5
    },
    "ece": 0.27544625324675326,
    "mean_tvd_gold_probs": 0.2619540757575758,
    "calibration": 59.35767088744588,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2662,
      "accuracy": 0.3415
     },
     "conf>=0.9": {
      "coverage": 0.0455,
      "accuracy": 0.2857
     }
    }
   },
   "v130_comparison_rank": 37,
   "v130_comparison_score": 63.33315017675085,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "jevact",
   "display": "JevAct (einptein, jev1-2b-v2)",
   "author": "einptein",
   "repo": "https://github.com/fstandhartinger/jevbench/issues/66",
   "class": "jev-rebuild",
   "licence": "not stated (the author published only an endpoint and the evaluation client)",
   "open": "no",
   "underlying": "the author's own checkpoint model/jev1-2b-v2/checkpoint-45000-67.53 over a Qwen3.5-2B base, served by the author on his own machine; one request per decision returns a native probability for every offered option",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "author-hosted endpoint (one machine at 115.190.124.200:18084, China), reached from a Hetzner server in Germany",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.6875,
    "judge": 0.589041095890411,
    "hard": 0.37272727272727274
   },
   "speed": {
    "p50_s_raw": 0.3992511257529259,
    "p95_s_raw": 1.4150911435484888,
    "p50_s_adjusted": 0.7985022515058517,
    "p95_s_adjusted": 2.8301822870969775,
    "adjustment": "x2 (not a production API; no +0.15 s)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.015491685393258427,
    "basis": "ESTIMATE: hosted-provider price, DeepInfra Qwen3.5-2B size-class reference list price $0.02/M in, $0.0/M out (a 2B one-pass decision model with nothing generated - between the <=0.6B reference the kev 0.5B row uses and the Qwen3.5-4B reference of the reflex 4B, Jobe and SemIf rows) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.009040000000000001,
    "usd_per_1000_hard": 0.0247
   },
   "calibration": {
    "score": 55.06434224219382,
    "note": null
   },
   "hard": {
    "run": "runs/jevact--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 142,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.6454545454545455,
    "accuracy": 0.37272727272727274,
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 0,
      "n": 38,
      "accuracy": 0.0
     },
     "multi_hop": {
      "correct": 1,
      "n": 35,
      "accuracy": 0.02857142857142857
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 9,
      "n": 10,
      "accuracy": 0.9
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6497616683189185,
    "ece": 0.2041986865896574,
    "probability_fidelity": 69.7478839531111,
    "calibration_score": 64.4540733175898,
    "onehot": {
     "ece": 0.4225352112676056,
     "probability_fidelity": 52.74769230769232,
     "calibration_score": 34.1203250270856
    },
    "latency_p50_s": 0.4214286766946316,
    "latency_p95_s": 1.666276295855641,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "release_evidence": {
    "sealed_aggregate_sha256": "9c76965bbd15d7b1400197d0c31318ce15441b43ef1a06edf996bdbc19627650",
    "detail_row_sha256": "f4f8bba7ee05c218629e1050c58e4e8f4d96229c47b8bb34d17be049fe82bfd5",
    "sealed_result_sha256": "492594c04be554b38a24b414be7f61faafc67a4d29a5e0a52c11fac3d64c2799"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official full-protocol measurement, jevbench-add-requests run 11, 24 Sep 2026)",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.6190476190476191,
   "sealed_accuracy": 0.23376623376623376,
   "public_minus_sealed_gap_pp": 38.528138528138534,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 237,
    "by_family": {
     "ambiguous_abstain": 0.2703,
     "judge_hard": 0.3415,
     "long_policy": 0.0,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.1429,
     "probability": 0.2143,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.1753,
     "2/3": 0.3431
    },
    "ece": 0.4263168474038442,
    "mean_tvd_gold_probs": 0.4216687033642745,
    "calibration": 36.28488009140186,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4416,
      "accuracy": 0.3088
     },
     "conf>=0.9": {
      "coverage": 0.211,
      "accuracy": 0.3077
     }
    }
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": 43.31953263358887,
   "scoring_note": "API measurement: sealed item text (no golds) was sent to the operator endpoint; all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) through JevBench's adapter.",
   "axes": {
    "intelligence": 29.266280238412953,
    "calibration": 55.06434224219382,
    "speed": 76.45909446386409,
    "cost": 64.29703993880327
   },
   "jevbench_score": 16.92690664787098,
   "presets": {
    "JevBench Score (25:25:25:25)": 16.92690664787098,
    "Balanced 33:33:33 (no calibration)": 16.366343940267765,
    "Emphasis on Accuracy 60:20:20": 13.062731525087502,
    "Emphasis on Speed 20:60:20": 19.25651120902886,
    "Emphasis on Cost 20:20:60": 18.24190047787369,
    "Intelligence only": 10.026805068618938
   },
   "rank": 67,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 67,
    "Balanced 33:33:33 (no calibration)": 68,
    "Emphasis on Accuracy 60:20:20": 72,
    "Emphasis on Speed 20:60:20": 67,
    "Emphasis on Cost 20:20:60": 65,
    "Intelligence only": 75
   }
  },
  {
   "key": "djev-thinking",
   "display": "djev (thinking)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "David Villalon / Maisa",
   "repo": "https://github.com/Davipar/djev-dev",
   "licence": "Apache-2.0",
   "underlying": "google/diffusiongemma-26b-a4b-it, BF16; full generation with thinking enabled",
   "has_distribution": true,
   "probability_source": [
    "verbalized"
   ],
   "endpoint_condition": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9583333333333334,
    "standard": 0.9895833333333334,
    "judge": 0.8013698630136986,
    "hard": 0.7772727272727272
   },
   "axes": {
    "intelligence": 71.62068248893371,
    "calibration": 87.77774635807619,
    "speed": 75.15117536916611,
    "cost": 26.85351129959409
   },
   "jevbench_score": 15.201198018956955,
   "speed": {
    "p50_s_raw": 0.4260098780505359,
    "p95_s_raw": 1.448969176481475,
    "p50_s_adjusted": 1.0020197561010717,
    "p95_s_adjusted": 3.04793835296295,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 1.061047127470374,
    "hard_tier_p95_s": 4.011991968052462
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.27429398876404487,
    "basis": "ESTIMATE: same-size hosted reference x 749 measured input and 690 measured output tokens per attempted decision across all 534, failures included",
    "usd_per_1000_v11_tiers": 0.159790127388535,
    "usd_per_1000_hard": 0.43772222727272725,
    "self_host_sensitivity": "Actual serial H200 rental equivalent: $0.846/1,000 decisions at measured 1.015s mean wall time and $3.00/hour; excludes idle/setup time."
   },
   "calibration": {
    "score": 87.77774635807619,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.06536279677768825,
    "probability_fidelity": 98.4813057049246,
    "brier_hard": 0.10142847386902532,
    "brier_standard_judge_v11": 0.020839209052624045,
    "note": null
   },
   "hard": {
    "run": "runs/djev-thinking--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 178,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.8090909090909091,
    "accuracy": 0.7772727272727272,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9090909090909091
     },
     "long_policy": {
      "correct": 23,
      "n": 38,
      "accuracy": 0.6052631578947368
     },
     "multi_hop": {
      "correct": 26,
      "n": 35,
      "accuracy": 0.7428571428571429
     },
     "probability": {
      "correct": 19,
      "n": 20,
      "accuracy": 0.95
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 21,
      "n": 30,
      "accuracy": 0.7
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.10142847386902532,
    "ece": 0.06536279677768825,
    "probability_fidelity": 98.4813057049246,
    "calibration_score": 92.70437317469347,
    "onehot": {
     "ece": 0.0393258426966292,
     "probability_fidelity": 65.78526315789472,
     "calibration_score": 78.96004730928445
    },
    "latency_p50_s": 1.061047127470374,
    "latency_p95_s": 4.011991968052462,
    "mean_input_tokens": 1249.5045454545455,
    "mean_output_tokens": 1084.2227272727273,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 15.201198018956955,
    "Balanced 33:33:33 (no calibration)": 13.414348413090133,
    "Emphasis on Accuracy 60:20:20": 15.602903817237578,
    "Emphasis on Speed 20:60:20": 15.82753899570648,
    "Emphasis on Cost 20:20:60": 10.376729581387078,
    "Intelligence only": 20.65858676820592
   },
   "rank": 68,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 68,
    "Balanced 33:33:33 (no calibration)": 72,
    "Emphasis on Accuracy 60:20:20": 65,
    "Emphasis on Speed 20:60:20": 73,
    "Emphasis on Cost 20:20:60": 79,
    "Intelligence only": 53
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8744588744588745,
   "sealed_accuracy": 0.6006493506493507,
   "public_minus_sealed_gap_pp": 27.380952380952383,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 220,
    "by_family": {
     "ambiguous_abstain": 0.5135,
     "judge_hard": 0.6585,
     "long_policy": 0.525,
     "multi_hop": 0.5789,
     "paraphrase_robustness": 0.7857,
     "probability": 0.7857,
     "safety_judge": 0.5625,
     "temporal_numeric": 0.6071,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.8333
    },
    "by_panel_stratum": {
     "0/3": 0.5229,
     "1/3": 0.6598,
     "2/3": 0.6275
    },
    "ece": 0.20398190438407426,
    "mean_tvd_gold_probs": 0.03354633673501975,
    "calibration": 77.92449272484158,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.6331,
      "accuracy": 0.8205
     },
     "conf>=0.9": {
      "coverage": 0.6071,
      "accuracy": 0.8289
     }
    }
   },
   "v130_comparison_rank": 41,
   "v130_comparison_score": 62.35960888483864,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 220/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "gliner2-large",
   "display": "GLiNER2 large (Fastino)",
   "class": "classifier",
   "open": "yes",
   "author": "Fastino",
   "repo": "https://huggingface.co/fastino/gliner2-large-v1",
   "licence": "Apache-2.0",
   "underlying": "GLiNER2 large schema-conditioned extractor (DeBERTa-v3-large encoder)",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.625,
    "judge": 0.6095890410958904,
    "hard": 0.36363636363636365
   },
   "axes": {
    "intelligence": 31.109577359763858,
    "calibration": 24.819443836109603,
    "speed": 61.65849142544844,
    "cost": 73.3279398087227
   },
   "jevbench_score": 15.138054851149672,
   "speed": {
    "p50_s_raw": 1.0967330671846867,
    "p95_s_raw": 14.48837994225323,
    "p50_s_adjusted": 2.3434661343693732,
    "p95_s_adjusted": 29.12675988450646,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 2.5048606134951115,
    "hard_tier_p95_s": 64.29295974373817
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.007745842696629214,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.004520000000000001,
    "usd_per_1000_hard": 0.01235,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 24.819443836109603,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.4661084924231875,
    "probability_fidelity": 41.85941495016359,
    "brier_hard": 1.0843070639080452,
    "brier_standard_judge_v11": 0.6502070879795632,
    "note": null
   },
   "hard": {
    "run": "runs/gliner2-large--hard (job jevbench-add-requests-20260919, run 4)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.36363636363636365,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 5,
      "n": 20,
      "accuracy": 0.25
     },
     "routing_hard": {
      "correct": 7,
      "n": 10,
      "accuracy": 0.7
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 2,
      "n": 12,
      "accuracy": 0.16666666666666666
     },
     "trap": {
      "correct": 11,
      "n": 16,
      "accuracy": 0.6875
     }
    },
    "has_distribution": true,
    "brier_mean": 1.0843070639080452,
    "ece": 0.4661084924231875,
    "probability_fidelity": 41.85941495016359,
    "calibration_score": 24.31885823276304,
    "onehot": {
     "ece": 0.6363636363636364,
     "probability_fidelity": 33.821,
     "calibration_score": 16.9105
    },
    "latency_p50_s": 2.5048606134951115,
    "latency_p95_s": 64.29295974373817,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 15.138054851149672,
    "Balanced 33:33:33 (no calibration)": 18.731672422749323,
    "Emphasis on Accuracy 60:20:20": 15.326835362123518,
    "Emphasis on Speed 20:60:20": 20.496346823061675,
    "Emphasis on Cost 20:20:60": 21.68147410129896,
    "Intelligence only": 12.043211805323642
   },
   "rank": 69,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 69,
    "Balanced 33:33:33 (no calibration)": 62,
    "Emphasis on Accuracy 60:20:20": 67,
    "Emphasis on Speed 20:60:20": 64,
    "Emphasis on Cost 20:20:60": 59,
    "Intelligence only": 69
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5670995670995671,
   "sealed_accuracy": 0.2857142857142857,
   "public_minus_sealed_gap_pp": 28.13852813852814,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4595,
     "judge_hard": 0.3659,
     "long_policy": 0.15,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.2857,
     "probability": 0.1786,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2385,
     "1/3": 0.2268,
     "2/3": 0.3922
    },
    "ece": 0.4493875919611423,
    "mean_tvd_gold_probs": 0.5848125152216609,
    "calibration": 25.820615042802725,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5584,
      "accuracy": 0.2907
     },
     "conf>=0.9": {
      "coverage": 0.4675,
      "accuracy": 0.2986
     }
    }
   },
   "v130_comparison_rank": 64,
   "v130_comparison_score": 29.55158409895235,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "openjev-thinking",
   "display": "OpenJev (thinking, BF16)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "razorback16",
   "repo": "https://github.com/razorback16/openjev",
   "licence": "Apache-2.0",
   "underlying": "google/diffusiongemma-26b-a4b-it, BF16; OpenJev think=512",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io H200 141 GB), reached over the internet from Germany; serial, one request at a time",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 1.0,
    "judge": 0.9452054794520548,
    "hard": 0.7818181818181819
   },
   "axes": {
    "intelligence": 58.08310902947819,
    "calibration": 58.097048339645674,
    "speed": 76.05530411890852,
    "cost": 27.82225297231227
   },
   "jevbench_score": 14.829064005607973,
   "speed": {
    "p50_s_raw": 0.46296589844860137,
    "p95_s_raw": 1.0775369299459272,
    "p50_s_adjusted": 1.0759317968972026,
    "p95_s_adjusted": 2.3050738598918543,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.8800784219056368,
    "hard_tier_p95_s": 1.2348620565375312
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.25463898876404495,
    "basis": "ESTIMATE: same hosted reference x 1778 billed input and 315 thought output tokens per decision",
    "usd_per_1000_v11_tiers": 0.15592996815286625,
    "usd_per_1000_hard": 0.39552368181818176,
    "self_host_sensitivity": "Actual serial H200 rental equivalent: $0.585/1,000 decisions at measured 0.702s mean wall time and $3.00/hour; excludes setup/idle time."
   },
   "calibration": {
    "score": 58.097048339645674,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.13293071724492195,
    "probability_fidelity": 65.70556085965308,
    "brier_hard": 0.3361950043526198,
    "brier_standard_judge_v11": 0.06627872347289883,
    "note": null
   },
   "hard": {
    "run": "runs/openjev-thinking--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7818181818181819,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 14,
      "n": 14,
      "accuracy": 1.0
     },
     "judge_hard": {
      "correct": 31,
      "n": 33,
      "accuracy": 0.9393939393939394
     },
     "long_policy": {
      "correct": 19,
      "n": 38,
      "accuracy": 0.5
     },
     "multi_hop": {
      "correct": 29,
      "n": 35,
      "accuracy": 0.8285714285714286
     },
     "probability": {
      "correct": 15,
      "n": 20,
      "accuracy": 0.75
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 14,
      "n": 30,
      "accuracy": 0.4666666666666667
     },
     "tradeoff": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3361950043526198,
    "ece": 0.13293071724492195,
    "probability_fidelity": 65.70556085965308,
    "calibration_score": 69.55970870533434,
    "onehot": {
     "ece": 0.21818181818181814,
     "probability_fidelity": 55.79700000000001,
     "calibration_score": 56.08031818181819
    },
    "latency_p50_s": 0.8800784219056368,
    "latency_p95_s": 1.2348620565375312,
    "mean_input_tokens": 2901.8136363636363,
    "mean_output_tokens": 447.8681818181818,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 14.829064005607973,
    "Balanced 33:33:33 (no calibration)": 14.008874936143108,
    "Emphasis on Accuracy 60:20:20": 15.367694897443487,
    "Emphasis on Speed 20:60:20": 16.71799429617309,
    "Emphasis on Cost 20:20:60": 11.202899721147364,
    "Intelligence only": 17.984337183128705
   },
   "rank": 70,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 70,
    "Balanced 33:33:33 (no calibration)": 71,
    "Emphasis on Accuracy 60:20:20": 66,
    "Emphasis on Speed 20:60:20": 70,
    "Emphasis on Cost 20:20:60": 75,
    "Intelligence only": 57
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8874458874458875,
   "sealed_accuracy": 0.42207792207792205,
   "public_minus_sealed_gap_pp": 46.536796536796544,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3784,
     "judge_hard": 0.6829,
     "long_policy": 0.25,
     "multi_hop": 0.3947,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3571,
     "safety_judge": 0.6875,
     "temporal_numeric": 0.3036,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.2661,
     "1/3": 0.3918,
     "2/3": 0.6176
    },
    "ece": 0.39430812212776295,
    "mean_tvd_gold_probs": 0.507949203579107,
    "calibration": 35.171727608268355,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.6883,
      "accuracy": 0.4906
     },
     "conf>=0.9": {
      "coverage": 0.5292,
      "accuracy": 0.5521
     }
    }
   },
   "v130_comparison_rank": 47,
   "v130_comparison_score": 59.98618116922599,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "mghafiri-qwen3.5-0.8b-decision-model",
   "display": "Qwen3.5-0.8B Decision Model (Mourad Ghafiri)",
   "author": "Mourad Ghafiri",
   "repo": "https://huggingface.co/mghafiri/qwen3.5-0.8B-decision-model",
   "class": "jev-rebuild",
   "licence": "MIT code and training data; Apache-2.0 model weights",
   "source": "v1.4.1 addition prepared from a local offline remeasurement",
   "open": "yes",
   "underlying": "Qwen3.5-0.8B-Base fine-tuned as a JevLite decision model with per-question calibration",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "Offline CPU inference using the pinned local JevLite model and calibration files.",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.5520833333333334,
    "judge": 0.3424657534246575,
    "hard": 0.509090909090909
   },
   "speed": {
    "p50_s_raw": 7.150214396417141,
    "p95_s_raw": 41.90147256627678,
    "p50_s_adjusted": 14.450428792834282,
    "p95_s_adjusted": 83.95294513255357,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial standard+judge tasks; CPU adjustment ×2 + 0.15 s",
    "hardware": "Ryzen 5 3600, 4 threads",
    "measured_where": "local CPU",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0064648501872659175,
    "basis": "same-size DeepInfra Qwen3.5-0.8B reference tariff; measured JevLite tokenizer usage; estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 68.22744227994228,
    "note": null
   },
   "hard": {
    "by_family": {
     "adversarial": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6429
     },
     "judge_hard": {
      "correct": 16,
      "n": 33,
      "accuracy": 0.4848
     },
     "long_policy": {
      "correct": 16,
      "n": 38,
      "accuracy": 0.4211
     },
     "multi_hop": {
      "correct": 16,
      "n": 35,
      "accuracy": 0.4571
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 11,
      "n": 16,
      "accuracy": 0.6875
     }
    }
   },
   "source_round": "v1.4.1 addition prepared from a local offline remeasurement",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5930735930735931,
   "sealed_accuracy": 0.3474025974025974,
   "public_minus_sealed_gap_pp": 24.56709956709957,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.40540540540540543,
     "judge_hard": 0.6829268292682927,
     "long_policy": 0.25,
     "multi_hop": 0.2631578947368421,
     "paraphrase_robustness": 0.2857142857142857,
     "probability": 0.32142857142857145,
     "safety_judge": 0.375,
     "temporal_numeric": 0.19642857142857142,
     "tradeoff": 0.34615384615384615,
     "trap_adversarial": 0.4166666666666667
    },
    "by_panel_stratum": {
     "0/3": 0.3669724770642202,
     "1/3": 0.3917525773195876,
     "2/3": 0.28431372549019607
    },
    "ece": 0.22153051948051955,
    "mean_tvd_gold_probs": 0.30575151515151516,
    "calibration": 62.559372294372295,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.19155844155844157,
      "accuracy": 0.4067796610169492
     },
     "conf>=0.9": {
      "coverage": 0.003246753246753247,
      "accuracy": 1.0
     }
    }
   },
   "v130_comparison_rank": 67,
   "v130_comparison_score": 23.945737520319526,
   "scoring_note": "Local CPU measurement on the existing 534-decision v1.3 set plus 308 sealed v1.4 decisions; no operator endpoint received sealed text.",
   "axes": {
    "intelligence": 28.073004486611502,
    "calibration": 68.22744227994228,
    "speed": 49.16083329619092,
    "cost": 75.68324604275398
   },
   "jevbench_score": 14.540627482985634,
   "presets": {
    "JevBench Score (25:25:25:25)": 14.540627482985634,
    "Balanced 33:33:33 (no calibration)": 13.216105451709785,
    "Emphasis on Accuracy 60:20:20": 10.851296757297625,
    "Emphasis on Speed 20:60:20": 13.869868192936037,
    "Emphasis on Cost 20:20:60": 15.938250270662172,
    "Intelligence only": 8.555100965797516
   },
   "rank": 71,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 71,
    "Balanced 33:33:33 (no calibration)": 74,
    "Emphasis on Accuracy 60:20:20": 75,
    "Emphasis on Speed 20:60:20": 75,
    "Emphasis on Cost 20:20:60": 67,
    "Intelligence only": 77
   }
  },
  {
   "key": "gemini-3.1-flash-lite",
   "display": "Gemini 3.1 Flash-Lite",
   "class": "llm-baseline",
   "open": "no",
   "author": "Google",
   "repo": null,
   "licence": "proprietary API",
   "underlying": "closed",
   "has_distribution": true,
   "probability_source": [
    "verbalized"
   ],
   "endpoint_condition": "production API (Google)",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.9315068493150684,
    "hard": 0.75
   },
   "axes": {
    "intelligence": 54.47139585292138,
    "calibration": 59.293625254911305,
    "speed": 81.78762581772229,
    "cost": 27.362399077259013
   },
   "jevbench_score": 14.26151759252857,
   "speed": {
    "p50_s_raw": 0.756230715662241,
    "p95_s_raw": 0.8761593606323003,
    "p50_s_adjusted": 0.756230715662241,
    "p95_s_adjusted": 0.8761593606323003,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.7880580946803093,
    "hard_tier_p95_s": 1.4522175978869198
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.26378698501872655,
    "basis": "public tariff x measured tokens (https://ai.google.dev/gemini-api/docs/pricing (paid tier, read 2026-09-19)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
    "usd_per_1000_v11_tiers": 0.17806050955414016,
    "usd_per_1000_hard": 0.38614204545454534,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 59.293625254911305,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2667577285845467,
    "probability_fidelity": 89.52777128427128,
    "brier_hard": 0.5096290133859069,
    "brier_standard_judge_v11": 0.09465247933884294,
    "note": null
   },
   "hard": {
    "run": "runs/gemini-3.1-flash-lite--hard",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.75,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857142857142857
     },
     "judge_hard": {
      "correct": 29,
      "n": 33,
      "accuracy": 0.8787878787878788
     },
     "long_policy": {
      "correct": 25,
      "n": 38,
      "accuracy": 0.6578947368421053
     },
     "multi_hop": {
      "correct": 27,
      "n": 35,
      "accuracy": 0.7714285714285715
     },
     "probability": {
      "correct": 18,
      "n": 20,
      "accuracy": 0.9
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 9,
      "n": 12,
      "accuracy": 0.75
     },
     "trap": {
      "correct": 15,
      "n": 16,
      "accuracy": 0.9375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5096290133859069,
    "ece": 0.2667577285845467,
    "probability_fidelity": 89.52777128427128,
    "calibration_score": 68.08811278368097,
    "onehot": {
     "ece": 0.25,
     "probability_fidelity": 63.314499999999995,
     "calibration_score": 56.65725
    },
    "latency_p50_s": 0.7880580946803093,
    "latency_p95_s": 1.4522175978869198,
    "mean_input_tokens": 1234.5045454545455,
    "mean_output_tokens": 51.67727272727273,
    "charged_usd": 0.08495124999999998
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 14.26151759252857,
    "Balanced 33:33:33 (no calibration)": 13.383290747534044,
    "Emphasis on Accuracy 60:20:20": 14.419158518746897,
    "Emphasis on Speed 20:60:20": 16.34983547001401,
    "Emphasis on Cost 20:20:60": 10.678598691148762,
    "Intelligence only": 16.313112875064334
   },
   "rank": 72,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 72,
    "Balanced 33:33:33 (no calibration)": 73,
    "Emphasis on Accuracy 60:20:20": 69,
    "Emphasis on Speed 20:60:20": 72,
    "Emphasis on Cost 20:20:60": 77,
    "Intelligence only": 63
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8701298701298701,
   "sealed_accuracy": 0.38636363636363635,
   "public_minus_sealed_gap_pp": 48.37662337662337,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 307,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.5122,
     "long_policy": 0.15,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.5,
     "probability": 0.6429,
     "safety_judge": 0.5,
     "temporal_numeric": 0.3036,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.2202,
     "1/3": 0.3299,
     "2/3": 0.6176
    },
    "ece": 0.5778164713767407,
    "mean_tvd_gold_probs": 0.1659069960525604,
    "calibration": 41.704650197371976,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.9026,
      "accuracy": 0.3633
     },
     "conf>=0.9": {
      "coverage": 0.8571,
      "accuracy": 0.3561
     }
    }
   },
   "v130_comparison_rank": 46,
   "v130_comparison_score": 60.08928259753951,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (Google API), as for every API measurement; 307/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "open-jev-deberta-v3-large",
   "display": "open-jev-deberta-v3-large (local CPU)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Kotoba Labs",
   "repo": "https://github.com/kotoba-lang/typed-decisions",
   "licence": "Apache-2.0 (model card); DeBERTa-v3 keeps its own terms",
   "underlying": "microsoft/deberta-v3-large",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.4895833333333333,
    "judge": 0.5342465753424658,
    "hard": 0.36363636363636365
   },
   "axes": {
    "intelligence": 25.580756971997335,
    "calibration": 66.6011163245795,
    "speed": 65.97921316871283,
    "cost": 74.02828721517697
   },
   "jevbench_score": 12.649316553397922,
   "speed": {
    "p50_s_raw": 1.767673410475254,
    "p95_s_raw": 3.349288306012749,
    "p50_s_adjusted": 3.685346820950508,
    "p95_s_adjusted": 6.848576612025498,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "local, 2 CPU threads of a Ryzen 5 3600",
    "hard_tier_p50_s": 2.63558766245842,
    "hard_tier_p95_s": 4.7398640830069665
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.007340468164794008,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) $0.01/M in, $0.0/M out x 1235 in / 0 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.003834076433121019,
    "usd_per_1000_hard": 0.012345045454545454,
    "self_host_sensitivity": {
     "usd_per_1000": 0.014168273865544099,
     "score": 71.21707642612242,
     "machine": "Hetzner CX22 (2 vCPU)",
     "usd_per_h": 0.0072,
     "concurrency": 1,
     "utilisation": 0.3,
     "p50_s_used": 2.1252410798316146,
     "decisions_per_hour": 508.1776417033919
    }
   },
   "calibration": {
    "score": 66.6011163245795,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.17293095619443358,
    "probability_fidelity": 67.35071130099892,
    "brier_hard": 0.7102063910056119,
    "brier_standard_judge_v11": 0.6512140520637422,
    "note": null
   },
   "hard": {
    "run": "runs/open-jev-deberta-v3-large--hard",
    "n_items": 220,
    "n_ok": 218,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.990909090909091,
    "accuracy": 0.36363636363636365,
    "by_family": {
     "adversarial": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 13,
      "n": 38,
      "accuracy": 0.34210526315789475
     },
     "multi_hop": {
      "correct": 13,
      "n": 35,
      "accuracy": 0.37142857142857144
     },
     "probability": {
      "correct": 6,
      "n": 20,
      "accuracy": 0.3
     },
     "routing_hard": {
      "correct": 2,
      "n": 10,
      "accuracy": 0.2
     },
     "temporal_numeric": {
      "correct": 4,
      "n": 30,
      "accuracy": 0.13333333333333333
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 5,
      "n": 16,
      "accuracy": 0.3125
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7102063910056119,
    "ece": 0.17293095619443358,
    "probability_fidelity": 67.35071130099892,
    "calibration_score": 66.3822600310561,
    "onehot": {
     "ece": 0.6330275229357798,
     "probability_fidelity": 35.785999999999994,
     "calibration_score": 17.892999999999997
    },
    "latency_p50_s": 2.63558766245842,
    "latency_p95_s": 4.7398640830069665,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 12.649316553397922,
    "Balanced 33:33:33 (no calibration)": 11.589292320629404,
    "Emphasis on Accuracy 60:20:20": 8.967710244104621,
    "Emphasis on Speed 20:60:20": 13.345184662618193,
    "Emphasis on Cost 20:20:60": 13.809283011947182,
    "Intelligence only": 6.695764439587149
   },
   "rank": 73,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 73,
    "Balanced 33:33:33 (no calibration)": 75,
    "Emphasis on Accuracy 60:20:20": 79,
    "Emphasis on Speed 20:60:20": 76,
    "Emphasis on Cost 20:20:60": 71,
    "Intelligence only": 80
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5238095238095238,
   "sealed_accuracy": 0.29545454545454547,
   "public_minus_sealed_gap_pp": 22.835497835497836,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 297,
    "by_family": {
     "ambiguous_abstain": 0.3784,
     "judge_hard": 0.439,
     "long_policy": 0.3,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.1429,
     "probability": 0.1429,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.3028,
     "1/3": 0.2165,
     "2/3": 0.3627
    },
    "ece": 0.16482671974885346,
    "mean_tvd_gold_probs": 0.32956998226976764,
    "calibration": 67.03882891162627,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1364,
      "accuracy": 0.4286
     },
     "conf>=0.9": {
      "coverage": 0.0032,
      "accuracy": 1.0
     }
    }
   },
   "v130_comparison_rank": 69,
   "v130_comparison_score": 23.065925583996304,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 297/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "smalljev",
   "display": "smalljev semantic-v9",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Aditya (isHeSatoshi)",
   "repo": "https://github.com/isHeSatoshi/smalljev",
   "licence": "Apache-2.0",
   "underlying": "MiniCPM5-2B-Base, 2.5B dense, with LoRA and native semantic decision heads",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), reached over the internet from Germany; serial, one request at a time",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9722222222222222,
    "standard": 0.6875,
    "judge": 0.4041095890410959,
    "hard": 0.38181818181818183
   },
   "axes": {
    "intelligence": 25.670293240126572,
    "calibration": 59.17561887769727,
    "speed": 79.81829102912181,
    "cost": 57.872598102653555
   },
   "jevbench_score": 12.308142930422852,
   "speed": {
    "p50_s_raw": 0.4138255603611469,
    "p95_s_raw": 0.45828209072351456,
    "p50_s_adjusted": 0.9776511207222939,
    "p95_s_adjusted": 1.066564181447029,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.42050960287451744,
    "hard_tier_p95_s": 0.8517502576112745
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.025365692883895133,
    "basis": "ESTIMATE: hosted-provider price, submitted Qwen/Qwen2.5-3B-Instruct hosted reference list price $0.04/M in, $0.0/M out (the author's documented reference for the same approximate size class; one forward pass, nothing generated) x 329 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.013142292993630575,
    "usd_per_1000_hard": 0.04281181818181819,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 59.17561887769727,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2262606866657734,
    "probability_fidelity": 63.012640314918,
    "brier_hard": 0.7819475766009777,
    "brier_standard_judge_v11": 0.5889297804477662,
    "note": null
   },
   "hard": {
    "run": "runs/smalljev--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.38181818181818183,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 19,
      "n": 33,
      "accuracy": 0.5757575757575758
     },
     "long_policy": {
      "correct": 9,
      "n": 38,
      "accuracy": 0.23684210526315788
     },
     "multi_hop": {
      "correct": 7,
      "n": 35,
      "accuracy": 0.2
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 4,
      "n": 10,
      "accuracy": 0.4
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 10,
      "n": 16,
      "accuracy": 0.625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7819475766009777,
    "ece": 0.2262606866657734,
    "probability_fidelity": 63.012640314918,
    "calibration_score": 58.88025149088166,
    "onehot": {
     "ece": 0.6181818181818182,
     "probability_fidelity": 39.76250000000001,
     "calibration_score": 19.881250000000005
    },
    "latency_p50_s": 0.42050960287451744,
    "latency_p95_s": 0.8517502576112745,
    "mean_input_tokens": 1070.2954545454545,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.009418599999999996
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 12.308142930422852,
    "Balanced 33:33:33 (no calibration)": 11.499687252431785,
    "Emphasis on Accuracy 60:20:20": 8.98539860206567,
    "Emphasis on Speed 20:60:20": 14.047373108201972,
    "Emphasis on Cost 20:20:60": 12.755540568146788,
    "Intelligence only": 6.76631918415945
   },
   "rank": 74,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 74,
    "Balanced 33:33:33 (no calibration)": 76,
    "Emphasis on Accuracy 60:20:20": 78,
    "Emphasis on Speed 20:60:20": 74,
    "Emphasis on Cost 20:20:60": 72,
    "Intelligence only": 79
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.6060606060606061,
   "sealed_accuracy": 0.2694805194805195,
   "public_minus_sealed_gap_pp": 33.65800865800866,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.6098,
     "long_policy": 0.15,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.3571,
     "probability": 0.25,
     "safety_judge": 0.1875,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.1154,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.2936,
     "1/3": 0.2784,
     "2/3": 0.2353
    },
    "ece": 0.226379418382784,
    "mean_tvd_gold_probs": 0.3519140902078625,
    "calibration": 59.76635365132847,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.224,
      "accuracy": 0.3623
     },
     "conf>=0.9": {
      "coverage": 0.039,
      "accuracy": 0.25
     }
    }
   },
   "v130_comparison_rank": 65,
   "v130_comparison_score": 27.444453253651357,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "gliner2",
   "display": "GLiNER2 (Fastino, gliner2.5-base)",
   "class": "classifier",
   "open": "yes",
   "author": "Fastino",
   "repo": "https://github.com/fastino-ai/GLiNER2",
   "licence": "Apache-2.0",
   "underlying": "DeBERTa-v3-base schema-conditioned extractor (GLiNER2.5 boundary architecture), 194M",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9722222222222222,
    "standard": 0.6666666666666666,
    "judge": 0.4589041095890411,
    "hard": 0.36363636363636365
   },
   "axes": {
    "intelligence": 27.41771615013038,
    "calibration": 25.196414331851887,
    "speed": 71.82859826025441,
    "cost": 83.05556459946699
   },
   "jevbench_score": 11.777645154812003,
   "speed": {
    "p50_s_raw": 0.31302378326654434,
    "p95_s_raw": 4.153845678269863,
    "p50_s_adjusted": 0.7760475665330887,
    "p95_s_adjusted": 8.457691356539726,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 0.9508068449795246,
    "hard_tier_p95_s": 24.590944871306384
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.00367125468164794,
    "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 383 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]",
    "usd_per_1000_v11_tiers": 0.0019170382165605096,
    "usd_per_1000_hard": 0.006175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 25.196414331851887,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.47177480337294664,
    "probability_fidelity": 41.6920350000529,
    "brier_hard": 1.0505351022898706,
    "brier_standard_judge_v11": 0.8269065970344461,
    "note": null
   },
   "hard": {
    "run": "runs/gliner2--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.36363636363636365,
    "by_family": {
     "adversarial": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "ambiguous": {
      "correct": 5,
      "n": 14,
      "accuracy": 0.35714285714285715
     },
     "judge_hard": {
      "correct": 15,
      "n": 33,
      "accuracy": 0.45454545454545453
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 13,
      "n": 35,
      "accuracy": 0.37142857142857144
     },
     "probability": {
      "correct": 6,
      "n": 20,
      "accuracy": 0.3
     },
     "routing_hard": {
      "correct": 7,
      "n": 10,
      "accuracy": 0.7
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 7,
      "n": 16,
      "accuracy": 0.4375
     }
    },
    "has_distribution": true,
    "brier_mean": 1.0505351022898706,
    "ece": 0.47177480337294664,
    "probability_fidelity": 41.6920350000529,
    "calibration_score": 23.668537162731784,
    "onehot": {
     "ece": 0.6363636363636364,
     "probability_fidelity": 33.6155,
     "calibration_score": 16.80775
    },
    "latency_p50_s": 0.9508068449795246,
    "latency_p95_s": 24.590944871306384,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 11.777645154812003,
    "Balanced 33:33:33 (no calibration)": 14.44828021553688,
    "Emphasis on Accuracy 60:20:20": 11.105459611699635,
    "Emphasis on Speed 20:60:20": 16.65351196833538,
    "Emphasis on Cost 20:20:60": 17.37801419347457,
    "Intelligence only": 8.244300614252186
   },
   "rank": 75,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 75,
    "Balanced 33:33:33 (no calibration)": 70,
    "Emphasis on Accuracy 60:20:20": 73,
    "Emphasis on Speed 20:60:20": 71,
    "Emphasis on Cost 20:20:60": 66,
    "Intelligence only": 78
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5800865800865801,
   "sealed_accuracy": 0.2922077922077922,
   "public_minus_sealed_gap_pp": 28.78787878787879,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4054,
     "judge_hard": 0.439,
     "long_policy": 0.225,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.1429,
     "probability": 0.1429,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.3846,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.2661,
     "1/3": 0.2165,
     "2/3": 0.3922
    },
    "ece": 0.4605469570144431,
    "mean_tvd_gold_probs": 0.513862712569272,
    "calibration": 28.252168670092093,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.5909,
      "accuracy": 0.3077
     },
     "conf>=0.9": {
      "coverage": 0.4091,
      "accuracy": 0.3492
     }
    }
   },
   "v130_comparison_rank": 66,
   "v130_comparison_score": 24.036445808742133,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "instinct",
   "display": "Instinct (ZooWork, Qwen3.8-27B)",
   "author": "rayrain-srp (ZooWork)",
   "repo": "https://github.com/fstandhartinger/jevbench/issues/69",
   "class": "jev-rebuild",
   "licence": "proprietary service over Apache-2.0 base weights (Qwen3.8-27B)",
   "open": "no",
   "underlying": "Qwen3.8-27B (Apache-2.0 base weights) with no additional fine-tuning; their own serving stack reads candidate-token scores in one forward pass per question and normalizes them into a decision probability, with no autoregressive decoding. The serving implementation is not public.",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "free evaluation/demo endpoint (ZooWork Instinct, api.zoowork.ai), reached from a Hetzner server in Germany",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.9583333333333334,
    "judge": 0.9452054794520548,
    "hard": 0.759090909090909
   },
   "speed": {
    "p50_s_raw": 0.25864607095718384,
    "p95_s_raw": 0.3919508829712868,
    "p50_s_adjusted": 0.5172921419143677,
    "p95_s_adjusted": 0.7839017659425735,
    "adjustment": "x2 (free evaluation/demo endpoint: no SLA, no status page, no terms, no bookable price)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.3032262772321701,
    "hard_tier_p95_s": 0.8084227547049523
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.3253253932584269,
    "basis": "ESTIMATE from the BASE MODEL's public reference price (Florian's price rule, 24 Sep 2026): OpenRouter's model-level input price for qwen/qwen3.8-27b, $0.42 per 1M input tokens (catalog snapshot 2026-09-24T19:52Z), zero output-token price for this direct-logit system, x the board-standard input token counts (452 per standard/judge decision, 1,235 per hard decision; 774.6 per decision over the 534 frozen items, the same convention as the other run-11 rows). ZooWork publishes no bookable price; the author-announced $0.03/M is not used.",
    "superseded_basis": "author-announced $0.03/M (board #824); superseded by the 20:30 price rule",
    "reference_model": "qwen/qwen3.8-27b",
    "input_usd_per_m": 0.42,
    "output_usd_per_m": 0.0
   },
   "calibration": {
    "score": 78.6725030936767,
    "note": null
   },
   "hard": {
    "run": "runs/instinct--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 217,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.9863636363636363,
    "accuracy": 0.759090909090909,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 12,
      "n": 14,
      "accuracy": 0.8571428571428571
     },
     "judge_hard": {
      "correct": 28,
      "n": 33,
      "accuracy": 0.8484848484848485
     },
     "long_policy": {
      "correct": 28,
      "n": 38,
      "accuracy": 0.7368421052631579
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 12,
      "n": 20,
      "accuracy": 0.6
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 10,
      "n": 12,
      "accuracy": 0.8333333333333334
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3030795173651991,
    "ece": 0.03778852692833676,
    "probability_fidelity": 76.84827661334029,
    "calibration_score": 84.64528561383646,
    "onehot": {
     "ece": 0.2304147465437788,
     "probability_fidelity": 52.4436842105263,
     "calibration_score": 53.18036745088527
    },
    "latency_p50_s": 0.3032262772321701,
    "latency_p95_s": 0.8084227547049523,
    "mean_input_tokens": 1242.857142857143,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.008091000000000006
   },
   "release_evidence": {
    "sealed_aggregate_sha256": "e9c9e1e56718064d7dfc1e413c94f7664f01d75502b608711e2c3f529997927e",
    "detail_row_sha256": "adf01c6492368359aa21293a0b225bb4d4a1c6166a45d657681f698c9f31ec1c",
    "sealed_result_sha256": "1e3603a23dfe201c6245ae1eb636af279b44b1354ac2b88ffb8ff8721923e257",
    "repriced_row_sha256": "f11e47a3a75664db3996b754cbdea31f39ae0c804b29acb5557c41b8ec163533"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official full-protocol measurement, jevbench-add-requests run 11, 24 Sep 2026)",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8658008658008658,
   "sealed_accuracy": 0.35064935064935066,
   "public_minus_sealed_gap_pp": 51.515151515151516,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4324,
     "judge_hard": 0.439,
     "long_policy": 0.25,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3214,
     "safety_judge": 0.5,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.3505,
     "2/3": 0.5686
    },
    "ece": 0.1805935393274521,
    "mean_tvd_gold_probs": 0.30427416027795257,
    "calibration": 66.72693805335716,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1851,
      "accuracy": 0.3333
     },
     "conf>=0.9": {
      "coverage": 0.026,
      "accuracy": 0.75
     }
    }
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": null,
   "scoring_note": "API measurement: sealed item text (no golds) was sent to the operator endpoint; all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) through JevBench's adapter.",
   "axes": {
    "intelligence": 51.1455873367657,
    "calibration": 78.6725030936767,
    "speed": 83.92002476015575,
    "cost": 24.630461096785695
   },
   "jevbench_score": 11.449217397802405,
   "presets": {
    "JevBench Score (25:25:25:25)": 11.449217397802405,
    "Balanced 33:33:33 (no calibration)": 10.101415587062302,
    "Emphasis on Accuracy 60:20:20": 10.913860152735934,
    "Emphasis on Speed 20:60:20": 12.651857290610904,
    "Emphasis on Cost 20:20:60": 7.916296875466699,
    "Intelligence only": 12.411184905342505
   },
   "rank": 76,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 76,
    "Balanced 33:33:33 (no calibration)": 78,
    "Emphasis on Accuracy 60:20:20": 74,
    "Emphasis on Speed 20:60:20": 77,
    "Emphasis on Cost 20:20:60": 83,
    "Intelligence only": 68
   }
  },
  {
   "key": "open-jev-zefan-9b",
   "display": "Open-Jev 9B (Zefan Cai)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Zefan Cai (@Zefan_Cai)",
   "repo": "https://github.com/Zefan-Cai/Open-Jev",
   "licence": "MIT (loader); Apache-2.0 (adapter and pinned Qwen base); CC0-1.0 public training projection",
   "underlying": "Qwen3.5-9B plus rank-8 LoRA and trained scalar decision head",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 80GB HBM3), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.90625,
    "judge": 0.815068493150685,
    "hard": 0.6090909090909091
   },
   "axes": {
    "intelligence": 44.18189841302913,
    "calibration": 61.780502096905636,
    "speed": 72.03286755510636,
    "cost": 28.123361021000306
   },
   "jevbench_score": 11.195362486014277,
   "speed": {
    "p50_s_raw": 0.754859171807766,
    "p95_s_raw": 1.8114654466509819,
    "p50_s_adjusted": 1.6597183436155318,
    "p95_s_adjusted": 3.7729308933019636,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.7952907048165798,
    "hard_tier_p95_s": 1.648427233844994
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.24882153558052428,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.14393662420382164,
    "usd_per_1000_hard": 0.39852090909090904,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 61.780502096905636,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.19026066635697753,
    "probability_fidelity": 64.60861999100311,
    "brier_hard": 0.5546258172920748,
    "brier_standard_judge_v11": 0.23654785588059785,
    "note": null
   },
   "hard": {
    "run": "runs/open-jev-zefan-9b--hard (job jevbench-zefan-openjev-20260921)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.6090909090909091,
    "by_family": {
     "adversarial": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 27,
      "n": 33,
      "accuracy": 0.8181818181818182
     },
     "long_policy": {
      "correct": 17,
      "n": 38,
      "accuracy": 0.4473684210526316
     },
     "multi_hop": {
      "correct": 24,
      "n": 35,
      "accuracy": 0.6857142857142857
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.5546258172920748,
    "ece": 0.19026066635697753,
    "probability_fidelity": 64.60861999100311,
    "calibration_score": 63.2782433598038,
    "onehot": {
     "ece": 0.3909090909090909,
     "probability_fidelity": 46.07800000000001,
     "calibration_score": 33.94809090909092
    },
    "latency_p50_s": 0.7952907048165798,
    "latency_p95_s": 1.648427233844994,
    "mean_input_tokens": 3985.2090909090907,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 11.195362486014277,
    "Balanced 33:33:33 (no calibration)": 10.28221087109429,
    "Emphasis on Accuracy 60:20:20": 10.52597456166002,
    "Emphasis on Speed 20:60:20": 12.37123378521091,
    "Emphasis on Cost 20:20:60": 8.62587381478816,
    "Intelligence only": 10.914090353651781
   },
   "rank": 77,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 77,
    "Balanced 33:33:33 (no calibration)": 77,
    "Emphasis on Accuracy 60:20:20": 76,
    "Emphasis on Speed 20:60:20": 78,
    "Emphasis on Cost 20:20:60": 80,
    "Intelligence only": 71
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.7748917748917749,
   "sealed_accuracy": 0.2987012987012987,
   "public_minus_sealed_gap_pp": 47.61904761904761,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.2927,
     "long_policy": 0.2,
     "multi_hop": 0.2895,
     "paraphrase_robustness": 0.5,
     "probability": 0.4643,
     "safety_judge": 0.25,
     "temporal_numeric": 0.25,
     "tradeoff": 0.4231,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.1927,
     "1/3": 0.2784,
     "2/3": 0.4314
    },
    "ece": 0.2739115176674875,
    "mean_tvd_gold_probs": 0.27647657324283903,
    "calibration": 58.7850195711093,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3409,
      "accuracy": 0.3048
     },
     "conf>=0.9": {
      "coverage": 0.1526,
      "accuracy": 0.2128
     }
    }
   },
   "v130_comparison_rank": 52,
   "v130_comparison_score": 54.95864804501749,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "open-jev-zefan-2b",
   "display": "Open-Jev 2B (Zefan Cai)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Zefan Cai (@Zefan_Cai)",
   "repo": "https://github.com/Zefan-Cai/Open-Jev",
   "licence": "MIT (loader); Apache-2.0 (adapter and pinned Qwen base); CC0-1.0 public training projection",
   "underlying": "Qwen3.5-2B plus rank-8 LoRA and trained scalar decision head",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (H100 80GB HBM3), reached over the internet",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.7916666666666666,
    "judge": 0.8835616438356164,
    "hard": 0.42727272727272725
   },
   "axes": {
    "intelligence": 42.33143216557079,
    "calibration": 55.27368440973887,
    "speed": 73.45427178913715,
    "cost": 28.123361021000306
   },
   "jevbench_score": 9.980246077108298,
   "speed": {
    "p50_s_raw": 0.664745207875967,
    "p95_s_raw": 1.450564834475517,
    "p50_s_adjusted": 1.479490415751934,
    "p95_s_adjusted": 3.051129668951034,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.6978476755321026,
    "hard_tier_p95_s": 1.2396420393139123
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.24882153558052428,
    "basis": "ESTIMATE: hosted-provider price, OpenRouter Qwen3.5-9B list price read 2026-09-21 list price $0.1/M in, $0.0/M out (the exact 9B base and a conservative same-family proxy for the unlisted 2B; the decision head generates no output tokens) x 1439 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.14393662420382164,
    "usd_per_1000_hard": 0.39852090909090904,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 55.27368440973887,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2573327936087847,
    "probability_fidelity": 61.57396445297609,
    "brier_hard": 0.7779483924445973,
    "brier_standard_judge_v11": 0.26468498350008274,
    "note": null
   },
   "hard": {
    "run": "runs/open-jev-zefan-2b--hard (job jevbench-zefan-openjev-20260921)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.42727272727272725,
    "by_family": {
     "adversarial": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 20,
      "n": 33,
      "accuracy": 0.6060606060606061
     },
     "long_policy": {
      "correct": 8,
      "n": 38,
      "accuracy": 0.21052631578947367
     },
     "multi_hop": {
      "correct": 18,
      "n": 35,
      "accuracy": 0.5142857142857142
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 8,
      "n": 10,
      "accuracy": 0.8
     },
     "temporal_numeric": {
      "correct": 5,
      "n": 30,
      "accuracy": 0.16666666666666666
     },
     "tradeoff": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "trap": {
      "correct": 14,
      "n": 16,
      "accuracy": 0.875
     }
    },
    "has_distribution": true,
    "brier_mean": 0.7779483924445973,
    "ece": 0.2573327936087847,
    "probability_fidelity": 61.57396445297609,
    "calibration_score": 55.05370286560958,
    "onehot": {
     "ece": 0.5727272727272728,
     "probability_fidelity": 41.585499999999996,
     "calibration_score": 20.792749999999998
    },
    "latency_p50_s": 0.6978476755321026,
    "latency_p95_s": 1.2396420393139123,
    "mean_input_tokens": 3985.2090909090907,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 9.980246077108298,
    "Balanced 33:33:33 (no calibration)": 9.345491678980963,
    "Emphasis on Accuracy 60:20:20": 9.445419286850528,
    "Emphasis on Speed 20:60:20": 11.335815540657029,
    "Emphasis on Cost 20:20:60": 7.878792691494776,
    "Intelligence only": 9.599382833639712
   },
   "rank": 78,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 78,
    "Balanced 33:33:33 (no calibration)": 79,
    "Emphasis on Accuracy 60:20:20": 77,
    "Emphasis on Speed 20:60:20": 80,
    "Emphasis on Cost 20:20:60": 84,
    "Intelligence only": 76
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.645021645021645,
   "sealed_accuracy": 0.262987012987013,
   "public_minus_sealed_gap_pp": 38.20346320346321,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.4146,
     "long_policy": 0.075,
     "multi_hop": 0.2368,
     "paraphrase_robustness": 0.4286,
     "probability": 0.25,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2202,
     "1/3": 0.2887,
     "2/3": 0.2843
    },
    "ece": 0.285145843407084,
    "mean_tvd_gold_probs": 0.3154353632258828,
    "calibration": 55.71364749799746,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3117,
      "accuracy": 0.2812
     },
     "conf>=0.9": {
      "coverage": 0.1169,
      "accuracy": 0.3056
     }
    }
   },
   "v130_comparison_rank": 57,
   "v130_comparison_score": 51.31393833805917,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "gliner2.5-multi",
   "display": "GLiNER2.5 multi (Fastino, 287M)",
   "class": "classifier",
   "open": "yes",
   "author": "Fastino",
   "repo": "https://huggingface.co/fastino/gliner2.5-multi-v1",
   "licence": "Apache-2.0",
   "underlying": "GLiNER2.5 multilingual schema-conditioned extractor, 287M",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9027777777777778,
    "standard": 0.5104166666666666,
    "judge": 0.4383561643835616,
    "hard": 0.37727272727272726
   },
   "axes": {
    "intelligence": 23.119657731293067,
    "calibration": 57.17734564391696,
    "speed": 67.79966551109104,
    "cost": 82.35883967864214
   },
   "jevbench_score": 9.759108258784872,
   "speed": {
    "p50_s_raw": 0.42792757973074913,
    "p95_s_raw": 8.17526703067124,
    "p50_s_adjusted": 1.0058551594614982,
    "p95_s_adjusted": 16.500534061342478,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 1.2314840070903301,
    "hard_tier_p95_s": 41.43838266544044
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.003872921348314607,
    "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.0022600000000000003,
    "usd_per_1000_hard": 0.006175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 57.17734564391696,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.2653272879394618,
    "probability_fidelity": 65.34895862362534,
    "brier_hard": 0.8370279252684946,
    "brier_standard_judge_v11": 0.769624024073474,
    "note": null
   },
   "hard": {
    "run": "runs/gliner2.5-multi--hard (job jevbench-add-requests-20260919, run 3)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.37727272727272726,
    "by_family": {
     "adversarial": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 17,
      "n": 33,
      "accuracy": 0.5151515151515151
     },
     "long_policy": {
      "correct": 14,
      "n": 38,
      "accuracy": 0.3684210526315789
     },
     "multi_hop": {
      "correct": 13,
      "n": 35,
      "accuracy": 0.37142857142857144
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 1,
      "n": 10,
      "accuracy": 0.1
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 5,
      "n": 16,
      "accuracy": 0.3125
     }
    },
    "has_distribution": true,
    "brier_mean": 0.8370279252684946,
    "ece": 0.2653272879394618,
    "probability_fidelity": 65.34895862362534,
    "calibration_score": 56.141750517866484,
    "onehot": {
     "ece": 0.6227272727272728,
     "probability_fidelity": 41.69349999999999,
     "calibration_score": 20.846749999999997
    },
    "latency_p50_s": 1.2314840070903301,
    "latency_p95_s": 41.43838266544044,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 9.759108258784872,
    "Balanced 33:33:33 (no calibration)": 9.144291507137781,
    "Emphasis on Accuracy 60:20:20": 6.824322525126999,
    "Emphasis on Speed 20:60:20": 10.72864819263668,
    "Emphasis on Cost 20:20:60": 11.32112414415609,
    "Intelligence only": 4.943154589172655
   },
   "rank": 79,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 79,
    "Balanced 33:33:33 (no calibration)": 80,
    "Emphasis on Accuracy 60:20:20": 80,
    "Emphasis on Speed 20:60:20": 82,
    "Emphasis on Cost 20:20:60": 74,
    "Intelligence only": 81
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.48917748917748916,
   "sealed_accuracy": 0.32792207792207795,
   "public_minus_sealed_gap_pp": 16.12554112554112,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4324,
     "judge_hard": 0.561,
     "long_policy": 0.275,
     "multi_hop": 0.2632,
     "paraphrase_robustness": 0.0714,
     "probability": 0.25,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.3077,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.3303,
     "1/3": 0.2887,
     "2/3": 0.3627
    },
    "ece": 0.23909397495837956,
    "mean_tvd_gold_probs": 0.33684133216288276,
    "calibration": 59.248535896017906,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3182,
      "accuracy": 0.4592
     },
     "conf>=0.9": {
      "coverage": 0.1071,
      "accuracy": 0.5455
     }
    }
   },
   "v130_comparison_rank": 70,
   "v130_comparison_score": 16.61305398280731,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "clm-8b",
   "display": "CLM-8B (Contrastive-LM, clm-latest)",
   "author": "Contrastive-LM (Kwok, Kang, Suresh, Saad-Falcon, Pavone, Ré, Mirhoseini)",
   "repo": "https://github.com/Contrastive-LM/CLM",
   "class": "system-one-open",
   "licence": "Apache-2.0 (code and CLM-v0.1-8B head); Qwen3-8B encoder Apache-2.0",
   "open": "yes",
   "underlying": null,
   "has_distribution": true,
   "probability_source": null,
   "endpoint_condition": "our evaluator-owned Lium GPU pod (RTX PRO 6000), offline read-only container, author's server on loopback",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.6666666666666666,
    "standard": 0.3854166666666667,
    "judge": 0.7123287671232876,
    "hard": 0.35909090909090907
   },
   "speed": {
    "p50_s_raw": 0.016774784307926893,
    "p95_s_raw": 0.04461275222711264,
    "p50_s_adjusted": 0.18354956861585378,
    "p95_s_adjusted": 0.23922550445422527,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "RTX PRO 6000 (evaluator-owned Lium pod)",
    "measured_where": "serial standard+judge tasks, in-process Engine.answer on RTX PRO 6000 with CLM's documented embedding and action caches on (default settings; repeated option texts are served from cache), all 842 items run serially in one process; own-GPU adjustment ×2 + 0.15 s",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.005234400749063671,
    "basis": "Qwen3-Embedding-8B hosted list price (OpenRouter/DeepInfra $0.01/M input, read 2026-09-24); CLM usage.input_tokens (encoder tokens on cache misses, documented default action cache); estimated, not charged",
    "usd_per_1000_v11_tiers": null,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 39.81252400740181,
    "note": null
   },
   "hard": null,
   "release_evidence": {
    "aggregate_source_sha256": "38f491d18216387eba0de63c64474d7ab434d55dfbbc58522ddd6e17dbba5814",
    "row_sha256": "aba3020880495c47a920b2895ac88153f597f11f340a56733e113bab592a69c8",
    "raw_results_sha256": "c839df117e42099d3c4d7d72ea51735e03a1ea557d164b106253d5dcaa4427cf",
    "input_sha256": "6b06782a8a9fadfae770cf88985ec188f3244f01510e9b3f15b34c9723937cfc"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official offline measurement on an evaluator-owned Lium pod, 24 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.4069264069264069,
   "sealed_accuracy": 0.24025974025974026,
   "public_minus_sealed_gap_pp": 16.666666666666664,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3243,
     "judge_hard": 0.2195,
     "long_policy": 0.05,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.4286,
     "probability": 0.5,
     "safety_judge": 0.375,
     "temporal_numeric": 0.1786,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.2018,
     "1/3": 0.2371,
     "2/3": 0.2843
    },
    "ece": 0.37410391604330606,
    "mean_tvd_gold_probs": 0.3857029820010355,
    "calibration": 43.30445929561762,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.3896,
      "accuracy": 0.2417
     },
     "conf>=0.9": {
      "coverage": 0.1786,
      "accuracy": 0.2364
     }
    }
   },
   "v130_comparison_rank": 65,
   "v130_comparison_score": 16.50861480797255,
   "scoring_note": "Offline local open-weight inference on the frozen 308-item v1.4 set in a network-disabled, read-only container on an evaluator-owned Lium RTX PRO 6000 pod; no operator endpoint; no golds were exposed. Latency is in-process Engine.answer time on the serial standard+judge items with the self-hosted adjustment. Cost is estimated at the Qwen3-Embedding-8B hosted list price ($0.01/M input, same-size 8B pooling encoder) over CLM's measured encoder tokens; it is not a GPU bill.",
   "axes": {
    "intelligence": 22.354059018750526,
    "calibration": 39.81252400740181,
    "speed": 93.5743915280099,
    "cost": 78.43399091681239
   },
   "jevbench_score": 8.570462299422305,
   "presets": {
    "JevBench Score (25:25:25:25)": 8.570462299422305,
    "Balanced 33:33:33 (no calibration)": 8.796202829578965,
    "Emphasis on Accuracy 60:20:20": 6.339808685609935,
    "Emphasis on Speed 20:60:20": 11.161052780870685,
    "Emphasis on Cost 20:20:60": 10.669454200225223,
    "Intelligence only": 4.468164677335805
   },
   "rank": 80,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 80,
    "Balanced 33:33:33 (no calibration)": 82,
    "Emphasis on Accuracy 60:20:20": 82,
    "Emphasis on Speed 20:60:20": 81,
    "Emphasis on Cost 20:20:60": 78,
    "Intelligence only": 83
   }
  },
  {
   "key": "simplejev-qwen3.5-0.8b",
   "display": "SimpleJev (Qwen3.5-0.8B, CPU)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "sabeel111 / Featherless AI",
   "repo": "https://github.com/featherless-ai/simple-jev",
   "licence": "Apache-2.0 server; Apache-2.0 Qwen3.5-0.8B checkpoint",
   "underlying": "Qwen3.5-0.8B through SimpleJev native assistant-prefill option-logit scorer",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU, Qwen3.5-0.8B float32, 4 inference threads, local loopback HTTP, serial, author-documented 4096-token limit",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.875,
    "standard": 0.5729166666666666,
    "judge": 0.1917808219178082,
    "hard": 0.4
   },
   "axes": {
    "intelligence": 21.482479668016303,
    "calibration": 49.07354475109724,
    "speed": 57.454265052488324,
    "cost": 68.25265889330862
   },
   "jevbench_score": 7.459762136724042,
   "speed": {
    "p50_s_raw": 3.993786856532097,
    "p95_s_raw": 10.967020866274833,
    "p50_s_adjusted": 8.137573713064194,
    "p95_s_adjusted": 22.084041732549665,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": "Sandy shared CPU (4 inference threads)",
    "measured_where": "our CPU, float32, four PyTorch threads, author-documented small-model command",
    "hard_tier_p50_s": 8.088544107973576,
    "hard_tier_p95_s": 25.18287063837051
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.011435224719101125,
    "basis": "Same nonzero hosted size-class reference and exact prompt-token accounting; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.011435224719101125,
    "usd_per_1000_hard": 0.011435224719101125,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 49.07354475109724,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.3060342388575625,
    "probability_fidelity": 58.27785757969319,
    "brier_hard": 0.847391268136483,
    "brier_standard_judge_v11": 1.132994899496268,
    "note": null
   },
   "hard": {
    "run": "run/results.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 213,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.9681818181818181,
    "accuracy": 0.4,
    "by_family": {
     "adversarial": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 15,
      "n": 33,
      "accuracy": 0.45454545454545453
     },
     "long_policy": {
      "correct": 15,
      "n": 38,
      "accuracy": 0.39473684210526316
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 5,
      "n": 10,
      "accuracy": 0.5
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 9,
      "n": 16,
      "accuracy": 0.5625
     }
    },
    "has_distribution": true,
    "brier_mean": 0.847391268136483,
    "ece": 0.3060342388575625,
    "probability_fidelity": 58.27785757969319,
    "calibration_score": 48.535504904090345,
    "onehot": {
     "ece": 0.5868544600938967,
     "probability_fidelity": 37.127500000000005,
     "calibration_score": 18.563750000000002
    },
    "latency_p50_s": 8.088544107973576,
    "latency_p95_s": 25.18287063837051,
    "mean_input_tokens": 1551.2300469483569,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 7.459762136724042,
    "Balanced 33:33:33 (no calibration)": 7.045203163888045,
    "Emphasis on Accuracy 60:20:20": 5.375454325252341,
    "Emphasis on Speed 20:60:20": 8.138095206232048,
    "Emphasis on Cost 20:20:60": 8.553443321561211,
    "Intelligence only": 3.965639389317859
   },
   "rank": 81,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 81,
    "Balanced 33:33:33 (no calibration)": 83,
    "Emphasis on Accuracy 60:20:20": 83,
    "Emphasis on Speed 20:60:20": 84,
    "Emphasis on Cost 20:20:60": 82,
    "Intelligence only": 84
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5454545454545454,
   "sealed_accuracy": 0.3474025974025974,
   "public_minus_sealed_gap_pp": 19.8051948051948,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.7073,
     "long_policy": 0.15,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5,
     "probability": 0.3571,
     "safety_judge": 0.5625,
     "temporal_numeric": 0.3214,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.3761,
     "1/3": 0.3814,
     "2/3": 0.2843
    },
    "ece": 0.326324301213026,
    "mean_tvd_gold_probs": 0.3443589086717277,
    "calibration": 50.149624445111016,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.4545,
      "accuracy": 0.4429
     },
     "conf>=0.9": {
      "coverage": 0.3214,
      "accuracy": 0.5253
     }
    }
   },
   "v130_comparison_rank": 73,
   "v130_comparison_score": 11.60209541534599,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "gliner2.5-small",
   "display": "GLiNER2.5 small (Fastino, 74M)",
   "class": "classifier",
   "open": "yes",
   "author": "Fastino",
   "repo": "https://huggingface.co/fastino/gliner2.5-small-v1",
   "licence": "Apache-2.0",
   "underlying": "GLiNER2.5 small schema-conditioned extractor, 74M",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (4 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8333333333333334,
    "standard": 0.4791666666666667,
    "judge": 0.5,
    "hard": 0.33181818181818185
   },
   "axes": {
    "intelligence": 20.495135641685263,
    "calibration": 50.706891113316594,
    "speed": 77.83454067554766,
    "cost": 82.35883967864214
   },
   "jevbench_score": 7.187800408767937,
   "speed": {
    "p50_s_raw": 0.11413825303316116,
    "p95_s_raw": 2.1012388937175266,
    "p50_s_adjusted": 0.37827650606632235,
    "p95_s_adjusted": 4.3524777874350535,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": 0.30439889430999756,
    "hard_tier_p95_s": 11.560181383416035
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.003872921348314607,
    "basis": "ESTIMATE: hosted-provider price, deepinfra base-size encoders (bge-base, e5-base, gte-base, all-mpnet-base) list price $0.005/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 452 input and 0 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts)",
    "usd_per_1000_v11_tiers": 0.0022600000000000003,
    "usd_per_1000_hard": 0.006175,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 50.706891113316594,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.3264417127452114,
    "probability_fidelity": 59.613436666939876,
    "brier_hard": 0.9006336546139394,
    "brier_standard_judge_v11": 0.7413305857792032,
    "note": null
   },
   "hard": {
    "run": "runs/gliner2.5-small--hard (job jevbench-add-requests-20260919, run 3)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.33181818181818185,
    "by_family": {
     "adversarial": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 12,
      "n": 38,
      "accuracy": 0.3157894736842105
     },
     "multi_hop": {
      "correct": 10,
      "n": 35,
      "accuracy": 0.2857142857142857
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 2,
      "n": 10,
      "accuracy": 0.2
     },
     "temporal_numeric": {
      "correct": 6,
      "n": 30,
      "accuracy": 0.2
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 4,
      "n": 16,
      "accuracy": 0.25
     }
    },
    "has_distribution": true,
    "brier_mean": 0.9006336546139394,
    "ece": 0.3264417127452114,
    "probability_fidelity": 59.613436666939876,
    "calibration_score": 47.1625470589488,
    "onehot": {
     "ece": 0.6681818181818182,
     "probability_fidelity": 37.12949999999999,
     "calibration_score": 18.564749999999997
    },
    "latency_p50_s": 0.30439889430999756,
    "latency_p95_s": 11.560181383416035,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 7.187800408767937,
    "Balanced 33:33:33 (no calibration)": 6.831773818130067,
    "Emphasis on Accuracy 60:20:20": 4.902380947581899,
    "Emphasis on Speed 20:60:20": 8.44515008364971,
    "Emphasis on Cost 20:20:60": 8.56670947865354,
    "Intelligence only": 3.443597486140591
   },
   "rank": 82,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 82,
    "Balanced 33:33:33 (no calibration)": 84,
    "Emphasis on Accuracy 60:20:20": 85,
    "Emphasis on Speed 20:60:20": 83,
    "Emphasis on Cost 20:20:60": 81,
    "Intelligence only": 85
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.4588744588744589,
   "sealed_accuracy": 0.2857142857142857,
   "public_minus_sealed_gap_pp": 17.31601731601732,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.2927,
     "long_policy": 0.25,
     "multi_hop": 0.1842,
     "paraphrase_robustness": 0.1429,
     "probability": 0.3929,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2321,
     "tradeoff": 0.3462,
     "trap_adversarial": 0.6667
    },
    "by_panel_stratum": {
     "0/3": 0.2569,
     "1/3": 0.1959,
     "2/3": 0.402
    },
    "ece": 0.24154998879734568,
    "mean_tvd_gold_probs": 0.3609884379642649,
    "calibration": 57.79557922205218,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2825,
      "accuracy": 0.4138
     },
     "conf>=0.9": {
      "coverage": 0.1916,
      "accuracy": 0.4746
     }
    }
   },
   "v130_comparison_rank": 72,
   "v130_comparison_score": 13.84974486725046,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "raw-qwen3-0.6b",
   "display": "Raw Qwen3 0.6B direct logits",
   "class": "raw-logit-control",
   "open": "yes",
   "author": "Alibaba Qwen / neutral reproduction",
   "repo": "https://huggingface.co/Qwen/Qwen3-0.6B",
   "licence": "Apache-2.0",
   "underlying": "Qwen3-0.6B BF16",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8888888888888888,
    "standard": 0.5416666666666666,
    "judge": 0.4657534246575342,
    "hard": 0.35
   },
   "axes": {
    "intelligence": 22.819571275888357,
    "calibration": 20.703384538799043,
    "speed": 89.94746620755858,
    "cost": 73.92667572688353
   },
   "jevbench_score": 7.135291650528435,
   "speed": {
    "p50_s_raw": 0.0677520353347063,
    "p95_s_raw": 0.10226013627834617,
    "p50_s_adjusted": 0.28550407066941264,
    "p95_s_adjusted": 0.35452027255669233,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io RTX A6000 48 GB; local in-process; serial; one forward pass",
    "hard_tier_p50_s": 0.07314195716753602,
    "hard_tier_p95_s": 0.1294608116615563
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.007397940074906366,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.007397940074906366,
    "usd_per_1000_hard": 0.007397940074906366,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 20.703384538799043,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.5672142430649385,
    "probability_fidelity": 43.834050653431625,
    "brier_hard": 1.2011873169824652,
    "brier_standard_judge_v11": 0.9346598494441297,
    "note": null
   },
   "hard": {
    "run": "raw-controls/runs/qwen3-06/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.35,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 10,
      "n": 38,
      "accuracy": 0.2631578947368421
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 8,
      "n": 20,
      "accuracy": 0.4
     },
     "routing_hard": {
      "correct": 1,
      "n": 10,
      "accuracy": 0.1
     },
     "temporal_numeric": {
      "correct": 8,
      "n": 30,
      "accuracy": 0.26666666666666666
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 6,
      "n": 16,
      "accuracy": 0.375
     }
    },
    "has_distribution": true,
    "brier_mean": 1.2011873169824652,
    "ece": 0.5672142430649385,
    "probability_fidelity": 43.834050653431625,
    "calibration_score": 21.917025326715812,
    "onehot": {
     "ece": 0.65,
     "probability_fidelity": 40.471500000000006,
     "calibration_score": 20.235750000000003
    },
    "latency_p50_s": 0.07314195716753602,
    "latency_p95_s": 0.1294608116615563,
    "mean_input_tokens": 1236.2227272727273,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 7.135291650528435,
    "Balanced 33:33:33 (no calibration)": 9.126783777413298,
    "Emphasis on Accuracy 60:20:20": 6.6713314882223385,
    "Emphasis on Speed 20:60:20": 11.482310442731109,
    "Emphasis on Cost 20:20:60": 10.903072437121299,
    "Intelligence only": 4.753160001301175
   },
   "rank": 83,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 83,
    "Balanced 33:33:33 (no calibration)": 81,
    "Emphasis on Accuracy 60:20:20": 81,
    "Emphasis on Speed 20:60:20": 79,
    "Emphasis on Cost 20:20:60": 76,
    "Intelligence only": 82
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.4805194805194805,
   "sealed_accuracy": 0.2564935064935065,
   "public_minus_sealed_gap_pp": 22.4025974025974,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2703,
     "judge_hard": 0.2927,
     "long_policy": 0.3,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.1429,
     "probability": 0.3214,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.125,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.2371,
     "2/3": 0.3529
    },
    "ece": 0.6510417528775192,
    "mean_tvd_gold_probs": 0.63447794074069,
    "calibration": 18.276102962965503,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8734,
      "accuracy": 0.2677
     },
     "conf>=0.9": {
      "coverage": 0.7175,
      "accuracy": 0.3032
     }
    }
   },
   "v130_comparison_rank": 71,
   "v130_comparison_score": 14.695767847925103,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "verdict-small",
   "display": "verdict-small (Manavarya09, multilingual-e5-small 118M)",
   "author": "Manavarya09 (Manav)",
   "repo": "https://github.com/Manavarya09/verdict",
   "class": "jev-rebuild",
   "licence": "Apache-2.0 (verdictml code and the Manav2op/verdict-small checkpoint over intfloat/multilingual-e5-small)",
   "open": "yes",
   "underlying": "intfloat/multilingual-e5-small (118M multilingual bi-encoder) fine-tuned on a typed-decision mix; each option is scored against the rendered state by cosine similarity, with temperature scaling and a conformal abstain set on top. One encoder pass per option, no tokens generated.",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our CPU (8 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.8888888888888888,
    "standard": 0.5104166666666666,
    "judge": 0.2808219178082192,
    "hard": 0.4
   },
   "speed": {
    "p50_s_raw": 0.0534759983420372,
    "p95_s_raw": 0.5461470346897828,
    "p50_s_adjusted": 0.2569519966840744,
    "p95_s_adjusted": 1.2422940693795654,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": "AMD Ryzen 5 3600 (Sandy), 4 threads",
    "measured_where": "local, 4 CPU threads of a Ryzen 5 3600, model loaded before timing",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0012943970037453184,
    "basis": "ESTIMATE: hosted-provider price, deepinfra small multilingual encoders (multilingual-e5-small class, 118M) list price $0.004/M in, $0.0/M out (an encoder of the same size class, below the base-size encoders the open-jev-deberta and OpenDecision rows use; one forward pass, nothing generated) x 114 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.00045771974522293,
    "usd_per_1000_hard": 0.0024885636363636368
   },
   "calibration": {
    "score": 66.93933658008659,
    "note": null
   },
   "hard": {
    "run": "runs/verdict-small--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4,
    "by_family": {
     "adversarial": {
      "correct": 5,
      "n": 12,
      "accuracy": 0.4166666666666667
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 15,
      "n": 33,
      "accuracy": 0.45454545454545453
     },
     "long_policy": {
      "correct": 17,
      "n": 38,
      "accuracy": 0.4473684210526316
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 5,
      "n": 20,
      "accuracy": 0.25
     },
     "routing_hard": {
      "correct": 8,
      "n": 10,
      "accuracy": 0.8
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 8,
      "n": 16,
      "accuracy": 0.5
     }
    },
    "has_distribution": true,
    "brier_mean": 0.770191638,
    "ece": 0.1801,
    "probability_fidelity": 62.575250000000004,
    "calibration_score": 63.277625,
    "onehot": {
     "ece": 0.6,
     "probability_fidelity": 27.659000000000013,
     "calibration_score": 13.829500000000007
    },
    "latency_p50_s": 0.16950783133506775,
    "latency_p95_s": 0.996096399798989,
    "mean_input_tokens": 622.1409090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "release_evidence": {
    "sealed_aggregate_sha256": "642280f9f6334e540f6ff8a2df0930d9ed3b76f20b57e953e683fdf2ac6b1455",
    "detail_row_sha256": "ebe9519b426e67b128b827b3de375a1347d910d2a6be8661516b0828ecce6fce",
    "sealed_result_sha256": "89fc5de11c7fcb47382830c8c0d00ce7eacd08ff0d7a6d8ba4fbda0f4ca2611e"
   },
   "new_in": "v1.4.2",
   "source_round": "v1.4.2 addition (official full-protocol measurement, jevbench-add-requests run 11, 24 Sep 2026)",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.5367965367965368,
   "sealed_accuracy": 0.288961038961039,
   "public_minus_sealed_gap_pp": 24.783549783549784,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1892,
     "judge_hard": 0.6829,
     "long_policy": 0.2,
     "multi_hop": 0.1579,
     "paraphrase_robustness": 0.5,
     "probability": 0.3571,
     "safety_judge": 0.5,
     "temporal_numeric": 0.1429,
     "tradeoff": 0.1538,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.2936,
     "1/3": 0.3814,
     "2/3": 0.1961
    },
    "ece": 0.13407922077922074,
    "mean_tvd_gold_probs": 0.24658636363636363,
    "calibration": 74.26275974025975,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.1851,
      "accuracy": 0.5439
     },
     "conf>=0.9": {
      "coverage": 0.0195,
      "accuracy": 0.5
     }
    }
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": 12.006979027169065,
   "scoring_note": "Offline measurement of all 842 decisions (534 frozen v1.2 + 308 sealed v1.4) on our own CPU through the author's server; no operator endpoint.",
   "axes": {
    "intelligence": 18.11199374812305,
    "calibration": 66.93933658008659,
    "speed": 84.95923591278107,
    "cost": 96.63797503090915
   },
   "jevbench_score": 5.688474707640219,
   "presets": {
    "JevBench Score (25:25:25:25)": 5.688474707640219,
    "Balanced 33:33:33 (no calibration)": 5.090543661575937,
    "Emphasis on Accuracy 60:20:20": 3.494399132997257,
    "Emphasis on Speed 20:60:20": 6.504235489044836,
    "Emphasis on Cost 20:20:60": 6.693000784267155,
    "Intelligence only": 2.376614651299102
   },
   "rank": 84,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 84,
    "Balanced 33:33:33 (no calibration)": 85,
    "Emphasis on Accuracy 60:20:20": 86,
    "Emphasis on Speed 20:60:20": 85,
    "Emphasis on Cost 20:20:60": 85,
    "Intelligence only": 86
   }
  },
  {
   "key": "deepseek-flash",
   "display": "DeepSeek V4.1 Flash (thinking default)",
   "class": "llm-baseline",
   "open": "weights",
   "author": "DeepSeek",
   "repo": null,
   "licence": "open weights, proprietary API route",
   "underlying": "DeepSeek-V4.1-Flash",
   "has_distribution": true,
   "probability_source": [
    "verbalized"
   ],
   "endpoint_condition": "production API (DeepSeek)",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.9895833333333334,
    "judge": 0.9315068493150684,
    "hard": 0.95
   },
   "axes": {
    "intelligence": 93.99556953048156,
    "calibration": 95.50293974977504,
    "speed": 71.59823070734842,
    "cost": 16.793370728606376
   },
   "jevbench_score": 4.768648223526218,
   "speed": {
    "p50_s_raw": 1.4163860343396664,
    "p95_s_raw": 4.886470635980367,
    "p50_s_adjusted": 1.4163860343396664,
    "p95_s_adjusted": 4.886470635980367,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 3.1479117795825005,
    "hard_tier_p95_s": 27.954364685714225
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.593682584269663,
    "basis": "public tariff x measured tokens (https://api-docs.deepseek.com/quick_start/pricing (cache-miss off-peak; the run is on a Saturday, off-peak all day)) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | public tariff x measured tokens (hard-tier run)",
    "usd_per_1000_v11_tiers": 0.2505138535031847,
    "usd_per_1000_hard": 1.083477954545455,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 95.50293974977504,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.03334125816558947,
    "probability_fidelity": 99.998601695287,
    "brier_hard": 0.043915329620841076,
    "brier_standard_judge_v11": 0.02778397179722692,
    "note": null
   },
   "hard": {
    "run": "runs/deepseek-flash--hard-mt16k",
    "n_items": 220,
    "n_ok": 212,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.9636363636363636,
    "accuracy": 0.95,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 13,
      "n": 14,
      "accuracy": 0.9285714285714286
     },
     "judge_hard": {
      "correct": 30,
      "n": 33,
      "accuracy": 0.9090909090909091
     },
     "long_policy": {
      "correct": 34,
      "n": 38,
      "accuracy": 0.8947368421052632
     },
     "multi_hop": {
      "correct": 35,
      "n": 35,
      "accuracy": 1.0
     },
     "probability": {
      "correct": 19,
      "n": 20,
      "accuracy": 0.95
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 28,
      "n": 30,
      "accuracy": 0.9333333333333333
     },
     "tradeoff": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.043915329620841076,
    "ece": 0.03334125816558947,
    "probability_fidelity": 99.998601695287,
    "calibration_score": 96.66517503108454,
    "onehot": {
     "ece": 0.014150943396226467,
     "probability_fidelity": 65.48263157894736,
     "calibration_score": 81.32622144985103
    },
    "latency_p50_s": 3.1479117795825005,
    "latency_p95_s": 27.954364685714225,
    "mean_input_tokens": 1189.15,
    "mean_output_tokens": 1752.8727272727272,
    "charged_usd": 0.23836515000000008
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 4.768648223526218,
    "Balanced 33:33:33 (no calibration)": 4.021496975714034,
    "Emphasis on Accuracy 60:20:20": 5.349822285145684,
    "Emphasis on Speed 20:60:20": 5.032133072942978,
    "Emphasis on Cost 20:20:60": 2.7751115004133564,
    "Intelligence only": 10.60335070848706
   },
   "rank": 85,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 85,
    "Balanced 33:33:33 (no calibration)": 86,
    "Emphasis on Accuracy 60:20:20": 84,
    "Emphasis on Speed 20:60:20": 86,
    "Emphasis on Cost 20:20:60": 87,
    "Intelligence only": 73
   },
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.9783549783549783,
   "sealed_accuracy": 0.948051948051948,
   "public_minus_sealed_gap_pp": 3.0303030303030276,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 298,
    "by_family": {
     "ambiguous_abstain": 0.8919,
     "judge_hard": 0.9512,
     "long_policy": 0.95,
     "multi_hop": 0.9737,
     "paraphrase_robustness": 0.8571,
     "probability": 0.8929,
     "safety_judge": 1.0,
     "temporal_numeric": 1.0,
     "tradeoff": 0.9615,
     "trap_adversarial": 0.9167
    },
    "by_panel_stratum": {
     "0/3": 0.9358,
     "1/3": 0.9485,
     "2/3": 0.9608
    },
    "ece": 0.053169160248450045,
    "mean_tvd_gold_probs": 0.030092295759979366,
    "calibration": 93.17846918715603,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.8766,
      "accuracy": 0.9815
     },
     "conf>=0.9": {
      "coverage": 0.8377,
      "accuracy": 0.9845
     }
    }
   },
   "v130_comparison_rank": 50,
   "v130_comparison_score": 57.54288064128547,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (DeepSeek API), as for every API measurement; 298/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "mirror",
   "display": "Mirror",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "Bluusun",
   "repo": "https://github.com/Bluusun/Decision-API",
   "licence": "Apache-2.0 wrapper and Mirror release; upstream DeBERTa/model assets retain their own terms",
   "underlying": "Mirror DeBERTa-v3-large 436M classification-span scorer",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "hosted submission endpoint, serial from Germany; strict 512-token context rejection",
   "endpoint_kind": "demo",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.875,
    "standard": 0.4583333333333333,
    "judge": 0.3150684931506849,
    "hard": 0.17272727272727273
   },
   "axes": {
    "intelligence": 13.583262842535385,
    "calibration": 25.99761921833361,
    "speed": 70.75829315240604,
    "cost": 73.3279398087227
   },
   "jevbench_score": 2.110814199749038,
   "speed": {
    "p50_s_raw": 0.8995787426829338,
    "p95_s_raw": 2.333842311054468,
    "p50_s_adjusted": 1.7991574853658676,
    "p95_s_adjusted": 4.667684622108936,
    "adjustment": "x2 (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "hosted submission endpoint, serial from Germany; strict 512-token context rejection",
    "hard_tier_p50_s": 1.6949163414537907,
    "hard_tier_p95_s": 2.7078006714582443
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.007745842696629214,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.007745842696629214,
    "usd_per_1000_hard": 0.007745842696629214,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 25.99761921833361,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.4635793043570804,
    "probability_fidelity": 41.36101260578538,
    "brier_hard": 1.0110485400854703,
    "brier_standard_judge_v11": 0.7293307591154035,
    "note": null
   },
   "hard": {
    "run": "mirror/runs/results.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 108,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 0.4909090909090909,
    "accuracy": 0.17272727272727273,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 2,
      "n": 14,
      "accuracy": 0.14285714285714285
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 0,
      "n": 38,
      "accuracy": 0.0
     },
     "multi_hop": {
      "correct": 1,
      "n": 35,
      "accuracy": 0.02857142857142857
     },
     "probability": {
      "correct": 2,
      "n": 20,
      "accuracy": 0.1
     },
     "routing_hard": {
      "correct": 1,
      "n": 10,
      "accuracy": 0.1
     },
     "temporal_numeric": {
      "correct": 3,
      "n": 30,
      "accuracy": 0.1
     },
     "tradeoff": {
      "correct": 1,
      "n": 12,
      "accuracy": 0.08333333333333333
     },
     "trap": {
      "correct": 3,
      "n": 16,
      "accuracy": 0.1875
     }
    },
    "has_distribution": true,
    "brier_mean": 1.0110485400854703,
    "ece": 0.4635793043570804,
    "probability_fidelity": 41.36101260578538,
    "calibration_score": 24.322575867184653,
    "onehot": {
     "ece": 0.6481481481481481,
     "probability_fidelity": 39.01428571428571,
     "calibration_score": 19.507142857142856
    },
    "latency_p50_s": 1.6949163414537907,
    "latency_p95_s": 2.7078006714582443,
    "mean_input_tokens": 327.212962962963,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0003533899999999999
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 2.110814199749038,
    "Balanced 33:33:33 (no calibration)": 2.183706836728038,
    "Emphasis on Accuracy 60:20:20": 1.4841733421103378,
    "Emphasis on Speed 20:60:20": 2.846086787834975,
    "Emphasis on Cost 20:20:60": 2.8679969296983217,
    "Intelligence only": 1.002472124312386
   },
   "rank": 86,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 86,
    "Balanced 33:33:33 (no calibration)": 87,
    "Emphasis on Accuracy 60:20:20": 87,
    "Emphasis on Speed 20:60:20": 87,
    "Emphasis on Cost 20:20:60": 86,
    "Intelligence only": 87
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.41125541125541126,
   "sealed_accuracy": 0.09090909090909091,
   "public_minus_sealed_gap_pp": 32.03463203463204,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 102,
    "by_family": {
     "ambiguous_abstain": 0.1081,
     "judge_hard": 0.2683,
     "long_policy": 0.0,
     "multi_hop": 0.0263,
     "paraphrase_robustness": 0.0714,
     "probability": 0.1429,
     "safety_judge": 0.0,
     "temporal_numeric": 0.0893,
     "tradeoff": 0.0,
     "trap_adversarial": 0.1667
    },
    "by_panel_stratum": {
     "0/3": 0.055,
     "1/3": 0.0309,
     "2/3": 0.1863
    },
    "ece": 0.46805769782243106,
    "mean_tvd_gold_probs": 0.4769304859425074,
    "calibration": 29.347705920631523,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.2013,
      "accuracy": 0.3387
     },
     "conf>=0.9": {
      "coverage": 0.1526,
      "accuracy": 0.3617
     }
    }
   },
   "v130_comparison_rank": 74,
   "v130_comparison_score": 5.198860594004477,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest; 102/308 sealed items answered validly (failures count as wrong)"
  },
  {
   "key": "mxbai-rerank-base-v2",
   "display": "Mixedbread mxbai-rerank-base-v2",
   "class": "reranker",
   "open": "yes",
   "author": "Mixedbread",
   "repo": "https://huggingface.co/mixedbread-ai/mxbai-rerank-base-v2",
   "licence": "Apache-2.0",
   "underlying": "494M cross-encoder",
   "has_distribution": true,
   "probability_source": [
    "public_calibrated_reranker_softmax"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.4444444444444444,
    "standard": 0.3333333333333333,
    "judge": 0.2671232876712329,
    "hard": 0.4
   },
   "axes": {
    "intelligence": 6.801766676742608,
    "calibration": 84.11217431743692,
    "speed": 87.50360537571527,
    "cost": 67.92989463663068
   },
   "jevbench_score": 0.3999944834827258,
   "speed": {
    "p50_s_raw": 0.06881788885220885,
    "p95_s_raw": 0.23386348099447787,
    "p50_s_adjusted": 0.2876357777044177,
    "p95_s_adjusted": 0.6177269619889557,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.07308950461447239,
    "hard_tier_p95_s": 0.31191255310550325
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.011722048450570633,
    "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
    "usd_per_1000_v11_tiers": 0.009748850236231591,
    "usd_per_1000_hard": 0.014538340447399992,
    "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21."
   },
   "calibration": {
    "score": 84.11217431743692,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.02692298605433799,
    "probability_fidelity": 71.68117369175145,
    "brier_hard": 0.6540462249946867,
    "brier_standard_judge_v11": 0.6896480856085945,
    "note": null
   },
   "hard": {
    "run": "runs/mxbai-rerank-base-v2--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.4,
    "by_family": {
     "adversarial": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 18,
      "n": 33,
      "accuracy": 0.5454545454545454
     },
     "long_policy": {
      "correct": 14,
      "n": 38,
      "accuracy": 0.3684210526315789
     },
     "multi_hop": {
      "correct": 16,
      "n": 35,
      "accuracy": 0.45714285714285713
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 2,
      "n": 10,
      "accuracy": 0.2
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "trap": {
      "correct": 7,
      "n": 16,
      "accuracy": 0.4375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6540462249946867,
    "ece": 0.02692298605433799,
    "probability_fidelity": 71.68117369175145,
    "calibration_score": 83.14828824044193,
    "onehot": {
     "ece": 0.6,
     "probability_fidelity": 33.347,
     "calibration_score": 16.6735
    },
    "latency_p50_s": 0.07308950461447239,
    "latency_p95_s": 0.31191255310550325,
    "mean_input_tokens": 4350.459090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 0.3999944834827258,
    "Balanced 33:33:33 (no calibration)": 0.32059192904261735,
    "Emphasis on Accuracy 60:20:20": 0.1980433934885754,
    "Emphasis on Speed 20:60:20": 0.47201935306947584,
    "Emphasis on Cost 20:20:60": 0.4566763236106434,
    "Intelligence only": 0.12587085482985666
   },
   "rank": 87,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 87,
    "Balanced 33:33:33 (no calibration)": 88,
    "Emphasis on Accuracy 60:20:20": 88,
    "Emphasis on Speed 20:60:20": 88,
    "Emphasis on Cost 20:20:60": 88,
    "Intelligence only": 88
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.3722943722943723,
   "sealed_accuracy": 0.34415584415584416,
   "public_minus_sealed_gap_pp": 2.813852813852813,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.4324,
     "judge_hard": 0.4878,
     "long_policy": 0.175,
     "multi_hop": 0.2105,
     "paraphrase_robustness": 0.6429,
     "probability": 0.5714,
     "safety_judge": 0.375,
     "temporal_numeric": 0.25,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.3333
    },
    "by_panel_stratum": {
     "0/3": 0.3028,
     "1/3": 0.299,
     "2/3": 0.4314
    },
    "ece": 0.02699784510577387,
    "mean_tvd_gold_probs": 0.22520538035991342,
    "calibration": 86.03994647142693,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0,
      "accuracy": null
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 75,
   "v130_comparison_score": 0.764251199794221,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "bge-reranker-v2-m3",
   "display": "BAAI bge-reranker-v2-m3",
   "class": "reranker",
   "open": "yes",
   "author": "BAAI",
   "repo": "https://huggingface.co/BAAI/bge-reranker-v2-m3",
   "licence": "Apache-2.0",
   "underlying": "568M multilingual cross-encoder",
   "has_distribution": true,
   "probability_source": [
    "public_calibrated_reranker_softmax"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.4305555555555556,
    "standard": 0.3645833333333333,
    "judge": 0.08904109589041095,
    "hard": 0.36818181818181817
   },
   "axes": {
    "intelligence": 5.010951521682325,
    "calibration": 84.22377218469961,
    "speed": 89.52513700533652,
    "cost": 73.3733106383488
   },
   "jevbench_score": 0.1700654625666753,
   "speed": {
    "p50_s_raw": 0.034560746513307095,
    "p95_s_raw": 0.17954895906150325,
    "p50_s_adjusted": 0.21912149302661418,
    "p95_s_adjusted": 0.5090979181230065,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.040032369550317526,
    "hard_tier_p95_s": 0.24636488300748166
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.007718915951068509,
    "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
    "usd_per_1000_v11_tiers": 0.005582540915812308,
    "usd_per_1000_hard": 0.010768105774115997,
    "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21."
   },
   "calibration": {
    "score": 84.22377218469961,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.022679951965887575,
    "probability_fidelity": 72.18897252368282,
    "brier_hard": 0.6627383026655216,
    "brier_standard_judge_v11": 0.7142212126429204,
    "note": null
   },
   "hard": {
    "run": "runs/bge-reranker-v2-m3--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.36818181818181817,
    "by_family": {
     "adversarial": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "ambiguous": {
      "correct": 6,
      "n": 14,
      "accuracy": 0.42857142857142855
     },
     "judge_hard": {
      "correct": 17,
      "n": 33,
      "accuracy": 0.5151515151515151
     },
     "long_policy": {
      "correct": 16,
      "n": 38,
      "accuracy": 0.42105263157894735
     },
     "multi_hop": {
      "correct": 9,
      "n": 35,
      "accuracy": 0.2571428571428571
     },
     "probability": {
      "correct": 9,
      "n": 20,
      "accuracy": 0.45
     },
     "routing_hard": {
      "correct": 0,
      "n": 10,
      "accuracy": 0.0
     },
     "temporal_numeric": {
      "correct": 9,
      "n": 30,
      "accuracy": 0.3
     },
     "tradeoff": {
      "correct": 3,
      "n": 12,
      "accuracy": 0.25
     },
     "trap": {
      "correct": 6,
      "n": 16,
      "accuracy": 0.375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6627383026655216,
    "ece": 0.022679951965887575,
    "probability_fidelity": 72.18897252368282,
    "calibration_score": 83.82649106525265,
    "onehot": {
     "ece": 0.6318181818181818,
     "probability_fidelity": 39.899499999999996,
     "calibration_score": 19.949749999999998
    },
    "latency_p50_s": 0.040032369550317526,
    "latency_p95_s": 0.24636488300748166,
    "mean_input_tokens": 4623.740909090909,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 0.1700654625666753,
    "Balanced 33:33:33 (no calibration)": 0.13429893934169462,
    "Emphasis on Accuracy 60:20:20": 0.08054573117331676,
    "Emphasis on Speed 20:60:20": 0.203562500680883,
    "Emphasis on Cost 20:20:60": 0.19958398034330965,
    "Intelligence only": 0.05032926579082464
   },
   "rank": 88,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 88,
    "Balanced 33:33:33 (no calibration)": 89,
    "Emphasis on Accuracy 60:20:20": 89,
    "Emphasis on Speed 20:60:20": 89,
    "Emphasis on Cost 20:20:60": 89,
    "Intelligence only": 89
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.3939393939393939,
   "sealed_accuracy": 0.2792207792207792,
   "public_minus_sealed_gap_pp": 11.471861471861471,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2432,
     "judge_hard": 0.6341,
     "long_policy": 0.225,
     "multi_hop": 0.1579,
     "paraphrase_robustness": 0.2857,
     "probability": 0.2857,
     "safety_judge": 0.25,
     "temporal_numeric": 0.1964,
     "tradeoff": 0.2308,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.3211,
     "1/3": 0.3299,
     "2/3": 0.1863
    },
    "ece": 0.03946008009412363,
    "mean_tvd_gold_probs": 0.2207131513398822,
    "calibration": 85.01833442359353,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0,
      "accuracy": null
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 76,
   "v130_comparison_score": 0.6763072957019215,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "gte-reranker-modernbert-base",
   "display": "Alibaba GTE Reranker ModernBERT-base",
   "class": "reranker",
   "open": "yes",
   "author": "Alibaba-NLP",
   "repo": "https://huggingface.co/Alibaba-NLP/gte-reranker-modernbert-base",
   "licence": "Apache-2.0",
   "underlying": "149M ModernBERT cross-encoder",
   "has_distribution": true,
   "probability_source": [
    "public_calibrated_reranker_softmax"
   ],
   "endpoint_condition": "our GPU (lium.io A6000 48 GB), serial, one option batch per decision",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.3333333333333333,
    "standard": 0.3958333333333333,
    "judge": 0.3013698630136986,
    "hard": 0.33636363636363636
   },
   "axes": {
    "intelligence": 4.827030765393093,
    "calibration": 78.88620538670396,
    "speed": 90.58799397367488,
    "cost": 69.64001789071294
   },
   "jevbench_score": 0.15201475507543563,
   "speed": {
    "p50_s_raw": 0.048107756301760674,
    "p95_s_raw": 0.10235980194993316,
    "p50_s_adjusted": 0.24621551260352134,
    "p95_s_adjusted": 0.35471960389986634,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.051893850322812796,
    "hard_tier_p95_s": 0.18488985453732257
   },
   "cost": {
    "kind": "measured",
    "usd_per_1000": 0.010280148864935354,
    "basis": "MEASURED model inference time x lium.io A6000 tariff USD 0.42/hour",
    "usd_per_1000_v11_tiers": 0.01057049607889699,
    "usd_per_1000_hard": 0.009865744205008289,
    "self_host_sensitivity": "Whole five-model rental including setup/download was USD 0.21."
   },
   "calibration": {
    "score": 78.88620538670396,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.0812949237783053,
    "probability_fidelity": 69.95417916897236,
    "brier_hard": 0.6771539067858408,
    "brier_standard_judge_v11": 0.805535942323157,
    "note": null
   },
   "hard": {
    "run": "runs/gte-reranker-modernbert-base--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.33636363636363636,
    "by_family": {
     "adversarial": {
      "correct": 7,
      "n": 12,
      "accuracy": 0.5833333333333334
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 15,
      "n": 33,
      "accuracy": 0.45454545454545453
     },
     "long_policy": {
      "correct": 11,
      "n": 38,
      "accuracy": 0.2894736842105263
     },
     "multi_hop": {
      "correct": 8,
      "n": 35,
      "accuracy": 0.22857142857142856
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 0,
      "n": 10,
      "accuracy": 0.0
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 4,
      "n": 12,
      "accuracy": 0.3333333333333333
     },
     "trap": {
      "correct": 8,
      "n": 16,
      "accuracy": 0.5
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6771539067858408,
    "ece": 0.0812949237783053,
    "probability_fidelity": 69.95417916897236,
    "calibration_score": 76.84759720665565,
    "onehot": {
     "ece": 0.6636363636363636,
     "probability_fidelity": 34.50550000000001,
     "calibration_score": 17.252750000000006
    },
    "latency_p50_s": 0.051893850322812796,
    "latency_p95_s": 0.18488985453732257,
    "mean_input_tokens": 4257.440909090909,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 0.15201475507543563,
    "Balanced 33:33:33 (no calibration)": 0.1202254938915665,
    "Emphasis on Accuracy 60:20:20": 0.0720367147880534,
    "Emphasis on Speed 20:60:20": 0.18300291400672078,
    "Emphasis on Cost 20:20:60": 0.17835147709583657,
    "Intelligence only": 0.04498836311645224
   },
   "rank": 89,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 89,
    "Balanced 33:33:33 (no calibration)": 90,
    "Emphasis on Accuracy 60:20:20": 90,
    "Emphasis on Speed 20:60:20": 90,
    "Emphasis on Cost 20:20:60": 90,
    "Intelligence only": 90
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.33766233766233766,
   "sealed_accuracy": 0.3344155844155844,
   "public_minus_sealed_gap_pp": 0.32467532467532756,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.3514,
     "judge_hard": 0.7073,
     "long_policy": 0.125,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5,
     "probability": 0.2857,
     "safety_judge": 0.4375,
     "temporal_numeric": 0.25,
     "tradeoff": 0.1923,
     "trap_adversarial": 0.25
    },
    "by_panel_stratum": {
     "0/3": 0.3853,
     "1/3": 0.3814,
     "2/3": 0.2353
    },
    "ece": 0.04756891580703539,
    "mean_tvd_gold_probs": 0.24559373344991753,
    "calibration": 82.96342174680058,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0162,
      "accuracy": 0.2
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 77,
   "v130_comparison_score": 0.32219057759911496,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "certo",
   "display": "Certo v1 (AltSlate Labs)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "AltSlate Labs",
   "repo": "https://huggingface.co/altslate/certo-decision-model",
   "licence": "MIT",
   "underlying": "ModernBERT-large with a per-option query/scoring head, ~400M parameters",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "our RunPod GPU (GeForce RTX 3090 24 GB, community cloud), reached over the internet from Germany",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.2777777777777778,
    "standard": 0.3020833333333333,
    "judge": 0.2191780821917808,
    "hard": 0.3181818181818182
   },
   "axes": {
    "intelligence": 0.06943551498007035,
    "calibration": 82.98944985569986,
    "speed": 93.97471093191521,
    "cost": 100.0
   },
   "jevbench_score": 5.344170110166062e-07,
   "speed": {
    "p50_s_raw": 0.019205978140234947,
    "p95_s_raw": 0.031265050335787234,
    "p50_s_adjusted": 0.1884119562804699,
    "p95_s_adjusted": 0.21253010067157446,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.019906828412786126,
    "hard_tier_p95_s": 0.02713130824267863
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.0009654868913857679,
    "basis": "ESTIMATE: hosted-provider price, deepinfra encoders of the same size (bge-large, e5-large, Qwen3-Embedding-0.6B) list price $0.01/M in, $0.0/M out (an encoder of the same size class; one forward pass, nothing generated) x 86 input and 0 output tokens per decision (input tokens measured (the system's own count))",
    "usd_per_1000_v11_tiers": 0.0008618471337579619,
    "usd_per_1000_hard": 0.001113409090909091,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 82.98944985569986,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.037609090909090884,
    "probability_fidelity": 71.58125,
    "brier_hard": 0.6629653915000002,
    "brier_standard_judge_v11": 0.6887553576859503,
    "note": null
   },
   "hard": {
    "run": "runs/certo--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.3181818181818182,
    "by_family": {
     "adversarial": {
      "correct": 2,
      "n": 12,
      "accuracy": 0.16666666666666666
     },
     "ambiguous": {
      "correct": 4,
      "n": 14,
      "accuracy": 0.2857142857142857
     },
     "judge_hard": {
      "correct": 15,
      "n": 33,
      "accuracy": 0.45454545454545453
     },
     "long_policy": {
      "correct": 13,
      "n": 38,
      "accuracy": 0.34210526315789475
     },
     "multi_hop": {
      "correct": 7,
      "n": 35,
      "accuracy": 0.2
     },
     "probability": {
      "correct": 7,
      "n": 20,
      "accuracy": 0.35
     },
     "routing_hard": {
      "correct": 2,
      "n": 10,
      "accuracy": 0.2
     },
     "temporal_numeric": {
      "correct": 12,
      "n": 30,
      "accuracy": 0.4
     },
     "tradeoff": {
      "correct": 2,
      "n": 12,
      "accuracy": 0.16666666666666666
     },
     "trap": {
      "correct": 6,
      "n": 16,
      "accuracy": 0.375
     }
    },
    "has_distribution": true,
    "brier_mean": 0.6629653915000002,
    "ece": 0.037609090909090884,
    "probability_fidelity": 71.58125,
    "calibration_score": 82.02971590909091,
    "onehot": {
     "ece": 0.6818181818181819,
     "probability_fidelity": 38.06900000000001,
     "calibration_score": 19.034500000000005
    },
    "latency_p50_s": 0.019906828412786126,
    "latency_p95_s": 0.02713130824267863,
    "mean_input_tokens": 111.3409090909091,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 5.344170110166062e-07,
    "Balanced 33:33:33 (no calibration)": 4.0114762986019647e-07,
    "Emphasis on Accuracy 60:20:20": 2.2307263219090737e-07,
    "Emphasis on Speed 20:60:20": 6.675942582993148e-07,
    "Emphasis on Cost 20:20:60": 6.676535327190674e-07,
    "Intelligence only": 1.3390752217542618e-07
   },
   "rank": 90,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 90,
    "Balanced 33:33:33 (no calibration)": 91,
    "Emphasis on Accuracy 60:20:20": 91,
    "Emphasis on Speed 20:60:20": 91,
    "Emphasis on Cost 20:20:60": 91,
    "Intelligence only": 91
   },
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.31601731601731603,
   "sealed_accuracy": 0.29545454545454547,
   "public_minus_sealed_gap_pp": 2.056277056277056,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.1622,
     "judge_hard": 0.6098,
     "long_policy": 0.175,
     "multi_hop": 0.3158,
     "paraphrase_robustness": 0.5,
     "probability": 0.3214,
     "safety_judge": 0.3125,
     "temporal_numeric": 0.2143,
     "tradeoff": 0.1154,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.2752,
     "1/3": 0.3402,
     "2/3": 0.2745
    },
    "ece": 0.037789610389610415,
    "mean_tvd_gold_probs": 0.22624242424242422,
    "calibration": 84.90891774891774,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.0,
      "accuracy": null
     },
     "conf>=0.9": {
      "coverage": 0.0,
      "accuracy": null
     }
    }
   },
   "v130_comparison_rank": 78,
   "v130_comparison_score": 0.0,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "open-jev-json-canvas-joshuasp",
   "display": "Open Jev JSON Canvas (JoshuaSP)",
   "class": "jev-rebuild",
   "open": "yes",
   "author": "JoshuaSP",
   "repo": "https://github.com/JoshuaSP/open-jev",
   "licence": "MIT code; Apache-2.0 DiffusionGemma weights",
   "underlying": "google/diffusiongemma-26B-A4B-it BF16, one-step JSON canvas",
   "has_distribution": false,
   "probability_source": [
    "label_only_no_calibrated_distribution"
   ],
   "endpoint_condition": "our lium.io H100 80 GB; local in-process; serial; seed 0; one denoising step",
   "endpoint_kind": "gpu",
   "partial": false,
   "ranked": true,
   "listing": "ranked",
   "not_ranked_because": null,
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.9246575342465754,
    "hard": 0.7045454545454546
   },
   "axes": {
    "intelligence": 48.158292382831725,
    "calibration": 0.0,
    "speed": 84.06148554434631,
    "cost": 45.616507411040736
   },
   "jevbench_score": 0.0,
   "speed": {
    "p50_s_raw": 0.2243216049973853,
    "p95_s_raw": 0.2528335442475509,
    "p50_s_adjusted": 0.5986432099947706,
    "p95_s_adjusted": 0.6556670884951018,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge subset of completed 534 run",
    "hardware": null,
    "measured_where": "our lium.io H100 80 GB; local in-process; serial; seed 0; one denoising step",
    "hard_tier_p50_s": 0.23627323599066585,
    "hard_tier_p95_s": 0.3619438183726741
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.06498050561797752,
    "basis": "Nonzero hosted-comparable estimate; see RESULT.md",
    "usd_per_1000_v11_tiers": 0.06498050561797752,
    "usd_per_1000_hard": 0.06498050561797752,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 0.0,
    "score_label_only_as_onehot": 49.026295454545455,
    "ece_hard": null,
    "probability_fidelity": null,
    "brier_hard": null,
    "brier_standard_judge_v11": null,
    "note": "returns a label, not a probability distribution: no calibration score (counts as 0 in the JevBench Score)"
   },
   "hard": {
    "run": "joshuasp/runs/canonical-results-v1.3.0.jsonl (hard-tier subset)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7045454545454546,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 9,
      "n": 14,
      "accuracy": 0.6428571428571429
     },
     "judge_hard": {
      "correct": 27,
      "n": 33,
      "accuracy": 0.8181818181818182
     },
     "long_policy": {
      "correct": 21,
      "n": 38,
      "accuracy": 0.5526315789473685
     },
     "multi_hop": {
      "correct": 26,
      "n": 35,
      "accuracy": 0.7428571428571429
     },
     "probability": {
      "correct": 15,
      "n": 20,
      "accuracy": 0.75
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 11,
      "n": 30,
      "accuracy": 0.36666666666666664
     },
     "tradeoff": {
      "correct": 8,
      "n": 12,
      "accuracy": 0.6666666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": false,
    "brier_mean": null,
    "ece": null,
    "probability_fidelity": null,
    "calibration_score": null,
    "onehot": {
     "ece": 0.2954545454545454,
     "probability_fidelity": 57.14349999999999,
     "calibration_score": 49.026295454545455
    },
    "latency_p50_s": 0.23627323599066585,
    "latency_p95_s": 0.3619438183726741,
    "mean_input_tokens": 1219.290909090909,
    "mean_output_tokens": 0.0,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 0.0,
    "Balanced 33:33:33 (no calibration)": 42.439636480559415,
    "Emphasis on Accuracy 60:20:20": 40.16948693800466,
    "Emphasis on Speed 20:60:20": 49.260539617014324,
    "Emphasis on Cost 20:20:60": 39.225079065063476,
    "Intelligence only": 37.18581305994612
   },
   "rank": 91,
   "rank_under": {
    "JevBench Score (25:25:25:25)": 91,
    "Balanced 33:33:33 (no calibration)": 20,
    "Emphasis on Accuracy 60:20:20": 18,
    "Emphasis on Speed 20:60:20": 19,
    "Emphasis on Cost 20:20:60": 25,
    "Intelligence only": 15
   },
   "source_round": "round 4 unpublished",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.8441558441558441,
   "sealed_accuracy": 0.3116883116883117,
   "public_minus_sealed_gap_pp": 53.246753246753244,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2162,
     "judge_hard": 0.4146,
     "long_policy": 0.275,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.5714,
     "probability": 0.3214,
     "safety_judge": 0.375,
     "temporal_numeric": 0.2679,
     "tradeoff": 0.1154,
     "trap_adversarial": 0.4167
    },
    "by_panel_stratum": {
     "0/3": 0.1835,
     "1/3": 0.2371,
     "2/3": 0.5196
    },
    "ece": 0.6883116883116883,
    "mean_tvd_gold_probs": 0.6473121212121212,
    "calibration": 17.634393939393938,
    "label_only": true,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 1.0,
      "accuracy": 0.3117
     },
     "conf>=0.9": {
      "coverage": 1.0,
      "accuracy": 0.3117
     }
    }
   },
   "v130_comparison_rank": 68,
   "v130_comparison_score": 23.76873427028549,
   "scoring_note": "re-run on a throwaway RunPod pod with the original recipe; deviations in its manifest"
  },
  {
   "key": "classifier-dev-fast",
   "display": "classifier.dev (fast tier)",
   "class": "jev-service",
   "open": "no",
   "author": "mrmps (@michael_chomsky)",
   "repo": "https://classifier.dev",
   "licence": "MIT (code); hosted service",
   "underlying": "Jev (TypeSafe) behind classifier.dev's zero-shot classification API",
   "has_distribution": true,
   "probability_source": [
    "native"
   ],
   "endpoint_condition": "production API (classifier.dev, fast tier)",
   "endpoint_kind": "api",
   "partial": false,
   "ranked": false,
   "listing": "honorable_mention",
   "not_ranked_because": "runs on Jev (TypeSafe) — listed, not ranked",
   "tiers": {
    "easy": 1.0,
    "standard": 0.9895833333333334,
    "judge": 0.9726027397260274,
    "hard": 0.7045454545454546
   },
   "axes": {
    "intelligence": 51.562124154634155,
    "calibration": 72.42569314355679,
    "speed": 87.59107265834845,
    "cost": 84.31363764158988
   },
   "jevbench_score": 70.82340862727794,
   "speed": {
    "p50_s_raw": 0.38636084645986557,
    "p95_s_raw": 0.4507125232368707,
    "p50_s_adjusted": 0.38636084645986557,
    "p95_s_adjusted": 0.4507125232368707,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 0.38397014886140823,
    "hard_tier_p95_s": 0.45760550089180463
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.003333333333333333,
    "basis": "ESTIMATE from the published paid plan (the free tier was used): classifier.dev Pro $20/month for 200,000 fast classifications a day (https://classifier.dev/pricing, read 2026-09-19) = $0.0033 per 1,000 decisions at full use; one decision = one classification. Lower use costs more per decision: at a tenth of that allowance it is $0.033 per 1,000, and the free tier (20,000 fast classifications a day, which is what this run used) costs nothing.",
    "usd_per_1000_v11_tiers": 0.003333333333333333,
    "usd_per_1000_hard": 0.003333333333333333,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": 72.42569314355679,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.09111937557392094,
    "probability_fidelity": 73.9555,
    "brier_hard": 0.3604363282039868,
    "brier_standard_judge_v11": 0.04268640347881518,
    "note": null
   },
   "hard": {
    "run": "runs/classifier-dev-fast--hard (job jevbench-add-requests-20260919)",
    "n_items": 220,
    "n_ok": 220,
    "n_attempted": 220,
    "coverage": 1.0,
    "success_rate": 1.0,
    "accuracy": 0.7045454545454546,
    "by_family": {
     "adversarial": {
      "correct": 12,
      "n": 12,
      "accuracy": 1.0
     },
     "ambiguous": {
      "correct": 11,
      "n": 14,
      "accuracy": 0.7857142857142857
     },
     "judge_hard": {
      "correct": 26,
      "n": 33,
      "accuracy": 0.7878787878787878
     },
     "long_policy": {
      "correct": 20,
      "n": 38,
      "accuracy": 0.5263157894736842
     },
     "multi_hop": {
      "correct": 28,
      "n": 35,
      "accuracy": 0.8
     },
     "probability": {
      "correct": 14,
      "n": 20,
      "accuracy": 0.7
     },
     "routing_hard": {
      "correct": 10,
      "n": 10,
      "accuracy": 1.0
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 11,
      "n": 12,
      "accuracy": 0.9166666666666666
     },
     "trap": {
      "correct": 16,
      "n": 16,
      "accuracy": 1.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.3604363282039868,
    "ece": 0.09111937557392094,
    "probability_fidelity": 73.9555,
    "calibration_score": 77.8658124426079,
    "onehot": {
     "ece": 0.2954545454545454,
     "probability_fidelity": 56.04099999999998,
     "calibration_score": 48.47504545454545
    },
    "latency_p50_s": 0.38397014886140823,
    "latency_p95_s": 0.45760550089180463,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 70.82340862727794,
    "Balanced 33:33:33 (no calibration)": 70.30495294014848,
    "Emphasis on Accuracy 60:20:20": 61.38026412849398,
    "Emphasis on Speed 20:60:20": 76.33048926745226,
    "Emphasis on Cost 20:20:60": 75.31004946963148,
    "Intelligence only": 51.562124154634155
   },
   "rank": null,
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.8528138528138528,
   "sealed_accuracy": 0.34415584415584416,
   "public_minus_sealed_gap_pp": 50.86580086580086,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 308,
    "by_family": {
     "ambiguous_abstain": 0.2973,
     "judge_hard": 0.3659,
     "long_policy": 0.175,
     "multi_hop": 0.3684,
     "paraphrase_robustness": 0.5714,
     "probability": 0.5,
     "safety_judge": 0.375,
     "temporal_numeric": 0.3036,
     "tradeoff": 0.2692,
     "trap_adversarial": 0.5833
    },
    "by_panel_stratum": {
     "0/3": 0.1193,
     "1/3": 0.2887,
     "2/3": 0.6373
    },
    "ece": 0.22765151515151522,
    "mean_tvd_gold_probs": 0.31378787878787867,
    "calibration": 61.54545454545454,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.289,
      "accuracy": 0.3933
     },
     "conf>=0.9": {
      "coverage": 0.1234,
      "accuracy": 0.3684
     }
    }
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": 83.64670446057299,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (classifier.dev), as for every API measurement",
   "rank_under": {}
  },
  {
   "key": "qwen3.8-27b",
   "display": "Qwen3.8 27B (Chutes TEE)",
   "class": "llm-baseline",
   "open": "weights",
   "author": "Qwen / Chutes",
   "repo": null,
   "licence": "open weights",
   "underlying": "Qwen3.8-27B",
   "has_distribution": true,
   "probability_source": [
    "verbalized"
   ],
   "endpoint_condition": "Chutes shared inference (TEE)",
   "endpoint_kind": "api",
   "partial": true,
   "ranked": false,
   "listing": "partial",
   "not_ranked_because": null,
   "tiers": {
    "easy": 0.9861111111111112,
    "standard": 0.9895833333333334,
    "judge": 0.952755905511811,
    "hard": 0.21363636363636362
   },
   "axes": {
    "intelligence": 40.40114828032634,
    "calibration": 93.55866776479952,
    "speed": 61.270489995780956,
    "cost": 0.0
   },
   "jevbench_score": 0.0,
   "speed": {
    "p50_s_raw": 5.753906108438969,
    "p95_s_raw": 12.97144114784895,
    "p50_s_adjusted": 5.753906108438969,
    "p95_s_adjusted": 12.97144114784895,
    "adjustment": "none (production API)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "from a Hetzner server in Germany, network included",
    "hard_tier_p50_s": 20.471472900360823,
    "hard_tier_p95_s": 74.47066541947424
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 2.669088141927965,
    "basis": "ESTIMATE: hosted-provider price, openrouter qwen/qwen3.8-27b list price $0.214/M in, $2.55/M out (same weights; our run used a flat-rate Chutes subscription) x 416 input and 393 output tokens per decision [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter qwen/qwen3.8-27b $0.214/M in, $2.55/M out x 1592 in / 1833 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 1.0249957201365187,
    "usd_per_1000_hard": 5.0156564166666655,
    "self_host_sensitivity": {
     "usd_per_1000": 3.8023331395736286,
     "score": 10.498745882580529,
     "machine": "1x A100 80 GB (Qwen3.8-27B bf16)",
     "usd_per_h": 1.39,
     "concurrency": 4,
     "utilisation": 0.3,
     "p50_s_used": 11.81732313881876,
     "decisions_per_hour": 365.5650225734472
    }
   },
   "calibration": {
    "score": 93.55866776479952,
    "score_label_only_as_onehot": null,
    "ece_hard": 0.07900871212083338,
    "probability_fidelity": 99.999484849,
    "brier_hard": 0.07885637798841445,
    "brier_standard_judge_v11": 0.019061104047845712,
    "note": null
   },
   "hard": {
    "run": "runs/qwen3.8-27b--hard-mt16k",
    "n_items": 220,
    "n_ok": 48,
    "n_attempted": 51,
    "coverage": 0.2318181818181818,
    "success_rate": 0.9411764705882353,
    "accuracy": 0.21363636363636362,
    "by_family": {
     "adversarial": {
      "correct": 0,
      "n": 12,
      "accuracy": 0.0
     },
     "ambiguous": {
      "correct": 7,
      "n": 14,
      "accuracy": 0.5
     },
     "judge_hard": {
      "correct": 0,
      "n": 33,
      "accuracy": 0.0
     },
     "long_policy": {
      "correct": 12,
      "n": 38,
      "accuracy": 0.3157894736842105
     },
     "multi_hop": {
      "correct": 5,
      "n": 35,
      "accuracy": 0.14285714285714285
     },
     "probability": {
      "correct": 10,
      "n": 20,
      "accuracy": 0.5
     },
     "routing_hard": {
      "correct": 0,
      "n": 10,
      "accuracy": 0.0
     },
     "temporal_numeric": {
      "correct": 7,
      "n": 30,
      "accuracy": 0.23333333333333334
     },
     "tradeoff": {
      "correct": 6,
      "n": 12,
      "accuracy": 0.5
     },
     "trap": {
      "correct": 0,
      "n": 16,
      "accuracy": 0.0
     }
    },
    "has_distribution": true,
    "brier_mean": 0.07885637798841445,
    "ece": 0.07900871212083338,
    "probability_fidelity": 99.999484849,
    "calibration_score": 92.09887121241667,
    "onehot": {
     "ece": 0.02083333333333337,
     "probability_fidelity": 66.726,
     "calibration_score": 81.27966666666666
    },
    "latency_p50_s": 20.471472900360823,
    "latency_p95_s": 74.47066541947424,
    "mean_input_tokens": 1591.6041666666667,
    "mean_output_tokens": 1833.3541666666667,
    "charged_usd": 0.0
   },
   "presets": {
    "JevBench Score (25:25:25:25)": 0.0,
    "Balanced 33:33:33 (no calibration)": 0.0,
    "Emphasis on Accuracy 60:20:20": 0.0,
    "Emphasis on Speed 20:60:20": 0.0,
    "Emphasis on Cost 20:20:60": 0.0,
    "Intelligence only": 0.0
   },
   "rank": null,
   "source_round": "v1.3.0 public",
   "api_flag": true,
   "api_exposure_note": "API — the operator's endpoint received sealed item text, without answers.",
   "public_accuracy": 0.7186147186147186,
   "sealed_accuracy": 0.21753246753246752,
   "public_minus_sealed_gap_pp": 50.1082251082251,
   "sealed_aggregate": {
    "n": 308,
    "answered_valid": 69,
    "by_family": {
     "ambiguous_abstain": 0.7838,
     "judge_hard": 0.9268,
     "long_policy": 0.0,
     "multi_hop": 0.0,
     "paraphrase_robustness": 0.0,
     "probability": 0.0,
     "safety_judge": 0.0,
     "temporal_numeric": 0.0,
     "tradeoff": 0.0,
     "trap_adversarial": 0.0
    },
    "by_panel_stratum": {
     "0/3": 0.1468,
     "1/3": 0.2165,
     "2/3": 0.2941
    },
    "ece": 0.017608695652173934,
    "mean_tvd_gold_probs": null,
    "calibration": 96.47826086956522,
    "label_only": false,
    "risk_coverage": {
     "conf>=0.7": {
      "coverage": 0.224,
      "accuracy": 0.971
     },
     "conf>=0.9": {
      "coverage": 0.224,
      "accuracy": 0.971
     }
    }
   },
   "v130_comparison_rank": null,
   "v130_comparison_score": 24.836695615027768,
   "scoring_note": "sealed item text (no golds) was sent to the operator endpoint (Chutes), as for every API measurement; 69/308 sealed items answered validly (failures count as wrong); partial: Chutes rate limit stopped the run after 81/308 items; unranked as in v1.3",
   "rank_under": {}
  },
  {
   "key": "needle-3-tools",
   "display": "Needle 3, options as tools (post-hoc adapter mode)",
   "class": "small-tool-model",
   "open": "yes",
   "author": "Cactus Compute",
   "repo": "https://github.com/cactus-compute/needle",
   "licence": "Apache-2.0 (model and package)",
   "underlying": "Needle 3 (121M parameters, 2-bit)",
   "has_distribution": false,
   "probability_source": [
    "label_only_no_calibrated_distribution"
   ],
   "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": true,
   "ranked": false,
   "listing": "partial",
   "not_ranked_because": "No completed sealed v1.4 run; retained as an unranked historical row",
   "tiers": {
    "easy": 0.6666666666666666,
    "standard": 0.3125,
    "judge": 0.3424657534246575,
    "hard": null
   },
   "axes": null,
   "jevbench_score": null,
   "speed": {
    "p50_s_raw": 3.778900783509016,
    "p95_s_raw": 33.63718595951795,
    "p50_s_adjusted": 7.707801567018032,
    "p95_s_adjusted": 67.4243719190359,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "local, 2 CPU threads of a Ryzen 5 3600",
    "hard_tier_p50_s": null,
    "hard_tier_p95_s": null
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.014372006369426754,
    "basis": "ESTIMATE: same per-token price as Needle 3 (openrouter meta-llama/llama-3.2-1b-instruct $0.027/M in, $0.201/M out) x 383 input and 20 output tokens per decision, over the 314 easy/standard/judge decisions it ran (no hard-tier run). The v1.2 score lab had no price for this row and scored it 100; fixed. [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json]",
    "usd_per_1000_v11_tiers": 0.014372006369426754,
    "usd_per_1000_hard": null,
    "self_host_sensitivity": null
   },
   "calibration": {
    "score": null,
    "score_label_only_as_onehot": null,
    "ece_hard": null,
    "probability_fidelity": null,
    "brier_hard": null,
    "brier_standard_judge_v11": null,
    "note": "returns a label, not a probability distribution: no calibration score (counts as 0 in the JevBench Score)"
   },
   "hard": null,
   "presets": {},
   "rank": null,
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.22077922077922077,
   "sealed_accuracy": null,
   "public_minus_sealed_gap_pp": null,
   "sealed_aggregate": null,
   "v130_comparison_rank": null,
   "v130_comparison_score": 1.0757743846539873,
   "scoring_note": null,
   "rank_under": {}
  },
  {
   "key": "needle-3",
   "display": "Needle 3 (Cactus, 2-bit, local CPU)",
   "class": "small-tool-model",
   "open": "yes",
   "author": "Cactus Compute",
   "repo": "https://github.com/cactus-compute/needle",
   "licence": "Apache-2.0 (model and package)",
   "underlying": "Needle 3 (121M parameters, 2-bit)",
   "has_distribution": false,
   "probability_source": [
    "label_only_no_calibrated_distribution"
   ],
   "endpoint_condition": "our CPU (2 threads, Ryzen 5 3600)",
   "endpoint_kind": "cpu",
   "partial": true,
   "ranked": false,
   "listing": "partial",
   "not_ranked_because": "No completed sealed v1.4 run; retained as an unranked historical row",
   "tiers": {
    "easy": 0.4722222222222222,
    "standard": 0.16666666666666666,
    "judge": 0.3150684931506849,
    "hard": 0.07727272727272727
   },
   "axes": null,
   "jevbench_score": null,
   "speed": {
    "p50_s_raw": 1.6870602630078793,
    "p95_s_raw": 14.364453018829225,
    "p50_s_adjusted": 3.5241205260157584,
    "p95_s_adjusted": 28.87890603765845,
    "adjustment": "x2 + 0.15 s (assumption, not measured)",
    "run": "serial 242-decision standard+judge run",
    "hardware": null,
    "measured_where": "local, 2 CPU threads of a Ryzen 5 3600",
    "hard_tier_p50_s": 19.284176252782345,
    "hard_tier_p95_s": 155.37855613604194
   },
   "cost": {
    "kind": "estimate",
    "usd_per_1000": 0.023839264044943822,
    "basis": "ESTIMATE: hosted-provider price, openrouter meta-llama/llama-3.2-1b-instruct list price $0.027/M in, $0.201/M out (no generative model under 1B is listed; the smallest listed one (1B) errs high; about 20 generated tokens for one tool call) x 383 input and 20 output tokens per decision (input tokens counted from the gemini-3.1-flash-lite run, same prompts) [corrected in v1.2.3: the price now averages each of the 314 v1.1 decisions once; see results/v1.2/cost-correction-v1.2.3.json] | ESTIMATE: openrouter meta-llama/llama-3.2-1b-instruct $0.027/M in, $0.201/M out x 1235 in / 20 out tokens per hard decision",
    "usd_per_1000_v11_tiers": 0.014372006369426754,
    "usd_per_1000_hard": 0.03735162272727273,
    "self_host_sensitivity": {
     "usd_per_1000": 0.059578722823927475,
     "score": 55.62272027242932,
     "machine": "Hetzner CX22 (2 vCPU)",
     "usd_per_h": 0.0072,
     "concurrency": 1,
     "utilisation": 0.3,
     "p50_s_used": 8.93680842358912,
     "decisions_per_hour": 120.84851199778322
    }
   },
   "calibration": {
    "score": null,
    "score_label_only_as_onehot": 22.869444444444444,
    "ece_hard": null,
    "probability_fidelity": null,
    "brier_hard": null,
    "brier_standard_judge_v11": null,
    "note": "returns a label, not a probability distribution: no calibration score (counts as 0 in the JevBench Score)"
   },
   "hard": {
    "run": "runs/needle-3--hard",
    "n_items": 220,
    "n_ok": 44,
    "n_attempted": 44,
    "coverage": 0.2,
    "success_rate": 1.0,
    "accuracy": 0.07727272727272727,
    "by_family": {
     "adversarial": {
      "correct": 0,
      "n": 12,
      "accuracy": 0.0
     },
     "ambiguous": {
      "correct": 3,
      "n": 14,
      "accuracy": 0.21428571428571427
     },
     "judge_hard": {
      "correct": 0,
      "n": 33,
      "accuracy": 0.0
     },
     "long_policy": {
      "correct": 5,
      "n": 38,
      "accuracy": 0.13157894736842105
     },
     "multi_hop": {
      "correct": 0,
      "n": 35,
      "accuracy": 0.0
     },
     "probability": {
      "correct": 4,
      "n": 20,
      "accuracy": 0.2
     },
     "routing_hard": {
      "correct": 0,
      "n": 10,
      "accuracy": 0.0
     },
     "temporal_numeric": {
      "correct": 3,
      "n": 30,
      "accuracy": 0.1
     },
     "tradeoff": {
      "correct": 2,
      "n": 12,
      "accuracy": 0.16666666666666666
     },
     "trap": {
      "correct": 0,
      "n": 16,
      "accuracy": 0.0
     }
    },
    "has_distribution": false,
    "brier_mean": null,
    "ece": null,
    "probability_fidelity": null,
    "calibration_score": null,
    "onehot": {
     "ece": 0.6136363636363636,
     "probability_fidelity": 45.73888888888889,
     "calibration_score": 22.869444444444444
    },
    "latency_p50_s": 19.284176252782345,
    "latency_p95_s": 155.37855613604194,
    "mean_input_tokens": null,
    "mean_output_tokens": null,
    "charged_usd": 0.0
   },
   "presets": {},
   "rank": null,
   "source_round": "v1.3.0 public",
   "api_flag": false,
   "api_exposure_note": null,
   "public_accuracy": 0.22510822510822512,
   "sealed_accuracy": null,
   "public_minus_sealed_gap_pp": null,
   "sealed_aggregate": null,
   "v130_comparison_rank": null,
   "v130_comparison_score": 0.09466799356370086,
   "scoring_note": null,
   "rank_under": {}
  }
 ],
 "sealed_chance": 0.293,
 "sealed_weight": 0.2,
 "top_five_note": "Imajev-4B leads the JevBench Score at 67.37, ahead of Plumb-4B (65.84). The v1.4.2 scoring code and earlier measurement rows are unchanged.",
 "cost_estimate_note": "Rows marked as an estimate have no public, bookable price for the measured system. Their Cost uses the system's own measured token usage at the public list price of its base model (or the nearest listed size class, named in each row). Author-announced tariffs and free tiers are never used. A row is re-scored when a bookable price is published."
}
